#!/usr/bin/env python3
import os
"""
crawl_zsshky.py - 中山市环境保护研究院-环境公示
https://zsshky.com/html/nav_list.html?cid=402880c8600a8d0401600ad797a80028
Spring Boot, API: /List + /getArticleById, 32条
"""
import os, re, sys, time, json, urllib.request, urllib.error
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE = "https://zsshky.com/officialhome/home"
HEADERS = {
    'User-Agent': 'Mozilla/5.0',
    'Content-Type': 'application/x-www-form-urlencoded'
}

CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
print(f"[zs] 3-year cutoff: {CUTOFF}")

def fetch_json(url, method='POST'):
    data = b'' if method == 'POST' else None
    req = urllib.request.Request(url, data=data, headers=HEADERS, method=method)
    try:
        resp = urllib.request.urlopen(req, timeout=20).read().decode('utf-8', errors='replace')
        return json.loads(resp)
    except Exception as e:
        print(f"[zs] fetch fail {url}: {e}")
        return None

def get_detail(article_id):
    """Fetch article detail via API."""
    url = f"{BASE}/getArticleById?articleId={article_id}"
    data = fetch_json(url)
    if not data:
        return None, None, None
    title = data.get('articleName', '')
    content = data.get('articleContent', '')
    date_str = data.get('createDate', '')[:10] if data.get('createDate') else ''
    return title, date_str, content

# --- fetch list ---
list_url = f"{BASE}/List?columnId=402880c8600a8d0401600ad797a80028"
articles = fetch_json(list_url)
if not articles or not isinstance(articles, list):
    print("[zs] Cannot fetch article list!")
    sys.exit(1)

print(f"[zs] Total articles from API: {len(articles)}")

# --- crawl ---
import sqlite3
conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()

total_new = 0
total_skip = 0
total_old = 0
total_empty = 0

for art in articles:
    article_id = art.get('articleId', '')
    title = art.get('articleName', '')
    raw_date = art.get('createDate', '') or ''
    date_str = raw_date[:10]
    
    # date filter
    if date_str and date_str < CUTOFF:
        total_old += 1
        continue
    
    detail_url = f"https://zsshky.com/html/nav_list_content.html?articleId={article_id}"
    site_name = "中山环保研究院-环境公示"
    
    # check existing
    cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, site_name))
    if cur.fetchone():
        total_skip += 1
        continue
    
    # get detail via API
    d_title, d_date, body = get_detail(article_id)
    if not d_title: d_title = title
    if not d_date and date_str: d_date = date_str
    
    if not body:
        total_empty += 1
    
    cur.execute(
        "INSERT OR IGNORE INTO gov_raw (title, content, page_url, site_name, publish_date) VALUES (?,?,?,?,?)",
        (d_title, body or '', detail_url, site_name, d_date or date_str)
    )
    if cur.rowcount > 0:
        total_new += 1
        if total_new % 10 == 0:
            conn.commit()
    
    print(f"[zs] {total_new}: {d_title[:40] if d_title else title[:40]}...", end=' ')
    print(f"date={d_date or date_str} body={len(body)}B")
    time.sleep(0.3)

conn.commit()
conn.close()
print(f"[zs] DONE: {total_new} new, {total_skip} existing, {total_old} old, {total_empty} empty body")
