#!/usr/bin/env python3
"""
鹤山市 gkmlpt - 优化的爬虫
Step 1: 快速收集API列表（仅URL和标题日期）
Step 2: 抓取详情页内容
"""
import os, re, sys, time, json
import requests, sqlite3
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

SITE_MAP = {
    "龙口镇": {"sid": "750257", "name": "江门鹤山市龙口镇人民政府", "category": "信息公开"},
    "共和镇": {"sid": "750259", "name": "江门鹤山市共和镇人民政府", "category": "信息公开"},
    "桃源镇": {"sid": "750258", "name": "江门鹤山市桃源镇人民政府", "category": "信息公开"},
}

def search_api(sid, page=1, pagesize=100):
    url = f"https://search.gd.gov.cn/jsonp/site/{sid}"
    params = {"callback":"jsonp","page":page,"pagesize":pagesize,"isgkml":1,"text":"","order":0,
              "including_url_doc":1,"including_attach_doc":1,"position":"","classify_mains_excluded":""}
    r = requests.get(url, params=params, headers=HEADERS, timeout=30)
    text = r.text
    if text.startswith("jsonp("):
        text = text[6:-1]
    return json.loads(text)

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
    except:
        return None, "", None
    soup = BeautifulSoup(r.text, 'html.parser')
    meta_t = soup.find("meta", attrs={"name": "ArticleTitle"})
    title = meta_t["content"].strip() if (meta_t and meta_t.get("content")) else None
    if not title:
        t = soup.find("title")
        title = t.get_text(strip=True) if t else "无标题"
        title = re.sub(r'[-_\s]*鹤山市.*', '', title).strip()
    date_str = None
    meta_d = soup.find("meta", attrs={"name": "PubDate"})
    if meta_d and meta_d.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_d["content"])
        if m: date_str = m.group(1)
    content_div = soup.select_one("div#zoom") or soup.find("div", class_=re.compile(r'(?i)content'))
    content_html = ""
    if content_div:
        for t in content_div.find_all(['script','style']): t.decompose()
        content_html = str(content_div).strip()
    return title, content_html, date_str

def main():
    if len(sys.argv) < 2 or sys.argv[1] not in SITE_MAP:
        print(f"Usage: python3 crawl_heshan_gkmlpt.py [{'|'.join(SITE_MAP.keys())}]")
        sys.exit(1)
    
    cfg = SITE_MAP[sys.argv[1]]
    SID, SITE_NAME, CATEGORY = cfg["sid"], cfg["name"], cfg["category"]
    print(f"Site: {SITE_NAME} (SID={SID})")
    
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    existing = set(r[0] for r in cur.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)))
    print(f"Existing: {len(existing)}")
    
    # Step 1: Collect API results
    data = search_api(SID, 1, 1)
    total = data.get("count", 0)
    total_pages = (total + 99) // 100
    print(f"Total: {total}, pages: {total_pages}")
    
    to_fetch = []  # (url, title, api_date)
    seen = set()
    for page in range(1, total_pages + 1):
        data = search_api(SID, page, 100)
        for item in data.get("results", []):
            url = item.get("url","").strip()
            if not url or url in seen or url in existing:
                continue
            seen.add(url)
            pub = item.get("pub_time") or item.get("publish_time") or ""
            d = ""
            m = re.search(r"(\d{4}-\d{2}-\d{2})", pub)
            if m: d = m.group(1)
            if d and d < CUTOFF:
                continue
            title = item.get("title","").replace("&ldquo;","\u201c").replace("&rdquo;","\u201d").replace("&mdash;","\u2014")
            to_fetch.append((url, title, d))
        print(f"  Page {page}/{total_pages}: {len(to_fetch)} collected so far", flush=True)
    
    print(f"Total to fetch: {len(to_fetch)}", flush=True)
    
    # Step 2: Fetch details
    inserted = 0
    for i, (url, title_api, api_date) in enumerate(to_fetch):
        title, content_html, detail_date = fetch_detail(url)
        final_title = title or title_api
        final_date = detail_date or api_date
        if final_date and final_date < CUTOFF:
            continue
        date_rank = 0
        if final_date:
            try: date_rank = int(datetime.strptime(final_date, "%Y-%m-%d").timestamp())
            except: pass
        try:
            cur.execute("INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                       (SITE_NAME, final_title, url, final_date, content_html, date_rank, CATEGORY))
            if cur.rowcount > 0: inserted += 1
        except Exception as e:
            print(f"  [ERR] {e}", flush=True)
        if (i+1) % 50 == 0:
            conn.commit()
            print(f"  Progress: {i+1}/{len(to_fetch)}, inserted: {inserted}", flush=True)
    
    conn.commit()
    print(f"\nDone: inserted {inserted} for {SITE_NAME}", flush=True)
    if inserted > 0:
        print("Rebuilding FTS...", flush=True)
        try:
            cur.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
            conn.commit()
            print("  FTS rebuilt OK", flush=True)
        except Exception as e:
            print(f"  [WARN] FTS: {e}", flush=True)
    conn.close()

if __name__ == "__main__":
    main()
