#!/usr/bin/env python3
"""万载县株潭镇-便民服务爬虫 (Wanzai Gov, JPaaS/政务公开系统)"""
import requests, json, os, re, sys
from datetime import datetime, timedelta

SITE_NAME = "wanzai.gov.cn-株潭镇便民服务"
API_URL = "https://www.wanzai.gov.cn/searchManuscript"
CHANNEL_ID = "1996821975816314880"
BASE_URL = "https://www.wanzai.gov.cn"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0",
    "Referer": "https://www.wanzai.gov.cn/wzxrmzf/jcbsls/pc/list.html",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
}

def fetch_page(page, page_size=10):
    data = {"current": page, "pageSize": page_size, "channelTreeIds[]": CHANNEL_ID}
    r = requests.post(API_URL, data=data, headers=HEADERS, timeout=30)
    r.raise_for_status()
    return r.json()

def crawl(incremental_days=None):
    db_path = os.getenv("SEARCH_DB", "/root/search.db")
    import sqlite3
    db = sqlite3.connect(db_path, timeout=60)
    c = db.cursor()
    c.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY, site_name TEXT, source_url TEXT,
        page_url TEXT, title TEXT, publish_date TEXT,
        date_rank INTEGER DEFAULT 0, summary TEXT, status TEXT,
        category TEXT DEFAULT '', visits INTEGER DEFAULT 0,
        content TEXT DEFAULT '', tags TEXT DEFAULT ''
    )""")
    c.execute("CREATE UNIQUE INDEX IF NOT EXISTS idx_temp_page_url ON gov_raw(page_url)")

    cutoff_date = None
    if incremental_days:
        cutoff_date = datetime.now() - timedelta(days=int(incremental_days))

    first = fetch_page(1)
    total = first["data"]["total"]
    page_size = first["data"]["rows"]
    total_pages = (total + page_size - 1) // page_size
    print(f"Total: {total} records, {total_pages} pages")

    count = 0
    for page in range(1, total_pages + 1):
        result = first if page == 1 else fetch_page(page)
        items = result["data"]["results"]
        for item in items:
            title = item.get("title", "").strip()
            if not title:
                continue
            pub_date_str = item.get("pubDate", "")
            pub_date = None
            if pub_date_str:
                for fmt in ["%Y-%m-%d %H:%M", "%Y-%m-%d"]:
                    try:
                        pub_date = datetime.strptime(pub_date_str, fmt)
                        break
                    except: pass
            if cutoff_date and pub_date and pub_date < cutoff_date:
                continue

            irn = item.get("irn", "") or ""
            page_url = BASE_URL + ("/wzxrmzf/jcbsls/detail/" + irn + ".html") if irn else (BASE_URL + "/wzxrmzf/jcbsls/pc/list.html")

            content_html = ""
            co = item.get("content", {})
            if isinstance(co, dict):
                content_html = co.get("content", "")
            elif isinstance(co, str):
                content_html = co
            if not content_html:
                print(f"  [SKIP] 空正文: {title[:30]}")
                continue

            content_html = re.sub(r'src="(?!http)(/[^"]+)"', lambda m: f'src="{BASE_URL}{m.group(1)}"', content_html)
            source_url = BASE_URL + "/wzxrmzf/jcbsls/pc/list.html"
            date_rank = int(pub_date.strftime("%Y%m%d")) if pub_date else 0

            try:
                c.execute("""INSERT OR IGNORE INTO gov_raw
                    (page_url, site_name, title, publish_date, source_url, content, category, date_rank, summary, status, visits, tags)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, '', 'published', 0, '')""",
                    (page_url, SITE_NAME, title, pub_date_str[:10] if pub_date_str else "",
                     source_url, content_html, "乡镇公示", date_rank))
                if c.rowcount > 0:
                    count += 1
            except Exception as e:
                print(f"  [ERR] {title[:30]}: {e}")

        db.commit()
        print(f"  Page {page}/{total_pages} done ({len(items)} items, +{count} new)")

    db.close()
    print(f"\nDone! {count} new records")

if __name__ == "__main__":
    incremental = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    crawl(incremental)
