#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""宣汉县人民政府 - 环评公示栏目爬虫 (xxgk-list-qthbxx.html)
站点: https://www.xuanhan.gov.cn (四川达州)
栏目: /xxgk-list-qthbxx.html (环评公示, 888条/30页)
列表: POST /api-ajax_list-{page}.html (页面内 fetch, jQuery 数组序列化 ajax_type[]=...; requests 直连返'无效操作'须 playwright session)
详情: GET /xxgk-show-{id}.html → div#content 正文 + 正文内 /uploadfile/ 附件相对路径绝对化
WAF: 无 (但 API 校验 session → 用 playwright 页面内 fetch)
入库: 自动检测 /root/search.db 存在则 INSERT OR IGNORE 直插 gov_raw (FTS 触发器自动同步 gov_search)
运行: python3 crawl_xuanhan_hpgs.py [--pages=N] [--no-db] [--sync-only]
"""
import json, os, re, sys, time, sqlite3
from datetime import datetime, timedelta
from playwright.sync_api import sync_playwright

SITE_NAME = "宣汉县-环评公示"
BASE = "https://www.xuanhan.gov.cn"
LIST_1 = "https://www.xuanhan.gov.cn/xxgk-list-qthbxx.html"
PAGE_SIZE = 30
TOTAL_PAGES = 30
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
OUT = "/tmp/xuanhan_hpgs.jsonl"
DB_PATH = "/root/search.db"
CATEGORY = "环评公示"

AJAX_BODY = ('ajax_type[]=45_xxgk&ajax_type[]=2498&ajax_type[]=45&ajax_type[]=xxgk'
             '&ajax_type[]=Y-m-d&ajax_type[]=36&ajax_type[]=30'
             '&ajax_type[]=is_top%20DESC&ajax_type[]=inputtime%20DESC'
             '&ajax_type[]=displayorder%20DESC&ajax_type[]=null&is_ds=1')

def parse_pages_arg():
    """支持 --pages=N 和纯数字"""
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            return int(a.split("=")[1])
        if a.isdigit():
            return int(a)
    return 0

def clean_title(t):
    t = re.sub(r"^[•··.\s]+", "", t or "").strip()
    t = t.replace("\u200b", "").replace("\ufeff", "")
    return t

def clean_content(html):
    """正文清洗: span/font unwrap + 嵌套p扁平化 + 保留表格"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    for tag in soup.find_all(lambda t: t.name and ":" in t.name):
        tag.decompose()
    for tag in soup.find_all(["span", "font"]):
        tag.unwrap()
    # 只提升含嵌套 p 的外层 p (如 <p 外壳><div><p>段1</p><p>段2</p></div></p>),
    # 保留内层真段落 p — 勿 unwrap 内层 p 否则段落被拍平成一坨 (永丰修复)
    while True:
        nested = [p for p in soup.find_all("p") if p.find("p")]
        if not nested:
            break
        nested[0].unwrap()
    for p in soup.find_all("p"):
        if p.get("style"):
            p.attrs = {"align": p.get("align")} if p.get("align") else {}
    return str(soup)

def extract_attachments(html_frag, detail_url):
    """正文内附件: <a href="/uploadfile/..."> + sudyfile-attr title"""
    atts = []
    for m in re.finditer(r'<a[^>]+href="([^"]+)"[^>]*>.*?</a>', html_frag, re.I | re.S):
        href = m.group(1).strip()
        if not href or href.startswith("javascript"):
            continue
        if re.search(r'\.(pdf|docx?|xlsx?|zip|rar|7z|doc)$', href, re.I):
            # 附件名: sudyfile-attr title 优先 (值内引号是 JS 转义 \'), 否则 a 文本
            m2 = re.search(r"title['\\]*:['\\]*([^'\\}]+)", m.group(0))
            if m2:
                name = m2.group(1)
            else:
                m3 = re.search(r'>([^<]{1,100})</a>', m.group(0))
                name = m3.group(1).strip() if m3 else ""
            if href.startswith("/"):
                href = BASE + href
            elif not href.startswith("http"):
                href = detail_url.rsplit("/", 1)[0] + "/" + href
            atts.append({"name": name or href.split("/")[-1], "url": href})
    seen = set()
    uniq = []
    for a in atts:
        if a["url"] not in seen:
            seen.add(a["url"])
            uniq.append(a)
    return uniq

def fetch_list(page, pn):
    """页面内 fetch 列表 API (jQuery 数组序列化 body)"""
    for _ in range(2):
        try:
            body = page.evaluate("""async (b) => {
                const r = await fetch('/api-ajax_list-' + '%d' + '.html', {
                    method: 'POST',
                    headers: {'Content-Type': 'application/x-www-form-urlencoded; charset=UTF-8', 'X-Requested-With': 'XMLHttpRequest'},
                    body: b
                });
                return await r.text();
            }""" % pn, AJAX_BODY)
            d = json.loads(body)
            if d.get("status") == 0:
                print(f"  页{pn} 无效操作 重试...", flush=True)
                time.sleep(3)
                continue
            return d
        except Exception as e:
            print(f"  页{pn} 失败: {str(e)[:80]} 重试...", flush=True)
            time.sleep(3)
    return None

def fetch_detail(page, detail_url):
    """页面内 fetch 详情页 (http→https 防混合内容拦截)"""
    fetch_url = detail_url.replace("http://", "https://")
    for _ in range(2):
        try:
            return page.evaluate("""async (u) => { const r = await fetch(u); return await r.text(); }""", fetch_url)
        except Exception:
            time.sleep(2)
    return None

def push_to_db(results):
    if not results:
        return 0, 0
    if not os.path.exists(DB_PATH):
        print(f"  未发现 {DB_PATH}, 跳过入库 (仅写 JSONL)", flush=True)
        return 0, 0
    conn = sqlite3.connect(DB_PATH, timeout=290)
    conn.execute("PRAGMA busy_timeout=290000")
    added = skipped = 0
    cur = conn.cursor()
    for r in results:
        try:
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name, attachments, group_name, industry)
                   VALUES (?,?,?,?,?,?,?,?,?,?,?,?)""",
                (SITE_NAME, r["url"], r["url"], r["title"], r["pub_date"], r["content"],
                 (r["content"] or "")[:500], CATEGORY, "crawl_xuanhan_hpgs.py", r["attachments"],
                 r.get("group_name", "四川"), r.get("industry", "other"))
            )
            if cur.rowcount > 0:
                added += 1
            else:
                skipped += 1
        except sqlite3.Error as e:
            print(f"  DB err: {e}", flush=True)
            skipped += 1
        conn.commit()
    conn.close()
    return added, skipped

def main():
    max_pages = parse_pages_arg()
    no_db = "--no-db" in sys.argv
    if "--sync-only" in sys.argv:
        print("sync-only: 抓取+入库一体, 无需单独同步", flush=True)
        return
    print(f"站点: {SITE_NAME} 最大页: {max_pages or '全量'} CUTOFF: {CUTOFF}", flush=True)

    results = []
    stats = {"total_api": 0, "kept": 0, "cutoff": 0, "empty": 0, "detail_fail": 0}
    with sync_playwright() as p:
        browser = p.chromium.launch(
            headless=True,
            args=["--no-sandbox", "--disable-dev-shm-usage"])
        ctx = browser.new_context(user_agent=UA, locale="zh-CN", viewport={"width": 1366, "height": 900}, ignore_https_errors=True)
        page = ctx.new_page()
        try:
            page.goto(LIST_1, timeout=45000, wait_until="domcontentloaded")
        except Exception:
            pass
        page.wait_for_timeout(3000)

        total_pages = max_pages if max_pages else TOTAL_PAGES
        cutoff_hit = False
        for pn in range(1, total_pages + 1):
            data = fetch_list(page, pn)
            if not data:
                print(f"  页{pn} 列表失败", flush=True)
                continue
            rows = data.get("data", [])
            total = data.get("total", 0)
            if not rows:
                break
            print(f"  页{pn}/{total_pages} {len(rows)} 条 (total {total})", flush=True)
            for item in rows:
                stats["total_api"] += 1
                title = clean_title(item.get("title", ""))
                pub_date = (item.get("inputtime") or "")[:10]
                detail_url = item.get("url", "")
                if pub_date and pub_date < CUTOFF:
                    stats["cutoff"] += 1
                    cutoff_hit = True
                    break
                detail_html = fetch_detail(page, detail_url)
                if not detail_html:
                    print(f"    详情失败: {title[:30]}", flush=True)
                    stats["detail_fail"] += 1
                    continue
                # 正文容器 div#content
                m = re.search(r'<div[^>]*id="content"[^>]*>(.*?)</div>\s*</div>', detail_html, re.S)
                content_html = m.group(1) if m else ""
                content = clean_content(content_html) if content_html else ""
                text = re.sub(r"<[^>]+>", "", content).strip()
                if len(text) < 10 and "<img" not in content:
                    print(f"    空正文跳过: {title[:30]}", flush=True)
                    stats["empty"] += 1
                    continue
                # img src 绝对化
                if "<img" in content:
                    content = re.sub(
                        r'(<img[^>]+src=")([^"]+)(")',
                        lambda mm: mm.group(1) + (mm.group(2) if mm.group(2).startswith("http") else (BASE + mm.group(2) if mm.group(2).startswith("/") else detail_url.rsplit("/", 1)[0] + "/" + mm.group(2))) + mm.group(3),
                        content)
                atts = extract_attachments(content_html, detail_url)
                rec = {
                    "site_name": SITE_NAME,
                    "title": title,
                    "pub_date": pub_date,
                    "content": content,
                    "source_url": detail_url,
                    "url": detail_url,
                    "group_name": "四川",
                    "industry": "other",
                    "attachments": json.dumps(atts, ensure_ascii=False) if atts else "",
                }
                results.append(rec)
                stats["kept"] += 1
            if cutoff_hit:
                print(f"    日期 < CUTOFF 停止", flush=True)
                break
            time.sleep(0.8)

        browser.close()

    save_results(results, stats)
    if not no_db:
        added, skipped = push_to_db(results)
        print(f"=== 入库: 新增{added} 跳过{skipped} ===", flush=True)
    print(f"=== 完成: API{stats['total_api']} 保留{stats['kept']} CUTOFF截{stats['cutoff']} 空{stats['empty']} 详情fail{stats['detail_fail']} ===", flush=True)

def save_results(results, stats=None):
    with open(OUT, "w", encoding="utf-8") as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + "\n")
    print(f"已写 {len(results)} 条到 {OUT}", flush=True)

if __name__ == "__main__":
    main()
