#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
唐山市人民政府 - 征询调查栏目爬虫
URL: https://www.tangshan.gov.cn/zhuzhan/zxdc/index.html
WAF: 瑞数 JS 挑战 (412) -> Playwright 过挑战
列表: /zhuzhan/zxdc/index_{N}.html (257条/26页)
详情: /zhuzhan/zxdc/20260720/1651636.html
  正文: div.main_div_text | 标题/日期/来源: meta ArticleTitle/PubDate/ContentSource
"""
import argparse
import json
import re
import sqlite3
import time

from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

SITE_NAME = "唐山市-征询调查"
BASE = "https://www.tangshan.gov.cn"
LIST_URL = "https://www.tangshan.gov.cn/zhuzhan/zxdc/index.html"
TOTAL_PAGES = 26
CUTOFF = "2023-08-16"
DB_PATH = "/root/search.db"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"

def clean_title(t):
    t = t.replace("&middot;", "").replace("&nbsp;", " ").replace("\u200b", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t).strip()
    return t

def clean_content(html_frag):
    soup = BeautifulSoup(html_frag, "html.parser")
    for tag in soup.find_all(["script", "style"]):
        tag.decompose()
    for tag in soup.find_all(lambda t: t.name and ":" in t.name):
        tag.decompose()
    for tag in soup.find_all(["span", "font"]):
        tag.unwrap()
    # 只提升含嵌套 p 的外层 p (如 <p 外壳><div><p>段1</p><p>段2</p></div></p>),
    # 保留内层真段落 p — 勿 unwrap 内层 p 否则段落被拍平成一坨 (永丰修复)
    while True:
        nested = [p for p in soup.find_all("p") if p.find("p")]
        if not nested:
            break
        nested[0].unwrap()
    return str(soup)

def fetch_list(page, pn):
    url = LIST_URL if pn == 1 else f"{BASE}/zhuzhan/zxdc/index_{pn}.html"
    html = page.evaluate("""async (u) => {
        const r = await fetch(u);
        return await r.text();
    }""", url)
    items = re.findall(r'href="(/zhuzhan/zxdc/\d+/\d+\.html)"[^>]*>([^<]{4,60})</a>', html)
    out = []
    for h, t in items:
        t = t.replace("&nbsp;", " ").strip()
        if t and not re.match(r"^(首页|上一页|下一页|尾页)", t):
            out.append((clean_title(t), h))
    return out

def fetch_detail(page, href):
    url = BASE + href
    html = page.evaluate("""async (u) => {
        const r = await fetch(u);
        return await r.text();
    }""", url)
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    m = soup.find("meta", attrs={"name": "ArticleTitle"})
    if m and m.get("content"):
        title = clean_title(m["content"])
    if not title and soup.title:
        title = clean_title(soup.title.get_text().split("-")[0])
    date = ""
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        dm = re.search(r"\d{4}-\d{2}-\d{2}", m["content"])
        date = dm.group(0) if dm else ""
    source = ""
    m = soup.find("meta", attrs={"name": "ContentSource"})
    if m and m.get("content"):
        source = m["content"].strip()

    content_div = soup.find("div", class_=re.compile("main_div_text"))
    if not content_div:
        return None
    attachments = []
    for a in content_div.find_all("a", href=True):
        h = a["href"]
        if re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps)(\?|$)", h, re.I):
            if not h.startswith("http"):
                h = BASE + h
            attachments.append({"name": a.get_text(strip=True) or h.split("/")[-1], "url": h})
            a["href"] = h
    content_html = clean_content(str(content_div))
    return {"title": title, "date": date, "source": source, "content_html": content_html,
            "attachments": attachments, "url": url}

def push_to_db(items):
    conn = sqlite3.connect(DB_PATH, timeout=290)
    conn.execute("PRAGMA busy_timeout=290000")
    added = skipped = 0
    cur = conn.cursor()
    for it in items:
        if not it:
            skipped += 1
            continue
        att_json = json.dumps([{"name": a["name"], "url": a["url"]} for a in it["attachments"]], ensure_ascii=False)
        try:
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name, attachments)
                   VALUES (?,?,?,?,?,?,?,?,?,?)""",
                (SITE_NAME, it["url"], it["url"], it["title"], it["date"], it["content_html"],
                 (it["content_html"] or "")[:500], "征询调查", "crawl_tangshan_zxdc.py", att_json)
            )
            if cur.rowcount > 0:
                added += 1
            else:
                skipped += 1
        except sqlite3.Error as e:
            print("  DB err:", e)
            skipped += 1
        conn.commit()
    conn.close()
    return added, skipped

def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1)
    args = ap.parse_args()
    pages = min(args.pages, TOTAL_PAGES)
    t0 = time.time()

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--no-sandbox", "--disable-blink-features=AutomationControlled"])
        ctx = browser.new_context(user_agent=UA, viewport={"width": 1366, "height": 900})
        page = ctx.new_page()
        print("[1/4] 过瑞数挑战...")
        page.goto(LIST_URL, timeout=45000, wait_until="domcontentloaded")
        for i in range(10):
            time.sleep(2)
            html = page.content()
            if "_$ts" not in html and len(html) > 5000:
                print(f"    挑战通过 ({i+1}轮)")
                break
        else:
            print("    ✗ 挑战未通过")
            browser.close()
            return

        print(f"[2/4] 抓列表 1-{pages}...")
        all_items = []
        for pn in range(1, pages + 1):
            try:
                items = fetch_list(page, pn)
                all_items.extend(items)
                print(f"    P{pn}: {len(items)} 条")
            except Exception as e:
                print(f"    P{pn} err: {e}")
            time.sleep(0.8)
        print(f"    列表共 {len(all_items)} 条")

        print("[3/4] 抓详情...")
        results = []
        n_empty = 0
        for i, (ltitle, href) in enumerate(all_items, 1):
            try:
                det = fetch_detail(page, href)
            except Exception as e:
                det = None
            if det is None:
                n_empty += 1
            else:
                if det["date"] and det["date"] < CUTOFF:
                    continue
                results.append(det)
            if i % 10 == 0:
                print(f"    [{i}/{len(all_items)}] 已收集 {len(results)}")
            time.sleep(0.4)

        print("[4/4] 入库...")
        added, skipped = push_to_db(results)
        print(f"\n=== 完成: 新增 {added} / 跳过 {skipped} / 空 {n_empty} / 耗时 {time.time()-t0:.0f}s ===")
        browser.close()

if __name__ == "__main__":
    main()
