#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
新津区人民政府门户网站 - 公示公告栏目爬虫
URL: https://www.xinjin.gov.cn/xjxrmzf/c1256263/zw_list.shtml
CMS: 瑞数WAF(412 JS挑战) + es-search 接口分页
破解方案: Playwright 无头浏览器执行JS挑战, 页面内 fetch 接口(自动带瑞数cookie)

列表: GET /es-search/search/6dff54b50943459f96774c6c9d5cadd2?page=N (53页/1045条/20条页)
  返回: <ul class="list-li"><li><span class="fr">2026-08-07</span><a href="..." title="...">标题</a></li>
详情: /xjxrmzf/c1256263/2026-08/14/content_xxx.shtml
  正文: div.content (保留 p/table 结构, unwrap span/font 样式)
  标题: meta ArticleTitle | 日期: meta PubDate | 来源: meta ContentSource
附件: 相对路径 urljoin(detail_url, href) 绝对化
"""
import argparse
import json
import os
import re
import sqlite3
import sys
import time
from urllib.parse import urljoin

from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

SITE_NAME = "新津区-公示公告"
BASE = "https://www.xinjin.gov.cn"
LIST_URL = "https://www.xinjin.gov.cn/xjxrmzf/c1256263/zw_list.shtml"
API_BASE = "/es-search/search/6dff54b50943459f96774c6c9d5cadd2?page="
TOTAL_PAGES = 53          # 1045条 / 20条页
PAGE_SIZE = 20
CUTOFF = "2023-08-16"     # 3年截断
DB_PATH = "/root/search.db"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"

def clean_title(t):
    """标题清洗: 实体/零宽/空白"""
    t = t.replace("&middot;", "").replace("&nbsp;", " ").replace("\u200b", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t).strip()
    return t

def clean_content_html(html_frag):
    """正文HTML清洗: 保留 p/table/a/img/br, unwrap span/font/b/strong 样式标签, 扁平化嵌套p"""
    soup = BeautifulSoup(html_frag, "html.parser")
    # 移除脚本/样式
    for tag in soup.find_all(["script", "style"]):
        tag.decompose()
    # 移除 Word 命名空间残留标签 (o:p 等)
    for tag in soup.find_all(lambda t: t.name and ":" in t.name):
        tag.decompose()
    # unwrap 纯样式标签 (保留其文本与子节点)
    for tag in soup.find_all(["span", "font"]):
        tag.unwrap()
    # 扁平化嵌套 p: 内层 p unwrap 提升内容到外层, 消除 p>p 非法嵌套
    # 只提升含嵌套 p 的外层 p (如 <p 外壳><div><p>段1</p><p>段2</p></div></p>),
    # 保留内层真段落 p — 勿 unwrap 内层 p 否则段落被拍平成一坨 (永丰修复)
    while True:
        nested = [p for p in soup.find_all("p") if p.find("p")]
        if not nested:
            break
        nested[0].unwrap()
    # 清理 p 里的 style 属性
    for p in soup.find_all("p"):
        for attr in list(p.attrs):
            if attr != "align":
                del p[attr]
    return str(soup)

def fetch_list_page(page, page_no):
    """抓列表页 N, 返回 [(date, title, href)]"""
    html = page.evaluate(
        """async (u) => {
            const r = await fetch(u, {headers: {'X-Requested-With': 'XMLHttpRequest'}});
            return await r.text();
        }""",
        API_BASE + str(page_no)
    )
    items = re.findall(r'<span class="fr">([^<]+)</span>\s*<a href="([^"]+)"[^>]*title="([^"]*)"', html)
    out = []
    for d, h, t in items:
        if not t:
            # title 属性缺失时取链接文本
            m = re.search(r'<a href="' + re.escape(h) + r'"[^>]*>([^<]+)</a>', html)
            t = m.group(1).strip() if m else ""
        out.append((d.strip(), clean_title(t), h))
    return out

def fetch_detail(page, href):
    """抓详情页, 返回 dict(title, date, source, content_html, attachments)"""
    url = href if href.startswith("http") else BASE + href
    html = page.evaluate(
        """async (u) => {
            const r = await fetch(u);
            return await r.text();
        }""",
        url
    )
    soup = BeautifulSoup(html, "html.parser")
    if "_$ts" in html and not soup.find("div", class_="content"):
        return None  # 挑战页失败

    title = ""
    m = soup.find("meta", attrs={"name": "ArticleTitle"})
    if m and m.get("content"):
        title = clean_title(m["content"])
    if not title and soup.title:
        title = clean_title(soup.title.get_text().split("-")[0])

    date = ""
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        date = m["content"].strip()
        date = re.match(r"\d{4}-\d{2}-\d{2}", date)
        date = date.group(0) if date else ""

    source = ""
    m = soup.find("meta", attrs={"name": "ContentSource"})
    if m and m.get("content"):
        source = m["content"].strip()

    content_div = soup.find("div", class_="content")
    if not content_div:
        return None
    # 收集附件 (在 content 内的 a)
    attachments = []
    for a in content_div.find_all("a", href=True):
        h = a["href"]
        if re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps|jpg|jpeg|png)(\?|$)", h, re.I):
            abs_url = urljoin(url, h)
            name = a.get_text(strip=True) or abs_url.split("/")[-1]
            attachments.append({"name": name, "url": abs_url})
            a["href"] = abs_url  # 正文内链接绝对化
    content_html = clean_content_html(str(content_div))
    return {
        "title": title,
        "date": date,
        "source": source,
        "content_html": content_html,
        "attachments": attachments,
        "url": url,
    }

def push_to_db(items):
    """直写 search.db (gov_raw 触发器自动同步 gov_search)"""
    conn = sqlite3.connect(DB_PATH, timeout=290)
    conn.execute("PRAGMA busy_timeout=290000")
    conn.execute("PRAGMA journal_mode=WAL")
    added = skipped = 0
    cur = conn.cursor()
    for it in items:
        if not it:
            skipped += 1
            continue
        # 生成附件 JSON
        att_json = json.dumps([{"name": a["name"], "url": a["url"]} for a in it["attachments"]], ensure_ascii=False)
        try:
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name, attachments)
                   VALUES (?,?,?,?,?,?,?,?,?,?)""",
                (SITE_NAME, it["url"], it["url"], it["title"], it["date"], it["content_html"],
                 (it["content_html"] or "")[:500], "公示公告", "crawl_xinjin_gsgg.py", att_json)
            )
            if cur.rowcount > 0:
                added += 1
            else:
                skipped += 1
        except sqlite3.Error as e:
            print("  DB err:", e, it["url"])
            skipped += 1
        conn.commit()
    conn.close()
    return added, skipped

def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1, help="抓取列表页数(1=测试 5=全量)")
    args = ap.parse_args()
    pages = min(args.pages, TOTAL_PAGES)

    t0 = time.time()
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--no-sandbox", "--disable-blink-features=AutomationControlled"])
        ctx = browser.new_context(user_agent=UA, viewport={"width": 1366, "height": 900})
        page = ctx.new_page()

        # 1. 加载列表页触发瑞数挑战
        print(f"[1/4] 加载列表页触发瑞数挑战...")
        page.goto(LIST_URL, timeout=45000, wait_until="domcontentloaded")
        for i in range(10):
            time.sleep(2)
            html = page.content()
            if "_$ts" not in html and len(html) > 5000:
                print(f"    挑战通过 ({i+1}轮) len={len(html)}")
                break
        else:
            print("    ✗ 挑战未通过, 退出")
            browser.close()
            return

        # 2. 抓列表
        print(f"[2/4] 抓列表页 1-{pages}...")
        all_items = []
        for pn in range(1, pages + 1):
            try:
                items = fetch_list_page(page, pn)
                all_items.extend(items)
                print(f"    P{pn}: {len(items)} 条")
            except Exception as e:
                print(f"    P{pn} err: {e}")
            time.sleep(0.8)  # 瑞数限频

        print(f"    列表共 {len(all_items)} 条 (CUTOFF {CUTOFF} 前过滤)")

        # 3. 抓详情
        print(f"[3/4] 抓详情页...")
        results = []
        n_empty = 0
        for i, (date, title, href) in enumerate(all_items, 1):
            if date < CUTOFF:
                print(f"    [{i}/{len(all_items)}] 跳过旧日期 {date} {title[:30]}")
                continue
            url = href if href.startswith("http") else BASE + href
            try:
                det = fetch_detail(page, href)
            except Exception as e:
                print(f"    [{i}/{len(all_items)}] 详情err: {e}")
                det = None
            if det is None:
                n_empty += 1
                print(f"    [{i}/{len(all_items)}] ✗ 空 {date} {title[:30]}")
            else:
                print(f"    [{i}/{len(all_items)}] ✓ {date} {det['title'][:30]} 正文{len(det['content_html'])}字 附件{len(det['attachments'])}")
            results.append(det)
            time.sleep(0.5)

        # 4. 入库
        print(f"[4/4] 入库...")
        added, skipped = push_to_db(results)
        print(f"\n=== 完成: 新增 {added} / 跳过 {skipped} / 空正文 {n_empty} / 耗时 {time.time()-t0:.0f}s ===")
        browser.close()

if __name__ == "__main__":
    main()
