#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""蒲江县人民政府 (www.pujiang.gov.cn) 环境保护栏目爬虫 — 瑞数WAF
栏目: c160163 环境保护 (空气质量日报等)
列表: /pjxrmzf/c160163/nav_list.shtml + nav_list_{N}.shtml (每页18条)
详情: /pjxrmzf/c160163/{YYYY-MM}/{DD}/content_{hash}.shtml
过盾: playwright + host-resolver-rules(IP) + 反检测; 瑞数JS动态cookie
注意: 四川蒲江县 (areaCode 510131), 与浙江浦江 pj.gov.cn 无关
用法: python3 crawl_pujiang_hjbh.py [--pages 5] [--out /tmp/pujiang_hjbh.jsonl]
"""
import argparse, json, os, re, sys, time, random
from datetime import datetime
from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

BASE = "http://www.pujiang.gov.cn"
LIST_URL = BASE + "/pjxrmzf/c160163/nav_list.shtml"
IP = "171.221.172.137"
SITE_NAME = "蒲江县-环境保护"
GROUP = "四川"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
SLEEP_MIN, SLEEP_MAX = 1.0, 1.8
MAX_FAIL = 8

KEEP = {"p", "table", "tr", "td", "th", "thead", "tbody", "a", "img", "br", "h1", "h2", "h3", "strong", "em", "b", "ul", "ol", "li"}
UNWRAP = {"span", "div", "font", "label"}
DROP = {"script", "style", "iframe", "form", "input", "button", "nav", "header", "footer"}


def clean_html(html):
    """保留段落/表格/链接, 清理噪声"""
    if not html:
        return ""
    html = re.sub(r"<!--.*?-->", "", html, flags=re.S)
    soup = BeautifulSoup(html, "html.parser")
    for tag in soup.find_all(True):
        if tag.name in DROP:
            tag.decompose()
            continue
        if tag.name == "img":
            src = tag.get("src") or tag.get("data-src") or ""
            if src:
                if src.startswith("/"):
                    src = BASE + src
                a = soup.new_tag("a")
                a["href"] = src
                a.string = "查看图片"
                tag.replace_with(a)
            else:
                tag.decompose()
            continue
        if tag.name == "a":
            href = tag.get("href") or ""
            if href.startswith("/"):
                tag["href"] = BASE + href
            elif href and not href.startswith("http"):
                tag["href"] = BASE + "/" + href
        if tag.name in UNWRAP:
            tag.unwrap()
        else:
            tag.attrs = {}
    return str(soup)


def parse_list(html):
    """列表: li > span(日期) + a[href](标题)"""
    items = []
    for m in re.finditer(r'<li[^>]*>\s*<span[^>]*>([^<]*)</span>\s*<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>', html, re.S):
        date, href, title = m.group(1).strip(), m.group(2).strip(), re.sub(r"<[^>]+>", "", m.group(3)).strip()
        if href.startswith("/"):
            href = BASE + href
        if title and "content_" in href:
            items.append({"title": title, "url": href, "date": date})
    seen, out = set(), []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            out.append(it)
    return out


def parse_detail(html):
    """详情: h1.data-title + div.publishedTime + div.source + div#NewsContent"""
    title = ""
    m = re.search(r'<h1[^>]*class="[^"]*data-title[^"]*"[^>]*>(.*?)</h1>', html, re.S)
    if m:
        title = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    date = ""
    m = re.search(r'class="[^"]*publishedTime[^"]*">日期：([^<]+)', html)
    if m:
        date = m.group(1).strip()
    source = ""
    m = re.search(r'class="[^"]*source[^"]*">来源：([^<]+)', html)
    if m:
        source = m.group(1).strip()
    body = ""
    soup = BeautifulSoup(html, "html.parser")
    nc = soup.find("div", id="NewsContent")
    if nc:
        body = clean_html(str(nc))
    # 附件
    attachments = []
    for am in re.finditer(r'<a[^>]+href="([^"]+)"[^>]*>([^<]{1,100})</a>', body):
        ah, at = am.group(1), am.group(2).strip()
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et)(\?|$)", ah, re.I):
            attachments.append({"fileName": at, "fileUrl": ah})
    return {"title": title, "date": date, "source": source, "body": body, "attachments": attachments}


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=5, help="前N页")
    ap.add_argument("--out", default="/tmp/pujiang_hjbh.jsonl")
    args = ap.parse_args()

    done_urls = set()
    if os.path.exists(args.out):
        for line in open(args.out, encoding="utf-8"):
            try:
                done_urls.add(json.loads(line)["page_url"])
            except Exception:
                pass
    print(f"[*] 已有 {len(done_urls)} 条, 断点续传", flush=True)

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled", "--no-sandbox",
            f"--host-resolver-rules=MAP www.pujiang.gov.cn {IP}, MAP pujiang.gov.cn {IP}",
        ])
        ctx = browser.new_context(
            user_agent=UA, locale="zh-CN",
            viewport={"width": 1920, "height": 1080},
            extra_http_headers={"Accept-Language": "zh-CN,zh;q=0.9"},
        )
        page = ctx.new_page()

        # 1. 列表前N页 (瑞数过盾)
        all_items = []
        for pno in range(1, args.pages + 1):
            url = LIST_URL if pno == 1 else LIST_URL.replace("nav_list.shtml", f"nav_list_{pno}.shtml")
            try:
                page.goto(url, timeout=10000, wait_until="domcontentloaded")
            except Exception:
                pass
            html = ""
            for i in range(10):
                time.sleep(2)
                html = page.content()
                if "_ts" not in html and len(html) > 5000:
                    break
            items = parse_list(html)
            new_items = [it for it in items if it["url"] not in done_urls]
            all_items.extend(items)
            print(f"[list] p{pno}: {len(items)} 条(新{len(new_items)})", flush=True)
            time.sleep(random.uniform(1.0, 1.5))

        print(f"[*] 列表共 {len(all_items)} 条, 开始抓详情", flush=True)
        f = open(args.out, "a", encoding="utf-8")
        ok = fail = 0
        for it in all_items:
            if it["url"] in done_urls:
                continue
            try:
                page.goto(it["url"], timeout=10000, wait_until="domcontentloaded")
            except Exception:
                pass
            html = ""
            for _r in range(6):
                time.sleep(1.5)
                try:
                    html = page.content()
                    if len(html) > 3000:
                        break
                except Exception:
                    continue
            det = parse_detail(html)
            if not det["body"] or len(det["body"]) < 10:
                fail += 1
                if fail >= MAX_FAIL:
                    print("[!] 空正文过多, 中止", flush=True)
                    break
                print(f"[warn] 空正文 {it['url'][-50:]}", flush=True)
            else:
                fail = 0
            rec = {
                "title": det["title"] or it["title"],
                "publish_date": det["date"] or it["date"],
                "content": det["body"],
                "page_url": it["url"],
                "source_url": it["url"],
                "site_name": SITE_NAME,
                "summary": re.sub(r"<[^>]+>", " ", det["body"]).strip()[:200],
                "date_rank": 0,
                "author": det["source"],
                "content_source": det["source"],
                "attachments": det["attachments"],
                "crawl_time": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
            }
            f.write(json.dumps(rec, ensure_ascii=False) + "\n")
            f.flush()
            done_urls.add(it["url"])
            ok += 1
            if ok % 20 == 0:
                print(f"[detail] +{ok} 条", flush=True)
            time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))
        f.close()
        browser.close()
    print(f"[✓] 完成: 本次新增 {ok} 条 -> {args.out}", flush=True)


if __name__ == "__main__":
    main()
