#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""上饶市广信区人民政府 (www.srx.gov.cn) 环评公示 — UCAP zoom
注意: srx.gov.cn = 上饶广信区 (不是上饶市), 路径 /srx/hpgs/
列表: /srx/hpgs/list.shtml + list_{N}.shtml (每页10条)
详情: /srx/hpgs/{YYYYMM}/{hash}.shtml, 正文 id=content 容器
用法: python3 crawl_srx_hpgs.py [--pages 5] [--out /tmp/srx_hpgs.jsonl]
"""
import argparse, json, os, re, sys, time, random, ssl
from urllib.request import Request, urlopen
from urllib.error import HTTPError, URLError
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SSL_CTX = ssl.create_default_context()
SSL_CTX.check_hostname = False
SSL_CTX.verify_mode = ssl.CERT_NONE

BASE = "https://www.srx.gov.cn"
LIST_URL = BASE + "/srx/hpgs/list.shtml"
SITE_NAME = "上饶广信区-环评公示"
GROUP = "江西"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
HEADERS = {
    "User-Agent": UA,
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
SLEEP_MIN, SLEEP_MAX = 1.0, 1.6
MAX_FAIL = 10


def fetch(url, referer=LIST_URL, timeout=25, retries=3):
    headers = dict(HEADERS)
    headers["Referer"] = referer
    for attempt in range(retries):
        try:
            req = Request(url, headers=headers)
            resp = urlopen(req, timeout=timeout, context=SSL_CTX)
            return resp.read().decode("utf-8", errors="ignore")
        except HTTPError as e:
            if e.code in (404, 410):
                return ""
            print(f"  [HTTP {e.code}] {url[-60:]}", flush=True)
            time.sleep(3 + attempt * 3)
        except (URLError, TimeoutError, OSError) as e:
            print(f"  [ERR {type(e).__name__}] {url[-60:]}", flush=True)
            time.sleep(3 + attempt * 3)
    return ""


def clean_html(html, base_url=None):
    base = base_url or BASE
    soup = BeautifulSoup(html, "html.parser")
    for t in soup.find_all(["script", "style", "iframe", "object", "embed", "form", "input", "button"]):
        t.decompose()
    # Word 命名空间标签 (o:p 等) 清理
    for t in soup.find_all(re.compile(r"^[a-zA-Z]+:[a-zA-Z]+$")):
        t.unwrap() if not t.get_text(strip=True) else t.decompose()
    # ucapcontent 包裹标签 (UCAP CMS 序列化残留) 剥掉, 保留内部内容
    for t in soup.find_all("ucapcontent"):
        t.unwrap()
    for img in soup.find_all("img"):
        src = img.get("src") or img.get("data-src") or ""
        if src:
            abs_src = src if src.startswith("http") else urljoin(base, src)
            a = soup.new_tag("a", href=abs_src)
            a.string = "查看图片"
            img.replace_with(a)
        else:
            img.decompose()
    for t in soup.find_all(True):
        for attr in ("style", "class", "align", "valign", "border", "cellpadding", "cellspacing",
                     "width", "height", "bgcolor", "face", "color", "size", "lang", "dir",
                     "setedaria", "tabindex", "role"):
            if attr in t.attrs:
                del t[attr]
        for attr in list(t.attrs):
            if attr.startswith("aria-") or attr.startswith("data-"):
                del t[attr]
    for t in soup.find_all(["div", "span", "font", "center", "b", "strong", "em", "i", "u", "s", "label"]):
        t.unwrap()
    for a in soup.find_all("a"):
        href = a.get("href", "")
        if href and not href.startswith("javascript"):
            a["href"] = href if href.startswith("http") else urljoin(base, href)
        elif href:
            a.decompose()
    for p in soup.find_all("p"):
        if not p.get_text(strip=True) and not p.find("table") and not p.find("a"):
            p.decompose()
    return str(soup)


def parse_list_html(html):
    """列表: a[href=/srx/hpgs/YYYYMM/hash.shtml]"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        m = re.search(r"/srx/hpgs/(\d{6})/([0-9a-f]+)\.shtml$", href)
        if not m:
            continue
        title = (a.get("title") or a.get_text(strip=True) or "").strip()
        date = m.group(1)
        date = f"{date[:4]}-{date[4:6]}-01"
        if title:
            items.append({"title": title, "url": urljoin(BASE, href), "date": date})
    seen, out = set(), []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            out.append(it)
    return out


def parse_detail_html(html, url):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    def meta(name):
        m = soup.find("meta", attrs={"name": name}) or soup.find("meta", attrs={"name": name.lower()})
        return (m.get("content") or "").strip() if m else ""
    title = meta("ArticleTitle") or ""
    pub = meta("PubDate") or ""
    source = meta("ContentSource") or ""
    zoom = (soup.find("div", id="content") or soup.find("div", class_="content") or
            soup.find("div", id="zoom") or soup.find("div", class_="zoom") or
            soup.find("div", class_="zwcon") or soup.find("div", class_="zwcontent") or
            soup.find("div", class_="TRS_Editor") or soup.find("td", id="content"))
    if zoom:
        # 删面包屑 div.path
        for path in zoom.find_all("div", class_="path"):
            path.decompose()
        # 优先 ucapcontent (纯正文) > mainContent/mainBox
        uc = zoom.find("ucapcontent") or zoom.find("div", class_="ucapcontent") or zoom.find("article")
        mc = zoom.find("div", class_="mainContent") or zoom.find("div", class_="mainBox")
        if uc:
            zoom = uc
        elif mc:
            zoom = mc
        # 去掉文章标题块 (h1/h2 在正文前)
        for t in zoom.find_all(["h1", "h2", "h3"]):
            if t.find_parent("table") is None:
                t.decompose()
        # 去掉 发布时间/字体 噪声
        for el in zoom.find_all("publishtime"):
            el.decompose()
        for m in re.findall(r"【字体[^】]*】", str(zoom)):
            pass
    body = ""
    att_links = []
    if zoom:
        for a in zoom.find_all("a", href=True):
            h = a["href"].lower()
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|caj|wps|et|jpg|jpeg|png|gif|mp4)$", h) or "download" in h:
                t = a.get_text(strip=True) or "附件"
                fu = h if h.startswith("http") else urljoin(url, h)
                att_links.append((t, fu))
                a.decompose()
        body = clean_html(str(zoom), url)
        # 去掉面包屑段落 (您的位置： 首页 > ...)
        body = re.sub(r"<p>\s*您的位置：.*?</p>", "", body, flags=re.S)
        body = re.sub(r"您的位置：.*?&gt;", "", body, flags=re.S)
        if att_links:
            att_p = "".join(f'<p><a href="{fu}">{t}</a></p>' for t, fu in att_links)
            body = (body + "\n" + att_p).strip()
    if not body:
        ps = soup.find_all("p")
        cands = []
        for p in ps:
            txt = p.get_text(strip=True)
            if len(txt) > 30:
                cands.append((len(txt), p))
        if cands:
            best = max(cands, key=lambda x: x[0])[1]
            body = clean_html(str(best), url)
    return {
        "title": title,
        "publish_date": pub[:10] if pub else "",
        "content": body,
        "source": source,
        "attachments": [{"fileName": t, "fileUrl": fu} for t, fu in att_links],
    }


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=0, help="0=全量")
    ap.add_argument("--start-page", type=int, default=1)
    ap.add_argument("--out", default="/tmp/srx_hpgs.jsonl")
    ap.add_argument("--max-detail-fail", type=int, default=15)
    args = ap.parse_args()

    stats = {"items": 0, "detail_ok": 0, "detail_fail": 0, "empty": 0, "skipped": 0}
    out_fp = open(args.out, "a", encoding="utf-8")
    done_urls = set()
    if os.path.exists(args.out):
        for line in open(args.out, encoding="utf-8"):
            try:
                done_urls.add(json.loads(line)["page_url"])
            except Exception:
                pass

    html = fetch(LIST_URL)
    if not html:
        print("首列表页抓取失败", flush=True)
        return
    pages_total = 100
    m = re.search(r"共\s*(\d+)\s*页", html)
    if m:
        pages_total = int(m.group(1))
    if args.pages > 0:
        pages_total = min(pages_total, args.pages)
    print(f"总页数={pages_total}", flush=True)

    all_items = []
    for pg in range(args.start_page, pages_total + 1):
        url = LIST_URL if pg == 1 else f"{LIST_URL.replace('.shtml','')}_{pg}.shtml"
        h = fetch(url)
        if not h:
            print(f"页{pg} 抓取失败, 停止翻页", flush=True)
            break
        items = parse_list_html(h)
        all_items.extend(items)
        print(f"页{pg}: {len(items)} 条, 累计 {len(all_items)}", flush=True)
        time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))
    stats["items"] = len(all_items)
    print(f"列表完成: {len(all_items)} 条 (已抓过 {len(done_urls)})", flush=True)

    new_items = [it for it in all_items if it["url"] not in done_urls]
    print(f"待抓详情: {len(new_items)}", flush=True)
    for idx, it in enumerate(new_items, 1):
        html = fetch(it["url"])
        if not html:
            stats["detail_fail"] += 1
            print(f"详情 {idx}/{len(new_items)}: 抓取失败 {it['url']}", flush=True)
            if stats["detail_fail"] >= args.max_detail_fail:
                print("失败过多, 中止", flush=True)
                break
            continue
        d = parse_detail_html(html, it["url"])
        if not d["title"]:
            d["title"] = it["title"]
        if not d["publish_date"]:
            d["publish_date"] = it["date"]
        content = d["content"]
        if len(content) < 10:
            stats["empty"] += 1
        rec = {
            "title": d["title"],
            "publish_date": d["publish_date"],
            "content": content,
            "page_url": it["url"],
            "source_url": it["url"],
            "site_name": SITE_NAME,
            "group_name": GROUP,
            "summary": re.sub(r"<[^>]+>", " ", content).strip()[:200],
            "date_rank": 0,
            "author": d["source"],
            "content_source": d["source"],
            "attachments": d["attachments"],
            "category": "环评",
            "industry": "other",
            "script_name": "crawl_srx_hpgs.py",
        }
        out_fp.write(json.dumps(rec, ensure_ascii=False) + "\n")
        out_fp.flush()
        stats["detail_ok"] += 1
        if idx % 20 == 0 or idx == len(new_items):
            print(f"详情 {idx}/{len(new_items)}: ok={stats['detail_ok']} fail={stats['detail_fail']}", flush=True)
        time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))
    out_fp.close()
    print(f"完成: {json.dumps(stats, ensure_ascii=False)}", flush=True)


if __name__ == "__main__":
    main()
