#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""上饶市广信区人民政府 (www.srx.gov.cn) 行政许可其他管理服务信息 — UCAP zoom
栏目: /srx/xzxkhqtglfwxx/gxzwgk_xxgklist.shtml (可选 ?query=关键词 过滤)
列表: gxzwgk_xxgklist.shtml + gxzwgk_xxgklist_{N}.shtml (每页20条)
详情: /{部门}/xzxkhqtglfwxx/{YYYYMM}/{hash}.shtml, 正文 ucapcontent 容器
用法:
  python3 crawl_srx_xzxk.py --pages 5 --out /tmp/srx_xzxk.jsonl
  python3 crawl_srx_xzxk.py --query 环境影响评价第一次 --pages 5 --out /tmp/srx_xzxk_1st.jsonl
"""
import argparse, json, os, re, sys, time, random, ssl
from urllib.request import Request, urlopen
from urllib.error import HTTPError, URLError
from urllib.parse import urljoin, quote
from bs4 import BeautifulSoup

SSL_CTX = ssl.create_default_context()
SSL_CTX.check_hostname = False
SSL_CTX.verify_mode = ssl.CERT_NONE

BASE = "https://www.srx.gov.cn"
LIST_PATH = "/srx/xzxkhqtglfwxx/gxzwgk_xxgklist.shtml"
SITE_NAME_BASE = "上饶广信区-行政许可"
GROUP = "江西"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
HEADERS = {
    "User-Agent": UA,
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
SLEEP_MIN, SLEEP_MAX = 1.0, 1.6
MAX_FAIL = 10


def fetch(url, referer=None, timeout=25, retries=3):
    headers = dict(HEADERS)
    headers["Referer"] = referer or url
    for attempt in range(retries):
        try:
            req = Request(url, headers=headers)
            resp = urlopen(req, timeout=timeout, context=SSL_CTX)
            return resp.read().decode("utf-8", errors="ignore")
        except HTTPError as e:
            if e.code in (404, 410):
                return ""
            print(f"  [HTTP {e.code}] {url[-60:]}", flush=True)
            time.sleep(3 + attempt * 3)
        except (URLError, TimeoutError, OSError) as e:
            print(f"  [ERR {type(e).__name__}] {url[-60:]}", flush=True)
            time.sleep(3 + attempt * 3)
    return ""


def clean_html(html, base_url=None):
    base = base_url or BASE
    soup = BeautifulSoup(html, "html.parser")
    for t in soup(["script", "style", "iframe", "object", "embed", "form", "input", "button"]):
        t.decompose()
    for img in soup.find_all("img"):
        src = img.get("src") or img.get("data-src") or ""
        if src:
            abs_src = src if src.startswith("http") else urljoin(base, src)
            a = soup.new_tag("a", href=abs_src)
            a.string = "查看图片"
            img.replace_with(a)
        else:
            img.decompose()
    for t in soup.find_all(True):
        for attr in ("style", "class", "align", "valign", "border", "cellpadding", "cellspacing",
                     "width", "height", "bgcolor", "face", "color", "size", "lang", "dir",
                     "setedaria", "tabindex", "role"):
            if attr in t.attrs:
                del t[attr]
        for attr in list(t.attrs):
            if attr.startswith("aria-") or attr.startswith("data-"):
                del t[attr]
    for t in soup.find_all(["div", "span", "font", "center", "b", "strong", "em", "i", "u", "s", "label"]):
        t.unwrap()
    for a in soup.find_all("a"):
        href = a.get("href", "")
        if href and not href.startswith("javascript"):
            a["href"] = href if href.startswith("http") else urljoin(base, href)
        elif href:
            a.decompose()
    for p in soup.find_all("p"):
        if not p.get_text(strip=True) and not p.find("table") and not p.find("a"):
            p.decompose()
    return str(soup)


def parse_list_html(html):
    """列表: a[href=/XX/xzxkhqtglfwxx/YYYYMM/hash.shtml]"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        m = re.search(r"/([a-z0-9]+)/xzxkhqtglfwxx/(\d{6})/([0-9a-f]+)\.shtml$", href)
        if not m:
            continue
        title = (a.get("title") or a.get_text(strip=True) or "").strip()
        date = m.group(2)
        date = f"{date[:4]}-{date[4:6]}-01"
        if title:
            items.append({"title": title, "url": urljoin(BASE, href), "date": date})
    seen, out = set(), []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            out.append(it)
    return out


def parse_detail_html(html, url):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    def meta(name):
        m = soup.find("meta", attrs={"name": name}) or soup.find("meta", attrs={"name": name.lower()})
        return (m.get("content") or "").strip() if m else ""
    title = meta("ArticleTitle") or ""
    pub = meta("PubDate") or ""
    source = meta("ContentSource") or ""
    zoom = (soup.find("div", id="content") or soup.find("div", class_="content") or
            soup.find("div", id="zoom") or soup.find("div", class_="zoom") or
            soup.find("div", class_="TRS_Editor") or soup.find("td", id="content"))
    if zoom:
        for path in zoom.find_all("div", class_="path"):
            path.decompose()
        uc = zoom.find("ucapcontent") or zoom.find("div", class_="ucapcontent") or zoom.find("article")
        mc = zoom.find("div", class_="mainContent") or zoom.find("div", class_="mainBox")
        if uc:
            zoom = uc
        elif mc:
            zoom = mc
        for t in zoom.find_all(["h1", "h2", "h3"]):
            if t.find_parent("table") is None:
                t.decompose()
        for el in zoom.find_all(["publishtime", "font"]):
            el.decompose()
    body = ""
    att_links = []
    if zoom:
        for a in zoom.find_all("a", href=True):
            h = a["href"].lower()
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|caj|wps|et|jpg|jpeg|png|gif|mp4)$", h) or "download" in h:
                t = a.get_text(strip=True) or "附件"
                fu = h if h.startswith("http") else urljoin(url, h)
                att_links.append((t, fu))
                a.decompose()
        body = clean_html(str(zoom), url)
        body = re.sub(r"<p>\s*您的位置：.*?</p>", "", body, flags=re.S)
        body = re.sub(r"您的位置：.*?&gt;", "", body, flags=re.S)
        if att_links:
            att_p = "".join(f'<p><a href="{fu}">{t}</a></p>' for t, fu in att_links)
            body = (body + "\n" + att_p).strip()
    if not body:
        ps = soup.find_all("p")
        cands = []
        for p in ps:
            txt = p.get_text(strip=True)
            if len(txt) > 30:
                cands.append((len(txt), p))
        if cands:
            best = max(cands, key=lambda x: x[0])[1]
            body = clean_html(str(best), url)
    return {
        "title": title,
        "publish_date": pub[:10] if pub else "",
        "content": body,
        "source": source,
        "attachments": [{"fileName": t, "fileUrl": fu} for t, fu in att_links],
    }


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--query", default="", help="搜索关键词过滤")
    ap.add_argument("--pages", type=int, default=0, help="0=全量")
    ap.add_argument("--start-page", type=int, default=1)
    ap.add_argument("--out", default="/tmp/srx_xzxk.jsonl")
    ap.add_argument("--max-detail-fail", type=int, default=15)
    args = ap.parse_args()

    site_name = SITE_NAME_BASE
    if args.query:
        site_name = SITE_NAME_BASE + "-" + args.query
    LIST_URL = BASE + LIST_PATH
    if args.query:
        LIST_URL += "?query=" + quote(args.query)

    stats = {"items": 0, "detail_ok": 0, "detail_fail": 0, "empty": 0, "skipped": 0}
    out_fp = open(args.out, "a", encoding="utf-8")
    done_urls = set()
    if os.path.exists(args.out):
        for line in open(args.out, encoding="utf-8"):
            try:
                done_urls.add(json.loads(line)["page_url"])
            except Exception:
                pass

    html = fetch(LIST_URL)
    if not html:
        print("首列表页抓取失败", flush=True)
        return
    pages_total = 100
    m = re.search(r"共\s*(\d+)\s*页", html)
    if m:
        pages_total = int(m.group(1))
    if args.pages > 0:
        pages_total = min(pages_total, args.pages)
    print(f"site_name={site_name} 总页数={pages_total}", flush=True)

    all_items = []
    for pg in range(args.start_page, pages_total + 1):
        if pg == 1:
            url = LIST_URL
        else:
            base_no_ext = LIST_PATH.replace(".shtml", "")
            url = f"{BASE}{base_no_ext}_{pg}.shtml"
            if args.query:
                url += "?query=" + quote(args.query)
        h = fetch(url, referer=LIST_URL)
        if not h:
            print(f"页{pg} 抓取失败, 停止翻页", flush=True)
            break
        items = parse_list_html(h)
        all_items.extend(items)
        print(f"页{pg}: {len(items)} 条, 累计 {len(all_items)}", flush=True)
        time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))
    stats["items"] = len(all_items)
    print(f"列表完成: {len(all_items)} 条 (已抓过 {len(done_urls)})", flush=True)

    new_items = [it for it in all_items if it["url"] not in done_urls]
    print(f"待抓详情: {len(new_items)}", flush=True)
    for idx, it in enumerate(new_items, 1):
        html = fetch(it["url"])
        if not html:
            stats["detail_fail"] += 1
            print(f"详情 {idx}/{len(new_items)}: 抓取失败 {it['url']}", flush=True)
            if stats["detail_fail"] >= args.max_detail_fail:
                print("失败过多, 中止", flush=True)
                break
            continue
        d = parse_detail_html(html, it["url"])
        if not d["title"]:
            d["title"] = it["title"]
        if not d["publish_date"]:
            d["publish_date"] = it["date"]
        content = d["content"]
        if len(content) < 10:
            stats["empty"] += 1
        rec = {
            "title": d["title"],
            "publish_date": d["publish_date"],
            "content": content,
            "page_url": it["url"],
            "source_url": it["url"],
            "site_name": site_name,
            "group_name": GROUP,
            "summary": re.sub(r"<[^>]+>", " ", content).strip()[:200],
            "date_rank": 0,
            "author": d["source"],
            "content_source": d["source"],
            "attachments": d["attachments"],
            "category": "行政许可",
            "industry": "other",
            "script_name": "crawl_srx_xzxk.py",
        }
        out_fp.write(json.dumps(rec, ensure_ascii=False) + "\n")
        out_fp.flush()
        stats["detail_ok"] += 1
        if idx % 20 == 0 or idx == len(new_items):
            print(f"详情 {idx}/{len(new_items)}: ok={stats['detail_ok']} fail={stats['detail_fail']}", flush=True)
        time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))
    out_fp.close()
    print(f"完成: {json.dumps(stats, ensure_ascii=False)}", flush=True)


if __name__ == "__main__":
    main()
