#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""温宿县人民政府 通知公告 爬虫 (wszf.gov.cn)
- 列表: POST /xjcms/openapi/t/info/list.do (参数 base64 编码, channelid=1250, 25条/页, 共25页616条)
- 详情: https://www.wszf.gov.cn/zwgk/tzgg/YYYYMMDD/i{iid}.html
- 正文: div#zoom, 标题 h1.main_5121, 日期 li.main_51221, 来源 li.main_51222
- 图片/附件相对路径转绝对 URL
用法: python3 crawl_wszf_tzgg.py [--pages N] [--out FILE]
"""
import argparse, base64, json, re, sys, time, urllib.request, urllib.parse, ssl, hashlib

HOST = "www.wszf.gov.cn"
BASE = f"https://{HOST}"
CHANNELID = "1250"
PAGE_SIZE = 25

UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
HEADERS = {
    "Host": HOST,
    "User-Agent": UA,
    "Referer": f"https://{HOST}/zwgk/tzgg/index.html",
    "Content-Type": "application/x-www-form-urlencoded",
}

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE


def fetch(url, data=None, timeout=20):
    req = urllib.request.Request(url, headers=HEADERS, data=data)
    with urllib.request.urlopen(req, timeout=timeout, context=ctx) as r:
        return r.read().decode("utf-8", "replace")


def get_list(pageno):
    params = {"channelid": CHANNELID, "pageno": str(pageno), "pagesize": str(PAGE_SIZE)}
    enc = {k: base64.b64encode(str(v).encode()).decode() for k, v in params.items()}
    body = urllib.parse.urlencode(enc).encode()
    raw = fetch(f"{BASE}/xjcms/openapi/t/info/list.do", data=body)
    return json.loads(raw)


def fetch_detail(url):
    """返回 (html, title, date, source)"""
    html = fetch(url)
    m = re.search(r'<h1[^>]*class="[^"]*main_5121[^"]*"[^>]*>\s*(.*?)\s*</h1>', html, re.S)
    title = re.sub(r'<[^>]+>', '', m.group(1)).strip() if m else None
    m = re.search(r'发布日期[：:]\s*([0-9\-: ]+)', html)
    date = m.group(1).strip() if m else None
    m = re.search(r'来源[：:]\s*([^<\s][^<]{0,30})', html)
    source = m.group(1).strip() if m else None
    return html, title, date, source


def clean_body(html):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    zoom = soup.find('div', id='zoom')
    if not zoom:
        return None
    # 正文 = zoom 的第一个直接子 div (main_51231); 分享/返回顶部等在 zoom 内其他位置
    container = None
    for child in zoom.find_all('div', recursive=False):
        container = child
        break
    if container is None:
        container = zoom
    for s in container.find_all(['script', 'style']):
        s.decompose()
    for a in container.find_all('a', href=True):
        if a['href'].startswith('javascript:'):
            a.decompose()
    # unwrap 纯包装 div (无 id/class 且无嵌套 div 的)
    for div in list(container.find_all('div')):
        if not div.get('id') and not div.get('class') and not div.find('div', recursive=False) and not div.find('table'):
            div.unwrap()
    seg = ''.join(str(c) for c in container.children)
    seg = re.sub(r'<!--.*?-->', '', seg, flags=re.S)
    seg = re.sub(r'<script.*?</script>', '', seg, flags=re.S)
    seg = re.sub(r'<style.*?</style>', '', seg, flags=re.S)
    # 相对路径转绝对
    seg = re.sub(r'(src|href)="(/DFS/)', rf'\1="https://{HOST}\2', seg)
    seg = re.sub(r'(src|href)="(/xjcms/)', rf'\1="https://{HOST}\2', seg)
    # 清 inline style + 无障碍属性 (保留 p/table/a/img/td)
    seg = re.sub(r'\sstyle="[^"]*"', '', seg)
    seg = re.sub(r'\s(width|height)="[^"]*"', '', seg)
    for attr in ['title', 'alt', 'align', 'border', 'cellspacing', 'cellpadding', 'class', 'id', 'name', 'target']:
        seg = re.sub(rf'\s{attr}="[^"]*"', '', seg)
    seg = re.sub(r'<br\s*/?>', '<br/>', seg)
    seg = re.sub(r'<span[^>]*>|</span>', '', seg)
    seg = re.sub(r'<strong[^>]*>|</strong>', '', seg)
    seg = re.sub(r'<font[^>]*>|</font>', '', seg)
    seg = re.sub(r'\s+', ' ', seg)
    return seg.strip()


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=25)
    ap.add_argument("--out", default="/tmp/wszf_tzgg.jsonl")
    args = ap.parse_args()

    out = open(args.out, "a", encoding="utf-8")
    seen = set()
    try:
        for line in open(args.out, encoding="utf-8"):
            d = json.loads(line)
            seen.add(d["url"])
    except FileNotFoundError:
        pass

    total_ok = 0
    for p in range(1, args.pages + 1):
        try:
            data = get_list(p)
        except Exception as e:
            print(f"[list] p{p} ERR {e}", flush=True)
            time.sleep(3)
            continue
        items = data.get("infolist", [])
        if not items:
            print(f"[list] p{p}: 空, 结束", flush=True)
            break
        print(f"[list] p{p}: {len(items)} 条", flush=True)
        for it in items:
            iid = it.get("iid")
            title_list = re.sub(r'<[^>]+>', '', it.get("title") or "").strip()
            ts = it.get("releasetime") or it.get("releaseTime") or 0
            date = time.strftime("%Y-%m-%d", time.localtime(ts / 1000)) if ts else ""
            url = it.get("url") or f"https://{HOST}/zwgk/tzgg/{time.strftime('%Y%m%d', time.localtime(ts/1000))}/i{iid}.html"
            if url.startswith("//"):
                url = "https:" + url
            if url in seen:
                continue
            try:
                html, title, ddate, source = fetch_detail(url)
                body = clean_body(html)
                if not body or len(body) < 50:
                    print(f"  [SKIP 空正文] {url} | {title_list[:30]}", flush=True)
                    seen.add(url)
                    continue
                rec = {
                    "id": str(iid),
                    "title": (title or title_list).strip(),
                    "date": (ddate or date)[:10],
                    "url": url,
                    "source": source or "温宿县人民政府",
                    "site": "wszf",
                    "body": body,
                    "attachments": [],
                    "crawl_time": time.strftime("%Y-%m-%d %H:%M:%S"),
                }
                out.write(json.dumps(rec, ensure_ascii=False) + "\n")
                out.flush()
                seen.add(url)
                total_ok += 1
            except Exception as e:
                print(f"  [ERR] {url} {e}", flush=True)
            time.sleep(0.5)
        time.sleep(0.8)
    out.close()
    print(f"[*] 完成, 共 {total_ok} 条新写入 {args.out}", flush=True)


if __name__ == "__main__":
    main()
