#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
三台县人民政府 - 公示公告（c111285）
https://www.santai.gov.cn/stxrmzf/c111285/list.shtml

站点特征（2026-08-11 探测）:
- 黑龙江政府站群 /common/search JSON API（同 jixilishu 模式）
- 列表: GET /common/search/{channelId}?_isAgg=true&_isJson=true&_pageSize=15&_template=index&page=N
  channelId 从页面 JS `channelId = "..."` 提取（非 meta）
- JSON 字段: title / url(相对路径) / publishedTimeStr / content(截断~1KB须补抓详情) / channelName
- 详情: <div class="article_content" id="zoomcon"> 正文 + meta ArticleTitle/PubDate
- 附件: 相对路径 {hash}/files/xxx.pdf → urljoin(详情页URL) 绝对化
- total=2076, pageSize=15 → 138 页；CUTOFF 3年窗口

用法:
  python3 crawl_santai_gsgg.py --pages=138   # 全量
  python3 crawl_santai_gsgg.py --pages=1     # 测试
"""
import re
import sys
import json
import html
import urllib3
import requests
from urllib.parse import urljoin

urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

from crawler_lib import push_to_searchdb

SITE_NAME = "三台县人民政府-公示公告"
BASE_URL = "https://www.santai.gov.cn"
LIST_URL = BASE_URL + "/stxrmzf/c111285/list.shtml"
CHANNEL_ID = "8027b3f0b2554c65aa1958ff5634aa37"
API_TPL = (BASE_URL + "/common/search/{ch}?_isAgg=true&_isJson=true&_pageSize=15&_template=index&page={page}")
CUTOFF = "2023-08-10"   # 3 年窗口
CATEGORY = "政府公开"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Referer": LIST_URL,
    "Accept": "application/json, text/javascript, */*; q=0.01",
}

def clean_title(t):
    t = html.unescape(t or "")
    t = re.sub(r"^(?:&middot;|&nbsp;|\s|•|·)+", "", t)
    t = t.replace("\u200b", "").replace("\ufeff", "")
    return t.strip()

def fetch_list_page(page):
    url = API_TPL.format(ch=CHANNEL_ID, page=page)
    r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
    if r.status_code != 200:
        return [], 0
    r.encoding = r.apparent_encoding or "utf-8"
    try:
        j = json.loads(r.text)
    except Exception:
        return [], 0
    total = j.get("data", {}).get("total", 0)
    items = []
    for row in j.get("data", {}).get("results", []) or []:
        title = clean_title(row.get("title", ""))
        if not title:
            continue
        url = row.get("url", "")
        if not url:
            continue
        detail_url = urljoin(LIST_URL, url)
        pub = (row.get("publishedTimeStr") or "")[:10]
        items.append({"title": title, "url": detail_url, "date": pub})
    return items, total

def fetch_detail(detail_url, list_title):
    """返回 (content_html, attachments, pub_date)"""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30, verify=False)
        r.raise_for_status()
        r.encoding = r.apparent_encoding or "utf-8"
        raw = r.text
    except Exception:
        return None, [], None
    # 标题/日期 meta
    title = list_title
    tm = re.search(r'<meta name="ArticleTitle" content="([^"]+)"', raw)
    if tm:
        title = clean_title(tm.group(1)) or title
    pub_date = None
    dm = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})"', raw)
    if dm:
        pub_date = dm.group(1)
    # 附件（正文内相对路径，含 files/）
    atts = []
    for am in re.finditer(r'<a[^>]+href="([^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar|wps|7z))"[^>]*>(.*?)</a>', raw, re.S):
        href, name = am.group(1), am.group(2)
        abs_url = href if href.startswith("http") else urljoin(detail_url, href)
        name = clean_title(re.sub(r"<[^>]+>", "", name)) or abs_url.split("/")[-1]
        atts.append({"name": name, "url": abs_url})
    # 正文 zoomcon
    body = ""
    m = re.search(r'<div[^>]*class="article_content"[^>]*id="zoomcon"[^>]*>(.*?)</div>\s*(?:<div|</div>)', raw, re.S)
    if not m:
        m = re.search(r'<div[^>]*class="article_content"[^>]*>(.*?)</div>\s*(?:<div|</div>)', raw, re.S)
    if m:
        body = m.group(1)
    if body:
        # 只删含附件链接的段落（防重复）
        body = re.sub(r'<p[^>]*>(?:(?!</p>).)*\.(?:pdf|doc|docx|xls|xlsx|zip|rar|wps|7z)(?:(?!</p>).)*</p>', '', body, flags=re.S|re.I)
        # 附件相对路径绝对化（body 内残留 img/a）
        body = re.sub(r'(src|href)="(?!https?://|/)([^"]+)"', lambda m2: f'{m2.group(1)}="{urljoin(detail_url, m2.group(2))}"', body)
        body = re.sub(r'<span[^>]*>', '', body)
        body = re.sub(r'</span>', '', body)
        body = re.sub(r'<p[^>]*>', '<p>', body)
        body = re.sub(r'<br\s*/?>', '', body)
        body = re.sub(r'<p>\s*</p>', '', body)
        body = re.sub(r'&ensp;', '', body)
    else:
        body = f"<p>{title}</p>"
    # 附件段追加（去重）
    seen = set()
    for a in atts:
        if a["url"] not in seen:
            seen.add(a["url"])
            if a["url"] not in body:
                body += f'<p><a href="{a["url"]}">{a["name"]}</a></p>'
    return body, atts, pub_date

def main():
    max_pages = None
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            max_pages = int(a.split("=")[1])
        elif a.isdigit():
            max_pages = int(a)
    all_items = []
    seen = set()
    page = 1
    total = 0
    while True:
        if max_pages and page > max_pages:
            break
        items, total = fetch_list_page(page)
        if not items:
            print(f"[P{page}] empty, stop")
            break
        new = 0
        for it in items:
            if it["url"] in seen:
                continue
            if it["date"] and it["date"] < CUTOFF:
                print(f"[P{page}] {it['date']} < CUTOFF {CUTOFF}, stop")
                page = 999999
                break
            seen.add(it["url"])
            all_items.append(it)
            new += 1
        print(f"[P{page}] got {len(items)} items, new {new}, total {len(all_items)} (api total={total})")
        if page == 999999:
            break
        page += 1
    print(f"[AutoPg] max_pages={max_pages}, collected {len(all_items)}")

    valid = []
    for i, it in enumerate(all_items):
        content, atts, pub_date = fetch_detail(it["url"], it["title"])
        if content is None:
            content = f"<p>{it['title']}</p>"
        valid.append({
            "site_name": SITE_NAME,
            "title": it["title"],
            "pub_date": pub_date or it["date"],
            "content": content,
            "source_url": it["url"],
            "url": it["url"],
            "attachments": json.dumps(atts, ensure_ascii=False) if atts else "[]",
        })
        if (i + 1) % 50 == 0:
            print(f"[Detail] {i+1}/{len(all_items)}")
    print(f"[Detail] done {len(valid)}")
    push_to_searchdb(valid, CATEGORY)

if __name__ == "__main__":
    main()
