#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
淮北市政府门户 - 信息公开（濉溪县栏目，监督保障/规范性文件）
https://www.huaibei.gov.cn/public/column/1981?type=4

站点特征（2026-08-11 探测）:
- 龙讯 Lonsun CMS；页面 HTML 是导航壳，列表经 Ls.ajax 异步加载
- 真数据源: GET /zwgk/site/label/8888?labelName=publicInfoList&type=6&siteId=4704141&catIds=4751791&pageIndex=N&pageSize=15&isJson=true
- JSON 直供完整字段: title / publishDate / content(完整正文) / attachPdfName+attachPdfTitle / attachDocName+attachDocTitle / link(详情URL) / fileNum(文号)
- 附件路径 /group1/M00/... 相对 → urljoin(BASE_URL) 绝对化（huaibei/sxx 双域均可下载）
- 页面 title=「信息公开_濉溪县人民政府信息公开网」；link 指向 www.sxx.gov.cn（濉溪县独立站）
- total=22, pageCount=2（2026-08-11 探测），数据 2021~2025

用法:
  python3 crawl_huaibei_xxgk.py --pages=2    # 全量
  python3 crawl_huaibei_xxgk.py --pages=1    # 测试
"""
import re
import sys
import json
import html
import urllib3
import requests
from urllib.parse import urljoin

urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

from crawler_lib import push_to_searchdb

SITE_NAME = "濉溪县政府信息公开-规范性文件"
BASE_URL = "https://www.huaibei.gov.cn"
API_URL = BASE_URL + "/zwgk/site/label/8888"
CUTOFF = "2015-01-01"   # 栏目仅 22 条（2018~2025），全量抓
CATEGORY = "政府公开"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Referer": "https://www.huaibei.gov.cn/public/column/1981?type=4",
    "X-Requested-With": "XMLHttpRequest",
    "Accept": "application/json, text/javascript, */*; q=0.01",
}

def clean_title(t):
    t = html.unescape(t or "")
    t = re.sub(r"^(?:&middot;|&nbsp;|\s|•|·)+", "", t)
    t = t.replace("\u200b", "").replace("\ufeff", "")
    return t.strip()

def fetch_list_page(page):
    """API 拉列表，返回 items"""
    params = {
        "labelName": "publicInfoList",
        "type": 6,
        "siteId": 4704141,
        "catIds": "4751791",
        "pageIndex": page,
        "pageSize": 15,
        "isJson": "true",
    }
    r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30, verify=False)
    if r.status_code != 200:
        return [], 0, 0
    r.encoding = r.apparent_encoding or "utf-8"
    try:
        j = json.loads(r.text)
    except Exception:
        return [], 0, 0
    total = j.get("total", 0)
    page_count = j.get("pageCount", 0)
    items = []
    for row in j.get("data", []) or []:
        title = clean_title(row.get("title", ""))
        if not title:
            continue
        items.append({"title": title, "row": row})
    return items, total, page_count

def build_content(row):
    """正文: content 完整正文 + 附件段"""
    content = row.get("content", "") or ""
    # 附件: attachPdf + attachDoc（去重）
    atts = []
    for key_name, key_title in [("attachPdfName", "attachPdfTitle"), ("attachDocName", "attachDocTitle")]:
        p = row.get(key_name, "")
        if not p:
            continue
        abs_url = p if p.startswith("http") else urljoin(BASE_URL, p)
        t = clean_title(row.get(key_title, "")) or abs_url.split("/")[-1]
        atts.append({"name": t, "url": abs_url})
    # 去重附件（pdf/doc 可能同名同 url）
    seen = set()
    atts_u = []
    for a in atts:
        if a["url"] not in seen:
            seen.add(a["url"])
            atts_u.append(a)
    # 正文: content 保留段落；若 content 无 <p> 包裹成 <p>
    if content:
        if "<p" not in content and "<table" not in content:
            content = re.sub(r"\n{2,}", "</p>\n<p>", content.strip())
            content = f"<p>{content}</p>"
        # 清理可能的 HTML 残留标签（保留 p/table/a）
        content = re.sub(r"<span[^>]*>", "", content)
        content = re.sub(r"</span>", "", content)
        content = re.sub(r"<br\s*/?>", "", content)
    body = content
    for a in atts_u:
        if a["url"] not in body:
            body += f'<p><a href="{a["url"]}">{a["name"]}</a></p>'
    return body, atts_u

def main():
    max_pages = None
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            max_pages = int(a.split("=")[1])
        elif a.isdigit():
            max_pages = int(a)
    all_items = []
    seen = set()
    page = 1
    total = 0
    while True:
        if max_pages and page > max_pages:
            break
        items, total, page_count = fetch_list_page(page)
        if not items:
            print(f"[P{page}] empty, stop")
            break
        new = 0
        for it in items:
            # 以 link 或 title 去重
            key = it["row"].get("link") or it["title"]
            if key in seen:
                continue
            pub = (it["row"].get("publishDate") or "")[:10]
            if pub and pub < CUTOFF:
                print(f"[P{page}] {pub} < CUTOFF {CUTOFF}, stop")
                page = 999999
                break
            seen.add(key)
            all_items.append(it)
            new += 1
        print(f"[P{page}] got {len(items)} items, new {new}, total {len(all_items)} (api total={total})")
        if page == 999999:
            break
        page += 1
    print(f"[AutoPg] max_pages={max_pages}, collected {len(all_items)}")

    valid = []
    for i, it in enumerate(all_items):
        row = it["row"]
        content, atts = build_content(row)
        pub = (row.get("publishDate") or "")[:10]
        link = row.get("link") or ""
        valid.append({
            "site_name": SITE_NAME,
            "title": it["title"],
            "pub_date": pub,
            "content": content,
            "source_url": link,
            "url": link,
            "attachments": json.dumps(atts, ensure_ascii=False) if atts else "[]",
        })
    print(f"[Detail] done {len(valid)}")
    push_to_searchdb(valid, CATEGORY)

if __name__ == "__main__":
    main()
