#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
南通市生态环境局 - 公告公示(gggs)栏目爬虫
系统: TRS TrueCMS, 前端 jquery.jpage.js
列表API: GET https://sthjj.nantong.gov.cn/truecms/messageController/getMessage.do
  参数(逆向自 jquery.jpage.js getRemoteData): startrecord / endrecord / perpage / columnId
  (范围式分页, 非页码式! page/pageNo/offset 均无效)
  返回 JSON: {"result": "<datastore><totalrecord>1820</totalrecord><recordset><record><![CDATA[<li>...<a href=...>标题</a><span>YYYY-MM-DD</span></li>]]></record>...</recordset></datastore>"}
  注意: JSON 内引号被转义为 \\", 需 json.loads 后解析内层 HTML
栏目ID: f53004c4-95fe-476f-8a97-88df5f6e76b3 (gggs.html 内 jpage ajaxParam, totalrecord=1820)
  注: 本栏目为「公告公示」聚合, 列表 href 混合指向 tzgg(通知公告)/jsxmgs(建设项目公示)/cgzbgg(采购招标) 的 content 详情页
详情页: https://sthjj.nantong.gov.cn/ntshbj/{栏}/content/{uuid}.html (requests 可直取, 无WAF拦截)
  与 tzgg 栏目 URL 相同的记录由 INSERT OR IGNORE 按 URL 去重自动跳过, 不重复入库
标题: 列表即完整标题 (容器 div.TRS_UEDITOR / #BodyLabel)
用法: python3 crawl_nantong_gggs.py [--limit=N] [--pages=N]
"""
import os, sys, re, json, time, html as html_mod
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "南通市生态环境局-公告公示"
CATEGORY = "公告公示"
API_URL = "https://sthjj.nantong.gov.cn/truecms/messageController/getMessage.do"
COLUMN_ID = "f53004c4-95fe-476f-8a97-88df5f6e76b3"
PER_PAGE = 10

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
HEADERS = {"User-Agent": UA, "Accept": "application/json, text/javascript, */*; q=0.01",
           "Referer": "https://sthjj.nantong.gov.cn/ntshbj/gggs/gggs.html", "X-Requested-With": "XMLHttpRequest"}
DETAIL_HEADERS = {"User-Agent": UA, "Referer": "https://sthjj.nantong.gov.cn/ntshbj/gggs/"}

CONTAINERS = ["div#zoom", "div.test-1", "div.TRS_UEDITOR", "div#BodyLabel", "div.zwcontent", "div.content", "div#content"]
ATTACH_RE = re.compile(r'\.(pdf|doc|docx|xls|xlsx|ppt|pptx|zip|rar|wps|et|ofd|txt)$', re.I)


def clean_title(t):
    t = html_mod.unescape(t or "")
    t = re.sub(r"[\u200b\u200e\u200f\ufeff\xa0]", "", t)
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"^[•·\-—]\s*", "", t)
    return t.strip()


def fetch_list_range(session, start, end):
    """GET 范围记录: startrecord~endrecord"""
    params = {"columnId": COLUMN_ID, "startrecord": start, "endrecord": end, "perpage": PER_PAGE}
    for attempt in range(3):
        try:
            r = session.get(API_URL, params=params, headers=HEADERS, timeout=25)
            if r.status_code != 200:
                print(f"    [WARN] list {start}-{end} status={r.status_code}, retry")
                time.sleep(3)
                continue
            j = json.loads(r.text)
            inner = j.get("result", "")
            total_m = re.search(r"<totalrecord>(\d+)<", inner)
            total = int(total_m.group(1)) if total_m else 0
            # 解析 record CDATA 块
            items = []
            for rec in re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", inner, re.S):
                am = re.search(r'<a href="([^"]+)"[^>]*>([^<]+)</a>', rec)
                if not am:
                    am = re.search(r"<a href='([^']+)'[^>]*>([^<]+)</a>", rec)
                if not am:
                    continue
                href, title = am.group(1), clean_title(am.group(2))
                if len(title) < 4:
                    continue
                date = ""
                dm = re.search(r"(\d{4})-(\d{2})-(\d{2})", rec)
                if dm:
                    date = "%s-%s-%s" % dm.groups()
                items.append({"title": title, "href": href, "date": date})
            return total, items
        except Exception as e:
            print(f"    [WARN] list {start}-{end} err {e}, retry {attempt}")
            time.sleep(3)
    return 0, []


def fetch_detail(session, page_url):
    """提取详情: 标题/日期/正文/附件"""
    try:
        r = session.get(page_url, headers=DETAIL_HEADERS, timeout=25)
        r.encoding = "utf-8"
        h = r.text
    except Exception as e:
        print(f"    [WARN] detail fetch {page_url}: {e}")
        return None
    if r.status_code != 200 or len(h) < 500:
        return None
    soup = BeautifulSoup(h, "html.parser")

    title = ""
    h1 = soup.find("h1")
    if h1:
        title = clean_title(h1.get_text(" ", strip=True))
    if not title:
        cmt = soup.select_one("div.container-main-title")
        if cmt:
            title = clean_title(cmt.get_text(" ", strip=True))
    if not title:
        m = re.search(r'<meta[^>]*name=["\']ArticleTitle["\'][^>]*content=["\']([^"\']*)["\']', h, re.I)
        if m:
            title = clean_title(m.group(1))
    if not title:
        t = soup.find("title")
        if t:
            title = clean_title(t.get_text(strip=True).split("-")[0])
    if not title:
        return None

    date = ""
    tm = re.search(r"(\d{4})[年\-/. ](\d{1,2})[月\-/. ](\d{1,2})", h)
    if tm:
        date = "%s-%02d-%02d" % (tm.group(1), int(tm.group(2)), int(tm.group(3)))

    body_el = None
    for sel in CONTAINERS:
        body_el = soup.select_one(sel)
        if body_el and body_el.get_text(strip=True):
            break
    if body_el is None:
        return {"title": title, "date": date, "content": "", "attachments": []}

    att_scope = soup.select_one("div#BodyLabel") or body_el
    attachments = []
    for a in att_scope.find_all("a", href=True):
        href = a["href"].strip()
        if ATTACH_RE.search(href) or "download" in href.lower() or "/attached/" in href or "/upload" in href.lower():
            attachments.append({"name": a.get_text(strip=True) or href.split("/")[-1],
                                "url": urljoin(page_url, href)})

    parts = []
    for el in body_el.find_all(["p", "table", "img"]):
        if el.name == "p":
            if el.find_parent("table") or el.find("table") or el.find("p"):
                continue
            for a in el.find_all("a", href=True):
                href = a["href"].strip()
                if ATTACH_RE.search(href) or "download" in href.lower():
                    a["href"] = urljoin(page_url, href)
            txt = el.get_text(" ", strip=True)
            txt = re.sub(r"\s+", " ", txt)
            if txt:
                parts.append("<p>%s</p>" % txt)
        elif el.name == "table":
            for a in el.find_all("a", href=True):
                href = a["href"].strip()
                if ATTACH_RE.search(href) or "download" in href.lower():
                    a["href"] = urljoin(page_url, href)
            parts.append(str(el))
        elif el.name == "img":
            src = el.get("src", "")
            if src:
                parts.append('<p><img src="%s" alt="%s"></p>' % (urljoin(page_url, src), el.get("alt", "")))
    for att in attachments:
        parts.append('<p><a href="%s">%s</a></p>' % (att["url"], att["name"]))
    content = "\n".join(parts)
    if not content:
        txt = body_el.get_text(" ", strip=True)
        if txt:
            content = "<p>%s</p>" % re.sub(r"\s+", " ", txt)
    return {"title": title, "date": date, "content": content, "attachments": attachments}


def parse_args():
    limit, pages = 0, 0
    for a in sys.argv[1:]:
        if a.startswith("--limit="):
            limit = int(a.split("=")[1])
        elif a.startswith("--pages="):
            pages = int(a.split("=")[1])
    return limit, pages


def main():
    limit, max_pages = parse_args()
    print("[Nantong-gggs] SITE=%s" % SITE_NAME)
    session = requests.Session()

    # 1) 列表全量 (范围式翻页; 部分record CDATA为空, 不因单页不足10条而提前停止)
    total, seen, list_items = 0, set(), []
    start = 1
    page_no = 1
    while True:
        if max_pages and page_no > max_pages:
            break
        end = start + PER_PAGE - 1
        t, items = fetch_list_range(session, start, end)
        if not items and page_no == 1:
            print("  [FAIL] 列表API无数据, 退出")
            return
        if t:
            total = t
        for it in items:
            if it["href"] in seen:
                continue
            seen.add(it["href"])
            it["url"] = "https://sthjj.nantong.gov.cn" + it["href"]
            list_items.append(it)
        print("  page %d: got %d items (total=%d, accumulated=%d)" % (page_no, len(items), total, len(list_items)))
        if start + PER_PAGE - 1 >= total:
            break
        start += PER_PAGE
        page_no += 1
        time.sleep(0.6)
    # 补漏: 若累计 < total, 用大步长整段重扫未覆盖区间 (受 max_pages 限制)
    gap_limit = total if not max_pages else min(total, max_pages * PER_PAGE)
    if total and len(list_items) < gap_limit:
        print("  gap-fill: accumulated %d < %d, rescanning ranges of 30 (up to %d)" % (len(list_items), gap_limit, gap_limit))
        s2 = 1
        while s2 <= gap_limit:
            e2 = min(s2 + 29, gap_limit)
            t2, items2 = fetch_list_range(session, s2, e2)
            for it in items2:
                if it["href"] in seen:
                    continue
                seen.add(it["href"])
                it["url"] = "https://sthjj.nantong.gov.cn" + it["href"]
                list_items.append(it)
            s2 = e2 + 1
            time.sleep(0.6)
        print("  after gap-fill: %d items" % len(list_items))
    print("  list done: %d items, total=%d" % (len(list_items), total))

    if limit and len(list_items) > limit:
        list_items = list_items[:limit]

    # 2) 详情
    results = []
    for i, it in enumerate(list_items, 1):
        d = fetch_detail(session, it["url"])
        if d is None:
            d = {"title": it["title"], "date": it["date"], "content": "", "attachments": []}
        results.append({
            "site_name": SITE_NAME,
            "category": CATEGORY,
            "source_url": it["url"],
            "url": it["url"],
            "title": d["title"] or it["title"],
            "pub_date": d["date"] or it["date"],
            "content": d["content"],
            "attachments": json.dumps(d["attachments"], ensure_ascii=False) if d["attachments"] else "",
            "group_name": "nanTong_gggs",
        })
        if i % 20 == 0:
            print("  detail %d/%d" % (i, len(list_items)))
        time.sleep(0.4)

    # 3) 入库
    print("  pushing %d items to searchdb..." % len(results))
    push_to_searchdb(results, batch_label="nantong_gggs")
    print("  DONE. pushed=%d" % len(results))


if __name__ == "__main__":
    main()
