#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""兰溪市人民政府 — 通知公告爬虫
URL: http://www.lanxi.gov.cn/col/col1229288169/index.html
CMS: JPAAS/JCMS
列表: API /api-gateway/jpaas-publish-server/front/page/build/unit
  参数: webId=3614, pageId=1229288169, tplSetId=O5SnyTEouC5PsGkqGVV0p
  分页: paramJson={"pageNo":N,"pageSize":15,"loadEnabled":true,"search":"{}"}
  共5776条/15=386页
详情: /col/{子栏目}/art/{year}/art_{id}.html
  标题: meta ArticleTitle
  日期: meta PubDate (2026-07-31 15:39)
  来源: meta ContentSource
  正文: div.main_section
"""
import requests, sqlite3, re, sys, time, json
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB = "/root/search.db"
BASE = "http://www.lanxi.gov.cn"
API_URL = BASE + "/api-gateway/jpaas-publish-server/front/page/build/unit"
SITE = "兰溪市-通知公告"
GROUP = "浙江"
SCRIPT = "crawl_lanxi_tzgg.py"
PAGE_ID = "1229288169"
MAX_PAGES = 386  # 5776/15
PER_PAGE = 15
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{BASE}/col/col{PAGE_ID}/index.html",
}


def _drop_container_parts(parts):
    """剔除「容器段」：find_all(['p','div']) 会同时收下容器 <div> 与它内部的 <p>，
    导致同一内容重复（容器那份常还被 get_text(strip=True) 拍平）。
    判据：某段被列表内另一段完全包含 → 它是容器段 → 剔除。保护：剔除后为空则原样返回。
    """
    if not parts:
        return parts
    ps = [p for p in parts if isinstance(p, str)]
    if len(ps) != len(parts):
        return parts
    # ① 先按值去重 —— 否则两段完全相同时 a in b 与 b in a 双向成立，
    #    会把两段都判成「容器段」剔光，再触发下面的保护而原样返回（等于没修）。
    seen, uniq = set(), []
    for p in parts:
        if p not in seen:
            seen.add(p); uniq.append(p)
    # ② 再剔容器段（被其它段完全包含的那个）
    keep = [a for a in uniq
            if not (len(a) >= 40 and any(b is not a and b and b in a for b in uniq))]
    return keep if keep else uniq


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def crawl_list(pg):
    """API分页抓列表, 返回 [{title, url, date}]"""
    param_json = json.dumps({
        "pageNo": pg, "pageSize": PER_PAGE, "loadEnabled": True, "search": "{}"
    }, ensure_ascii=False)
    params = {
        "webId": "3614",
        "pageId": PAGE_ID,
        "parseType": "bulidstatic",
        "pageType": "column",
        "tagId": "新闻列表",
        "tplSetId": "O5SnyTEouC5PsGkqGVV0p",
        "unitUrl": "/api-gateway/jpaas-publish-server/front/page/build/unit",
        "paramJson": param_json,
    }
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        data = r.json()
        html = data.get("data", {}).get("html", "")
    except Exception as e:
        log(f"  [WARN] 第{pg}页请求失败: {e}")
        return []
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.find_all("li"):
        a = li.find("a")
        date_div = li.find("div", class_="list-date")
        if not a or not date_div:
            continue
        href = a.get("href", "").strip()
        if not href or href == "#":
            continue
        if not href.startswith("http"):
            href = urljoin(BASE, href)
        title = a.get_text(strip=True)
        # strip 实体前缀
        title = re.sub(r"^(?:&middot;|·|&nbsp;|\s)+", "", title).strip()
        date = date_div.get_text(strip=True)
        dm = re.search(r"(20\d{2}-\d{2}-\d{2})", date)
        if not dm:
            continue
        items.append({"title": title, "url": href, "date": dm.group(1)})
    return items


def extract_detail(url):
    """返回 (title, date, content)"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        log(f"  [WARN] 详情失败 {url}: {e}")
        return None, None, None
    soup = BeautifulSoup(html, "html.parser")
    # 标题
    title = ""
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()
    if not title:
        mh = soup.find("div", class_="main_title")
        if mh:
            title = mh.get_text(strip=True)
    # 日期
    date = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        m = re.search(r"(20\d{2}-\d{2}-\d{2})", md["content"])
        if m:
            date = m.group(1)
    # 正文: div.main_section
    content = ""
    dc = soup.select_one("div.main_section")
    if not dc:
        dc = soup.find("div", id="zoom")
    if dc:
        # 附件
        attachments = []
        seen_att = set()
        for a in dc.find_all("a", href=True):
            href = a["href"]
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|txt)$", href, re.I):
                if not href.startswith("http"):
                    href = urljoin(BASE, href)
                name = a.get_text(strip=True) or "附件"
                if href not in seen_att:
                    seen_att.add(href)
                    attachments.append({"name": name, "url": href})

        parts = []
        for node in dc.find_all(["p", "div", "table"], recursive=True):
            # 跳过表格内部节点(表格已整体保留, 内部 p/div 不拍平, 避免重复)
            if node.name != "table" and node.find_parent("table"):
                continue
            if node.name == "div" and node.get("id") in ("xgwdP", "qrcode"):
                continue
            if node.find_parent("div", id="xgwdP"):
                continue
            if node.find_parent(class_="filelink") or (node.get("class") and "filelink" in node.get("class")):
                continue
            # 跳过含已提取附件的p(避免正文+附件区重复)
            if node.name == "p":
                def _att_url(el):
                    u = ""
                    if el.name == "a":
                        u = el.get("href", "")
                    if u and not u.startswith("http"):
                        u = urljoin(BASE, u)
                    return u
                has_att = any(
                    _att_url(el) in seen_att
                    for el in node.find_all(["a"]) if _att_url(el))
                if has_att:
                    continue
            html_str = str(node).strip()
            if not html_str:
                continue
            # 去嵌套: p 内再套 p 只取最内层文本段落
            if node.name == "p":
                # 外层p包含表格则跳过(表格内部节点独立处理)
                if node.find("table"):
                    continue
                inner_p = node.find("p", recursive=False)
                if inner_p:
                    html_str = str(inner_p).strip()
            if node.name == "div" and node.find_parent("div") and not node.find(["p", "table"]):
                parts.append(html_str)
            elif node.name in ("p", "table"):
                parts.append(html_str)
        parts = _drop_container_parts(parts)
        content = "\n\n".join(parts).strip()
        # 去重 + 过滤空段落
        uniq_parts = []
        for part in content.split("\n\n"):
            part_stripped = part.strip()
            if not part_stripped:
                continue
            if not re.sub(r"<[^>]+>", "", part_stripped).strip():
                continue
            if part_stripped not in uniq_parts:
                uniq_parts.append(part_stripped)
        content = "\n\n".join(uniq_parts).strip()
        if attachments:
            att_links = "\n".join(f'<p><a href="{att["url"]}">{att["name"]}</a></p>' for att in attachments)
            content = content + "\n\n" + att_links if content else att_links
    return title, date, content


def main():
    import argparse
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1)
    args = ap.parse_args()
    pages = max(1, min(args.pages, MAX_PAGES))

    conn = sqlite3.connect(DB, timeout=15)
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    c.execute("CREATE TABLE IF NOT EXISTS gov_raw (id INTEGER PRIMARY KEY, site_name TEXT, source_url TEXT, page_url TEXT, title TEXT, publish_date TEXT, date_rank INTEGER DEFAULT 0, summary TEXT, status TEXT, category TEXT DEFAULT '', visits INTEGER DEFAULT 0, content TEXT DEFAULT '', tags TEXT DEFAULT '', industry TEXT DEFAULT 'other', attachments TEXT DEFAULT '', group_name TEXT, has_table INTEGER DEFAULT 0, script_name TEXT DEFAULT '')")
    c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(title, site_name, summary, tokenize='trigram')")

    total_new = total_dup = total_skip = 0
    seen = set()
    for pg in range(1, pages + 1):
        log(f"--- 第{pg}页 ---")
        items = crawl_list(pg)
        if not items:
            log("列表为空, 停止翻页")
            break
        log(f"找到{len(items)}条")
        for item in items:
            if item["url"] in seen:
                continue
            seen.add(item["url"])
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if c.fetchone():
                total_dup += 1
                continue
            log(f"  [{item['date']}] {item['title'][:45]}")
            title, date, content = extract_detail(item["url"])
            if not title:
                title = item["title"]
            if not date:
                date = item["date"]
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                log(f"    正文过短({len(content) if content else 0}), 跳过")
                total_skip += 1
                continue
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if re.search(r"<table", content) else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (title, summary, content, item["url"], item["url"], date, SITE, SCRIPT, GROUP, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, title, SITE, summary))
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.3)
    print(f"\n新增: {total_new}  重复: {total_dup}  过短: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
