#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""秦皇岛市政府信息公开平台 — 公告公示 (qhd.gov.cn, code=293)
URL: http://www.qhd.gov.cn/list.jsp?pages=1&deptid=null&code=293
CMS: 自定义JSP (qhd.gov.cn:81)
列表: ul.info-list-xxgk.mt15 > li > a + span.time
  分页: list.jsp?pages=N&deptid=null&code=293, 20条/页, 354页~7000条
  注意: 最新1-11条(约2周内)详情常404(源站发布延迟), 跳过不写占位符
详情: /article/293/{id}.html
  标题: meta ArticleTitle / div.top-info h1
  日期: meta PubDate / p.info
  来源: meta ContentSource
  正文: div.detail-2#content (直接子节点 p/div/table)
"""
import requests, sqlite3, re, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB = "/root/search.db"
BASE = "http://www.qhd.gov.cn:81"
SITE = "秦皇岛市-公告公示"
GROUP = "河北"
SCRIPT = "crawl_qhd_gggs.py"
MAX_PAGES = 354
ATTACH_RE = re.compile(r"\.(pdf|doc|docx|xls|xlsx|rar|zip|txt|wps|et|ofd)$", re.I)
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{BASE}/list.jsp?pages=1&deptid=null&code=293",
}
# Session 保持 cookie (避免302 COLLCC 重定向)
SESSION = requests.Session()
SESSION.headers.update(HEADERS)


def _drop_container_parts(parts):
    """剔除「容器段」：原 find_all(['p','div']) 同时收下容器 <div> 和它内部的 <p>，
    导致同一内容出现两次（容器那份往往还被 get_text(strip=True) 拍平）。
    判据：某段文本被列表内另一段完全包含 → 它是容器段 → 剔除。
    保护：若剔除后为空，则原样返回（不冒删空的风险）。
    """
    if not parts:
        return parts
    ps = [p for p in parts if isinstance(p, str)]
    if len(ps) != len(parts):
        return parts
    keep = []
    for a in parts:
        if len(a) >= 40 and any(b is not a and b and b in a for b in parts):
            continue
        keep.append(a)
    return keep if keep else parts


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def crawl_list(pg):
    url = f"{BASE}/list.jsp?pages={pg}&deptid=null&code=293"
    html = ""
    for attempt in range(4):
        try:
            r = SESSION.get(url, timeout=30)
            r.encoding = "utf-8"
            html = r.text
            if len(html) > 3000:
                break
            log(f"  [WARN] 第{pg}页响应过小({len(html)}), 第{attempt+1}次重试")
        except Exception as e:
            log(f"  [WARN] 第{pg}页请求失败({attempt+1}): {e}")
        time.sleep(4)
    if not html or len(html) < 1000:
        log(f"  [WARN] 第{pg}页最终失败, 跳过")
        return []
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.find("ul", class_="info-list-xxgk")
    if not ul:
        return []
    for li in ul.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        if not href or not re.search(r"/article/\d+/\d+\.html$", href):
            continue
        if not href.startswith("http"):
            href = urljoin(BASE, href)
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date = ""
        span = li.find("span", class_="time")
        if span:
            dm = re.search(r"(20\d{2}-\d{2}-\d{2})", span.get_text())
            if dm:
                date = dm.group(1)
        items.append({"title": title, "url": href, "date": date})
    return items


def extract_detail(url):
    """返回 (title, date, content); 404返回 (None, None, None)"""
    html = ""
    for attempt in range(2):
        try:
            r = SESSION.get(url, timeout=25)
            if r.status_code == 404:
                log(f"  [WARN] 详情404: {url}")
                return None, None, None
            r.encoding = "utf-8"
            html = r.text
            if len(html) > 1500:
                break
        except Exception as e:
            log(f"  [WARN] 详情失败 {url}: {e}")
        time.sleep(1.5)
    if not html:
        return None, None, None
    soup = BeautifulSoup(html, "html.parser")
    # 标题
    title = ""
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()
    if not title:
        h1 = soup.select_one("div.top-info h1")
        if h1:
            title = h1.get_text(strip=True)
    # 日期
    date = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        m = re.search(r"(20\d{2}-\d{2}-\d{2})", md["content"])
        if m:
            date = m.group(1)
    # 正文
    content = ""
    dc = soup.find("div", class_="detail-2", id="content")
    if not dc:
        dc = soup.find("div", id="content")
    if dc:
        for s in dc.find_all(["script", "style"]):
            s.decompose()
        # 附件
        attachments = []
        seen_att = set()
        for a in dc.find_all("a", href=True):
            href = a["href"].strip()
            if ATTACH_RE.search(href):
                if not href.startswith("http"):
                    href = urljoin(BASE, href)
                name = a.get_text(strip=True) or "附件"
                if href not in seen_att:
                    seen_att.add(href)
                    attachments.append({"name": name, "url": href})

        parts = []
        for node in dc.find_all(["p", "div", "table"], recursive=True):
            if node.name != "table" and node.find_parent("table"):
                continue
            if node.name == "p":
                has_att = any(
                    a.get("href", "").strip() in seen_att
                    for a in node.find_all(["a"], href=True))
                if has_att:
                    continue
            html_str = str(node).strip()
            if not html_str:
                continue
            if node.name == "p":
                if node.find("table"):
                    continue
                inner_p = node.find("p", recursive=False)
                if inner_p:
                    html_str = str(inner_p).strip()
            if node.name == "div" and node.find_parent("div") and not node.find(["p", "table"]):
                continue
            if node.name in ("p", "table"):
                parts.append(html_str)
        parts = _drop_container_parts(parts)
        content = "\n\n".join(parts).strip()
        uniq_parts = []
        for part in content.split("\n\n"):
            part_stripped = part.strip()
            if not part_stripped:
                continue
            if not re.sub(r"<[^>]+>", "", part_stripped).strip():
                continue
            if part_stripped not in uniq_parts:
                uniq_parts.append(part_stripped)
        content = "\n\n".join(uniq_parts).strip()
        if attachments:
            att_links = "\n".join(f'<p><a href="{att["url"]}">{att["name"]}</a></p>' for att in attachments)
            content = content + "\n\n" + att_links if content else att_links
    return title, date, content


def main():
    import argparse
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1)
    args = ap.parse_args()
    pages = max(1, min(args.pages, MAX_PAGES))

    conn = sqlite3.connect(DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    c.execute("CREATE TABLE IF NOT EXISTS gov_raw (id INTEGER PRIMARY KEY, site_name TEXT, source_url TEXT, page_url TEXT, title TEXT, publish_date TEXT, date_rank INTEGER DEFAULT 0, summary TEXT, status TEXT, category TEXT DEFAULT '', visits INTEGER DEFAULT 0, content TEXT DEFAULT '', tags TEXT DEFAULT '', industry TEXT DEFAULT 'other', attachments TEXT DEFAULT '', group_name TEXT, has_table INTEGER DEFAULT 0, script_name TEXT DEFAULT '')")
    c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(title, site_name, summary, tokenize='trigram')")

    total_new = total_dup = total_skip = 0
    seen = set()
    for pg in range(1, pages + 1):
        log(f"--- 第{pg}页 ---")
        items = crawl_list(pg)
        if not items:
            log(f"第{pg}页失败/为空, 跳过(继续下一页)")
            continue
        log(f"找到{len(items)}条")
        for item in items:
            if item["url"] in seen:
                continue
            seen.add(item["url"])
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if c.fetchone():
                total_dup += 1
                continue
            log(f"  [{item['date']}] {item['title'][:45]}")
            title, date, content = extract_detail(item["url"])
            if not title:
                title = item["title"]
            if not date:
                date = item["date"]
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                log(f"    正文过短/404({len(content) if content else 0}), 跳过")
                total_skip += 1
                continue
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if re.search(r"<table", content) else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (title, summary, content, item["url"], item["url"], date, SITE, SCRIPT, GROUP, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, title, SITE, summary))
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.5)
    print(f"\n新增: {total_new}  重复: {total_dup}  过短/404: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
