#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""老边区人民政府 — 通知公告爬虫
URL: http://www.laobian.gov.cn/004/about.html
CMS: webBuilder (ewb- 前缀, ivs_content)
列表: ul > li.ewb-list-node > a.ewb-list-name + span.ewb-list-date
  第1页: /004/about.html
  第N页: /004/{N}.html
  共656条/15=44页
详情: /004/{YYYYMMDD}/{UUID}.html
  标题: meta ArticleTitle
  日期: meta PubDate (2026-07-23)
  来源: meta ContentSource
  正文: div#ivs_content (多p段落 + table)
  附件: ul.ewb-attach (正文容器外, 单独提取)
"""
import requests, sqlite3, re, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB = "/root/search.db"
BASE = "http://www.laobian.gov.cn"
SITE = "老边区-通知公告"
GROUP = "辽宁"
SCRIPT = "crawl_laobian_tzgg.py"
MAX_PAGES = 44  # 656/15
PER_PAGE = 15
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
}


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def crawl_list(pg):
    """抓列表页, 返回 [{title, url, date}]"""
    url = f"{BASE}/004/about.html" if pg == 1 else f"{BASE}/004/{pg}.html"
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        log(f"  [WARN] 第{pg}页请求失败: {e}")
        return []
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    for li in soup.find_all("li", class_="ewb-list-node"):
        a = li.find("a", class_="ewb-list-name")
        if not a:
            continue
        href = a.get("href", "").strip()
        if not href or href == "#" or "{{" in href:
            continue
        if not href.startswith("http"):
            href = urljoin(BASE, href)
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date = ""
        dspan = li.find("span", class_="ewb-list-date")
        if dspan:
            dm = re.search(r"(20\d{2}-\d{2}-\d{2})", dspan.get_text())
            if dm:
                date = dm.group(1)
        items.append({"title": title, "url": href, "date": date})
    return items


def extract_detail(url):
    """返回 (title, date, content)"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        log(f"  [WARN] 详情失败 {url}: {e}")
        return None, None, None
    soup = BeautifulSoup(html, "html.parser")
    # 标题
    title = ""
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()
    # 日期
    date = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        m = re.search(r"(20\d{2}-\d{2}-\d{2})", md["content"])
        if m:
            date = m.group(1)
    # 正文
    content = ""
    dc = soup.find("div", id="ivs_content")
    if not dc:
        dc = soup.find("div", class_="ewb-article-content")
    if dc:
        # 附件区 (ul.ewb-attach 在正文容器外)
        attachments = []
        seen_att = set()
        for ul in soup.find_all("ul", class_="ewb-attach"):
            for a in ul.find_all("a", href=True):
                href = a["href"].strip()
                if not href.startswith("http"):
                    href = urljoin(BASE, href)
                name = a.get_text(strip=True) or "附件"
                if href not in seen_att:
                    seen_att.add(href)
                    attachments.append({"name": name, "url": href})
        # 正文内附件链接也提取(去重)
        for a in dc.find_all("a", href=True):
            href = a["href"].strip()
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|txt)$", href, re.I):
                if not href.startswith("http"):
                    href = urljoin(BASE, href)
                name = a.get_text(strip=True) or "附件"
                if href not in seen_att:
                    seen_att.add(href)
                    attachments.append({"name": name, "url": href})

        parts = []
        for node in dc.find_all(["p", "div", "table"], recursive=True):
            # 跳过表格内部节点(表格整体保留)
            if node.name != "table" and node.find_parent("table"):
                continue
            # 跳过含已提取附件的p(避免正文+附件区重复)
            if node.name == "p":
                has_att = any(
                    a.get("href", "").strip() in seen_att
                    for a in node.find_all(["a"], href=True))
                if has_att:
                    continue
            html_str = str(node).strip()
            if not html_str:
                continue
            # 去嵌套: 外层p包含表格则跳过; p内再套p只取最内层
            if node.name == "p":
                if node.find("table"):
                    continue
                inner_p = node.find("p", recursive=False)
                if inner_p:
                    html_str = str(inner_p).strip()
            if node.name == "div" and node.find_parent("div") and not node.find(["p", "table"]):
                continue
            if node.name in ("p", "table"):
                parts.append(html_str)
        content = "\n\n".join(parts).strip()
        # 去重 + 过滤空段落
        uniq_parts = []
        for part in content.split("\n\n"):
            part_stripped = part.strip()
            if not part_stripped:
                continue
            if not re.sub(r"<[^>]+>", "", part_stripped).strip():
                continue
            if part_stripped not in uniq_parts:
                uniq_parts.append(part_stripped)
        content = "\n\n".join(uniq_parts).strip()
        if attachments:
            att_links = "\n".join(f'<p><a href="{att["url"]}">{att["name"]}</a></p>' for att in attachments)
            content = content + "\n\n" + att_links if content else att_links
    return title, date, content


def main():
    import argparse
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1)
    args = ap.parse_args()
    pages = max(1, min(args.pages, MAX_PAGES))

    conn = sqlite3.connect(DB, timeout=15)
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    c.execute("CREATE TABLE IF NOT EXISTS gov_raw (id INTEGER PRIMARY KEY, site_name TEXT, source_url TEXT, page_url TEXT, title TEXT, publish_date TEXT, date_rank INTEGER DEFAULT 0, summary TEXT, status TEXT, category TEXT DEFAULT '', visits INTEGER DEFAULT 0, content TEXT DEFAULT '', tags TEXT DEFAULT '', industry TEXT DEFAULT 'other', attachments TEXT DEFAULT '', group_name TEXT, has_table INTEGER DEFAULT 0, script_name TEXT DEFAULT '')")
    c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(title, site_name, summary, tokenize='trigram')")

    total_new = total_dup = total_skip = 0
    seen = set()
    for pg in range(1, pages + 1):
        log(f"--- 第{pg}页 ---")
        items = crawl_list(pg)
        if not items:
            log("列表为空, 停止翻页")
            break
        log(f"找到{len(items)}条")
        for item in items:
            if item["url"] in seen:
                continue
            seen.add(item["url"])
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if c.fetchone():
                total_dup += 1
                continue
            log(f"  [{item['date']}] {item['title'][:45]}")
            title, date, content = extract_detail(item["url"])
            if not title:
                title = item["title"]
            if not date:
                date = item["date"]
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                log(f"    正文过短({len(content) if content else 0}), 跳过")
                total_skip += 1
                continue
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if re.search(r"<table", content) else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (title, summary, content, item["url"], item["url"], date, SITE, SCRIPT, GROUP, has_table, 0))
                rid = cur.lastrowid
                # 2026-09-22: 先提交 gov_raw —— 下面手动写 FTS 会因触发器已写过同一
                #   rowid 而 IntegrityError，若不先 commit，这条记录会被一并回滚（静默丢数据）
                conn.commit()
                c.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, title, SITE, summary))
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.3)
    print(f"\n新增: {total_new}  重复: {total_dup}  过短: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
