#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""耒阳市人民政府 — 夏塘镇 通知公告爬虫
URL: http://www.leiyang.gov.cn/xxgk/bmxxgkml/gxzjdbsc/xtz/tzgg/
CMS: 博山科技 (boshan) 政府门户系统(老版模板)
列表: div.submiInfor > ul > li, 15条/页, 共2页29条(2021-03~2026-07)
分页: /tzgg/pages/{N}.html (第1页=index.html)
详情: /tzgg/{YYYYMMDD}/i{ID}.html
  标题: h3.general_title (无meta ArticleTitle)
  日期: label.dt_time "时间：2025-07-10 17:26:08"
  正文: div.general_article#div_content
"""
import requests, sqlite3, re, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB = "/root/search.db"
BASE = "http://www.leiyang.gov.cn"
LIST_DIR = BASE + "/xxgk/bmxxgkml/gxzjdbsc/xtz/tzgg"
SITE = "耒阳夏塘镇-通知公告"
GROUP = "湖南"
SCRIPT = "crawl_leiyang_xtz.py"
MAX_PAGES = 2
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Referer": LIST_DIR + "/index.html",
}


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def fetch(url, retries=3):
    for attempt in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=60)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            log(f"  [WARN] {url} 请求失败({attempt+1}/{retries}): {e}")
            time.sleep(2)
    return ""


def crawl_list(pg):
    """抓列表页, 返回 [{title, url, date}]"""
    if pg <= 1:
        url = LIST_DIR + "/index.html"
    else:
        url = f"{LIST_DIR}/pages/{pg}.html"
    html = fetch(url)
    if not html:
        return []
    soup = BeautifulSoup(html, "html.parser")
    infos = soup.find_all("div", class_="submiInfor")
    if not infos:
        return []
    items = []
    for info in infos:
        for li in info.find_all("li"):
            a = li.find("a")
            if not a:
                continue
            href = a.get("href", "")
            if not href.startswith("http"):
                href = urljoin(BASE, href)
            title = a.get_text(strip=True)
            # strip 实体前缀
            title = re.sub(r"^(?:&middot;|·|&nbsp;|\s)+", "", title).strip()
            # 日期
            dm = re.search(r"(20\d{2}-\d{2}-\d{2})", li.get_text())
            date = dm.group(1) if dm else ""
            items.append({"title": title, "url": href, "date": date})
    return items


def extract_detail(url):
    """返回 (title, date, content)"""
    html = fetch(url)
    if not html:
        return None, None, None
    soup = BeautifulSoup(html, "html.parser")
    # 标题: h3.general_title
    title = ""
    h3 = soup.find("h3", class_="general_title")
    if h3:
        title = h3.get_text(strip=True)
    # 日期: label.dt_time "时间：2025-07-10 17:26:08"
    date = ""
    dt = soup.find("label", class_="dt_time")
    if dt:
        m = re.search(r"(20\d{2}-\d{2}-\d{2})", dt.get_text())
        if m:
            date = m.group(1)
    # 正文: div.general_article#div_content
    content = ""
    dc = soup.find("div", class_="general_article")
    if not dc:
        dc = soup.find("div", id="div_content")
    if dc:
        # 附件: a[href] 指向文件
        attachments = []
        seen_att = set()
        for a in dc.find_all("a", href=True):
            href = a["href"]
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|txt)$", href, re.I):
                if not href.startswith("http"):
                    href = urljoin(BASE, href)
                name = a.get_text(strip=True) or "附件"
                if href not in seen_att:
                    seen_att.add(href)
                    attachments.append({"name": name, "url": href})

        parts = []
        # 排除相关阅读/二维码/分享区块 + 附件下载重复区(filelink)
        for node in dc.find_all(["p", "div", "table", "h1", "h2", "h3"], recursive=True):
            # 关键: 跳过表格内部节点(表格已整体保留, 内部 p/div/h 不拍平, 避免重复)
            if node.name != "table" and node.find_parent("table"):
                continue
            if node.name == "div" and node.get("id") in ("xgwdP", "qrcode"):
                continue
            if node.find_parent("div", id="xgwdP"):
                continue
            if node.find_parent(class_="filelink") or (node.get("class") and "filelink" in node.get("class")):
                continue
            # 跳过含已提取附件的p(避免正文+附件区重复)
            if node.name == "p":
                def _att_url(el):
                    u = ""
                    if el.name == "a":
                        u = el.get("href", "")
                    if u and not u.startswith("http"):
                        u = urljoin(BASE, u)
                    return u
                has_att = any(
                    _att_url(el) in seen_att
                    for el in node.find_all(["a"]) if _att_url(el))
                if has_att:
                    continue
            html_str = str(node).strip()
            if not html_str:
                continue
            # 去嵌套: p 内再套 p 只取最内层文本段落
            if node.name == "p":
                # 外层p包含表格则跳过(表格内部节点独立处理, 避免表格内容被p包裹提取)
                if node.find("table"):
                    continue
                inner_p = node.find("p", recursive=False)
                if inner_p:
                    # 保留内层 p 的HTML, 去掉外层包装
                    html_str = str(inner_p).strip()
                parts.append(html_str)
            elif node.name == "div" and node.find_parent("div") and not node.find(["p", "table"]):
                parts.append(html_str)
            elif node.name in ("table", "h1", "h2", "h3"):
                parts.append(html_str)
        content = "\n\n".join(parts).strip()
        # 去重: 相同段落只保留一次 + 过滤空/纯<br>段落(含 span 包裹)
        uniq_parts = []
        for part in content.split("\n\n"):
            part_stripped = part.strip()
            if not part_stripped:
                continue
            # 剥标签后无文本(仅空白/br) → 过滤
            if not re.sub(r"<[^>]+>", "", part_stripped).strip():
                continue
            if part_stripped not in uniq_parts:
                uniq_parts.append(part_stripped)
        content = "\n\n".join(uniq_parts).strip()
        if attachments:
            att_links = "\n".join(f'<p><a href="{att["url"]}">{att["name"]}</a></p>' for att in attachments)
            content = content + "\n\n" + att_links if content else att_links
    return title, date, content


def main():
    import argparse
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1)
    args = ap.parse_args()
    pages = max(1, min(args.pages, MAX_PAGES))

    conn = sqlite3.connect(DB, timeout=15)
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    c.execute("CREATE TABLE IF NOT EXISTS gov_raw (id INTEGER PRIMARY KEY, site_name TEXT, source_url TEXT, page_url TEXT, title TEXT, publish_date TEXT, date_rank INTEGER DEFAULT 0, summary TEXT, status TEXT, category TEXT DEFAULT '', visits INTEGER DEFAULT 0, content TEXT DEFAULT '', tags TEXT DEFAULT '', industry TEXT DEFAULT 'other', attachments TEXT DEFAULT '', group_name TEXT, has_table INTEGER DEFAULT 0, script_name TEXT DEFAULT '')")
    c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(title, site_name, summary, tokenize='trigram')")

    total_new = total_dup = total_skip = 0
    seen = set()
    for pg in range(1, pages + 1):
        log(f"--- 第{pg}页 ---")
        items = crawl_list(pg)
        if not items:
            log("列表为空, 停止翻页")
            break
        log(f"找到{len(items)}条")
        for item in items:
            if item["url"] in seen:
                continue
            seen.add(item["url"])
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if c.fetchone():
                total_dup += 1
                continue
            log(f"  [{item['date']}] {item['title'][:45]}")
            title, date, content = extract_detail(item["url"])
            if not title:
                title = item["title"]
            if not date:
                date = item["date"]
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                log(f"    正文过短({len(content) if content else 0}), 跳过")
                total_skip += 1
                continue
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if re.search(r"<table", content) else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (title, summary, content, item["url"], item["url"], date, SITE, SCRIPT, GROUP, has_table, 0))
                rid = cur.lastrowid
                # 2026-09-22: 先提交 gov_raw —— 下面手动写 FTS 会因触发器已写过同一
                #   rowid 而 IntegrityError，若不先 commit，这条记录会被一并回滚（静默丢数据）
                conn.commit()
                c.execute("INSERT INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, title, SITE, summary))
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.5)
    print(f"\n新增: {total_new}  重复: {total_dup}  过短: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
