#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""仙桃市生态环境局 — 公示公告爬虫 (修复版 v2)
修复: 
1. 日期从 span.date 提取(原取第一个span, 实为list_text标题→date错位+title被清空)
2. title 优先 a[title] 属性, 其次 list_text span
3. 写 gov_search FTS (原写废弃表 gov_search_v3 → 搜索不到)
4. 补 source_url/script_name/group_name
"""
import requests, sqlite3, re
from bs4 import BeautifulSoup

DB = "/root/search.db"
BASE = "https://www.xiantao.gov.cn"
LIST_TPL = BASE + "/bmxxgk/shbj/zfxxgk/fdzdgknr/qtzdgknr_37950/gsgg/index{}.shtml"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36"}
SITE = "xiantao_gsgg"
GROUP = "湖北"
SCRIPT = "crawl_xiantao.py"
# createPageHTML(20, 0, "index", "shtml", "pages clearfix", 311) -> 16 pages
MAX_PAGES = 20
seen_urls = set()

def _drop_container_parts(parts):
    """剔除「容器段」：find_all(['p','div']) 会同时收下容器 <div> 与它内部的 <p>，
    导致同一内容重复（容器那份常还被 get_text(strip=True) 拍平）。
    判据：先按值去重，再剔除被其它段完全包含的段。保护：剔除后为空则返回去重结果。
    """
    if not parts:
        return parts
    ps = [p for p in parts if isinstance(p, str)]
    if len(ps) != len(parts):
        return parts
    seen, uniq = set(), []
    for p in parts:
        if p not in seen:
            seen.add(p); uniq.append(p)
    keep = [a for a in uniq
            if not (len(a) >= 40 and any(b is not a and b and b in a for b in uniq))]
    return keep if keep else uniq


def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception:
        return None, None, None
    soup = BeautifulSoup(r.text, "html.parser")
    # 标题
    title = ""
    tt = soup.select_one("div.xl_tit")
    if tt:
        title = tt.get_text(strip=True)
    # 日期
    date = ""
    for td in soup.find_all("td"):
        if "发文日期" in td.get_text():
            nxt = td.find_next_sibling("td")
            if nxt:
                m = re.search(r"(\d{4}-\d{2}-\d{2})", nxt.get_text())
                if m: date = m.group(1); break
    # 正文 - TRS UEditor
    content_div = soup.find("div", class_="view") or soup.find("div", class_="TRS_Editor")
    content = ""
    if content_div:
        _parts = []
        for tag in content_div.find_all(["p", "div", "table"]):
            txt = str(tag).strip()
            if txt: _parts.append(txt)
        content = "\n".join(_drop_container_parts(_parts))
    return title, date, content

def crawl_page(pg):
    if pg == 1:
        url = LIST_TPL.format("")
    else:
        url = LIST_TPL.format(f"_{pg}")
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception:
        return [], 0
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if "/shbj/" not in href or "/gsgg/" not in href:
            continue
        # 跳过栏目首页链接
        if href.rstrip("/").endswith("/gsgg") or href.rstrip("/").endswith("/gsgg/"):
            continue
        if not href.startswith("http"):
            href = BASE + href
        if href in seen_urls:
            continue
        seen_urls.add(href)
        # 标题: 优先 a[title] 属性, 其次 list_text span, 最后纯文本
        title = (a.get("title") or "").strip()
        if not title:
            lt = a.select_one("span.list_text")
            if lt:
                title = lt.get_text(strip=True)
        if not title:
            title = a.get_text(strip=True)
        # 日期: 从 span.date 提取(修复: 原逻辑取第一个span实为list_text标题)
        date = ""
        sp = a.select_one("span.date")
        if sp:
            date = sp.get_text(strip=True)
        # 清理标题中的日期残留
        if sp and sp.get_text(strip=True) in title:
            title = title.replace(sp.get_text(strip=True), "").strip()
        items.append({"title": title.strip(), "url": href, "date": date})
    return items

def main():
    import argparse
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1)
    args = ap.parse_args()
    pages = max(1, min(args.pages, MAX_PAGES))

    conn = sqlite3.connect(DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()
    c.execute("CREATE TABLE IF NOT EXISTS gov_raw (id INTEGER PRIMARY KEY, site_name TEXT, source_url TEXT, page_url TEXT, title TEXT, publish_date TEXT, date_rank INTEGER DEFAULT 0, summary TEXT, status TEXT, category TEXT DEFAULT '', visits INTEGER DEFAULT 0, content TEXT DEFAULT '', tags TEXT DEFAULT '', industry TEXT DEFAULT 'other', attachments TEXT DEFAULT '', group_name TEXT, has_table INTEGER DEFAULT 0, script_name TEXT DEFAULT '')")
    c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(title, site_name, summary, tokenize='trigram')")

    total_new = total_dup = total_skip = 0
    for pg in range(1, pages + 1):
        print(f"--- 第{pg}页 ---")
        items = crawl_page(pg)
        if not items:
            print("无结果，停止翻页"); break
        print(f"找到{len(items)}条")
        for item in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if c.fetchone():
                total_dup += 1; continue
            print(f"  [{item['date']}] {item['title'][:50]}")
            title, date, content = extract_detail(item["url"])
            if not title: title = item["title"]
            if not date: date = item["date"]
            if not content or len(content) < 200:
                print(f"    正文过短({len(content) if content else 0}), 跳过")
                total_skip += 1; continue
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name) VALUES (?,?,?,?,?,?,?,?,?)",
                    (title, summary, content, item["url"], item["url"], date, SITE, SCRIPT, GROUP))
                rid = cur.lastrowid
                c.execute("INSERT INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                    (rid, title, SITE, summary))
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_dup += 1
    print(f"\n新增: {total_new}  重复: {total_dup}  过短: {total_skip}  总计: {total_new+total_dup+total_skip}")

if __name__ == "__main__":
    main()
