#!/usr/bin/env python3
"""爬取 xiushui.gov.cn - 城投集团通知公告 (2026-09-08 改为 data_<year>.xml + x-api-key)"""
import requests
from bs4 import BeautifulSoup
import sqlite3
import os
import re
import xml.etree.ElementTree as ET
from datetime import datetime

BASE_URL = "https://www.xiushui.gov.cn"
LIST_PATH = "/xsctgg/ctgg/"
XML_API_KEY = "********^^^^^^^^"
SITE_NAME = "xiushui.gov.cn-城投集团通知公告"
CUTOFF_DATE = "2023-06-16"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "x-api-key": XML_API_KEY,
}


def get_items_xml(year):
    """拉 data_<year>.xml 的 ITEM 列表"""
    url = f"{BASE_URL}{LIST_PATH}data_{year}.xml"
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        raw = resp.text
    except Exception as e:
        print(f"  [WARN] {year} fetch: {e}")
        return []
    try:
        root = ET.fromstring(raw)
    except Exception:
        try:
            root = ET.fromstring(re.sub(r'^[^<]*<', '<', raw, count=1))
        except Exception as e:
            print(f"  [WARN] {year} parse: {e}")
            return []
    items = []
    for it in root.iter("ITEM"):
        def g(tag):
            e = it.find(tag)
            return "".join(e.itertext()).strip() if e is not None else ""
        title, pub, rel = g("TITLE"), g("PUBURL"), g("RELTIME")
        if not title or not pub:
            continue
        if pub and not pub.startswith("http"):
            pub = BASE_URL + LIST_PATH + pub.lstrip("./")
        items.append({"title": title, "url": pub, "publish_date": rel[:10]})
    return items


def get_detail(url):
    """获取详情页的正文HTML"""
    resp = requests.get(url, headers=HEADERS, timeout=30)
    resp.encoding = "utf-8"
    soup = BeautifulSoup(resp.text, "html.parser")

    content_parts = []
    for sel in ["div.view.TRS_UEDITOR", "div.Article_zw", "div#article-box"]:
        div = soup.select_one(sel)
        if div:
            for a in div.find_all("a"):
                href = a.get("href", "")
                if href and not href.startswith("http"):
                    a["href"] = BASE_URL + href
            content_parts.append(str(div))
            break

    if content_parts:
        return "".join(content_parts)
    return ""


def crawl():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    c.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            title TEXT,
            page_url TEXT UNIQUE,
            content TEXT,
            publish_date TEXT,
            site_name TEXT DEFAULT '',
            crawl_time TEXT DEFAULT (datetime('now', '+8 hours')),
            similar TEXT
        )
    """)
    conn.commit()

    now_year = datetime.now().year
    cutoff_year = int(CUTOFF_DATE[:4])
    years = list(range(now_year, cutoff_year - 1, -1))
    print(f"年份: {years} (截止 {CUTOFF_DATE})")

    items_all = []
    for y in years:
        recs = get_items_xml(y)
        print(f"  {y}: {len(recs)} items from xml")
        items_all.extend(recs)
    items_all.sort(key=lambda x: x["publish_date"] or "", reverse=True)

    total_added = 0
    total_skipped = 0

    for item in items_all:
        pub = item["publish_date"]
        if pub and pub < CUTOFF_DATE:
            continue

        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
        if c.fetchone():
            total_skipped += 1
            continue

        content = get_detail(item["url"])
        if not content:
            print(f"    警告: 空正文 {item['title'][:30]}")
            total_skipped += 1
            continue

        c.execute(
            "INSERT OR IGNORE INTO gov_raw (title, page_url, content, publish_date, site_name) VALUES (?, ?, ?, ?, ?)",
            (item["title"], item["url"], content, pub, SITE_NAME)
        )
        if c.rowcount > 0:
            total_added += 1
        conn.commit()

    conn.close()
    print(f"完成: 新增 {total_added} 条, 跳过 {total_skipped} 条")


if __name__ == "__main__":
    crawl()
