#!/usr/bin/env python3
"""
crawl_xiushui_hp.py -- Xiushui county EIA project crawler (TRS)

2026-09-08 修复: 栏目列表已改 JS 渲染, 静态 index.html 无行 →
改从同目录 data_<年份>.xml 拉取(需 x-api-key 头, 见 enc.js/document_year.js)。
"""
import sys, re, time, os
import requests
import xml.etree.ElementTree as ET
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

def extract_plain_text(html_text):
    if not html_text:
        return ""
    soup = BeautifulSoup(html_text, "html.parser")
    for tag in soup(["script", "style"]):
        tag.decompose()
    text = soup.get_text(separator=" ", strip=True)
    text = re.sub(r"\s+", " ", text)
    return text[:500]

BASE_URL = "https://www.xiushui.gov.cn"
XML_DIR = "/xxgk/bmxxgk/sthjj/sthj/zdmsxm/xmhp/"
XML_API_KEY = "********^^^^^^^^"   # enc.js/document_year.js 里的 x-api-key 值
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "修水县-项目环评"
TOTAL_PAGES = 38
MAX_PAGES = 5

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "x-api-key": XML_API_KEY,
})

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
stats = {"new": 0, "skip": 0, "errors": 0}


def fetch(url, max_retries=3, extra_headers=None):
    for attempt in range(max_retries):
        try:
            resp = session.get(url, timeout=30, headers=extra_headers or {})
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return resp.text
        except Exception:
            if attempt < max_retries - 1:
                time.sleep(2)
    return None


def fetch_items_xml(year):
    """拉 data_<year>.xml → [{title,url,date}], 失败返回空 list"""
    url = f"{BASE_URL}{XML_DIR}data_{year}.xml"
    raw = fetch(url)
    if not raw:
        print(f"  [WARN] data_{year}.xml fetch failed")
        return []
    try:
        root = ET.fromstring(raw)
    except Exception as e:
        # 尝试去掉前导非 xml 字节(BOM/乱码)再解析
        try:
            s = re.sub(r'^[^<]*<', '<', raw, count=1)
            root = ET.fromstring(s)
        except Exception as e2:
            print(f"  [WARN] data_{year}.xml parse failed: {e} / {e2}")
            return []
    records = []
    for it in root.iter("ITEM"):
        def g(tag):
            e = it.find(tag)
            return "".join(e.itertext()).strip() if e is not None else ""
        title = g("TITLE")
        pub = g("PUBURL")
        rel = g("RELTIME")
        if not title or not pub:
            continue
        if not pub.startswith("http"):
            pub = BASE_URL + XML_DIR + pub.lstrip("./")
        records.append({"title": title, "url": pub, "date": rel[:10]})
    return records


def fetch_detail(url):
    html_text = fetch(url)
    if not html_text:
        return "", ""
    soup = BeautifulSoup(html_text, 'html.parser')

    meta_date = soup.find('meta', attrs={"name": "PubDate"})
    date_str = ''
    if meta_date:
        raw_date = meta_date.get('content', '')
        date_str = raw_date[:10] if raw_date else ''

    content_div = soup.find('div', class_=lambda c: c and 'TRS_UEDITOR' in str(c) and 'view' in str(c))
    if not content_div:
        content_div = soup.find('div', class_=lambda c: c and 'TRS_UEDITOR' in str(c))
    if not content_div:
        content_div = soup.find('div', class_=lambda c: c and 'trs_editor_view' in str(c))
    if not content_div:
        content_div = soup.find('div', id='article-box')

    content_html = ""
    if content_div:
        content_html = str(content_div)

    return content_html, date_str


def store_item(title, url, content, date_str):
    import sqlite3
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        summary = extract_plain_text(content)
        sql = "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, content, source_url, summary, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)"
        c.execute(sql,
                  (SITE_NAME, title, url, date_str, content, url, summary, "xmhp"))
        if c.rowcount > 0:
            stats["new"] += 1
        else:
            stats["skip"] += 1
        conn.commit()
        conn.close()
    except Exception as e:
        print("  [DB]", e)
        stats["errors"] += 1


def crawl(full=False, test_n=None):
    print("=" * 60)
    print("Xiushui EIA project crawler (XML mode)")
    print("=" * 60)

    now_year = datetime.now().year
    cutoff_year = int(THREE_YEARS_AGO[:4])
    years = list(range(now_year, cutoff_year - 1, -1))
    print(f"Years: {years} (cutoff {THREE_YEARS_AGO})")

    items_all = []
    for y in years:
        recs = fetch_items_xml(y)
        print(f"  {y}: {len(recs)} items from xml")
        items_all.extend(recs)
    print(f"Total items: {len(items_all)}")

    seen = 0
    for idx, item in enumerate(items_all):
        if item['date'] and item['date'] < THREE_YEARS_AGO:
            stats["skip"] += 1
            continue
        print(f"  [{idx+1}] {item['title'][:40]} | {item['date']}")

        if test_n and seen >= test_n:
            print(f"\n  Test mode: {test_n} items done")
            break

        content, detail_date = fetch_detail(item['url'])
        date_str = detail_date or item['date']
        store_item(item['title'], item['url'], content, date_str)
        seen += 1
        time.sleep(0.5)

    print(f"\nDone! New: {stats['new']}, Skip: {stats['skip']}, Errors: {stats['errors']}")


if __name__ == '__main__':
    full = '--full' in sys.argv
    test_n = None
    for arg in sys.argv:
        if arg.startswith('--test='):
            test_n = int(arg.split('=')[1])
    crawl(full=full, test_n=test_n)
