#!/usr/bin/env python3
"""
昔阳县-公告公示 (www.xiyang.gov.cn)
==============================
Power CMS — 政府信息公开平台
工信局 → 法定主动公开 → 政策文件 → 通知公告

列表: https://www.xiyang.gov.cn/zwgk/zfxxgkml/xzgzbm/23gxj/fdzdgknrrsj11111/gwfg3rsj11111/gggs3rsj11111
详情: /.../content_XXXXX
正文: article.articleCon
分页: 后缀 _N

用法:
    python3 crawl_xiyang.py          # 全量（2页，近3年）
    python3 crawl_xiyang.py --test   # 测试 5 条
"""

import re, sys, os, json, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

# ── 配置 ──
SITE_NAME  = "昔阳县-公告公示"
BASE_URL   = "https://www.xiyang.gov.cn"
LIST_URL   = "https://www.xiyang.gov.cn/zwgk/zfxxgkml/xzgzbm/23gxj/fdzdgknrrsj11111/gwfg3rsj11111/gggs3rsj11111"
SEARCH_DB  = "/root/search.db"
MAX_PAGES  = 5  # 实际只有2页
CUTOFF     = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS    = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  ⚠ fetch fail: {e}")
        return None

def parse_list(html):
    """提取列表条目 (url, title, pub_date)"""
    items = []
    area = re.search(r'<ul class="infoList">(.*?)</ul>', html, re.DOTALL)
    if not area:
        return items
    for m in re.finditer(r'<li[^>]*>(.*?)</li>', area.group(1), re.DOTALL):
        li = m.group(1)
        date_m = re.search(r'<span class="date"[^>]*>(\d{4}-\d{2}-\d{2})</span>', li)
        link_m = re.search(r'<a[^>]*href="([^"]*)"[^>]*>([^<]+)</a>', li)
        if link_m:
            href = link_m.group(1).strip()
            title = link_m.group(2).strip()
            if not href.startswith('http'):
                href = BASE_URL + href
            pub_date = date_m.group(1) if date_m else ""
            if title and href:
                items.append({"title": title, "url": href, "pub_date": pub_date})
    return items

def fetch_detail(url):
    """获取详情页的 title, content, pub_date"""
    html = fetch(url)
    if not html:
        return {"title": "", "content": "", "pub_date": ""}

    # 标题
    title = ""
    m = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]*)"', html, re.I)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r'<title>([^<]+)', html)
        if m:
            title = re.sub(r'[-_—|].*', '', m.group(1)).strip()

    # 日期
    pub_date = ""
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="([^"]+)', html, re.I)
    if m:
        pub_date = m.group(1)[:10]
    if not pub_date:
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
        if m:
            pub_date = m.group(1)

    # 正文
    content = ""
    art = re.search(r'<article class="articleCon">(.*?)</article>', html, re.DOTALL)
    if art:
        content = art.group(1)
    else:
        m = re.search(r'<div class="printArea"[^>]*>(.*?)</div>\s*<div class="others"', html, re.DOTALL)
        if m:
            content = m.group(1)

    if content:
        # 清理无用标签
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'<link[^>]*>', '', content, flags=re.I)
        content = re.sub(r' style="[^"]*"', '', content)
        content = content.strip()

    return {"title": title, "content": content, "pub_date": pub_date}

def to_db(items):
    """批量写入 search.db（gov_raw 表 + FTS 同步）"""
    if not items:
        print("  ⏭ 无数据")
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, title, page_url, content, publish_date, summary, tags) "
                "VALUES (?,?,?,?,?,?,?)",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "通知公告",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    # FTS 同步
    # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
    #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
    db.commit()
    db.execute(
        "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    db.commit()
    db.close()
    print(f"  💾 入库: 新增{ok}, 跳过{fail}, FTS已同步")
    return ok, fail

def crawl(test=False):
    """全量爬取"""
    # 生成列表页 URL 列表
    urls = [LIST_URL]
    for n in range(2, MAX_PAGES + 1):
        urls.append(f"{LIST_URL}_{n}")

    # 提取列表条目
    all_items = []
    seen_urls = set()
    for page_url in urls:
        html = fetch(page_url)
        if not html:
            continue
        items = parse_list(html)
        if not items:
            print(f"  ⚠ 第{urls.index(page_url)+1}页无数据，可能已到末页")
            break
        # 过滤：近3年
        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])
            if item["pub_date"] and item["pub_date"] < CUTOFF:
                continue
            all_items.append(item)
        first_date = items[0].get("pub_date", "?")
        last_date = items[-1].get("pub_date", "?")
        print(f"  📄 第{urls.index(page_url)+1}页: {len(items)}条 ({first_date} ~ {last_date})")

    if test:
        all_items = all_items[:5]
        print(f"  🧪 测试模式: 只处理前 {len(all_items)} 条")

    print(f"\n  📊 列表汇总: {len(all_items)} 条（近3年）")
    if not all_items:
        return 0, 0

    # 爬详情
    results = []
    for i, item in enumerate(all_items, 1):
        detail = fetch_detail(item["url"])
        entry = {
            "title": detail["title"] or item["title"],
            "url": item["url"],
            "content": detail["content"],
            "pub_date": detail["pub_date"] or item.get("pub_date", ""),
        }
        results.append(entry)
        if i % 5 == 0 or i == len(all_items):
            print(f"  [{i}/{len(all_items)}] 📝 全部完成" if i == len(all_items) else f"  [{i}/{len(all_items)}] ...")

    # 入库
    ok, fail = to_db(results)
    return ok, fail

if __name__ == "__main__":
    test = "--test" in sys.argv
    print(f"\n📡 [{SITE_NAME}] {'测试模式' if test else '全量'} (最多{MAX_PAGES}页, 近3年)")
    t0 = time.time()
    ok, fail = crawl(test=test)
    print(f"  ⏱ 耗时: {time.time()-t0:.1f}s")
    print(f"  ✅ 新增: {ok}  ❌ 跳过: {fail}")
