#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""D批 9 站存量重刷 v2：支持单站参数
用法: python3 reback_d9.py [site_keyword]
  不带参数 = 全部 8 站；带参数 = 只跑 site_name 包含该关键词的站
"""
import sys, os, time, sqlite3, importlib, re

DB = "/root/search.db"
SITES = [
    ("crawl_bx_gggs",         "本溪满族自治县-公示公告"),
    ("crawl_bincheng_tzgg",   "滨城区人民政府-通知公告"),
    ("crawl_baofengenergy_tzz","宝丰能源集团-投资者关系公示"),
    ("crawl_aygov_tzgg",      "安远县-最新公告"),
    ("crawl_anyang_hjbh",     "安阳市-建设项目环评信息"),
    ("crawl_arq_tzgg",        "阿荣旗-通知公告"),
    ("crawl_anshan_tzgg",     "鞍山市-通知公告"),
    ("crawl_wedz_hpgs",       "望城经开区-环评公示"),
]
FILTER = sys.argv[1] if len(sys.argv) > 1 else None

def clean_ellipsis(t):
    if not t:
        return t
    return re.sub(r'[\.]{3,}|…\s*$', '', t).strip()

def fetch_html(mod, url):
    if hasattr(mod, "fetch_detail"):
        return mod.fetch_detail(url)
    if hasattr(mod, "fetch"):
        return mod.fetch(url)
    import requests
    r = requests.get(url, headers=getattr(mod, "HEADERS", {}), timeout=30, verify=False)
    r.encoding = getattr(mod, "ENCODING", "utf-8")
    return r.text

conn = sqlite3.connect(DB, timeout=60)
conn.execute("PRAGMA busy_timeout=30000")

total_updated = total_skip = total_fail = 0
for mod_name, site in SITES:
    if FILTER and FILTER not in site:
        continue
    mod = importlib.import_module(mod_name)
    cur = conn.execute(
        "SELECT id, page_url, title FROM gov_raw WHERE site_name=? AND page_url IS NOT NULL AND page_url!='' ORDER BY id",
        (site,))
    rows = cur.fetchall()
    print(f"\n=== {site} ({mod_name}): {len(rows)} 条待重刷 ===", flush=True)
    site_ok = site_skip = site_fail = 0
    for idx, (rid, url, old_title) in enumerate(rows, 1):
        try:
            html = fetch_html(mod, url)
            if mod_name == "crawl_wedz_hpgs":
                detail = mod.parse_detail(html, page_url=url)
            else:
                detail = mod.parse_detail(html)
        except Exception as e:
            print(f"  [{idx}/{len(rows)}] FAIL fetch {url[-50:]}: {e}", flush=True)
            site_fail += 1; total_fail += 1
            time.sleep(0.3)
            continue
        title = clean_ellipsis(detail.get("title") or old_title)
        content = detail.get("content", "")
        pub_date = detail.get("publish_date", "")
        if not content:
            print(f"  [{idx}/{len(rows)}] SKIP 空正文 {url[-50:]}", flush=True)
            site_skip += 1; total_skip += 1
            time.sleep(0.3)
            continue
        conn.execute(
            "UPDATE gov_raw SET title=?, content=?, publish_date=? WHERE id=?",
            (title, content, pub_date, rid))
        conn.commit()
        site_ok += 1; total_updated += 1
        if idx % 20 == 0 or idx == len(rows):
            print(f"  [{idx}/{len(rows)}] ok={site_ok} skip={site_skip} fail={site_fail}", flush=True)
        time.sleep(0.4)
    print(f"  → {site}: ok={site_ok} skip={site_skip} fail={site_fail}", flush=True)

conn.close()
print(f"\n{'='*50}\nTOTAL: updated={total_updated} skip={total_skip} fail={total_fail}")
