#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""存量回填：把 8 个脚本已入库的「无日期 / 错日期」记录，重抓详情页修好 publish_date。

遵循铁律：
  · 先备份受影响行（id + 旧值）到 bak_datefix_20260921
  · 只做 UPDATE publish_date WHERE id=?，绝不 DELETE / 不重建行
  · 拿不到新日期就保持原样（不写空、不写错）
用法: python3 backfill_dates_20260921.py [--dry-run]
"""
import importlib.util
import os
import re
import sqlite3
import sys
import time

DRY = "--dry-run" in sys.argv
D = "/root/gov_crawler"
os.chdir(D)
sys.path.insert(0, D)
DB = "/root/search.db"

# 脚本 → 它的详情页解析函数（返回里第几个元素是日期）
TARGETS = [
    ("crawl_dongyingnews_bzgg.py", "parse_detail", 1),
    ("crawl_zx_tzgg.py", "fetch_detail", 2),
    ("crawl_cjgjgxq_tzgg.py", "fetch_detail", 1),
    ("crawl_zibo_epb.py", "fetch_detail", 1),
    ("crawl_yantai.py", "fetch_detail", 1),
    ("crawl_juxian_gggs.py", None, None),      # 用列表页 + URL 兜底，见下
    ("crawl_fqlook.py", None, None),           # 用列表页
    ("crawl_yichun_jjxw.py", None, None),      # 用 URL 兜底
]

db = sqlite3.connect(DB, timeout=300)
db.execute("PRAGMA busy_timeout=300000")
c = db.cursor()
db.execute("""CREATE TABLE IF NOT EXISTS bak_datefix_20260921 (
                id INTEGER PRIMARY KEY, script_name TEXT, old_date TEXT, new_date TEXT)""")
db.commit()

DATE_OK = re.compile(r"^\d{4}-\d{1,2}-\d{1,2}$")


def norm(v):
    m = re.search(r"(\d{4})[/-](\d{1,2})[/-](\d{1,2})", str(v or ""))
    return "%04d-%02d-%02d" % (int(m.group(1)), int(m.group(2)), int(m.group(3))) if m else ""


def url_date(u):
    m = re.search(r"/(20\d{2})[/-](\d{1,2})[/-](\d{1,2})[/.]", u or "") or \
        re.search(r"/(20\d{2})(\d{2})(\d{2})[/.]", u or "")
    if not m:
        return ""
    return "%04d-%02d-%02d" % (int(m.group(1)), int(m.group(2)), int(m.group(3)))


def page_date(u, mod=None):
    """抓页面，从「发表于 <span title=...>」等标记里取日期（fqlook 等 URL 无日期时用）。"""
    try:
        html = mod.fetch(u) if (mod and hasattr(mod, "fetch")) else \
            mod.requests.get(u, headers=getattr(mod, "HEADERS", {}), timeout=25).text
    except Exception:
        return ""
    for pat in (r'发表于\s*<span\s+title="(\d{4}-\d{1,2}-\d{1,2})',
                r'<b>\s*时间:\s*</b>\s*(\d{4}-\d{1,2}-\d{1,2})',
                r'作者:.*?时间:\s*(\d{4}-\d{1,2}-\d{1,2})',
                r'<meta[^>]*name=["\']?PubDate["\']?[^>]*content=["\'](\d{4}-\d{1,2}-\d{1,2})',
                r'发布日期[：:]\s*(\d{4}-\d{1,2}-\d{1,2})'):
        m = re.search(pat, html or "", re.S)
        if m:
            return norm(m.group(1))
    return ""


total_fixed = 0
for fn, func, idx in TARGETS:
    rows = list(c.execute("""SELECT id, COALESCE(publish_date,''), COALESCE(page_url,'')
                             FROM gov_raw WHERE script_name=?""", (fn,)))
    bad = []
    for rid, pd_, u in rows:
        if not DATE_OK.match(pd_.strip()):
            bad.append((rid, pd_, u))            # 无日期 / 日期格式非法
            continue
        ud = url_date(u)
        if ud:                                    # URL 内嵌日期与库内日期差 >30 天 → 错日期
            try:
                import datetime as _dt
                d1 = _dt.date(*[int(x) for x in ud.split("-")])
                d2 = _dt.date(*[int(x) for x in norm(pd_).split("-")])
                if abs((d1 - d2).days) > 30:
                    bad.append((rid, pd_, u))
            except Exception:
                pass
    print("=" * 92)
    print("%s: 共 %d 条，日期异常 %d 条" % (fn, len(rows), len(bad)))
    if not bad:
        continue

    mod = None
    if func:
        spec = importlib.util.spec_from_file_location(fn[:-3], os.path.join(D, fn))
        mod = importlib.util.module_from_spec(spec)
        try:
            spec.loader.exec_module(mod)
        except SystemExit:
            pass

    fixed = 0
    for rid, old, url in bad[:400]:
        new = ""
        try:
            if mod and func:
                f = getattr(mod, func, None)
                if f:
                    import inspect
                    ps = list(inspect.signature(f).parameters.values())
                    names = [p.name for p in ps]
                    if "session" in names:
                        r = f(mod.requests.Session(), url)
                    elif "tid" in names and len(ps) > 1:
                        r = f(mod.fetch(url), url.rstrip("/").split("-")[1])
                    elif len(ps) > 1 and any("url" in n for n in names):
                        # 注意用子串匹配：参数名可能是 page_url / detail_url，不是恰好 "url"
                        # （2026-09-21 踩过：写成 "url" in names 导致 dongyingnews 一条都没修）
                        if ps[1].default is inspect.Parameter.empty:
                            r = f(mod.fetch(url), url)      # 如 dongyingnews: parse_detail(html, page_url)
                        else:
                            r = f(url)                      # 如 yantai: fetch_detail(url, list_title='')
                    else:
                        r = f(url)
                    if isinstance(r, (tuple, list)) and len(r) > (idx or 0):
                        new = norm(r[idx])
        except Exception as e:
            print("   ! id=%s %s" % (rid, str(e)[:70]))
        # ⚠️ 兜底必须放在 try 外面：模块调用抛异常时（如详情页已 404）仍要能靠 URL 日期修好
        #   （2026-09-21 踩过：放在 try 内 → dongyingnews 的 25 条全是 404，一条没修）
        if not new:
            try:
                new = page_date(url, mod)    # 抓页面从标记里取（URL 无日期时用）
            except Exception:
                new = ""
        if not new:
            new = url_date(url)              # 最后兜底：URL 内嵌日期
        if new and DATE_OK.match(new) and new != old:
            if not DRY:
                c.execute("INSERT OR REPLACE INTO bak_datefix_20260921 VALUES (?,?,?,?)",
                          (rid, fn, old, new))
                c.execute("UPDATE gov_raw SET publish_date=? WHERE id=?", (new, rid))
            fixed += 1
            if fixed <= 3:
                print("   ✓ id=%s  %r → %s" % (rid, old, new))
        time.sleep(0.15)
    if not DRY:
        db.commit()
    print("   → 修复 %d 条%s" % (fixed, "  [DRY-RUN]" if DRY else ""))
    total_fixed += fixed

print("=" * 92)
print("合计修复 %d 条%s" % (total_fixed, "  [DRY-RUN 未写库]" if DRY else ""))
if not DRY:
    n = c.execute("SELECT COUNT(*) FROM bak_datefix_20260921").fetchone()[0]
    print("备份表 bak_datefix_20260921 现有 %d 行（可完整回滚）" % n)
db.close()
