#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""patch_binyang_title.py —— 剥掉宾阳标题里的站点后缀"""
import ast
import io
import os
import shutil
import sys

P = "/root/gov_crawler/crawl_binyang_hjsp.py"
BAK = "/root/gov_crawler/Archive/crawl_binyang_hjsp.py.bak_20260924_title"

src = io.open(P, encoding="utf-8").read()
old = '''    if not title:
        m = re.search(r"<title>(.*?)</title>", html_text, re.S | re.I)
        if m:
            title = clean_title(m.group(1))'''
new = '''    if not title:
        m = re.search(r"<title>(.*?)</title>", html_text, re.S | re.I)
        if m:
            _t0 = clean_title(m.group(1))
            # 该站无 meta ArticleTitle，回退 <title> 会带站点后缀：
            #   "标题_栏目_广西南宁宾阳县人民政府门户网站" → 剥掉尾巴
            t = re.sub(r"_[^_]{2,40}_[^_]{0,30}(?:人民政府|政府)[^_]*$", "", _t0).strip()
            t = re.sub(r"[-—|]\\s*[^-—|]{0,30}人民政府[^-—|]*$", "", t).strip()
            title = t or _t0'''

n = src.count(old)
if n != 1:
    print("❌ 锚点命中 %d 次（期望 1）" % n)
    sys.exit(1)
src2 = src.replace(old, new)
try:
    ast.parse(src2)
except SyntaxError as e:
    print("❌ 语法错误，未落盘:", e)
    sys.exit(1)
os.makedirs(os.path.dirname(BAK), exist_ok=True)
if not os.path.exists(BAK):
    shutil.copy2(P, BAK)
io.open(P, "w", encoding="utf-8").write(src2)
print("✅ 已修（备份 %s）" % BAK)

# 立即验证标题
import importlib.util
import re as _re
spec = importlib.util.spec_from_file_location("m2", P)
m2 = importlib.util.module_from_spec(spec)
spec.loader.exec_module(m2)
h = m2.fetch(m2.LIST_URL)
its = m2.parse_list(h)
ok = True
for u, t, d in its[:5]:
    hd = m2.fetch(u)
    r = m2.parse_detail(hd, u)
    tt = r[0]
    bad = ("人民政府门户网站" in tt) or ("建设项目环境影响评价审批" in tt and tt.count("_") > 0)
    print("   %s %s" % ("❌" if bad else "✅", tt[:62]))
    ok = ok and not bad
print("结论:", "✅ 标题已干净" if ok else "❌ 仍有后缀")
