#!/usr/bin/env python3
"""
fix_names_v3.py — 从脚本 SITE_NAME 常量 + URL 精确修复名称
第3版：不再用拼音猜测，而是直接从脚本读取 SITE_NAME
"""

import json, re, os, sys

CONFIG_PATH = "/root/gov_crawler/daily_crawl_config.json"
GOV_CRAWLER_DIR = "/root/gov_crawler"
BACKUP_DIR = os.path.join(GOV_CRAWLER_DIR, "backups")

# 站点名 → 简写映射（对一些过长的站点名做精简）
SITE_SHORTEN = {
    "黄陵县人民政府": "黄陵县政府",
    "黄陵县环境保护局": "黄陵县环保局",
}

def extract_sitename_from_script(script_name):
    """从脚本文件提取 SITE_NAME 常量"""
    if not script_name:
        return ""
    if script_name.startswith("node ") or script_name.startswith("bash ") or script_name.startswith("python3 "):
        script_name = script_name.split(" ", 1)[1]
    script_path = os.path.join(GOV_CRAWLER_DIR, script_name)
    if not os.path.isfile(script_path):
        return ""
    try:
        with open(script_path, "r", errors="replace") as f:
            head = f.read(5000)
    except:
        return ""
    m = re.search(r'^SITE_NAME\s*=\s*[\"\']([^\"\']+)[\"\']', head, re.MULTILINE)
    if m:
        sn = m.group(1).strip()
        # 过滤：SITE_NAME 必须包含中文，且不是纯脚本文件名
        if any('\u4e00' <= c <= '\u9fff' for c in sn):
            # 进一步过滤：不能是域名形式（xxx.gov.cn）
            if not re.match(r'^[a-z0-9.]+\.(gov|com|cn|net|org)', sn):
                return sn
    return ""


def load_config():
    with open(CONFIG_PATH) as f:
        cfg = json.load(f)
    return cfg if isinstance(cfg, list) else cfg.get("crawlers", cfg)


def save_config(config, path=None):
    path = path or CONFIG_PATH
    with open(path, "w") as f:
        json.dump(config, f, ensure_ascii=False, indent=2)

def backup():
    os.makedirs(BACKUP_DIR, exist_ok=True)
    from datetime import datetime
    ts = datetime.now().strftime("%Y%m%d_%H%M%S")
    bp = os.path.join(BACKUP_DIR, f"config_before_v3_{ts}.json")
    cfg = load_config()
    save_config(cfg, bp)
    return bp

def main():
    cfg = load_config()
    
    print(f"=== 第3轮名称修复：从脚本SITE_NAME提取精确站点名 ===\n")
    print(f"总条目: {len(cfg)}")
    
    # 先备份
    backup()
    
    # Step 1: 提取所有脚本的 SITE_NAME
    print("\n📥 正在扫描所有脚本的 SITE_NAME...")
    sitename_map = {}
    for item in cfg:
        script = item.get("script", "")
        if script:
            sn = extract_sitename_from_script(script)
            if sn:
                sitename_map[script] = sn
    
    print(f"   找到 {len(sitename_map)} 个脚本有 SITE_NAME\n")
    
    # Step 2: 对每一条，如果脚本有 SITE_NAME 且当前名称不是精确站点名，则替换
    changes = []
    skip_count = 0
    
    for i, item in enumerate(cfg):
        old_name = item.get("name", "") or ""
        script = item.get("script", "")
        group = item.get("group", "") or ""
        
        if not script:
            continue
        
        # 脚本的 SITE_NAME
        script_sitename = sitename_map.get(script, "")
        if not script_sitename:
            continue
        
        # 如果当前名称已经是完整站点名（包含 SITE_NAME），跳过
        if script_sitename in old_name:
            skip_count += 1
            continue
        
        # 从当前名称提取后缀/栏目部分
        # 旧名称可能是 "黄陵县" 或 "黄陵县-公示公告"
        suffix = ""
        for sep in ["-", "—", "·"]:
            parts = old_name.split(sep, 1)
            if len(parts) > 1:
                potential_suffix = parts[-1]
                # 如果后缀看起来像栏目名（中文且不算太长），保留
                if potential_suffix and any('\u4e00' <= c <= '\u9fff' for c in potential_suffix) and len(potential_suffix) <= 20:
                    suffix = potential_suffix
                    break
        
        # 构造新名称
        new_name = script_sitename
        if suffix:
            new_name = f"{script_sitename}-{suffix}"
        
        if new_name != old_name:
            changes.append((i, old_name, new_name, group, script))
    
    # Step 3: 打印预览
    print(f"共 {len(changes)} 条名称可以优化\n")
    print(f"{'Idx':>4} {'旧名称':<35} → {'新名称':<40} {'分组':<12}")
    print("=" * 95)
    for idx, (i, old_n, new_n, group, script) in enumerate(changes):
        old_d = (old_n[:34] + "…") if len(old_n) > 35 else old_n
        new_d = (new_n[:39] + "…") if len(new_n) > 40 else new_n
        print(f"{i:>4} {old_d:<35} → {new_d:<40} {group:<12}")
    
    print(f"\n跳过（已有精确名）: {skip_count}")
    print(f"待修改: {len(changes)}")
    print(f"\n确认应用以上修改? (yes/no): ", end="")
    
    confirm = input().strip().lower()
    if confirm in ("yes", "y", "是"):
        for i, _, new_n, _, _ in changes:
            cfg[i]["name"] = new_n
        save_config(cfg)
        print(f"✅ 已保存 {len(changes)} 条名称修改")
    else:
        print("❌ 已取消")


if __name__ == "__main__":
    main()
