#!/usr/bin/env python3
"""
validate_crawler_config.py
校验 daily_crawl_config.json 与磁盘脚本文件的双向一致性。
在服务器上每天跑一次，发现不一致就告警。

用法: python3 validate_crawler_config.py
"""

import json, os, sys

GOV_DIR = '/root/gov_crawler/'
CONFIG_FILE = os.path.join(GOV_DIR, 'daily_crawl_config.json')


def main():
    errors = []

    # 1. Load config — it's a list of crawler entries directly
    with open(CONFIG_FILE) as f:
        crawlers = json.load(f)

    if not isinstance(crawlers, list):
        crawlers = crawlers.get('crawlers', [])

    # 2. Build config index (script -> entry) + name_path set
    config_scripts = {}  # script name -> entry
    name_path_scripts = set()
    for entry in crawlers:
        script = entry.get('script', '')
        if script:
            config_scripts[script] = entry
        name = entry.get('name', '')
        path = entry.get('path', '')
        if name and path:
            s = name if name.endswith('.py') else name + '.py'
            name_path_scripts.add(s)

    # 3. Scan disk for all crawl scripts (Python + Node.js)
    disk_scripts = set()
    for fname in os.listdir(GOV_DIR):
        if fname == 'Archive' or fname.startswith('.'):
            continue
        if fname.startswith('crawl_') and (fname.endswith('.py') or fname.endswith('.js')):
            disk_scripts.add(fname)

    # 4. Check: scripts in config but MISSING on disk
    config_missing = []
    for script in sorted(config_scripts.keys()):
        if script.startswith('node '):
            actual_file = script[5:]
        else:
            actual_file = script
        # FIX: normalize absolute paths to basename
        basename = os.path.basename(actual_file)
        if actual_file not in disk_scripts and basename not in disk_scripts:
            config_missing.append(script)

    if config_missing:
        errors.append('配置中存在但磁盘上缺失 ({}个):'.format(len(config_missing)))
        for s in config_missing:
            errors.append('    ' + s)

    # 5. Check: scripts on disk but NOT in config
    skip_list = {"crawl_TEMPLATE.py", "crawl_yidu_backup.py", "crawl_scheduler_v2.py"}
    # 扫描Archive目录——已主动归档的脚本跳过孤儿检测
    archive_dir = os.path.join(GOV_DIR, "Archive")
    if os.path.isdir(archive_dir):
        for fname in os.listdir(archive_dir):
            if fname.startswith("crawl_") and (fname.endswith(".py") or fname.endswith(".js")):
                skip_list.add(fname)
    config_script_names = set()
    for s in config_scripts.keys():
        if s.startswith('node '):
            config_script_names.add(s[5:])
        else:
            config_script_names.add(os.path.basename(s))
    # Also include name+path registered scripts
    config_script_names.update(name_path_scripts)

    disk_orphans = sorted(disk_scripts - config_script_names - skip_list)
    if disk_orphans:
        n = len(disk_orphans)
        errors.append('磁盘上存在但配置中缺失 ({}个):'.format(n))
        for s in disk_orphans[:10]:
            errors.append('    ' + s)
        if n > 10:
            errors.append('    ... 还有{}个'.format(n - 10))

    # 6. Summary
    print('爬虫配置校验报告')
    print('=' * 40)
    print('配置爬虫数: {}'.format(len(crawlers)))
    print('磁盘脚本数: {} (排除模板/备份后: {})'.format(
        len(disk_scripts), len(disk_scripts - skip_list)))
    print('配置唯一脚本: {}'.format(len(config_scripts)))

    if errors:
        print()
        print('发现 {} 个问题:'.format(len(errors)))
        for e in errors:
            print('  ' + e)
        return 1
    else:
        print()
        print('配置与磁盘完全一致，无孤儿脚本')
        return 0


if __name__ == '__main__':
    sys.exit(main())
