#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""check_jcx_dup.py —— 精确查重（区分 jcx.gov.cn 与 zjcx.gov.cn 的子串陷阱）"""
import os
import re
import sqlite3

PAT = re.compile(r"://(?:www\.)?jcx\.gov\.cn/", re.I)

print("=== ① 脚本里精确引用（://jcx.gov.cn/ 或 ://www.jcx.gov.cn/）===")
hits = []
for f in sorted(os.listdir("/root/gov_crawler")):
    if not f.endswith(".py"):
        continue
    p = "/root/gov_crawler/" + f
    try:
        t = open(p, encoding="utf-8", errors="ignore").read()
    except Exception:
        continue
    if PAT.search(t):
        hits.append(f)
print("  命中脚本:", hits or "无 ✅")

print("\n=== ② 脚本里含 zjcx（对照，属长兴县，非目标）===")
zj = [f for f in sorted(os.listdir("/root/gov_crawler"))
      if f.endswith(".py") and "zjcx.gov.cn" in open("/root/gov_crawler/" + f, encoding="utf-8", errors="ignore").read()]
print(" ", zj[:8])

print("\n=== ③ 库内精确匹配 ===")
c = sqlite3.connect("/root/search.db", timeout=20)
c.execute("PRAGMA busy_timeout=20000")
q = "SELECT COUNT(*) FROM gov_raw WHERE page_url LIKE 'https://www.jcx.gov.cn/%' OR page_url LIKE 'https://jcx.gov.cn/%' OR page_url LIKE 'http://www.jcx.gov.cn/%' OR page_url LIKE 'http://jcx.gov.cn/%'"
print("  jcx.gov.cn（精确）:", c.execute(q).fetchone()[0], "条")
print("  其中 site_name 分布:", c.execute(
    "SELECT site_name, COUNT(*) FROM gov_raw WHERE page_url LIKE 'https://www.jcx.gov.cn/%' OR page_url LIKE 'https://jcx.gov.cn/%' GROUP BY site_name LIMIT 5").fetchall())
print("  zjcx.gov.cn（对照，长兴）:", c.execute(
    "SELECT COUNT(*) FROM gov_raw WHERE page_url LIKE '%://www.zjcx.gov.cn/%'").fetchone()[0], "条")
print("  site_name 含 泾川:", c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name LIKE '%泾川%'").fetchone()[0])
print("  site_name 含 长兴:", c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name LIKE '%长兴%'").fetchone()[0])
c.close()

print("\n=== ④ config 里精确 ===")
import json
cfg = json.load(open("/root/gov_crawler/daily_crawl_config.json", encoding="utf-8"))
for i, x in enumerate(cfg):
    blob = json.dumps(x, ensure_ascii=False)
    if PAT.search(blob) or "jcx.gov.cn" in blob.replace("zjcx.gov.cn", ""):
        print("  #%d %s" % (i + 1, blob[:200]))
    if "泾川" in blob:
        print("  #%d (泾川) %s" % (i + 1, blob[:200]))
print("  （无输出 = config 无该站）")
