#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""batch6_probe.py —— 6 个新站：一次性查重 + 结构探测"""
import json
import os
import re
import urllib.parse
from collections import Counter

import requests

requests.packages.urllib3.disable_warnings()
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
H = {"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"}

SITES = [
    ("JY-A", "缙云-col1229425888", "https://www.jinyun.gov.cn/col/col1229425888/index.html"),
    ("JY-B", "缙云-col1229856959", "https://www.jinyun.gov.cn/col/col1229856959/index.html"),
    ("JXST", "江西生态环境厅-col42221", "https://sthjt.jiangxi.gov.cn/jxssthjt/col/col42221/index.html"),
    ("BY", "宾阳-项目环评审批", "http://www.binyang.gov.cn/gk/xxgkml/shgysyjslygk/hjbhly/jsxmhjyxpjsp/"),
    ("AHQY", "青阳-基层政务公开", "https://www.ahqy.gov.cn/Jczwgk/opennessList/650/102002008/page_1.html"),
    ("DZ", "大足-某镇政务公开", "https://www.dazu.gov.cn/qzfjz/glz_101897/zwgk_53321/fdzdgknr_53323/lzyj_100471/qzfjz/"),
]

print("=" * 100)
print("【第 1 步 查重】")
doms = ["jinyun.gov.cn", "sthjt.jiangxi.gov.cn", "binyang.gov.cn", "ahqy.gov.cn", "dazu.gov.cn"]
fj = set(os.listdir("/root/gov_crawler"))
scripts = [f for f in fj if f.endswith(".py")]
cfg = json.load(open("/root/gov_crawler/daily_crawl_config.json", encoding="utf-8"))
blobs = [json.dumps(c, ensure_ascii=False) for c in cfg]

import sqlite3
db = sqlite3.connect("/root/search.db", timeout=30)
db.execute("PRAGMA busy_timeout=30000")

for d in doms:
    pat = re.compile(r"://(?:www\.)?" + re.escape(d) + "/", re.I)
    hit_s, hit_c = [], []
    for f in scripts:
        try:
            t = open("/root/gov_crawler/" + f, encoding="utf-8", errors="ignore").read()
        except Exception:
            continue
        if pat.search(t) or (d.split(".")[0] + ".") in t and d in t:
            if pat.search(t):
                hit_s.append(f)
    for i, b in enumerate(blobs):
        if pat.search(b):
            hit_c.append(i + 1)
    n_db = db.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url LIKE ?", ("%://" + d + "/%",)).fetchone()[0]
    print("  %-26s 脚本:%-30s config编号:%-14s 库内:%d 条" % (
        d, (",".join(hit_s[:3]) or "无"), (",".join(map(str, hit_c[:3])) or "无"), n_db))

print()
print("=" * 100)
print("【第 2 步 结构探测】")
for tag, name, url in SITES:
    print("\n" + "-" * 100)
    print("【%s】%s" % (tag, url))
    try:
        r = requests.get(url, headers=H, timeout=30, verify=False, allow_redirects=True)
        r.encoding = r.apparent_encoding or "utf-8"
        h = r.text
    except Exception as e:
        print("   ❌ 失败:", str(e)[:90])
        continue
    m = re.search(r"<title[^>]*>(.*?)</title>", h, re.S | re.I)
    print("   HTTP %s | 最终URL %s | 字节 %d" % (r.status_code, r.url[:70], len(h)))
    print("   title: %s" % (m.group(1).strip()[:90] if m else "(无)"))
    fp = {k: [p for p in v if p in h] for k, v in {
        "TRS": ["trs_editor", "createPageHTML", "dynclicks", "TRS_Editor"],
        "jcms(浙江)": ["jsearch", "col/col", "__NEXT_DATA", "jcms"],
        "JPAAS": ["unitbuild", "paramJson", "api-gateway"],
        "Lonsun": ["label/8888", "site/label"],
        "安徽Jczwgk": ["Jczwgk", "opennessList", "OpennessContent"],
        "Hanweb": ["dataproxy.jsp", "jpage"],
        "UCAP": ["UCAPCONTENT", "zoomcon"],
        "vue/spa": ["__NUXT__", "window.__INITIAL", "app.js"],
    }.items()}
    print("   指纹: %s" % {k: v for k, v in fp.items() if v})
    # 分页线索
    pgs = {}
    for pat in [r"createPageHTML\([^)]*\)", r"page_\d+\.html", r"pagecount[\"'\s:=]+(\d+)",
                r"totalpage[\"'\s:=]+(\d+)", r"col\d+/index_\d+\.html", r"共\s*(\d+)\s*条",
                r"index_\d+\.html", r"pageIndex", r"\.jsp\?.*page", r"jpage"]:
        mm = re.findall(pat, h, re.I)
        if mm:
            pgs[pat[:26]] = list(dict.fromkeys([str(x) for x in mm]))[:4]
    print("   分页: %s" % pgs)
    # 条目
    lis = re.findall(r"<li[^>]*>.*?</li>", h, re.S | re.I)
    dated = [x for x in lis if re.search(r"20\d{2}[-/.]\d{1,2}[-/.]\d{1,2}", x) and "<a" in x]
    print("   li 总数 %d | 含日期+链接 %d" % (len(lis), len(dated)))
    for x in dated[:2]:
        print("      %s" % re.sub(r"\s+", " ", x)[:250])
    # URL 形态
    hrefs = re.findall(r'href=["\']([^"\']+)["\']', h)
    pc = Counter()
    for x in hrefs:
        if x.startswith(("javascript:", "#", "mailto:", "http://www.w3.org")):
            continue
        pc[re.sub(r"\d+", "N", x)[:58]] += 1
    print("   URL 形态 top6: %s" % pc.most_common(6))
db.close()
