#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""probe_zhcqhj_gsl.py —— 为什么 Found 0 articles"""
import re, ssl, urllib.request
D = "/root/gov_crawler/"; f = "crawl_zhcqhj_gsl.py"
src = open(D + f, encoding="utf-8", errors="ignore").read()
print("=== 脚本里的 URL 与解析 ===")
for i, ln in enumerate(src.split("\n"), 1):
    if re.search(r"BASE_URL|LIST|url|re\.compile|find_all|select|href|page_num|INDEX", ln):
        print("  L%-4d %s" % (i, ln.strip()[:112]))
ctx = ssl.create_default_context(); ctx.check_hostname = False; ctx.verify_mode = ssl.CERT_NONE
UA = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/141.0.0.0 Safari/537.36"}
for u in ("https://www.zhcqhj.com/html/news/gsl/", "https://www.zhcqhj.com/html/news/gsl/index.html",
          "https://www.zhcqhj.com/", "https://www.zhcqhj.com/html/news/gsl/index_1.html"):
    try:
        r = urllib.request.urlopen(urllib.request.Request(u, headers=UA), timeout=12, context=ctx)
        b = r.read(200000).decode("utf-8", "ignore")
        lis = len(re.findall(r"<li", b)); as_ = len(re.findall(r"<a\s", b))
        print("\n  %-52s HTTP=%s len=%d <li>=%d <a>=%d" % (u[:52], r.status, len(b), lis, as_))
        if len(b) > 200:
            print("    片段:", re.sub(r"\s+", " ", b[:260]))
    except Exception as e:
        print("\n  %-52s ERR %s" % (u[:52], type(e).__name__))
