#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""diag_jxth_parse.py —— 对比脚本的解析 与 接口真实返回结构"""
import json, re, ssl, urllib.parse, urllib.request
D = "/root/gov_crawler/"; f = "crawl_jxth.py"
src = open(D + f, encoding="utf-8").read()
print("=== 脚本里的 fetch_api + 取列表的代码 ===")
for i, ln in enumerate(src.split("\n"), 1):
    if re.search(r"def fetch_api|ajax_type|json\.loads|\.get\(|\[\"|for .* in .*items|page_size|pagesize|current", ln):
        print("  L%-4d %s" % (i, ln.rstrip()[:118]))
print("\n=== 接口真实返回结构 ===")
ctx = ssl.create_default_context(); ctx.check_hostname = False; ctx.verify_mode = ssl.CERT_NONE
B = "https://www.jxth.gov.cn"
H = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 Chrome/141.0.0.0 Safari/537.36",
     "X-Requested-With": "XMLHttpRequest", "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
     "Referer": B + "/"}
body = [("ajax_type[]", "9_xxgk"), ("ajax_type[]", "167445"), ("ajax_type[]", "9"),
        ("ajax_type[]", "xxgk"), ("ajax_type[]", "Y-m-d"), ("ajax_type[]", "50"), ("ajax_type[]", "20"),
        ("ajax_type[]", "is_top DESC"), ("ajax_type[]", "displayorder DESC"),
        ("ajax_type[]", "inputtime DESC"), ("ajax_type[]", ""), ("is_ds", "1")]
r = urllib.request.urlopen(urllib.request.Request(B + "/api-ajax_list-1.html", data=urllib.parse.urlencode(body).encode(), headers=H), timeout=20, context=ctx)
t = r.read(300000).decode("utf-8", "ignore")
j = json.loads(t)
def shape(o, dep=0):
    if dep > 2: return type(o).__name__
    if isinstance(o, dict):
        return "{" + ", ".join("%s:%s" % (k, shape(v, dep + 1)) for k, v in list(o.items())[:8]) + "}"
    if isinstance(o, list):
        return "[%s x%d]" % (shape(o[0], dep + 1) if o else "?", len(o))
    return type(o).__name__
print("  顶层:", shape(j)[:600])
if isinstance(j, dict):
    for k, v in j.items():
        if isinstance(v, list) and v:
            print("  列表字段 %r 首元素键: %s" % (k, list(v[0].keys())[:12] if isinstance(v[0], dict) else type(v[0]).__name__))
