#!/usr/bin/env python3
"""Analyze col1384 and col1611 article list structure."""
import requests, re

headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

# col1384 - look for all article links
r = requests.get("http://www.jxlc.gov.cn/col/col1384/index.html",
                 timeout=30, headers=headers, verify=False)
html = r.text

print("=== col1384: article links ===")
# Find all links containing art_1384
for m in re.finditer(r'<a[^>]*href="(/art/\d+/\d+/art_1384_\d+\.html)"[^>]*>(.*?)</a>', html):
    href = m.group(1)
    text = re.sub(r'<[^>]+>', '', m.group(2)).strip()
    print(f"  {text[:50]:52s} -> {href}")

# Check total count
total = re.search(r'totalRecord[=:"]*(\d+)', html)
if total:
    print(f"\nTotal record hint: {total.group(1)}")

# Look for proxyUrl pattern
for m in re.finditer(r'proxyUrl[:\s=]+["\']([^"\']+)["\']', html):
    print(f"proxyUrl: {m.group(1)[:120]}")

print("\n=== col1611: article links ===")
r2 = requests.get("http://www.jxlc.gov.cn/col/col1611/index.html",
                  timeout=30, headers=headers, verify=False)
html2 = r2.text
for m in re.finditer(r'<a[^>]*href="(/art/\d+/\d+/art_1611_\d+\.html)"[^>]*>(.*?)</a>', html2):
    href = m.group(1)
    text = re.sub(r'<[^>]+>', '', m.group(2)).strip()
    print(f"  {text[:50]:52s} -> {href}")

total2 = re.search(r'totalRecord[=:"]*(\d+)', html2)
if total2:
    print(f"\nTotal record hint: {total2.group(1)}")

# Now check the detail page structure
print("\n=== Sample detail page ===")
r3 = requests.get("http://www.jxlc.gov.cn/art/2025/12/15/art_1384_4395067.html",
                  timeout=30, headers=headers, verify=False)
html3 = r3.text
# Find meta tags
for m in re.finditer(r'<meta[^>]+name=["\'](ArticleTitle|PubDate|ContentSource)["\'][^>]*>', html3):
    print(f"  {m.group()}")

# Check for content div
for div_id in ['zoom', 'content', 'article', 'text', 'mainText', 'UCAP-CONTENT']:
    if f'id="{div_id}"' in html3 or f"id='{div_id}'" in html3:
        print(f"  Has div#{div_id}")

# Check for TRS_Editor
if '.TRS_Editor' in html3 or 'TRS_Editor' in html3:
    print("  Has .TRS_Editor")

# Look for date
for m in re.finditer(r'\d{4}-\d{2}-\d{2}', html3):
    ctx = html3[max(0,m.start()-30):m.end()+30]
    if any(x in ctx for x in ['pub', 'date', '时间', '日期']):
        print(f"  Date context: {ctx}")
        break
