#!/usr/bin/env python3
"""Only collect URLs from zh.gov.cn via Playwright, save to file"""
import json, time, re
from playwright.sync_api import sync_playwright

BASE = "https://www.zh.gov.cn"
OUTPUT = "/root/gov_crawler/zh_urls.json"

with sync_playwright() as p:
    browser = p.chromium.launch(headless=True, args=["--no-sandbox", "--disable-gpu"])
    page = browser.new_page()
    page.goto("https://www.zh.gov.cn/col/col1229821392/index.html", 
              wait_until="commit", timeout=60000)
    time.sleep(3)
    
    all_urls = set()
    for pg in range(1, 100):
        html = page.content()
        urls = re.findall(r'href="(/col/col1229821392/art/[^"]+\.html)"', html)
        for u in urls:
            all_urls.add(BASE + u)
        
        dates = re.findall(r'su_bright[^>]*>([^<]+)</b>', html)
        fd = dates[0] if dates else "N/A"
        ld = dates[-1] if dates else "N/A"
        print(f"Page {pg}: {len(urls)} items [{fd} ~ {ld}] total: {len(all_urls)}")
        
        next_btn = page.query_selector("a.layui-laypage-next")
        if not next_btn:
            break
        dp = next_btn.get_attribute("data-page")
        if not dp or int(dp) <= pg:
            break
        next_btn.click()
        time.sleep(2)
    
    browser.close()
    
    with open(OUTPUT, "w") as f:
        json.dump(list(all_urls), f)
    print(f"\nSaved {len(all_urls)} URLs to {OUTPUT}")
