#!/usr/bin/env python3
"""Fetch detail pages for zh.gov.cn from saved URLs list"""
import json, re, time, os, sqlite3
from datetime import datetime, date
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests

BASE = "https://www.zh.gov.cn"
CUTOFF_DATE = date(2023, 6, 17)
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "镇海区-生态环境分局-通知公告"
URLS_FILE = "/root/gov_crawler/zh_urls.json"
MAX_WORKERS = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}


def fetch_url(url):
    for i in range(3):
        try:
            r = requests.get(url, timeout=30, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code == 200 and r.text:
                return r.text
        except:
            if i < 2:
                time.sleep(2)
    return None


def fetch_detail(url):
    html = fetch_url(url)
    if not html:
        return (url, None, None, None)
    title = ""
    content = ""
    pub_date = ""
    m = re.search(r'<title>(.*?)</title>', html)
    if m:
        title = m.group(1).strip()
    m = re.search(r'<div class="content"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    # Try meta date
    m2 = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    if m2:
        pub_date = m2.group(1)
    return (url, title, content, pub_date)


def main():
    print("=== 镇海区-生态环境分局-通知公告 (详情抓取) ===", flush=True)
    
    with open(URLS_FILE) as f:
        urls = json.load(f)
    print(f"  从文件加载 {len(urls)} 条URL", flush=True)
    
    conn = sqlite3.connect(DB_PATH)
    cursor = conn.cursor()
    total_new = 0
    total_skip = 0
    done = 0
    
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        fut_map = {executor.submit(fetch_detail, url): url for url in urls}
        for fut in as_completed(fut_map):
            url, title, content, pub_date = fut.result()
            done += 1
            if done % 50 == 0:
                print(f"  详情 {done}/{len(urls)}...", flush=True)
            
            if title is None:
                continue
            if not content:
                content = title
            if not pub_date:
                from urllib.parse import urlparse
                dm = re.search(r'/art/(\d{4})/', url)
                pub_date = f"{dm.group(1)}-01-01" if dm else "2026-01-01"
            
            try:
                d = date.fromisoformat(pub_date[:10])
                if d < CUTOFF_DATE:
                    total_skip += 1
                    continue
            except:
                pass
            
            try:
                cursor.execute("""
                    INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (SITE_NAME, title, url, content, pub_date[:10], content,
                      0))
                if cursor.rowcount > 0:
                    total_new += 1
            except:
                pass
            conn.commit()
    
    cursor.execute("SELECT COUNT(*), MIN(publish_date), MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    cnt, min_d, max_d = cursor.fetchone()
    conn.close()
    
    print(f"\n=== 完成 ===", flush=True)
    print(f"  新增: {total_new}条", flush=True)
    print(f"  跳过(超出3年): {total_skip}条", flush=True)
    print(f"  累计: {cnt}条 ({min_d} ~ {max_d})", flush=True)


if __name__ == "__main__":
    main()
