#!/usr/bin/env python3
"""广德市人民政府 - 建设项目环评文件审批 (guangde.gov.cn)"""
import os, sys, re, json, time, traceback, sqlite3
from datetime import datetime, timedelta
from urllib.parse import urljoin
import requests
# 支持 --pages 参数
import argparse as _AP
_AP_PARSER = _AP.ArgumentParser()
_AP_PARSER.add_argument("--pages", type=int, default=0, help="限制页数")
_AP_ARGS, _ = _AP_PARSER.parse_known_args()
_MAX_PAGES_ARG = _AP_ARGS.pages

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "https://www.guangde.gov.cn"
LIST_URL = BASE_URL + "/Jczwgk/showList/0/111001001/page_{n}.html"
SITE_NAME = "广德市建设项目环评文件审批"

CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}

def log(msg):
    print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}", flush=True)

def safe_get(url, retries=3):
    for attempt in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
            log(f"  HTTP {r.status_code} for {url}")
        except Exception as e:
            log(f"  Attempt {attempt+1} failed: {e}")
            time.sleep(2)
    return None

def parse_list(html):
    """Parse list page, return list of (title, url, date)"""
    items = []
    parts = re.split(r'<li[^>]*>', html)
    for part in parts:
        if '</li>' not in part:
            continue
        li = part[:part.index('</li>')]
        date_m = re.search(r'<span[^>]*>(\d{4}-\d{2}-\d{2})</span>', li)
        if not date_m:
            continue
        date_str = date_m.group(1)
        a_m = re.search(r'<a\s+href="([^"]+)"[^>]*title="([^"]*)"', li)
        if not a_m:
            continue
        href, title = a_m.group(1), a_m.group(2)
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)
        items.append((title.strip(), href, date_str))
    return items

def fetch_detail(url):
    """Fetch detail page, return (content_html, date_str)"""
    html = safe_get(url)
    if not html:
        return None, None
    date_m = re.search(r'<meta[^>]*PubDate[^>]*content="(\d{4}-\d{2}-\d{2}\s*\d{2}:\d{2})"', html)
    pub_date = date_m.group(1)[:10] if date_m else None
    # Extract content from div#zoom
    content_m = re.search(r'<div[^>]*id="zoom"[^>]*>(.*?)</div>\s*</div>\s*</div>', html, re.DOTALL)
    if not content_m:
        content_m = re.search(r'<div[^>]*id="zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
    content_html = content_m.group(1).strip() if content_m else ""
    return content_html, pub_date

def parse_total_pages(html):
    m = re.search(r'pagecount="(\d+)"', html)
    return int(m.group(1)) if m else 1

def main():
    log(f"Starting crawl: {SITE_NAME}")
    log(f"Cutoff date: {CUTOFF}")

    conn = sqlite3.connect(DB_PATH)

    html = safe_get(LIST_URL.format(n=1))
    if not html:
        log("ERROR: Cannot fetch page 1")
        return
    total_pages = parse_total_pages(html)
    if _MAX_PAGES_ARG > 0:
        total_pages = min(total_pages, _MAX_PAGES_ARG)
        log(f"Total pages: {total_pages}")

    new_count = 0
    skip_old = 0
    skip_empty = 0
    errors = 0

    for page in range(1, total_pages + 1):
        url = LIST_URL.format(n=page)
        log(f"Page {page}/{total_pages}...")

        html = safe_get(url)
        if not html:
            errors += 1
            continue

        items = parse_list(html)
        if not items:
            log(f"  No items on page {page}")
            continue

        log(f"  {len(items)} items")

        all_old = True
        for title, link, date_str in items:
            if date_str < CUTOFF:
                skip_old += 1
                continue
            all_old = False

            content, pub_date = fetch_detail(link)
            use_date = pub_date if pub_date else date_str

            if not content:
                log(f"  Empty: {title[:40]}")
                skip_empty += 1
                continue

            try:
                conn.execute(
                    "INSERT OR IGNORE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary) VALUES (?, ?, ?, ?, ?, ?, ?)",
                    (link, title, content, use_date, SITE_NAME, link, title)
                )
                if conn.total_changes > 0:
                    new_count += 1
            except Exception as e:
                log(f"  DB error: {e}")
                errors += 1

        if all_old and page > 1:
            log(f"  All items past cutoff, stopping")
            break

        # Commit every page
        conn.commit()

        time.sleep(0.5)

    conn.commit()
    conn.close()

    log(f"\n=== Summary ===")
    log(f"New entries: {new_count}")
    log(f"Skipped (old): {skip_old}")
    log(f"Skipped (empty): {skip_empty}")
    log(f"Errors: {errors}")
    log(f"Done!")

if __name__ == "__main__":
    main()
