#!/usr/bin/env python3
import os
"""crawl_qdncg.py - 岑巩县人民政府 项目环评"""
import re, time, sys, os
import urllib.request, urllib.error
from datetime import datetime

BASE_URL = "https://www.qdncg.gov.cn/ztzl/rdzt/cgxhbzt/xmhp"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "岑巩县-项目环评"

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}

def fetch(url, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            with urllib.request.urlopen(req, timeout=30) as resp:
                return resp.read().decode('utf-8', errors='replace')
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print("  [WARN] Failed to fetch %s: %s" % (url, e))
                return None

def extract_items_from_page(html):
    items = []
    pattern = r'<a[^>]*TARGET="_blank"[^>]*title="([^"]*)"[^>]*href="([^"]*)"[^>]*>.*?<span>([^<]*)</span>'
    for m in re.finditer(pattern, html, re.DOTALL):
        title = m.group(1).strip()
        href = m.group(2).strip()
        date_str = m.group(3).strip()
        items.append((title, href, date_str))
    return items

def extract_content(html):
    m = re.search(r'class="trs_editor_view[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        return m.group(1).strip()
    # Fallback: try nry
    m = re.search(r'class="nry"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        return m.group(1).strip()
    return ""

def extract_date(html, default_date):
    m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    if m:
        return m.group(1)
    return default_date

def extract_title(html, fallback_title):
    m = re.search(r'<title>(.*?)</title>', html)
    if m:
        title = m.group(1)
        for suffix in ['-项目环评', '-岑巩县人民政府']:
            idx = title.rfind(suffix)
            if idx > 0:
                title = title[:idx]
                break
        return title.strip()
    return fallback_title

def main():
    print("[%s] Starting crawl: %s" % (datetime.now().isoformat(), SITE_NAME))

    all_items = []

    # Page 1
    print("Fetching page 1...")
    html = fetch(BASE_URL + '/')
    if html:
        items = extract_items_from_page(html)
        print("  Found %d items" % len(items))
        all_items.extend(items)

    # Page 2
    page2_url = BASE_URL + '/index_2.html'
    print("Fetching page 2...")
    html = fetch(page2_url)
    if html:
        items = extract_items_from_page(html)
        print("  Found %d items" % len(items))
        all_items.extend(items)

    print("\nTotal items to process: %d" % len(all_items))

    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    c = conn.cursor()

    inserted = 0
    skipped = 0

    for title, page_url, date_str in all_items:
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,))
        if c.fetchone():
            skipped += 1
            continue

        detail_html = fetch(page_url)
        if not detail_html:
            skipped += 1
            continue

        content = extract_content(detail_html)
        if not content:
            print("  [SKIP] No content: %s" % title[:30])
            skipped += 1
            continue

        clean_title = extract_title(detail_html, title)
        pub_date = extract_date(detail_html, date_str)

        try:
            c.execute("""INSERT OR REPLACE INTO gov_raw (site_name, source_url, page_url, title, publish_date, content, status, summary, script_name) VALUES (?, ?, ?, ?, ?, ?, 'published', ?, 'crawl_qdncg.py')""",
                (SITE_NAME, BASE_URL, page_url, clean_title, pub_date, content,
                 content[:200].replace('\n', ' ').strip()))
            inserted += 1
        except Exception as e:
            print("  [ERR] DB error for %s: %s" % (page_url, e))
            skipped += 1

        time.sleep(0.3)

    conn.commit()
    conn.close()

    print("\n=== Completed ===")
    print("Inserted: %d" % inserted)
    print("Skipped: %d" % skipped)
    print("Total: %d" % len(all_items))

if __name__ == '__main__':
    main()
