#!/usr/bin/env python3
import os
"""crawl_pyjkq.py - 濮阳经济技术开发区 公告公示"""
import re, time, sys, os
import urllib.request, urllib.error
from datetime import datetime

BASE_URL = "http://www.pyjkq.gov.cn"
CHANNEL_URL = BASE_URL + "/channel/list/16771"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "濮阳经济技术开发区-公告公示"

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}

MAX_PAGES = 2  # Only pages 1-2 have working details
TIMEOUT_FAST = 5  # Fast timeout for 404 pages

def fetch(url, retries=2):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            with urllib.request.urlopen(req, timeout=10) as resp:
                return resp.read().decode('utf-8', errors='replace')
        except urllib.error.HTTPError as e:
            if e.code == 404:
                return None  # Don't retry 404
            if i < retries - 1:
                time.sleep(1)
        except Exception:
            if i < retries - 1:
                time.sleep(1)
            else:
                return None
    return None

def extract_items_from_page(html):
    items = []
    lis = re.findall(r'<li>(.*?)</li>', html, re.DOTALL)
    for li in lis:
        m = re.search(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', li)
        m2 = re.search(r'<b>(.*?)</b>', li)
        if m and m2:
            href = m.group(1)
            title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
            date_str = m2.group(1).strip()
            if title and href.startswith('http'):
                items.append((title, href, date_str))
    return items

def extract_content(html):
    m = re.search(r'class="xw"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        return m.group(1).strip()
    return ""

def extract_date(html, default_date):
    m = re.search(r'时间：(\d{4}-\d{2}-\d{2})', html)
    if m:
        return m.group(1)
    return default_date

def extract_title(html, fallback_title):
    m = re.search(r'<h1>(.*?)</h1>', html)
    if m:
        title = m.group(1).strip()
        if title:
            return title
    return fallback_title

def main():
    print("[%s] Starting crawl: %s" % (datetime.now().isoformat(), SITE_NAME))

    all_items = []

    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            url = CHANNEL_URL + '.html'
        else:
            url = CHANNEL_URL + '_%d.html' % page
        print("Fetching page %d..." % page)
        html = fetch(url)
        if not html:
            print("  Failed, stopping")
            break
        items = extract_items_from_page(html)
        print("  Found %d items" % len(items))
        all_items.extend(items)
        time.sleep(0.5)

    print("\nTotal items to process: %d" % len(all_items))

    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    c = conn.cursor()

    inserted = 0
    skipped = 0

    for idx, (title, page_url, date_str) in enumerate(all_items):
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,))
        if c.fetchone():
            skipped += 1
            continue

        detail_html = fetch(page_url)
        if not detail_html:
            print("  [SKIP] 404/No response: %s" % title[:30])
            skipped += 1
            continue

        content = extract_content(detail_html)
        if not content:
            print("  [SKIP] No content: %s" % title[:30])
            skipped += 1
            continue

        clean_title = extract_title(detail_html, title)
        pub_date = extract_date(detail_html, date_str)

        try:
            c.execute("""INSERT OR REPLACE INTO gov_raw (site_name, source_url, page_url, title, publish_date, content, status, summary, script_name) VALUES (?, ?, ?, ?, ?, ?, 'published', ?, 'crawl_pyjkq.py')""",
                (SITE_NAME, CHANNEL_URL, page_url, clean_title, pub_date, content,
                 content[:200].replace('\n', ' ').strip()))
            inserted += 1
            if inserted % 10 == 0:
                conn.commit()
                print("  Progress: %d inserted" % inserted)
        except Exception as e:
            print("  [ERR] DB error for %s: %s" % (page_url, e))
            skipped += 1

        time.sleep(0.3)

    conn.commit()
    conn.close()

    print("\n=== Completed ===")
    print("Inserted: %d" % inserted)
    print("Skipped: %d" % skipped)
    print("Total: %d" % len(all_items))

if __name__ == '__main__':
    main()
