#!/usr/bin/env python3
"""
Crawl 衡水高新区 - 生态环境 (hskfq.gov.cn/col/col11730/)
Data loaded via AJAX proxy: /module/web/jpage/dataproxy.jsp
Detail: div.neirong / div.nr
"""

import sys, re, time
import os, requests, urllib3
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

BASE_URL = "http://www.hskfq.gov.cn"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "衡水高新区 - 生态环境"

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
})
session.verify = False
urllib3.disable_warnings()

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

AJAX_URL = "/module/web/jpage/dataproxy.jsp"
AJAX_PARAMS = "page={}&appid=1&webid=15&path=/&columnid=11730&unitid=4063&webname={}&permissiontype=0"
WEBNAME = "河北衡水高新技术产业开发区管理委员会"

def fetch_list_page(page=1):
    """Call AJAX proxy, return items"""
    import urllib.parse
    params = AJAX_PARAMS.format(page, urllib.parse.quote(WEBNAME))
    url = f"{BASE_URL}{AJAX_URL}?{params}"
    try:
        resp = session.get(url, timeout=20)
        if resp.status_code != 200:
            return None, 0, 0
        xml = resp.text
        import xml.etree.ElementTree as ET
        root = ET.fromstring(xml)
        total_records = int(root.findtext('totalrecord', '0'))
        total_pages = int(root.findtext('totalpage', '0'))
        records = []
        for record in root.findall('.//record'):
            cdata = record.text.strip() if record.text else ''
            if cdata:
                records.append(cdata)
        return records, total_records, total_pages
    except Exception as e:
        print(f"  AJAX error: {e}")
        return None, 0, 0

def parse_list_items(cdata_list):
    """Parse CDATA HTML items"""
    records = []
    for cdata in cdata_list:
        soup = BeautifulSoup(cdata, 'html.parser')
        a = soup.find('a')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        if not href or not title:
            continue
        if href.startswith('/'):
            href = BASE_URL + href
        elif not href.startswith('http'):
            href = BASE_URL + '/' + href.lstrip('/')
        # Date from span [2025-12-01]
        span = soup.find('span')
        date_str = span.get_text(strip=True).strip('[]') if span else ''
        records.append({"title": title, "url": href, "date": date_str})
    return records

def fetch_detail(url):
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = 'utf-8'
    except Exception:
        return "", "", ""
    soup = BeautifulSoup(resp.text, 'html.parser')
    content_div = soup.find('div', class_='nr')
    if not content_div:
        content_div = soup.find('div', class_='neirong')
    content_html = str(content_div) if content_div else ""
    date_str = ""
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        date_str = meta['content'].strip()[:10]
    full_title = ""
    h1 = soup.find('h1')
    if h1:
        full_title = h1.get_text(strip=True)
    return content_html, date_str, full_title

def insert_to_db(records, conn):
    cursor = conn.cursor()
    inserted = 0
    skipped = 0
    for rec in records:
        if rec["date"] and rec["date"] < THREE_YEARS_AGO:
            skipped += 1
            continue
        try:
            cursor.execute("""
                INSERT OR IGNORE INTO gov_raw 
                (page_url, source_url, title, summary, content, publish_date, site_name)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (rec["url"], BASE_URL, rec["title"], "", rec.get("content", ""), rec.get("date", ""), SITE_NAME))
            if cursor.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  DB error: {e}")
    conn.commit()
    return inserted, skipped

def main():
    incremental = 'incremental' in sys.argv or sys.argv[-1] == '1'
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    print(f"=== {SITE_NAME} ===")

    # Fetch page 1
    cdata_list, total, total_pages = fetch_list_page(1)
    if cdata_list is None:
        print("Failed to fetch data")
        conn.close()
        return

    all_records = parse_list_items(cdata_list)
    print(f"  Page 1: {len(all_records)} records (total: {total}, pages: {total_pages})")

    # Fetch remaining pages if not incremental
    if not incremental:
        max_pages = min(5, total_pages)
        for page in range(2, max_pages + 1):
            cdata_list2, _, _ = fetch_list_page(page)
            if not cdata_list2:
                break
            records = parse_list_items(cdata_list2)
            if not records:
                break
            print(f"  Page {page}: {len(records)} records")
            all_records.extend(records)
            time.sleep(0.3)

    print(f"\nTotal collected: {len(all_records)}")
    print("Fetching details...")
    for i, rec in enumerate(all_records):
        if i % 10 == 0:
            print(f"  {i}/{len(all_records)}", flush=True)
        content, date_from_detail, full_title = fetch_detail(rec["url"])
        if content:
            rec["content"] = content
        if date_from_detail:
            rec["date"] = date_from_detail
        if full_title:
            rec["title"] = full_title

    inserted, skipped = insert_to_db(all_records, conn)
    conn.close()
    print(f"\n=== SUMMARY ===")
    print(f"Inserted: {inserted}, Skipped (old): {skipped}")

if __name__ == "__main__":
    main()
