#!/usr/bin/env python3
"""
Crawl 潢川县人民政府 - 环境保护 (huangchuan.gov.cn/zfxxgk/zdlyxxgk/hjbh/)
Static HTML with TRS pagination: index.html, index_2.html, ...
"""

import sys, re, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
import os

BASE_URL = "https://www.huangchuan.gov.cn"
LIST_PATH = "/zfxxgk/zdlyxxgk/hjbh"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "潢川县人民政府 - 环境保护"

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
})

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def parse_list_page(html_text):
    records = []
    soup = BeautifulSoup(html_text, 'html.parser')
    list_mian = soup.find('div', class_='list_mian')
    if not list_mian:
        return records
    ul = list_mian.find('ul')
    if not ul:
        return records
    for li in ul.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get_text(strip=True)
        if not href or not title:
            continue
        title = re.sub(r'<!--.*?-->', '', title).strip()
        if href.startswith('//'):
            href = 'https:' + href
        elif href.startswith('/'):
            href = BASE_URL + href
        elif not href.startswith('http'):
            href = BASE_URL + '/' + href.lstrip('/')
        span = li.find('span')
        date_str = span.get_text(strip=True) if span else ''
        records.append({"title": title, "url": href, "date": date_str})
    return records

def fetch_detail(url):
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = 'utf-8'
    except Exception:
        return "", "", ""
    soup = BeautifulSoup(resp.text, 'html.parser')
    # Content
    content_div = soup.find('div', class_='content', id='content')
    if not content_div:
        content_div = soup.find('div', class_=lambda c: c and 'content' in str(c).lower())
    content_html = str(content_div) if content_div else ""
    # Date
    date_str = ""
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        date_str = meta['content'].strip()[:10]
    # Full title from h1
    full_title = ""
    h1 = soup.find('h1')
    if h1:
        full_title = h1.get_text(strip=True)
    return content_html, date_str, full_title

def insert_to_db(records, conn):
    cursor = conn.cursor()
    inserted = 0
    skipped = 0
    for rec in records:
        if rec["date"] and rec["date"] < THREE_YEARS_AGO:
            skipped += 1
            continue
        try:
            cursor.execute("""
                INSERT OR IGNORE INTO gov_raw 
                (page_url, source_url, title, summary, content, publish_date, site_name)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (rec["url"], BASE_URL, rec["title"], "", rec.get("content", ""), rec.get("date", ""), SITE_NAME))
            if cursor.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  DB error: {e}")
    conn.commit()
    return inserted, skipped

def main():
    incremental = 'incremental' in sys.argv or sys.argv[-1] == '1'
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    print(f"=== {SITE_NAME} ===")
    if incremental:
        print("Mode: incremental (page 1 only)")
        pages_to_crawl = 1
    else:
        pages_to_crawl = 5

    url1 = f"{BASE_URL}{LIST_PATH}/index.html"
    resp = session.get(url1, timeout=30)
    resp.encoding = 'utf-8'
    all_records = parse_list_page(resp.text)
    print(f"  Page 1: {len(all_records)} records")

    if not incremental:
        soup = BeautifulSoup(resp.text, 'html.parser')
        page_div = soup.find('div', id='pageDec')
        if page_div:
            total = int(page_div.get('pagecount', '0'))
            per_page = int(page_div.get('pagesize', '24'))
            total_pages = (total + per_page - 1) // per_page if per_page > 0 else 1
            if total_pages > 0 and total_pages < pages_to_crawl:
                pages_to_crawl = total_pages
            print(f"  Total: {total} records, {total_pages} pages, crawling {pages_to_crawl}")

    for page in range(2, pages_to_crawl + 1):
        page_url = f"{BASE_URL}{LIST_PATH}/index_{page}.html"
        try:
            resp = session.get(page_url, timeout=30)
            resp.encoding = 'utf-8'
            if resp.status_code != 200:
                break
            records = parse_list_page(resp.text)
            if not records:
                break
            print(f"  Page {page}: {len(records)} records")
            all_records.extend(records)
            time.sleep(0.3)
        except Exception as e:
            print(f"  Page {page} error: {e}")
            break

    print(f"\nTotal collected: {len(all_records)}")
    print("Fetching details...")
    for i, rec in enumerate(all_records):
        if i % 10 == 0:
            print(f"  {i}/{len(all_records)}", flush=True)
        content, date_from_detail, full_title = fetch_detail(rec["url"])
        if content:
            rec["content"] = content
        if date_from_detail:
            rec["date"] = date_from_detail
        if full_title:
            rec["title"] = full_title

    inserted, skipped = insert_to_db(all_records, conn)
    conn.close()
    print(f"\n=== SUMMARY ===")
    print(f"Inserted: {inserted}, Skipped (old): {skipped}")

if __name__ == "__main__":
    main()
