#!/usr/bin/env python3
"""西固区人民政府 — 环评信息 (xigu.gov.cn)
http://www.xigu.gov.cn/col/col13286/index.html
TRS jPage, dataStore embedded XML, 32 records total
Detail: meta ArticleTitle + meta PubDate + div.bt_content
"""

import requests, re, sys, os, time, sqlite3
from bs4 import BeautifulSoup

BASE_URL = "http://www.xigu.gov.cn"
LIST_URL = BASE_URL + "/col/col13286/index.html"
SITE_NAME = "西固区人民政府-环评信息"
GROUP = "甘肃"
INDUSTRY = "环评公示"
SCRIPT_NAME = "crawl_xigu_hj.py"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

session = requests.Session()
session.headers.update(HEADERS)
session.verify = False

import urllib3
urllib3.disable_warnings()


def fetch(url):
    for i in range(3):
        try:
            r = session.get(url, timeout=15)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            print(f"[WARN] {url[:60]} failed: {e}")
        time.sleep(2)
    return None


def parse_list(html):
    """Extract items from embedded XML dataStore"""
    items = []
    m = re.search(r'<div id="81919">.*?<script[^>]*type="text/xml">(.*?)</script>', html, re.DOTALL)
    if not m:
        return items
    
    xml_data = m.group(1)
    records = re.findall(
        r'<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>.*?</a><b>([^<]*)</b>',
        xml_data, re.DOTALL
    )
    for href, title, date in records:
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append((title.strip(), href.strip(), date.strip()))
    return items


def parse_detail(html, url):
    """Parse detail page"""
    soup = BeautifulSoup(html, "html.parser")
    
    # Title
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        m = re.search(r"<title>(.*?)</title>", html, re.DOTALL)
        if m:
            title = m.group(1).replace("西固区人民政府 环评信息 ", "").strip()
    
    # Date
    date_str = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
        if m:
            date_str = m.group(1)
    if not date_str:
        m = re.search(r"发布日期[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})", html)
        if m:
            date_str = m.group(1).replace("/", "-")
    
    # Content
    content = ""
    body = soup.find("div", id="zoom") or soup.find("div", class_="zfxxgk_pageCon")
    if body:
        content = str(body)
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
        content = content.strip()
    
    return title, date_str, content


def crawl():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cursor = conn.cursor()
    
    print(f"[{SCRIPT_NAME}] Fetching list page...")
    html = fetch(LIST_URL)
    if not html:
        print("[ERROR] Cannot load list page")
        return
    
    items = parse_list(html)
    print(f"[{SCRIPT_NAME}] Found {len(items)} items")
    
    all_count = 0
    skip_count = 0
    
    for title, item_url, date_str in items:
        existing = cursor.execute(
            "SELECT id FROM gov_raw WHERE page_url = ?", (item_url,)
        ).fetchone()
        if existing:
            skip_count += 1
            continue
        
        detail_html = fetch(item_url)
        if not detail_html:
            skip_count += 1
            continue
        
        detail_title, detail_date, content = parse_detail(detail_html, item_url)
        if not detail_title:
            detail_title = title
        if not detail_date:
            detail_date = date_str
        if not content or len(content.strip()) < 50:
            content = "正文为空"
        
        try:
            cursor.execute("""
                INSERT INTO gov_raw (source_url, page_url, title, publish_date, site_name, content, group_name, industry)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (item_url, item_url, detail_title, detail_date, SITE_NAME, content, GROUP, INDUSTRY))
            conn.commit()
            all_count += 1
            print(f"[{SCRIPT_NAME}] +{all_count}: {detail_title[:40]} ({detail_date})")
        except Exception as e:
            conn.rollback()
            skip_count += 1
        
        time.sleep(0.5)
    
    conn.close()
    print(f"\n[{SCRIPT_NAME}] Done. New: {all_count}, Skipped: {skip_count}")

if __name__ == "__main__":
    crawl()
