#!/usr/bin/env python3
"""商水县人民政府 — 生态环境 (shangshui.gov.cn)
https://www.shangshui.gov.cn/sitesources/ssx/page_pc/zwgk/zdxxgk/sthj/list1.html
CMS: AJAX script.json pagination, embedded static HTML list
Total: 2 articles only
Detail: meta ArticleTitle, meta PubDate, div.article-detail/div.articleDetail
"""

import requests, re, sys, os, time, sqlite3, json
from bs4 import BeautifulSoup

BASE_URL = "https://www.shangshui.gov.cn"
LIST_URL = BASE_URL + "/sitesources/ssx/page_pc/zwgk/zdxxgk/sthj/list1.html"
SITE_NAME = "商水县人民政府-生态环境"
GROUP = "河南"
INDUSTRY = "环评公示"
SCRIPT_NAME = "crawl_shangshui_sthj.py"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

session = requests.Session()
session.headers.update(HEADERS)
session.verify = False

import urllib3
urllib3.disable_warnings()


def fetch(url):
    for i in range(3):
        try:
            r = session.get(url, timeout=15)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            print(f"[WARN] {url[:60]} failed: {e}")
        time.sleep(2)
    return None


def parse_list(html):
    """Extract items from static HTML"""
    items = []
    # colRightOne divs with article links
    pattern = r'<div class="colRightOne">\s*<a href="([^"]*)"[^>]*title="([^"]*)"[^>]*>.*?<font>([^<]*)</font>'
    for m in re.finditer(pattern, html, re.DOTALL):
        href, title, date = m.group(1), m.group(2).strip(), m.group(3).strip()
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append((title, href, date))
    return items


def parse_detail(html, url):
    """Parse detail page"""
    soup = BeautifulSoup(html, "html.parser")
    
    # Title
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    
    # Date
    date_str = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
        if m:
            date_str = m.group(1)
    if not date_str:
        m = re.search(r"时间[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})", html)
        if m:
            date_str = m.group(1).replace("/", "-")
    
    # Content
    content = ""
    body = soup.find("div", class_="article-detail") or soup.find("div", class_="articleDetail")
    if body:
        content = str(body)
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
        content = content.strip()
    
    return title, date_str, content


def crawl():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cursor = conn.cursor()
    
    print(f"[{SCRIPT_NAME}] Fetching list page...")
    html = fetch(LIST_URL)
    if not html:
        print("[ERROR] Cannot load list page")
        return
    
    items = parse_list(html)
    print(f"[{SCRIPT_NAME}] Found {len(items)} items")
    
    all_count = 0
    skip_count = 0
    
    for title, item_url, date_str in items:
        existing = cursor.execute(
            "SELECT id FROM gov_raw WHERE page_url = ?", (item_url,)
        ).fetchone()
        if existing:
            skip_count += 1
            continue
        
        detail_html = fetch(item_url)
        if not detail_html:
            skip_count += 1
            continue
        
        detail_title, detail_date, content = parse_detail(detail_html, item_url)
        if not detail_title:
            detail_title = title
        if not detail_date:
            detail_date = date_str
        if not content or len(content.strip()) < 50:
            content = "正文为空"
        
        try:
            cursor.execute("""
                INSERT INTO gov_raw (source_url, page_url, title, publish_date, site_name, content, group_name, industry)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (item_url, item_url, detail_title, detail_date, SITE_NAME, content, GROUP, INDUSTRY))
            conn.commit()
            all_count += 1
            print(f"[{SCRIPT_NAME}] +{all_count}: {detail_title[:40]} ({detail_date})")
        except Exception as e:
            conn.rollback()
            skip_count += 1
        
        time.sleep(0.5)
    
    conn.close()
    print(f"\n[{SCRIPT_NAME}] Done. New: {all_count}, Skipped: {skip_count}")

if __name__ == "__main__":
    crawl()
