#!/usr/bin/env python3
"""武川县人民政府 — 通知公告
https://www.wuchuan.gov.cn/zwgk/tzgg/
CMS: TRS，静态分页 index_N.html
详情: meta ArticleTitle + meta PubDate + div.TRS_UEDITOR
"""

import requests, re, sys, os, time, sqlite3
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

BASE_URL = "https://www.wuchuan.gov.cn"
LIST_BASE = "https://www.wuchuan.gov.cn/zwgk/tzgg/"
SITE_NAME = "武川县人民政府-通知公告"
GROUP = "内蒙古"
INDUSTRY = "政府公告"
SCRIPT_NAME = "crawl_wuchuan_tzgg.py"
MAX_PAGES = 5

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

session = requests.Session()
session.headers.update(HEADERS)
session.verify = False


def fetch(url):
    for i in range(3):
        try:
            r = session.get(url, timeout=15, allow_redirects=True)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if i < 2:
                time.sleep(2)
    return None


def parse_list(html):
    """提取 (title, full_url, date) 从列表页"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for a in soup.find_all("a"):
        href = a.get("href", "")
        txt = a.get_text(strip=True)
        if not txt or len(txt) < 10:
            continue
        # relative path like ./202607/t20260724_2023449.html
        if href.startswith("./") and ".html" in href:
            full_url = LIST_BASE + href[2:]
            items.append((txt.strip(), full_url))
    return items


def parse_detail(html):
    """提取标题、日期、正文"""
    soup = BeautifulSoup(html, "html.parser")
    
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()
    if not title:
        h = soup.find("h2")
        if h:
            title = h.get_text(strip=True)
    
    date_str = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
        if m:
            date_str = m.group(1)
    
    content = ""
    body = (
        soup.find("div", class_="TRS_UEDITOR")
        or soup.find("div", class_="trs_editor_view")
        or soup.find("div", class_="article_content")
        or soup.find("div", id="zoom")
    )
    if body:
        content = str(body)
        content = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', content, flags=re.DOTALL|re.I)
        content = content.strip()
    
    return title, date_str, content


def crawl():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cursor = conn.cursor()
    
    all_items = []
    for pg in range(1, MAX_PAGES + 1):
        url = LIST_BASE + (f"index_{pg}.html" if pg > 1 else "")
        html = fetch(url)
        if not html:
            break
        items = parse_list(html)
        if not items:
            break
        all_items.extend(items)
        print(f"[{SCRIPT_NAME}] Page {pg}: {len(items)} items")
        time.sleep(0.3)
    
    print(f"[{SCRIPT_NAME}] Total: {len(all_items)} items")
    
    new_count = skip_count = 0
    for title, item_url in all_items:
        existing = cursor.execute(
            "SELECT id FROM gov_raw WHERE page_url = ?", (item_url,)
        ).fetchone()
        if existing:
            skip_count += 1
            continue
        
        detail_html = fetch(item_url)
        if not detail_html:
            skip_count += 1
            continue
        
        real_title, real_date, content = parse_detail(detail_html)
        if not real_title:
            real_title = title
        if not content or len(content.strip()) < 50:
            content = "正文为空"
        
        try:
            cursor.execute("""
                INSERT INTO gov_raw (source_url, page_url, title, publish_date, site_name, content, group_name, industry)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (item_url, item_url, real_title, real_date, SITE_NAME, content, GROUP, INDUSTRY))
            conn.commit()
            new_count += 1
            print(f"[{SCRIPT_NAME}] +{new_count}: {real_title[:40]} ({real_date})")
        except Exception as e:
            conn.rollback()
            skip_count += 1
        
        time.sleep(0.5)
    
    conn.close()
    print(f"\n[{SCRIPT_NAME}] Done. New: {new_count}, Skipped: {skip_count}")


if __name__ == "__main__":
    crawl()
