#!/usr/bin/env python3
"""
武穴市-政府信息公开 爬虫
http://www.wuxue.gov.cn/zwgk/public/column/6636855?type=4&catId=7025450&action=list

CMS: Hanweb (大汉) 政府信息公开系统
列表: ul > li > a.title + span.date (单页20条，无分页)
详情: div.wzcon.j-fontContent > p
"""
import sys
import time
import re
import requests
import sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/mnt/data/search.db"
LIST_URL = "http://www.wuxue.gov.cn/zwgk/public/column/6636855?type=4&catId=7025450&action=list"
BASE_URL = "http://www.wuxue.gov.cn"
SITE_NAME = "武穴市-政府信息公开"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

def crawl_list():
    """Parse the list page"""
    r = requests.get(LIST_URL, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    if r.status_code != 200:
        print(f"  HTTP {r.status_code}")
        return None
    
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # Find the main content ul
    items = []
    for ul in soup.find_all('ul'):
        for li in ul.find_all('li'):
            a = li.find('a', class_='title', href=True)
            if not a:
                continue
            href = a['href'].strip()
            if not href.startswith(BASE_URL) and not href.startswith('/zwgk/public/'):
                continue
            if href.startswith('/'):
                href = urljoin(BASE_URL, href)
            
            title = a.get_text(strip=True)
            if len(title) < 5:
                continue
            
            span = li.find('span', class_='date')
            date = span.get_text(strip=True) if span else ''
            
            items.append({'title': title, 'url': href, 'date': date})
    
    return items

def fetch_detail(item):
    """Parse detail page content"""
    r = requests.get(item['url'], headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    if r.status_code != 200:
        return None
    
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # Content div - try multiple selectors
    content_div = soup.select_one('div.wzcon')
    if not content_div:
        content_div = soup.select_one('div#wenzhang')
    if not content_div:
        content_div = soup.select_one('div.ls-lmcontent')
    if not content_div:
        content_div = soup.select_one('div.gkwz_contnet')
    
    if not content_div:
        print(f"  No content div: {item['url']}")
        return None
    
    for tag in content_div.find_all(['script', 'style', 'iframe']):
        tag.decompose()
    
    # Recursive extraction preserving tables and paragraph breaks
    parts = []
    def extract(node):
        for child in node.children:
            t = getattr(child, 'name', None)
            if t == 'p':
                txt = child.get_text(separator='', strip=True)
                if txt:
                    parts.append(txt)
            elif t == 'table':
                parts.append(str(child))
            elif t == 'div':
                extract(child)
            elif t is None and isinstance(child, str):
                txt = child.strip()
                if txt and len(txt) > 3:
                    parts.append(txt)
    
    extract(content_div)
    content = '\n\n'.join(parts) if parts else content_div.get_text(separator='', strip=True)
    
    if not content or len(content) < 20:
        print(f"  Content too short: {item['url']}")
        return None
    
    # Title
    title_tag = soup.find('title')
    full_title = title_tag.get_text(strip=True) if title_tag else ''
    full_title = re.sub(r'\s*[-_―]\s*武穴市人民政府.*', '', full_title).strip()
    if not full_title or len(full_title) < 5:
        full_title = item['title']
    
    return {
        'title': full_title,
        'content': content,
        'date': item['date'],
        'url': item['url'],
        'site_name': SITE_NAME,
    }

def save_to_db(records):
    if not records:
        return 0
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for rec in records:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, content, page_url, publish_date, site_name) "
                "VALUES (?, ?, ?, ?, ?)",
                (rec['title'], rec['content'], rec['url'], rec['date'], rec['site_name'])
            )
            if c.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  DB error: {e}")
    conn.commit()
    conn.commit()
    conn.close()
    return inserted

def main():
    print(f"=== {SITE_NAME} ===")
    print(f"List URL: {LIST_URL}")
    
    items = crawl_list()
    if not items:
        print("  No items found")
        return
    
    print(f"  Found {len(items)} items")
    
    batch = []
    for i, item in enumerate(items):
        print(f"  [{i+1}/{len(items)}] {item['title'][:40]}...")
        detail = fetch_detail(item)
        if detail:
            batch.append(detail)
        time.sleep(0.3)
    
    inserted = save_to_db(batch)
    print(f"\n{'='*40}")
    print(f"Inserted: {inserted}/{len(batch)} new")

if __name__ == '__main__':
    main()
