#!/usr/bin/env python3
"""
crawl_siyang.py - 泗阳县政府政务公告爬虫
CMS: UCAPCONTENT, 分页 list_{n}.shtml
"""
import requests
import re
import sqlite3
import time
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
SITE_NAME = "泗阳县人民政府-政务公告"
BASE_URL = "http://www.siyang.gov.cn"
LIST_URL = BASE_URL + "/siyang/zhwgg/list.shtml"
TOTAL_PAGES = 67
DELAY = 0.5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

def get_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  [ERROR] {e}")
        return ""

def parse_list(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="news-list")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href or "zhwgg" not in href:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date_span = li.find("span", class_="date")
        date = date_span.get_text(strip=True) if date_span else ""
        items.append((href, title, date))
    return items

def extract_detail(html, url, list_date):
    soup = BeautifulSoup(html, "html.parser")
    
    # title
    title = ""
    h2 = soup.find("h2", id="zoomsubtitl")
    if h2:
        title = h2.get_text(strip=True)
    if not title:
        meta = soup.find("meta", {"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()
    
    # date
    publish_date = list_date
    pt = soup.find("publishtime")
    if pt:
        d = pt.get_text(strip=True)
        if d:
            publish_date = d
    
    # content
    content_html = ""
    ucap = soup.find("ucapcontent")
    if ucap:
        parts = []
        for child in ucap.children:
            if child.name == "table":
                parts.append(str(child))
            elif child.name == "p":
                parts.append(str(child))
            elif child.name == "img":
                parts.append(str(child))
            elif child.name == "div":
                for sub in child.find_all(["p", "table", "img"], recursive=False):
                    parts.append(str(sub))
        content_html = "\n".join(parts)
    
    # attachments
    attachments = []
    for link in soup.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|rar|zip)$", re.I)):
        att_url = link.get("href", "")
        if att_url and not att_url.startswith("http"):
            base_dir = url[:url.rfind("/")]
            att_url = base_dir + "/" + att_url.lstrip("/")
        att_title = link.get_text(strip=True)
        attachments.append(f"[{att_title}]({att_url})")
    
    if not content_html.strip() and ucap:
        content_html = str(ucap)
    
    return title, publish_date, content_html, attachments

def render_markdown(content_html, url):
    soup = BeautifulSoup(content_html, "html.parser")
    parts = []
    
    for child in soup.children:
        if child.name == "p":
            text = child.get_text("\n", strip=True)
            imgs = child.find_all("img")
            for img in imgs:
                src = img.get("src", "")
                alt = img.get("alt", "")
                if src and not src.startswith("http"):
                    # Image relative to page URL dir
                    base_dir = url[:url.rfind("/")]
                    src = base_dir + "/" + src.lstrip("/")
                text += f"\n![{alt}]({src})"
            if text.strip():
                parts.append(text)
        elif child.name == "table":
            rows = []
            for tr in child.find_all("tr"):
                cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                if cells:
                    rows.append("| " + " | ".join(cells) + " |")
            if rows:
                parts.append("\n".join(rows))
        elif child.name == "img":
            src = child.get("src", "")
            alt = child.get("alt", "")
            if src and not src.startswith("http"):
                base_dir = url[:url.rfind("/")]
                src = base_dir + "/" + src.lstrip("/")
            parts.append(f"![{alt}]({src})")
        elif child.name is None:
            text = str(child).strip()
            if text:
                parts.append(text)
    
    return "\n\n".join(parts)

def run():
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    total_new = 0
    
    for page in range(1, TOTAL_PAGES + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = f"{BASE_URL}/siyang/zhwgg/list_{page}.shtml"
        
        print(f"--- Page {page}/{TOTAL_PAGES} ---")
        html = get_page(url)
        if not html:
            continue
        
        items = parse_list(html)
        if not items:
            continue
        
        print(f"  Found {len(items)} items")
        
        for href, title, date in items:
            try:
                detail_html = get_page(href)
                if not detail_html:
                    continue
                
                full_title, pub_date, content_html, attachments = extract_detail(detail_html, href, date)
                if not full_title:
                    full_title = title
                if not pub_date:
                    pub_date = date
                
                content_md = render_markdown(content_html, href)
                plain = re.sub(r"<[^>]+>", "", content_html).strip()
                summary = plain[:200] if len(plain) > 200 else plain
                
                c.execute("""
                    INSERT OR REPLACE INTO gov_raw 
                    (page_url, site_name, title, publish_date, summary, content, source_url, date_rank, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
                """, (
                    href, SITE_NAME, full_title, pub_date, summary, content_md,
                    href, int(time.time()), "\n".join(attachments)
                ))
                conn.commit()
                total_new += 1
                time.sleep(DELAY)
            except Exception as e:
                print(f"  [ERR] {href.split('/')[-1][:24]}: {e}")
                continue
        time.sleep(DELAY)
    
    conn.close()
    print(f"\nDone! Upserted: {total_new}")

if __name__ == "__main__":
    run()
