#!/usr/bin/env python3
"""
滕州市-木石镇-通知公告 爬虫
http://www.tengzhou.gov.cn/zwgk/xxgkml/zj/msz/

CMS: TRS (天泽) with iframe-based search
列表: govsearch/searPageNewTengZhou.jsp (单页7条，分页为JS控制)
详情: div.view.TRS_UEDITOR
"""
import sys
import time
import re
import requests
import sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/mnt/data/search.db"
SEARCH_URL = "http://www.tengzhou.gov.cn/govsearch/searPageNewTengZhou.jsp"
BASE_URL = "http://www.tengzhou.gov.cn"
SITE_NAME = "滕州市-木石镇-通知公告"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

def crawl_list():
    """Parse the list page (single page)"""
    data = {
        "page": "1",
        "siteid": "189",
        "classinfoid": "2714",
        "channelid": "12633",
    }
    r = requests.post(SEARCH_URL, headers=HEADERS, data=data, timeout=30)
    r.encoding = "utf-8"
    if r.status_code != 200:
        return None
    
    soup = BeautifulSoup(r.text, "html.parser")
    
    items = []
    for ul in soup.find_all("ul"):
        for li in ul.find_all("li"):
            a = li.find("a", href=True)
            if not a:
                continue
            href = a["href"].strip()
            title = a.get_text(strip=True)
            if len(title) < 5:
                continue
            
            if not href.startswith("http"):
                href = urljoin(BASE_URL, href)
            if not href.endswith(".html"):
                href += ".html"
            
            # Extract date from URL
            date = ""
            m = re.search(r'/(20\d{4})/t(\d{8})_', href)
            if m:
                d = m.group(2)
                date = f"{d[:4]}-{d[4:6]}-{d[6:8]}"
            
            items.append({"title": title, "url": href, "date": date})
    
    return items

def fetch_detail(item):
    """Parse detail page content"""
    r = requests.get(item["url"], headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    if r.status_code != 200:
        return None
    
    soup = BeautifulSoup(r.text, "html.parser")
    
    content_div = soup.select_one("div.view.TRS_UEDITOR")
    if not content_div:
        content_div = soup.select_one("div.zwnr")
    if not content_div:
        content_div = soup.select_one("div.news-cont")
    if not content_div:
        content_div = soup.find("div", class_=lambda x: x and "TRS_Editor" in (x or ""))
    
    if not content_div:
        return None
    
    for tag in content_div.find_all(["script", "style", "iframe"]):
        tag.decompose()
    
    # Recursive extraction
    def extract(node):
        parts = []
        for child in node.children:
            t = getattr(child, "name", None)
            if t == "p":
                txt = child.get_text(separator="", strip=True)
                if txt:
                    parts.append(txt)
            elif t == "table":
                parts.append(str(child))
            elif t == "div":
                sub = extract(child)
                if sub:
                    parts.append(sub)
            elif t is None and isinstance(child, str):
                txt = child.strip()
                if txt and len(txt) > 3:
                    parts.append(txt)
        return "\n\n".join(parts) if parts else ""
    
    content = extract(content_div)
    if not content or len(content) < 20:
        content = content_div.get_text(separator="", strip=True)
    if not content or len(content) < 20:
        return None
    
    title_tag = soup.find("title")
    full_title = title_tag.get_text(strip=True) if title_tag else ""
    full_title = re.sub(r"^\s*滕州市政府信息公开\s*[-_―]+\s*", "", full_title).strip()  # strip prefix
    full_title = re.sub(r"\s*[-_―]+\s*滕州市政府信息公开.*", "", full_title).strip()     # strip suffix
    if not full_title or len(full_title) < 5:
        full_title = item["title"]
    
    return {
        "title": full_title,
        "content": content,
        "date": item["date"],
        "url": item["url"],
        "site_name": SITE_NAME,
    }

def save_to_db(records):
    if not records:
        return 0
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for rec in records:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, content, page_url, publish_date, site_name) "
                "VALUES (?, ?, ?, ?, ?)",
                (rec["title"], rec["content"], rec["url"], rec["date"], rec["site_name"])
            )
            if c.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  DB error: {e}")
    conn.commit()
    conn.commit()
    conn.close()
    return inserted

def main():
    print(f"=== {SITE_NAME} ===")
    
    items = crawl_list()
    if not items:
        print("  No items found")
        return
    
    print(f"  Found {len(items)} items")
    
    batch = []
    for item in items:
        detail = fetch_detail(item)
        if detail:
            batch.append(detail)
        time.sleep(0.3)
    
    inserted = save_to_db(batch)
    print(f"\n{'='*40}")
    print(f"Inserted: {inserted}/{len(batch)} new")

if __name__ == "__main__":
    main()
