#!/usr/bin/env python3
"""
crawl_sixian.py - 泗县人民政府 通知公告 爬虫
CMS: 龙讯科技(Lonsun), API分页 content/column/{id}?pageIndex={n}
"""
import requests
import re
import sqlite3
import time
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
SITE_NAME = "泗县人民政府-通知公告"
BASE_URL = "https://www.sixian.gov.cn"
LIST_URL = BASE_URL + "/xwzx/tzgg/index.html"
API_URL = BASE_URL + "/content/column/11465623?pageIndex={}"
TOTAL_PAGES = 139
PAGE_SIZE = 20
DELAY = 0.5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

def get_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  [ERROR] {e}")
        return ""

def parse_list(html):
    """从列表页提取 (url, title, date)"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    
    # Find the doc_list ul
    ul = soup.find("ul", class_=re.compile(r"doc_list"))
    if not ul:
        # Try the whole page for links
        for li in soup.find_all("li"):
            a = li.find("a", href=re.compile(r"/xwzx/tzgg/\d+\.html"))
            if a:
                href = a.get("href", "")
                if href and not href.startswith("http"):
                    href = BASE_URL + href
                title = a.get("title", "").strip() or a.get_text(strip=True)
                date_span = li.find("span", class_="date")
                date = date_span.get_text(strip=True) if date_span else ""
                items.append((href, title, date))
        return items
    
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href or "tzgg" not in href or not href.endswith(".html"):
            continue
        if not href.startswith("http"):
            href = BASE_URL + href
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date_span = li.find("span", class_="date")
        date = date_span.get_text(strip=True) if date_span else ""
        items.append((href, title, date))
    
    return items

def extract_detail(html, url, list_date):
    """从详情页提取完整信息"""
    soup = BeautifulSoup(html, "html.parser")
    
    # 标题
    title = ""
    h1 = soup.find("h1", class_=re.compile(r"newstitl"))
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        meta = soup.find("meta", {"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            t = title_tag.get_text(strip=True)
            # Remove trailing site name
            if "_" in t:
                title = t.split("_")[0].strip()
            elif "-" in t:
                title = t.split("-")[0].strip()
            else:
                title = t
    
    # 发布日期
    publish_date = list_date
    # From span.fbsj
    fbsj = soup.find("span", class_="fbsj")
    if fbsj:
        txt = fbsj.get_text(strip=True)
        dm = re.search(r"(\d{4}-\d{2}-\d{2})", txt)
        if dm:
            publish_date = dm.group(1)
    # From meta PubDate
    if not publish_date:
        meta = soup.find("meta", {"name": "PubDate"})
        if meta and meta.get("content"):
            dm = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
            if dm:
                publish_date = dm.group(1)
    
    # 正文 contentbox
    content_html = ""
    content_box = soup.find("div", class_="contentbox")
    if content_box:
        # Find the article div inside
        article_div = content_box.find("div", role="article", attrs={"aria-label": "内容"})
        if not article_div:
            article_div = content_box
        
        # 找正文容器 newscontnet
        content_area = article_div.find("div", class_="newscontnet")
        if not content_area:
            content_area = article_div
        
        # 递归提取所有 p/table/img 标签，按文档顺序保留
        parts = []
        seen_positions = set()
        
        def extract_ordered(tag):
            for child in tag.children:
                if not hasattr(child, "name") or child.name is None:
                    continue
                cname = child.name.lower()
                # 跳过标题/信息/分享栏
                cls = " ".join(child.get("class", [])) if child.get("class") else ""
                if cname == "h1" and "newstitle" in cls:
                    continue
                if cname == "div":
                    if "newsinfo" in cls or "sharebox" in cls or "wzewmbox" in cls:
                        continue
                    # 递归进div
                    extract_ordered(child)
                elif cname in ("p", "table", "img"):
                    # 跳过已被父级包含的项（通过id或position去重）
                    child_id = id(child)
                    if child_id in seen_positions:
                        continue
                    seen_positions.add(child_id)
                    parts.append(str(child))
        
        extract_ordered(content_area)
        content_html = "\n".join(parts)
    
    # 附件
    attachments = []
    for link in soup.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|rar|zip)$", re.I)):
        att_url = link.get("href", "")
        if att_url and not att_url.startswith("http"):
            att_url = BASE_URL + att_url
        att_title = link.get_text(strip=True)
        attachments.append(f"[{att_title}]({att_url})")
    
    if not content_html.strip():
        content_html = str(article_div) if 'article_div' in dir() else ""
    
    return title, publish_date, content_html, attachments

def render_markdown(content_html, url):
    """渲染正文为Markdown"""
    soup = BeautifulSoup(content_html, "html.parser")
    parts = []
    
    for child in soup.children:
        if child.name == "p":
            text = child.get_text("", strip=True)
            imgs = child.find_all("img")
            for img in imgs:
                src = img.get("src", "")
                alt = img.get("alt", "")
                if src and not src.startswith("http"):
                    src = BASE_URL + src
                text += f"\n![{alt}]({src})"
            # Skip title/info paragraphs
            if text.strip():
                parts.append(text)
        elif child.name == "table":
            # 保留原始表格HTML，search_app渲染
            parts.append(str(child))
        elif child.name == "img":
            src = child.get("src", "")
            alt = child.get("alt", "")
            if src and not src.startswith("http"):
                src = BASE_URL + src
            parts.append(f"![{alt}]({src})")
        elif child.name is None:
            text = str(child).strip()
            if text:
                parts.append(text)
    
    return "\n\n".join(parts)

def run():
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    total_new = 0
    
    for page in range(1, TOTAL_PAGES + 1):
        url = API_URL.format(page)
        
        print(f"--- Page {page}/{TOTAL_PAGES} ---")
        html = get_page(url)
        if not html:
            continue
        
        items = parse_list(html)
        if not items:
            print(f"  [SKIP] no items")
            continue
        
        print(f"  Found {len(items)} items")
        
        for href, title, date in items:
            try:
                detail_html = get_page(href)
                if not detail_html:
                    continue
                
                full_title, pub_date, content_html, attachments = extract_detail(detail_html, href, date)
                if not full_title:
                    full_title = title
                if not pub_date:
                    pub_date = date
                
                content_md = render_markdown(content_html, href)
                plain = re.sub(r"<[^>]+>", "", content_html).strip()
                summary = plain[:200] if len(plain) > 200 else plain
                
                c.execute("""
                    INSERT OR REPLACE INTO gov_raw
                    (page_url, site_name, title, publish_date, summary, content, source_url, date_rank, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
                """, (
                    href, SITE_NAME, full_title, pub_date, summary, content_md,
                    href, int(time.time()), "\n".join(attachments)
                ))
                conn.commit()
                total_new += 1
                time.sleep(DELAY)
            except Exception as e:
                print(f"  [ERR] {href.split('/')[-1][:24]}: {e}")
                continue
        time.sleep(DELAY)
    
    conn.close()
    print(f"\nDone! Upserted: {total_new}")

if __name__ == "__main__":
    run()
