#!/usr/bin/env python3
"""
新余市-政务公告爬虫
http://www.xinyu.gov.cn/xinyu/zwgg/list.shtml
共9页，每页约20条，总计177条
分页: list.shtml(第1页), list_2.shtml ... list_9.shtml
"""

import requests
import sqlite3
import json
import os
import argparse
from bs4 import BeautifulSoup

BASE_URL = "http://www.xinyu.gov.cn"
LIST_URL = BASE_URL + "/xinyu/zwgg/list.shtml"
TOTAL_PAGES = 9

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "xinyu_zwgg"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def get_soup(url, timeout=30):
    resp = requests.get(url, headers=HEADERS, timeout=timeout)
    resp.encoding = "utf-8"
    return BeautifulSoup(resp.text, "html.parser")


def parse_list_page(soup):
    """从列表页提取 (url, title, pub_date)"""
    items = []
    for a in soup.select("a[href*='/xinyu/zwgg/']"):
        href = a.get("href", "")
        if "content_" not in href:
            continue
        full_url = BASE_URL + href if href.startswith("/") else href
        title = a.get_text(strip=True)
        parent_li = a.find_parent("li")
        pub_date = ""
        if parent_li:
            ts = parent_li.find("span", class_="time")
            if ts:
                pub_date = ts.get_text(strip=True)
        items.append((full_url, title, pub_date))
    return items


def extract_detail(soup, url):
    """提取详情页内容"""
    title = ""
    pub_date = ""
    source = ""
    content = ""
    attachments = []

    # title - from UCAPTITLE or h1.article-title
    ucap = soup.find("ucaptitle")
    if ucap:
        title = ucap.get_text(strip=True)
    if not title:
        h1 = soup.select_one("h1.article-title")
        if h1:
            title = h1.get_text(strip=True)

    # pub_date - from PUBLISHTIME or meta
    pt = soup.find("publishtime")
    if pt:
        pub_date = pt.get_text(strip=True)
    if not pub_date:
        meta = soup.find("meta", attrs={"name": "PubDate"})
        if meta and meta.get("content"):
            pub_date = meta["content"].strip()

    # source - from meta or span.ly
    meta_src = soup.find("meta", attrs={"name": "ContentSource"})
    if meta_src and meta_src.get("content"):
        source = meta_src["content"].strip()
    if not source:
        ly_spans = soup.select("span.ly")
        for sp in ly_spans:
            b = sp.find("b")
            if b:
                txt = b.get_text(strip=True)
                if txt and txt not in ("", "来源"):
                    source = txt
                    break

    # main content
    content_div = soup.select_one("div.article-content#zoomcon")
    if content_div:
        ucap_content = content_div.find("ucapcontent")
        if ucap_content:
            content_div = ucap_content

        # 提取附件
        for a_tag in content_div.find_all("a"):
            a_href = a_tag.get("href", "")
            if a_href.endswith((".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar")):
                if not a_href.startswith("http"):
                    a_href = BASE_URL + a_href
                attachments.append({
                    "name": a_tag.get_text(strip=True) or os.path.basename(a_href),
                    "url": a_href,
                })

        # 提取图片
        for img in content_div.find_all("img"):
            src = img.get("src", "")
            if src and not src.startswith("http"):
                src = BASE_URL + src

        # 提取纯文本：直系子元素，并去重
        target = content_div.find("ucapcontent") or content_div
        paragraphs = []
        seen_texts = set()
        for child in target.children:
            if not hasattr(child, "name") or not child.name:
                continue
            if child.name == "table":
                paragraphs.append(str(child))
                paragraphs.append("")
            elif child.name in ("p", "div"):
                # 如果内嵌 <table>，取原始 table HTML 而非纯文本
                inner_table = child.find("table")
                if inner_table:
                    paragraphs.append(str(inner_table))
                    paragraphs.append("")
                    continue
                # 跳过分页符 span
                page_break = child.find("span", attrs={"mso-break-type": "section-break"})
                if page_break:
                    continue
                txt = child.get_text(strip=True)
                if txt and len(txt) > 2:
                    if paragraphs and txt in paragraphs[-1]:
                        continue
                    if txt not in seen_texts:
                        seen_texts.add(txt)
                        paragraphs.append(txt)
            elif child.name == "span":
                # 跳过分页符、空 span
                continue

        if paragraphs:
            content = "\n".join(paragraphs)

        # 如果提取的内容太杂乱，直接用get_text
        if len(content.strip()) < 50:
            content = content_div.get_text(strip=True, separator="\n\n")

    return title, pub_date, source, content, attachments


def crawl(max_pages=None):
    """主爬取逻辑，返回 (new_count, total_count, errors)"""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()

    all_items = []
    pages_to_crawl = TOTAL_PAGES
    if max_pages and max_pages < pages_to_crawl:
        pages_to_crawl = max_pages

    for page in range(1, pages_to_crawl + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = f"{BASE_URL}/xinyu/zwgg/list_{page}.shtml"

        print(f"[列表] 第{page}/{pages_to_crawl}页: {url}")
        try:
            soup = get_soup(url)
            items = parse_list_page(soup)
            print(f"  -> 提取 {len(items)} 条")
            all_items.extend(items)
        except Exception as e:
            print(f"  -> 失败: {e}")

    print(f"\n列表共 {len(all_items)} 条，开始抓详情...")

    new_count = 0
    error_count = 0

    for idx, (page_url, list_title, list_date) in enumerate(all_items, 1):
        # 查重
        cur.execute(
            "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
            (page_url, SITE_NAME),
        )
        existing = cur.fetchone()
        if existing:
            print(f"  [{idx}/{len(all_items)}] 跳过(已有): {list_title[:30]}")
            continue

        print(f"  [{idx}/{len(all_items)}] 抓取: {list_title[:40]}...")
        try:
            soup = get_soup(page_url)
            title, pub_date, source, content, attachments = extract_detail(soup, page_url)

            if not title:
                title = list_title
            if not pub_date:
                pub_date = list_date

            # attachments to JSON
            attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""

            cur.execute(
                """INSERT OR REPLACE INTO gov_raw
                (page_url, site_name, title, publish_date, source_url, content, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?)""",
                (
                    page_url,
                    SITE_NAME,
                    title,
                    pub_date,
                    page_url,
                    content,
                    attachments_json,
                ),
            )
            conn.commit()
            new_count += 1
            print(f"    -> 新增: {title[:30]}")
        except Exception as e:
            error_count += 1
            print(f"    -> 异常: {e}")

    conn.close()
    return new_count, len(all_items), error_count


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None, help="最大抓取页数，日跑用1")
    args = parser.parse_args()

    total_list = 0
    for p in range(1, min(args.max_pages or TOTAL_PAGES, TOTAL_PAGES) + 1):
        u = LIST_URL if p == 1 else f"{BASE_URL}/xinyu/zwgg/list_{p}.shtml"
        try:
            s = get_soup(u)
            total_list += len(parse_list_page(s))
        except:
            pass
    print(f"列表共约 {total_list} 条")

    new, total, errors = crawl(args.max_pages)
    print(f"\n完成: 新增 {new}, 列表 {total}, 异常 {errors}")
