#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""渝水区人民政府 — 政务公告爬虫 (UCAP CMS)"""
import requests, sqlite3, re, sys, os, argparse, subprocess
from bs4 import BeautifulSoup

DB = "/root/search.db"
BASE = "http://www.yushui.gov.cn"
LIST_URL = BASE + "/yushui/zwgg/list.shtml"
SITE = "渝水区-政务公告"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
PER_PAGE = 25
MAX_PAGES = 5

seen_urls = set()

def extract_detail(url):
    """提取详情页的标题、日期、内容"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] 详情请求失败: {e}")
        return None, None, None
    soup = BeautifulSoup(r.text, "html.parser")
    # 标题 - UCAPTITLE优先
    title = ""
    ucap_title = soup.find("ucaptitle")
    if ucap_title:
        title = ucap_title.get_text(strip=True)
    if not title:
        h1 = soup.find("h1", class_="article-title")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        t_tag = soup.find("title")
        if t_tag:
            t = t_tag.get_text(strip=True)
            t = re.sub(r"\s*[|]\s*渝水区人民政府\s*$", "", t)
            if t:
                title = t
    # 日期
    date = ""
    d_span = soup.find("span", class_="date")
    if d_span:
        date = d_span.get_text(strip=True)
    # 正文 - UCAPCONTENT内的HTML
    content = ""
    ucap = soup.find("ucapcontent")
    if ucap:
        content = str(ucap).strip()
        # 去掉<UCAPCONTENT>标签本身
        content = re.sub(r'^<UCAPCONTENT[^>]*>', '', content)
        content = re.sub(r'</UCAPCONTENT>$', '', content)
        content = content.strip()
    if len(content) < 50:
        # 兜底: 用article-content
        ac = soup.find("div", class_="article-content")
        if ac:
            content = str(ac).strip()
    return title, date, content


def parse_list(html, page_num):
    """解析列表页"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.select("div.page_list ul li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        if not href.startswith("http"):
            href = BASE + href
        title = a.get("title", "") or a.get_text(strip=True)
        title = title.strip().replace("\u00a0", " ").strip()
        if not title or href in seen_urls:
            continue
        seen_urls.add(href)
        sp = li.find("span", class_="time")
        date = sp.get_text(strip=True) if sp else ""
        items.append({"title": title, "url": href, "date": date})
    return items


def main():
    parser = argparse.ArgumentParser(description="渝水区-政务公告爬虫")
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="最大爬取页数")
    parser.add_argument("--start", type=int, default=1, help="起始页码")
    args = parser.parse_args()
    max_pages = args.pages

    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()

    total_new = total_dup = 0

    for pg in range(args.start, args.start + max_pages):
        if pg == 1:
            url = LIST_URL
        else:
            url = BASE + f"/yushui/zwgg/list_{pg}.shtml"
        print(f"[PAGE {pg}] {url}")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERR] 列表请求失败: {e}")
            break

        # 检查是否有数据（空页面 = 到底了）
        if "暂无数据" in r.text or "没有数据" in r.text:
            print("  [END] 无更多数据")
            break

        items = parse_list(r.text, pg)
        if not items:
            # 也可能是最后一页没有li了
            print("  [END] 无更多条目")
            break

        print(f"  [LIST] 本页 {len(items)} 条")

        for item in items:
            title = item["title"]
            url = item["url"]
            date = item["date"]

            # 查重
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                print(f"  [DUP] {title[:40]} - {date}")
                total_dup += 1
                continue

            # 提取详情
            dt, dd, dc = extract_detail(url)
            det_title = dt or title
            det_date = dd or date
            content = dc or ""

            if len(content) < 20:
                print(f"  [WARN] 正文过短 ({len(content)}字符): {det_title[:30]}")

            # 入库
            try:
                c.execute(
                    "INSERT INTO gov_raw (title, content, publish_date, source_url, page_url, site_name, group_name, industry) VALUES (?,?,?,?,?,?,?,?)",
                    (det_title, content, det_date, BASE, url, SITE, "江西", "政府公告"),
                )
                conn.commit()
                print(f"  [NEW] {det_title[:40]} - {det_date}")
                total_new += 1
            except sqlite3.IntegrityError as e:
                print(f"  [DUP] {det_title[:40]} - {e}")
                total_dup += 1

    # FTS同步
    print(f"\n{'='*50}")
    print(f"新增: {total_new} | 重复: {total_dup}")
    if total_new > 0:
        print("同步FTS...")
        rows = c.execute(
            "SELECT id, title, site_name, content FROM gov_raw WHERE id NOT IN (SELECT rowid FROM gov_search)"
        ).fetchall()
        if rows:
            sql = "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES\n"
            vals = []
            for row_id, t, sn, _ in rows:
                summary = (_ or "")[:500].replace("'", "''")
                t_esc = (t or "").replace("'", "''")
                sn_esc = (sn or "").replace("'", "''")
                vals.append(f"({row_id},'{t_esc}','{sn_esc}','{summary}')")
            sql += ",\n".join(vals) + ";"
            proc = subprocess.run(
                ["sqlite3", "-cmd", ".timeout 60000", DB],
                input=sql,
                capture_output=True,
                text=True,
                timeout=30,
            )
            if proc.returncode == 0:
                print(f"FTS同步完成: {len(rows)}条")
            else:
                print(f"FTS同步失败: {proc.stderr}")
        else:
            print("FTS无需同步")
    conn.close()
    print(f"完成! 共 {total_new} 新, {total_dup} 重复")


if __name__ == "__main__":
    main()
