#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""邹城市人民政府 — 太平镇通知公告爬虫 (col24376)"""
import requests, sqlite3, re, sys, os, argparse
from bs4 import BeautifulSoup

DB = "/root/search.db"
BASE = "http://www.zoucheng.gov.cn"
DATAPROXY = BASE + "/module/web/jpage/dataproxy.jsp?page={}&webid=103&path=/&columnid=24376&unitid=455568&permissiontype=0"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36"}
SITE = "邹城市太平镇-通知公告"
MAX_PAGES = 5

seen_urls = set()

def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] 详情请求失败: {e}")
        return None, None, None
    soup = BeautifulSoup(r.text, "html.parser")
    # 标题 - ArticleTitle meta优先, 其次main_tit
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        tt = soup.select_one("div.main_tit")
        if tt:
            title = tt.get_text(strip=True)
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    # 日期
    date = ""
    meta_date = soup.find("meta", attrs={"name": "pubdate"})
    if meta_date and meta_date.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_date["content"])
        if m:
            date = m.group(1)
    if not date:
        riqi_div = soup.select_one("div.main_riqi")
        if riqi_div:
            d = riqi_div.get_text()
            m = re.search(r"(\d{4}-\d{2}-\d{2})", d)
            if m: date = m.group(1)
    # 正文
    content_div = soup.find("div", class_="neirong")
    content = ""
    if content_div:
        parts = []
        # 注意: html.parser 会把<meta>当成容器,导致children迭代失效
        # 改用find_all直接从content_div下搜索所有p和table
        for p in content_div.find_all("p"):
            if p.find_parent("table"):
                continue  # 跳过表格内的p（避免重复）
            txt = str(p).strip()
            if txt:
                parts.append(txt)
        for table in content_div.find_all("table"):
            txt = str(table).strip()
            if txt:
                parts.append(txt)
        content = "\n".join(parts).strip()
        # 安全兜底: 如果正文过短,回退到直接提取
        if len(content) < 50:
            content = ""
            for tag in content_div.find_all(["p", "table"]):
                txt = str(tag).strip()
                if txt: content += txt + "\n"
            content = content.strip()
    return title, date, content


def parse_dataproxy(xml_text):
    items = []
    records = re.findall(r'<record><!\[CDATA\[(.*?)\]\]></record>', xml_text, re.DOTALL)
    for rec in records:
        soup = BeautifulSoup(rec, "html.parser")
        a = soup.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        if not href.startswith("http"):
            href = BASE + href
        title = a.get("title", "") or a.get_text(strip=True)
        title = title.strip()
        if not title or href in seen_urls:
            continue
        seen_urls.add(href)
        sp = soup.find("span", class_="bt-right")
        date = sp.get_text(strip=True) if sp else ""
        items.append({"title": title, "url": href, "date": date})
    return items


def main():
    parser = argparse.ArgumentParser(description="邹城市太平镇-通知公告爬虫")
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="最大爬取页数")
    args = parser.parse_args()
    max_pages = args.pages

    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()

    total_new = total_dup = 0

    for pg in range(1, max_pages + 1):
        url = DATAPROXY.format(pg)
        print(f"[PAGE {pg}] {url}")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERR] 列表请求失败: {e}")
            break

        # 检查总页数
        m_totalpage = re.search(r"<totalpage>(\d+)</totalpage>", r.text)
        totalpage = int(m_totalpage.group(1)) if m_totalpage else 1
        m_totalrec = re.search(r"<totalrecord>(\d+)</totalrecord>", r.text)
        totalrec = int(m_totalrec.group(1)) if m_totalrec else 0

        items = parse_dataproxy(r.text)
        if not items:
            print("  [EMPTY] 无更多条目")
            break

        print(f"  [LIST] 本页 {len(items)} 条 (总记录 {totalrec}, 总页 {totalpage})")

        for item in items:
            title = item["title"]
            url = item["url"]
            date = item["date"]

            # 查重
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                print(f"  [DUP] {title} - {date}")
                total_dup += 1
                continue

            # 提取详情
            dt, dd, dc = extract_detail(url)
            det_title = dt or title
            det_date = dd or date
            content = dc or ""

            if len(content) < 20:
                print(f"  [WARN] 正文过短 ({len(content)}字符): {det_title}")

            # 入库
            try:
                c.execute(
                    "INSERT INTO gov_raw (title, content, publish_date, source_url, page_url, site_name, group_name, industry) VALUES (?,?,?,?,?,?,?,?)",
                    (det_title, content, det_date, BASE, url, SITE, "山东省", "政府公告"),
                )
                conn.commit()
                print(f"  [NEW] {det_title} - {det_date}")
                total_new += 1
            except sqlite3.IntegrityError as e:
                print(f"  [DUP] {det_title} - {e}")
                total_dup += 1

        if pg >= totalpage:
            print("  [END] 已到最后一页")
            break

    # FTS同步
    print(f"\n{'='*50}")
    print(f"新增: {total_new} | 重复: {total_dup}")
    if total_new > 0:
        print("同步FTS...")
        import subprocess
        rows = c.execute(
            "SELECT id, title, site_name, content FROM gov_raw WHERE id NOT IN (SELECT rowid FROM gov_search)"
        ).fetchall()
        if rows:
            sql = "INSERT INTO gov_search(rowid, title, site_name, summary) VALUES\n"
            vals = []
            for row_id, t, sn, _ in rows:
                summary = (_ or "")[:500].replace("'", "''")
                t_esc = (t or "").replace("'", "''")
                sn_esc = (sn or "").replace("'", "''")
                vals.append(f"({row_id},'{t_esc}','{sn_esc}','{summary}')")
            sql += ",\n".join(vals) + ";"
            proc = subprocess.run(
                ["sqlite3", "-cmd", ".timeout 60000", DB],
                input=sql,
                capture_output=True,
                text=True,
                timeout=30,
            )
            if proc.returncode == 0:
                print(f"FTS同步完成: {len(rows)}条")
            else:
                print(f"FTS同步失败: {proc.stderr}")
        else:
            print("FTS无需同步")
    conn.close()
    print(f"完成! 共 {total_new} 新, {total_dup} 重复")


if __name__ == "__main__":
    main()
