#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""张家港市人民政府 — 通知公告爬虫 (UCAP CMS, /zjg/tzgg/)"""
import requests, sqlite3, re, sys, os, argparse, subprocess
from bs4 import BeautifulSoup

DB = "/root/search.db"
BASE = "https://www.zjg.gov.cn"
SITE = "张家港市-通知公告"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
MAX_PAGES = 5

seen_urls = set()


def extract_detail(url):
    """提取详情页"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] 详情请求失败: {e}")
        return None, None, None
    soup = BeautifulSoup(r.text, "html.parser")
    # 标题
    title = ""
    meta_t = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_t and meta_t.get("content"):
        title = meta_t["content"].strip()
    if not title:
        h1 = soup.find("h1", class_="article-title")
        if h1:
            title = h1.get_text(strip=True)
    # 日期
    date = ""
    meta_pub = soup.find("meta", attrs={"name": "PubDate"})
    if meta_pub and meta_pub.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_pub["content"])
        if m:
            date = m.group(1)
    # 正文
    content = ""
    zoom = soup.find("div", id="zoomcon")
    if zoom:
        content = str(zoom).strip()
    if not content or len(content) < 50:
        ac = soup.find("div", class_="article-content")
        if ac:
            content = str(ac).strip()
    return title, date, content


def parse_list(html):
    """解析列表页"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.select("div.pageList li"):
        h4 = li.find("h4")
        if not h4:
            continue
        a = h4.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        if not href.startswith("http"):
            href = BASE + href
        title = a.get("title", "") or a.get_text(strip=True)
        title = title.strip()
        if not title or href in seen_urls:
            continue
        seen_urls.add(href)
        sp = li.find("span", class_="time")
        date = sp.get_text(strip=True) if sp else ""
        items.append({"title": title, "url": href, "date": date})
    return items


def main():
    parser = argparse.ArgumentParser(description="张家港市-通知公告爬虫")
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="最大爬取页数")
    args = parser.parse_args()

    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    total_new = total_dup = 0

    for pg in range(1, args.pages + 1):
        if pg == 1:
            url = BASE + "/zjg/tzgg/common_list.shtml"
        else:
            url = BASE + f"/zjg/tzgg/common_list_{pg}.shtml"
        print(f"[PAGE {pg}] {url}")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERR] 列表请求失败: {e}")
            break
        if "暂无数据" in r.text:
            print("  [END] 无更多数据")
            break
        items = parse_list(r.text)
        if not items:
            print("  [END] 无更多条目")
            break
        print(f"  [LIST] 本页 {len(items)} 条")

        for item in items:
            title = item["title"]
            url = item["url"]
            date = item["date"]
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                print(f"  [DUP] {title[:40]} - {date}")
                total_dup += 1
                continue
            dt, dd, dc = extract_detail(url)
            det_title = dt or title
            det_date = dd or date
            content = dc or ""
            if len(content) < 20:
                print(f"  [WARN] 正文过短 ({len(content)}字符)")
            try:
                c.execute(
                    "INSERT INTO gov_raw (title, content, publish_date, source_url, page_url, site_name, group_name, industry) VALUES (?,?,?,?,?,?,?,?)",
                    (det_title, content, det_date, BASE, url, SITE, "江苏", "政府公告"),
                )
                conn.commit()
                print(f"  [NEW] {det_title[:40]} - {det_date}")
                total_new += 1
            except sqlite3.IntegrityError as e:
                print(f"  [DUP] {det_title[:40]} - {e}")
                total_dup += 1

    # FTS
    print(f"\n{'='*50}")
    print(f"新增: {total_new} | 重复: {total_dup}")
    if total_new > 0:
        print("同步FTS...")
        rows = c.execute(
            "SELECT id, title, site_name, content FROM gov_raw WHERE id NOT IN (SELECT rowid FROM gov_search)"
        ).fetchall()
        if rows:
            sql = "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES\n"
            vals = []
            for row_id, t, sn, _ in rows:
                summary = (_ or "")[:500].replace("'", "''")
                t_esc = (t or "").replace("'", "''")
                sn_esc = (sn or "").replace("'", "''")
                vals.append(f"({row_id},'{t_esc}','{sn_esc}','{summary}')")
            sql += ",\n".join(vals) + ";"
            proc = subprocess.run(
                ["sqlite3", "-cmd", ".timeout 60000", DB], input=sql, capture_output=True, text=True, timeout=30
            )
            if proc.returncode == 0:
                print(f"FTS同步完成: {len(rows)}条")
            else:
                print(f"FTS同步失败: {proc.stderr}")
        else:
            print("FTS无需同步")
    conn.close()
    print(f"完成! 共 {total_new} 新, {total_dup} 重复")


if __name__ == "__main__":
    main()
