#!/usr/bin/env python3
"""
延津县 - 公告公示
https://www.yanjin.gov.cn/news/31.html

CMS: 自定义CMS（延津县）
WAF: Apache 403 IP阻断 - 需自定义User-Agent绕过
列表: <div class="news-list mt-20"> -> <div class="news-item">
分页: /news/31.html?page=N (共57页, 每页15条)
详情: /newsinfo/{id}.html
详情标题: <title>标签
详情日期: 页面文本中YYYY-MM-DD
正文容器: <div class="news-content">
"""

import sys
import os
import re
import subprocess
import json
from bs4 import BeautifulSoup

# ---------- 数据库配置 ----------
DB_PATH = "/mnt/data/search.db" if os.path.exists("/mnt/data/search.db") else os.path.expanduser("~/data/search.db")

# ---------- 站点配置 ----------
BASE_URL = "https://www.yanjin.gov.cn"
LIST_PATH = "/news/31.html"
DOMAIN = "www.yanjin.gov.cn"
PAGES = 5
GROUP_NAME = "河南省新乡市"
INDUSTRY = "政府公告"
SITE_NAME = "延津县"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"


def fetch_page(url):
    """获取页面，用User-Agent绕过Apache 403"""
    cmd = [
        "curl", "-sS", "--connect-timeout", "10", "-m", "30",
        "-A", UA,
        url
    ]
    try:
        result = subprocess.run(cmd, capture_output=True, text=True, timeout=35)
        if result.returncode != 0:
            print(f"  [WARN] curl返回码 {result.returncode}")
            return None
        if "403 Forbidden" in result.stdout[:500]:
            print(f"  [WARN] 403 Forbidden")
            return None
        return result.stdout
    except subprocess.TimeoutExpired:
        print(f"  [WARN] 超时: {url}")
        return None


def fetch_detail(detail_url):
    """获取详情页并提取标题、日期、正文"""
    html = fetch_page(detail_url)
    if not html:
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # 标题: 从<title>提取
    title = ""
    t = soup.find("title")
    if t:
        title = t.get_text(strip=True)
        # 去掉站点后缀
        title = re.sub(r"\s*-\s*公告公示\s*-\s*延津县人民政府\s*$", "", title).strip()

    # strip &middot; &nbsp; 实体
    title = title.replace("&middot;", "").replace("&nbsp;", "").replace("\xa0", "").replace("\u3000", "").strip()

    # 日期: 页面文本中YYYY-MM-DD
    date_str = ""
    all_text = soup.get_text()
    m = re.search(r"(\d{4}-\d{2}-\d{2})", all_text)
    if m:
        date_str = m.group(1)

    # 正文: div.news-content
    content_div = soup.select_one("div.news-content")

    text = ""
    if content_div:
        paragraphs = []
        for child in content_div.children:
            name = getattr(child, "name", None)
            if name == "p":
                inner_table = child.find("table")
                if inner_table:
                    for node in child.children:
                        nname = getattr(node, "name", None)
                        if nname is None:
                            t = str(node).strip()
                            if t:
                                paragraphs.append(t)
                        elif nname == "table":
                            paragraphs.append(str(node))
                        elif nname == "br":
                            pass
                        else:
                            t2 = node.get_text(strip=True)
                            if t2:
                                paragraphs.append(t2)
                else:
                    p_text = child.get_text(strip=True)
                    if p_text:
                        paragraphs.append(p_text)
            elif name == "table":
                paragraphs.append(str(child))
            elif name is None:
                t = str(child).strip()
                if t:
                    paragraphs.append(t)
        if paragraphs:
            text = "\n\n".join(paragraphs)

    # 附件/图片嵌入
    if len(text) < 500 and content_div:
        links = []
        for img in content_div.select("img[src]"):
            src = img.get("src", "")
            if src and not src.startswith("http"):
                src = BASE_URL + src if src.startswith("/") else src
            links.append(f"[图片] {src}")
        for iframe in content_div.select("iframe[src]"):
            src = iframe.get("src", "")
            if src and not src.startswith("http"):
                src = BASE_URL + src if src.startswith("/") else src
            links.append(f"[附件PDF] {src}")
        for obj in content_div.select("object[data]"):
            data = obj.get("data", "")
            if data and not data.startswith("http"):
                data = BASE_URL + data if data.startswith("/") else data
            links.append(f"[附件PDF] {data}")
        for a in content_div.select("a[href]"):
            href = a.get("href", "")
            ext = href.lower().split("?")[0].rsplit(".", 1)[-1] if "." in href else ""
            if ext in ("pdf", "doc", "docx", "xls", "xlsx", "jpg", "png", "gif", "zip", "rar"):
                if not href.startswith("http"):
                    href = BASE_URL + href if href.startswith("/") else href
                fname = a.get_text(strip=True) or href.split("/")[-1]
                links.append(f"[附件: {fname}] {href}")
        if links:
            text = "\n".join(links) if not text else text + "\n" + "\n".join(links)

    return title, date_str, text


def parse_list_page(html, page_num):
    """解析列表页，返回 (url, title, date) 元组列表"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for item in soup.select("div.news-item"):
        a_tag = item.find("a")
        if not a_tag:
            continue
        href = a_tag.get("href", "").strip()
        # title在 div.news-title.text-truncate
        title_div = item.find("div", class_="news-title")
        title = title_div.get_text(strip=True) if title_div else a_tag.get("title", "")
        # date在 div.news-time
        time_div = item.find("div", class_="news-time")
        date_str = time_div.get_text(strip=True) if time_div else ""
        if href and title:
            full_href = href if href.startswith("http") else BASE_URL + href
            items.append((full_href, title, date_str))

    print(f"  第{page_num}页: {len(items)}条")
    return items


def main():
    import argparse
    parser = argparse.ArgumentParser(description="延津县-公告公示爬虫")
    parser.add_argument("--pages", type=int, default=PAGES, help=f"爬取页数(默认{PAGES})")
    args = parser.parse_args()
    max_pages = args.pages

    # ---------- 1. 爬取列表页 ----------
    print(f"🌐 爬取列表页 1-{max_pages}...")
    all_items = []
    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = BASE_URL + LIST_PATH
        else:
            list_url = f"{BASE_URL}{LIST_PATH}?page={page}"
        html = fetch_page(list_url)
        if not html:
            print(f"  [ERROR] 无法获取第{page}页")
            continue
        items = parse_list_page(html, page)
        all_items.extend(items)

    print(f"\n📋 共获取 {len(all_items)} 条列表数据")

    # ---------- 2. 去重 ----------
    seen_urls = set()
    unique_items = []
    for href, list_title, date in all_items:
        if href not in seen_urls:
            seen_urls.add(href)
            unique_items.append((href, list_title, date))
    print(f"🔍 去重后 {len(unique_items)} 条")

    # ---------- 3. 爬取详情 ----------
    print("\n📄 爬取详情页...")
    results = []
    for idx, (href, list_title, list_date) in enumerate(unique_items, 1):
        print(f"  [{idx}/{len(unique_items)}] {list_title[:30]}...", end=" ")
        sys.stdout.flush()

        title, date_str, text = fetch_detail(href)
        if not title:
            title = list_title
            title = title.replace("&middot;", "").replace("&nbsp;", "").replace("\xa0", "").replace("\u3000", "").strip()
        if not date_str:
            date_str = list_date

        if text:
            print(f"✅ {len(text)}字")
        else:
            print(f"⚠️ 正文为空")

        results.append({
            "title": title,
            "date": date_str,
            "content": text,
            "url": href,
            "source_url": href,
        })

    # ---------- 4. 入库 ----------
    print(f"\n💾 入库 {len(results)} 条到 {DB_PATH}...")
    if not os.path.exists(DB_PATH):
        print(f"  [ERROR] 数据库不存在: {DB_PATH}")
        print("\n📊 结果预览:")
        for r in results[:5]:
            print(f"  {r['date']} | {r['title'][:40]} | {len(r['content'])}字")
        return

    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    added = 0
    for r in results:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, content, publish_date, page_url, source_url, site_name, group_name, industry) VALUES (?,?,?,?,?,?,?,?)",
                (r["title"], r["content"], r["date"], r["url"], r["source_url"], SITE_NAME, GROUP_NAME, INDUSTRY)
            )
            if c.rowcount > 0:
                added += 1
        except Exception as e:
            print(f"  [ERROR] 入库失败: {e}")

    conn.commit()

    # ---------- 5. FTS同步 ----------
    if added > 0:
        print(f"\n🔍 FTS同步 {added} 条...")
        script = """
import sqlite3, sys
db = sys.argv[1]
conn = sqlite3.connect(db)
c = conn.cursor()
added = int(sys.argv[2])
c.execute("SELECT id, title, content FROM gov_raw ORDER BY id DESC LIMIT ?", (added,))
rows = c.fetchall()
site = '""" + SITE_NAME + """'
for rowid, title, content in rows:
    c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
              (rowid, title, site, (content or '')[:500]))
conn.commit()
c.execute("SELECT COUNT(*) FROM gov_search")
print(f"FTS条数: {c.fetchone()[0]}")
conn.close()
"""
        result = subprocess.run(["python3", "-c", script, DB_PATH, str(added)],
                                capture_output=True, text=True, timeout=30)
        print(f"  FTS: {result.stdout.strip()}")
        if result.stderr:
            print(f"  FTS stderr: {result.stderr[:100]}")
    else:
        print("\n  ⚠️ 无新增记录，跳过FTS同步")

    conn.close()

    # ---------- 6. 输出结果 ----------
    print(f"\n📊 结果总览:")
    empty_count = sum(1 for r in results if not r["content"])
    print(f"  总条数: {len(results)}")
    print(f"  新增入库: {added}")
    print(f"  空正文: {empty_count}")
    print(f"  站点: {SITE_NAME}")
    print(f"  栏目: 公告公示")

    print(f"\n{'─' * 60}")
    print(f"{'日期':<14} {'标题':<40} {'字数':>6}")
    print(f"{'─' * 60}")
    for r in results[:5]:
        title_short = r['title'][:38] if len(r['title']) > 38 else r['title']
        print(f"{r['date']:<14} {title_short:<40} {len(r['content']):>6}")
    if len(results) > 5:
        print(f"  ... 共 {len(results)} 条")
    print(f"{'─' * 60}")


if __name__ == "__main__":
    main()
