#!/usr/bin/env python3
"""
方城县民政局 - 公示公告爬虫
URL: https://www.fangcheng.gov.cn/fcxmzj/gsgg/
CMS: 南阳市政府网站群 (t.nanyang.gov.cn)
列表: 静态HTML分页 index.html → index_1.html (25条/页, 共45条2页)
详情: /YYYY/MM-DD/NUMBER.html, 正文在 <div class="content" id="content">
日跑: --max-pages 1 (增量取最新一页)
"""

import sys, os, re, time, sqlite3, argparse, json
from datetime import datetime
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

import warnings
warnings.filterwarnings("ignore")

BASE_URL = "https://www.fangcheng.gov.cn"
SITE_NAME = "方城县民政局"
CATEGORY = "公示公告"
GROUP = "河南省"
DB_PATH = os.environ.get("DB_PATH", "/mnt/data/search.db")
PAGE_SIZE = 25

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def get_session():
    s = requests.Session()
    s.headers.update(HEADERS)
    # 先访问首页获取 cookie
    s.get(f"{BASE_URL}/fcxmzj/gsgg/", timeout=30)
    return s


def fetch_list(session, page):
    """获取列表页"""
    if page == 1:
        url = f"{BASE_URL}/fcxmzj/gsgg/index.html"
    else:
        url = f"{BASE_URL}/fcxmzj/gsgg/index_{page-1}.html"

    items = []
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text

        soup = BeautifulSoup(html, "html.parser")
        for li in soup.select("div.zfxxgk_zdgkc ul li"):
            a_tag = li.select_one("a[href]")
            if not a_tag:
                continue
            href = a_tag.get("href", "")
            title = (a_tag.get("title") or a_tag.get_text(strip=True) or "").strip()
            if not title or not href or len(title) < 5:
                continue
            if href.startswith("/"):
                href = f"{BASE_URL}{href}"
            elif not href.startswith("http"):
                href = f"{BASE_URL}/{href.lstrip('/')}"

            # 日期: <b>YYYY-MM-DD</b>
            publish_date = ""
            b_tag = li.select_one("b")
            if b_tag:
                txt = b_tag.get_text(strip=True)
                m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
                if m:
                    publish_date = m.group(1)

            items.append({
                "url": href,
                "title": title,
                "date": publish_date,
            })
        return items
    except Exception as e:
        print(f"  [错误] 列表页 {page} 请求失败: {e}", flush=True)
        return []


def fetch_detail(session, detail_url):
    """获取详情页标题、日期和正文"""
    try:
        resp = session.get(detail_url, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text

        # 标题: <div class="title"> 在 detailBox 内
        title = ""
        m = re.search(r'<div class="title">([^<]+)', html)
        if m:
            title = m.group(1).strip()

        # 日期: 时间：YYYY-MM-DD
        publish_date = ""
        m = re.search(r'时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
        if m:
            publish_date = m.group(1).strip()

        # 正文: <div class="content" id="content">
        body = ""
        idx = html.find('class="content" id="content"')
        if idx >= 0:
            start = html.find('>', idx) + 1
            depth = 1
            i = start
            while i < len(html) and depth > 0:
                if html[i:i+4] == '<div':
                    depth += 1
                    i += 4
                elif html[i:i+6] == '</div>':
                    depth -= 1
                    i += 6
                else:
                    i += 1
            raw = html[start:i-6] if depth == 0 else html[start:start+5000]
            if raw.strip():
                body = clean_content_html(raw.strip())

        if not body or len(body) < 20:
            body = ""

        return title, publish_date, body

    except Exception as e:
        print(f"  [错误] 详情页 {detail_url}: {e}", flush=True)
        return "", "", ""


def clean_content_html(html_content):
    """清洗正文 HTML"""
    if not html_content:
        return ""

    soup = BeautifulSoup(html_content, "html.parser")

    # 处理图片路径 (OSS)
    for img in soup.find_all("img"):
        src = img.get("src", "")
        if src and not src.startswith("http"):
            img["src"] = urljoin(BASE_URL, src)

    # 处理附件链接
    for a in soup.find_all("a"):
        href = a.get("href", "")
        if href and not href.startswith("http") and not href.startswith("#") and not href.startswith("javascript"):
            a["href"] = urljoin(BASE_URL, href)

    result = str(soup)
    # 清理多余样式标签
    result = re.sub(r'\sstyle="[^"]*"', '', result)
    result = re.sub(r'<span[^>]*>|</span>', '', result)
    result = re.sub(r'<strong[^>]*>|</strong>', '', result)
    result = re.sub(r'<b[^>]*>|</b>', '', result)
    result = re.sub(r'<font[^>]*>|</font>', '', result)
    result = re.sub(r'\n{3,}', '\n\n', result)
    return result.strip()


def crawl(max_pages=None):
    s = get_session()

    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    existing = set()
    cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    for row in cur:
        existing.add(row[0])

    print(f"已有 {len(existing)} 条 {SITE_NAME} 记录", flush=True)

    # 共 2 页 (45 条)
    total_pages = 2
    pages_to_fetch = min(max_pages, total_pages) if max_pages else 2
    print(f"共约 {total_pages} 页, 爬取前 {pages_to_fetch} 页", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, pages_to_fetch + 1):
        print(f"[第 {page}/{pages_to_fetch} 页] ", end="", flush=True)
        items = fetch_list(s, page)
        if not items:
            print("无数据", flush=True)
            continue

        page_new = 0
        for item in items:
            full_url = item["url"]
            if full_url in existing:
                total_skip += 1
                continue

            title = item["title"]
            publish_date = item["date"]

            if not title:
                total_skip += 1
                continue

            # 获取详情
            det_title, det_date, body = fetch_detail(s, full_url)
            final_title = det_title if det_title else title
            if det_date and not publish_date:
                publish_date = det_date

            # 摘要
            summary = re.sub(r'<[^>]+>', ' ', body) if body else final_title
            summary = re.sub(r'\s+', ' ', summary).strip()[:300]

            has_table = 1 if body and '<table' in body else 0

            try:
                cur = db.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date,
                     summary, content, status, category, group_name, has_table)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                    SITE_NAME, full_url, full_url, final_title[:500], publish_date[:10],
                    summary[:500], body if body else '',
                    'active', CATEGORY, GROUP, has_table
                ))
                if cur.rowcount > 0:
                    row = db.execute("SELECT id FROM gov_raw WHERE page_url=?", (full_url,)).fetchone()
                    if row:
                        rid = row[0]
                        try:
                            # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                            #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                            db.commit()
                            db.execute("INSERT INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                                       (rid, final_title[:500], SITE_NAME, summary[:200]))
                        except sqlite3.IntegrityError:
                            pass
                    existing.add(full_url)
                    db.commit()
                    total_new += 1
                    page_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"x", end="", flush=True)
                total_error += 1

        print(f"+{page_new} 条 (累计 {total_new})", flush=True)
        if page < pages_to_fetch:
            time.sleep(0.5)

    db.close()
    print(f"\n完成! 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error}", flush=True)
    return total_new


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None)
    args = parser.parse_args()
    crawl(max_pages=args.max_pages)
