#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""湛江经济技术开发区 — 意见征集 (yjzj) 爬虫"""
import requests, sqlite3, re, sys, os, argparse, subprocess
from bs4 import BeautifulSoup
import urllib.parse

DB = "/root/search.db"
BASE = "http://www.zetdz.gov.cn"
LIST_API = BASE + "/hdjlpt/letter/cms/classify/articles"
SITE_NAME = "湛江经济技术开发区-意见征集"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "X-Requested-With": "XMLHttpRequest",
}
PER_PAGE = 20
MAX_PAGES = 5

# 共享session，保持cookie一致性
sess = requests.Session()

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_csrf():
    """获取CSRF token（复用session保持cookie）"""
    try:
        r = sess.get(
            BASE + "/hdjlpt/c_cat?name=yjzj&via=pc",
            headers=HEADERS,
            timeout=20,
        )
        m = re.search(r"_CSRF\s*=\s*'([^']+)'", r.text)
        if m:
            return m.group(1)
    except Exception as e:
        print(f"  [ERR] CSRF获取失败: {e}")
    return ""

def fetch_list(page, csrf):
    """获取列表页数据（复用session）"""
    payload = {
        "offset": page * PER_PAGE,
        "limit": PER_PAGE,
        "classify": "YJZJ",
        "_token": csrf,
    }
    try:
        r = sess.post(LIST_API, data=payload, headers=HEADERS, timeout=20)
        return r.json()
    except Exception as e:
        print(f"  [ERR] 列表请求失败: {e}")
    return None

def extract_detail(url):
    """从详情页提取标题、日期、内容"""
    try:
        r = sess.get(url, headers=HEADERS, timeout=20)

        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] 详情请求失败: {e}")
        return None, None, None
    soup = BeautifulSoup(r.text, "html.parser")
    # 标题
    title = ""
    t_tag = soup.find("title")
    if t_tag:
        t = t_tag.get_text(strip=True)
        # 去掉 " 湛江经济技术开发区门户网站" 后缀
        t = re.sub(r"\s*[—\-–]\s*湛江经济技术开发区门户网站\s*$", "", t)
        if t:
            title = t
    # 日期从URL或meta提取
    date = ""
    meta_date = soup.find("meta", attrs={"name": "pubdate"})
    if meta_date and meta_date.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_date["content"])
        if m:
            date = m.group(1)
    # 正文 - 该平台内容由JS动态加载，#article-content
    # 尝试从question-content获取文本（有部分静态内容）
    content = ""
    qc = soup.find("div", id="question-content")
    if qc:
        content = body_text(qc)
    if not content:
        ac = soup.find("div", id="article-content")
        if ac:
            content = body_text(ac)
    return title, date, content

def main():
    parser = argparse.ArgumentParser(description="湛江经开区-意见征集爬虫")
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="最大爬取页数")
    args = parser.parse_args()
    max_pages = args.pages

    print("获取CSRF token...")
    csrf = get_csrf()
    if not csrf:
        print("[ERR] 无法获取CSRF token")
        sys.exit(1)
    print(f"  CSRF: {csrf[:20]}...")

    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()

    total_new = total_dup = 0

    for pg in range(max_pages):
        print(f"\n[PAGE {pg+1}] offset={pg*PER_PAGE}")
        data = fetch_list(pg, csrf)
        if not data or data.get("errcode") != 0:
            print(f"  [ERR] API返回错误: {data}")
            break

        items = data.get("data", {}).get("list", [])
        total = data.get("data", {}).get("total", 0)
        if not items:
            print("  [EMPTY] 无更多条目")
            break

        print(f"  [LIST] 本页 {len(items)} 条 (总共 {total})")

        for item in items:
            title = item.get("title", "").strip()
            item_id = item.get("id", "")
            raw_content = item.get("content", "")
            api_url = item.get("url", "") or item.get("post_url", "")
            pub_ts = item.get("publish_time", 0)
            display_ts = item.get("display_publish_time", 0) or item.get("first_publish_time", 0)

            # URL去重的key: 用API的url字段
            page_url = api_url or f"{BASE}/hdjlpt/yjzj/answer/{item_id}"
            source_url = BASE

            # 日期
            date = ""
            import datetime
            if display_ts:
                date = datetime.datetime.fromtimestamp(display_ts).strftime("%Y-%m-%d")
            elif pub_ts:
                date = datetime.datetime.fromtimestamp(pub_ts).strftime("%Y-%m-%d")

            if not title:
                continue

            # 查重
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,))
            if c.fetchone():
                print(f"  [DUP] {title[:40]} - {date}")
                total_dup += 1
                continue

            # 内容: 如果API返回的就是文本（非URL），直接使用
            content = ""
            if raw_content and not raw_content.startswith("http"):
                content = raw_content
            elif page_url:
                # 从详情页提取
                dt, dd, dc = extract_detail(page_url)
                content = dc or ""

            if len(content) < 10:
                print(f"  [WARN] 正文过短 ({len(content)}字符)")

            # 入库
            try:
                c.execute(
                    "INSERT INTO gov_raw (title, content, publish_date, source_url, page_url, site_name, group_name, industry) VALUES (?,?,?,?,?,?,?,?)",
                    (title, content, date, source_url, page_url, SITE_NAME, "广东", "政府公告"),
                )
                conn.commit()
                print(f"  [NEW] {title[:40]} - {date}")
                total_new += 1
            except sqlite3.IntegrityError as e:
                print(f"  [DUP] {title[:40]} - {e}")
                total_dup += 1

        if pg * PER_PAGE + PER_PAGE >= total:
            print("  [END] 已到最后一页")
            break

    # FTS
    print(f"\n{'='*50}")
    print(f"新增: {total_new} | 重复: {total_dup}")
    if total_new > 0:
        print("同步FTS...")
        rows = c.execute(
            "SELECT id, title, site_name, content FROM gov_raw WHERE id NOT IN (SELECT rowid FROM gov_search)"
        ).fetchall()
        if rows:
            sql = "INSERT INTO gov_search(rowid, title, site_name, summary) VALUES\n"
            vals = []
            for row_id, t, sn, _ in rows:
                summary = (_ or "")[:500].replace("'", "''")
                t_esc = (t or "").replace("'", "''")
                sn_esc = (sn or "").replace("'", "''")
                vals.append(f"({row_id},'{t_esc}','{sn_esc}','{summary}')")
            sql += ",\n".join(vals) + ";"
            proc = subprocess.run(
                ["sqlite3", "-cmd", ".timeout 60000", DB],
                input=sql,
                capture_output=True,
                text=True,
                timeout=30,
            )
            if proc.returncode == 0:
                print(f"FTS同步完成: {len(rows)}条")
            else:
                print(f"FTS同步失败: {proc.stderr}")
        else:
            print("FTS无需同步")
    conn.close()
    print(f"完成! 共 {total_new} 新, {total_dup} 重复")


if __name__ == "__main__":
    main()
