#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
康定市人民政府(www.kangding.gov.cn) - 公示公告栏目 (column 1840)
URL: https://www.kangding.gov.cn/wzlb/c/1840 (旧入口 /kdsrmzf/c100353/common_list.shtml 301 至此)
CMS: 自建 Vue.js SPA + SSR
列表: POST /site-service/c/manuscript/getManuscriptListByColumnId
  参数 {"columnId":"1840","current":N,"size":15,"type":false} -> data.total/data.records[]
  1667 条, 112 页, 15条/页
详情: https://www.kangding.gov.cn/gsgg/article/{id} (SSR HTML)
  标题 div.txt-title, 日期 span[title], 正文 div#contentBox
生产库 schema: gov_raw(id, site_name, source_url, page_url, title, publish_date, date_rank,
  summary, status, category, visits, content, tags, industry, attachments, group_name, has_table, script_name)
gov_search: fts5(title, site_name, summary, tokenize=trigram) rowid 对齐 gov_raw.id
"""
import argparse
import hashlib
import os
import re
import sqlite3
import time
from html import unescape
from urllib.parse import urljoin

import requests

BASE_URL = "https://www.kangding.gov.cn"
LIST_API = BASE_URL + "/site-service/c/manuscript/getManuscriptListByColumnId"
COLUMN_ID = "1840"
SITE_NAME = "康定市-公示公告"
SCRIPT_NAME = "crawl_kangding.py"
GROUP_NAME = "四川"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "application/json, text/plain, */*",
    "Content-Type": "application/json",
    "Referer": BASE_URL + "/gsgg/c/" + COLUMN_ID,
}


def clean_title(t):
    """清洗标题: strip &middot;&nbsp;&#32; 实体前缀 和站点后缀"""
    t = unescape(t or "")
    t = re.sub(r"^[\s\xa0·\u00b7]+", "", t)
    t = re.sub(r"\s*\.{3,}\s*$", "", t)
    t = re.sub(r"\s+", " ", t).strip()
    return t


def fetch_list(session, current=1, size=15):
    """POST 列表 API, 返回 records 列表"""
    payload = {"columnId": COLUMN_ID, "current": current, "size": size, "type": False}
    r = session.post(LIST_API, json=payload, headers=HEADERS, timeout=30, verify=False)
    if r.status_code != 200:
        return None, 0
    data = r.json()
    if data.get("code") != 0:
        return None, 0
    d = data.get("data") or {}
    return d.get("records") or [], d.get("total", 0)


def parse_detail(html_text, url):
    """解析详情页, 返回 (title, date, content_html, attachments)"""
    # 标题
    title = ""
    m = re.search(r'<div class="txt-title"[^>]*>(.*?)</div>', html_text, re.S)
    if m:
        title = clean_title(re.sub(r"<[^>]+>", "", m.group(1)))
    # 日期
    date = ""
    dm = re.search(r'<span[^>]*title="(\d{4}-\d{2}-\d{2})', html_text)
    if dm:
        date = dm.group(1)
    # 正文 contentBox
    cm = re.search(r'<div id="contentBox"[^>]*>(.*?)</div>\s*</div>', html_text, re.S)
    if not cm:
        cm = re.search(r'<div id="contentBox"[^>]*>(.*)', html_text, re.S)
    if not cm:
        return title, date, "", []
    body = cm.group(1)
    # 附件
    atts = []
    for a in re.finditer(r'<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>', body, re.S):
        href, txt = a.group(1), re.sub(r"<[^>]+>", "", a.group(2)).strip()
        if re.search(r"\.(?:pdf|docx?|xlsx?|zip|rar|wps|et|dps)(?:\?|$)", href, re.I) or txt.startswith("附件"):
            atts.append((urljoin(url, href), txt or href.split("/")[-1]))
    return title, date, body, atts


def html_to_text(html_text, attachments):
    """HTML -> 文本, 保留表格、\\n\\n 分段、附件嵌入"""
    tables = []

    def _save_tbl(m):
        tables.append(m.group(0))
        return f"@@TABLE{len(tables) - 1}@@"

    html_text = re.sub(r"<table[\s\S]*?</table>", _save_tbl, html_text)
    html_text = re.sub(r"</?(?:p|div|tr|br|h[1-6]|li)[^>]*>", "\n", html_text)
    html_text = re.sub(r"</?(?:span|font|b|strong|i|em|u|o:p|o|a)[^>]*>", "", html_text)
    html_text = html_text.replace("&nbsp;", " ").replace("\xa0", " ")
    html_text = re.sub(r"\n{3,}", "\n\n", html_text)
    lines = [ln.strip() for ln in html_text.split("\n")]
    out = [ln for ln in lines if ln]
    text = "\n\n".join(out)
    for i, t in enumerate(tables):
        text = text.replace(f"@@TABLE{i}@@", t)
    text = re.sub(r"\n{3,}", "\n\n", text)
    # 附件嵌入: <p><a href="完整URL">附件名</a></p>
    for href, txt in attachments:
        if href not in text:
            text += f'\n\n<p><a href="{href}" target="_blank">{txt}</a></p>'
    return text.strip()


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1, help="抓取页数")
    ap.add_argument("--db", default="/mnt/data/search.db", help="SQLite 数据库路径")
    ap.add_argument("--script-name", default=SCRIPT_NAME)
    args = ap.parse_args()

    db_path = args.db
    script_name = args.script_name
    if not db_path.startswith("/mnt/"):
        db_path = os.path.expanduser(db_path)

    conn = sqlite3.connect(db_path, timeout=60)
    cur = conn.cursor()
    new_count = 0
    dup_count = 0
    err_count = 0

    session = requests.Session()
    # 先访问栏目页获取 cookie
    try:
        session.get(BASE_URL + "/gsgg/c/" + COLUMN_ID, headers=HEADERS, timeout=30, verify=False)
    except Exception:
        pass

    # 第一页确定总页数
    _, total = fetch_list(session, current=1, size=15)
    total_pages = (total + 14) // 15 if total else 1
    pages_to_crawl = min(args.pages, total_pages)
    print(f"共 {total} 条 / {total_pages} 页, 本次爬 {pages_to_crawl} 页")

    for page in range(1, pages_to_crawl + 1):
        records, _ = fetch_list(session, current=page, size=15)
        if records is None:
            print(f"[{page}] 列表请求失败")
            err_count += 1
            continue
        print(f"[{page}] 列表 {len(records)} 条")
        if not records:
            continue
        for rec in records:
            try:
                article_id = rec.get("id")
                title = clean_title(rec.get("title", ""))
                pub_date = (rec.get("releaseTime") or rec.get("time") or "")[:10]
                if not title or not article_id:
                    continue
                art_url = f"{BASE_URL}/gsgg/article/{article_id}"
                cur.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=? AND script_name=?", (art_url, script_name))
                if cur.fetchone()[0] > 0:
                    dup_count += 1
                    continue
                dr = session.get(art_url, headers={**HEADERS, "Content-Type": "text/html"}, timeout=30, verify=False)
                dr.encoding = "utf-8"
                d_title, d_date, content_html, atts = parse_detail(dr.text, art_url)
                if not content_html.strip():
                    print(f"  跳过空正文: {title[:30]}")
                    dup_count += 1
                    continue
                if d_title:
                    title = d_title
                if d_date:
                    pub_date = d_date
                content = html_to_text(content_html, atts)
                att_txt = "\n\n".join(h for h, _ in atts) if atts else ""
                md5 = hashlib.md5((title + content[:500]).encode("utf-8")).hexdigest()
                summary = content[:500] if content else ""
                cur.execute(
                    "INSERT INTO gov_raw (site_name, source_url, page_url, title, publish_date, date_rank, summary, status, category, visits, content, tags, industry, attachments, group_name, has_table, script_name) "
                    "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
                    (SITE_NAME, art_url, art_url, title, pub_date, 0, summary, "", "", 0, content, "",
                     "other", att_txt, GROUP_NAME, 1 if "<table" in content else 0, script_name),
                )
                # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                conn.commit()
                cur.execute(
                    "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                    (cur.lastrowid, title, SITE_NAME, summary),
                )
                conn.commit()
                new_count += 1
                print(f"  + {pub_date} {title[:40]}")
            except Exception as e:
                err_count += 1
                print(f"  ERR {title[:30]}: {e}")
        time.sleep(0.3)

    conn.close()
    print(f"完成: 新增 {new_count} / 重复 {dup_count} / 错误 {err_count}")


if __name__ == "__main__":
    main()
