#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
昆明市晋宁区人民政府(www.kmjn.gov.cn) - 生态环境栏目
URL: http://www.kmjn.gov.cn/zfxxgk/fdzdgknr/zdlyxxgk/shgysyjsly/hjbh/
CMS: zcms (index_N.shtml 静态分页)
列表: div.data-table-item > p.w659 > a (标题) + p.w80 > a[title] (日期)
分页: index.shtml / index_N.shtml, 23页, 20条/页
详情: div.content > h1 + div.activity (正文, 含表格)
生产库 schema: gov_raw(id, site_name, source_url, page_url, title, publish_date, date_rank,
  summary, status, category, visits, content, tags, industry, attachments, group_name, has_table, script_name)
gov_search: fts5(title, site_name, summary, tokenize=trigram) rowid 对齐 gov_raw.id
"""
import argparse
import hashlib
import os
import re
import sqlite3
import time
from html import unescape
from urllib.parse import urljoin

import requests

BASE_URL = "http://www.kmjn.gov.cn"
LIST_URL = "http://www.kmjn.gov.cn/zfxxgk/fdzdgknr/zdlyxxgk/shgysyjsly/hjbh/"
SITE_NAME = "昆明市晋宁区人民政府-生态环境"
SCRIPT_NAME = "crawl_kmjn_hjbh.py"
GROUP_NAME = "云南"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def clean_title(t):
    """清洗标题: strip &middot;&nbsp; 等实体前缀"""
    t = unescape(t or "")
    t = re.sub(r"^\s*(?:&middot;|·|&nbsp;|\s)+", "", t)
    t = re.sub(r"\s*_\s*昆明市晋宁区人民政府.*$", "", t)
    t = re.sub(r"\s+", " ", t).strip()
    return t


def parse_list(html_text, base_url):
    """解析列表页, 返回 [(title, url, date), ...]"""
    items = []
    # 结构: <p class="w659"><a href="...">标题</a></p><p class="w80"><a href="..." title="2026-06-09 14:41:51">2026.06.09</a></p>
    pat = re.compile(
        r'<p class="w659">\s*<a href="([^"]+)"[^>]*>\s*([^<]+?)\s*</a>\s*</p>\s*'
        r'<p class="w80">\s*<a href="[^"]*"[^>]*title="([^"]*)"',
        re.S,
    )
    for m in pat.finditer(html_text):
        href, title, ts = m.group(1), m.group(2), m.group(3).strip()
        if not href.startswith(("http://", "https://")):
            href = urljoin(base_url, href)
        if "/c/" not in href:
            continue
        title = clean_title(title)
        if not title:
            continue
        # 日期: title 属性完整时间戳取前10位
        date = ts[:10] if ts else ""
        if not date:
            dm = re.search(r'(\d{4})\.(\d{2})\.(\d{2})', html_text)
            date = f"{dm.group(1)}-{dm.group(2)}-{dm.group(3)}" if dm else ""
        items.append((title, href, date))
    seen = set()
    out = []
    for it in items:
        if it[1] not in seen:
            seen.add(it[1])
            out.append(it)
    return out


def parse_detail(html_text, url):
    """解析详情页, 返回 (content_html, attachments)"""
    m = re.search(r'<div class="content">.*?<div class="activity">(.*?)</div>\s*</div>\s*</div>', html_text, re.S)
    if not m:
        m = re.search(r'<div class="activity">(.*?)</div>\s*</div>\s*</div>', html_text, re.S)
    if not m:
        return "", []
    body = m.group(1)
    atts = re.findall(r'<a[^>]+href="([^"]+)"[^>]*>', body)
    atts = [urljoin(url, a) for a in atts if re.search(r'\.(?:pdf|docx?|xlsx?|zip|rar|wps|et|dps)(?:\?|$)', a, re.I)]
    return body, atts


def html_to_text(html_text):
    """HTML -> 文本, 保留表格、\n\n 分段"""
    tables = []
    def _save_tbl(m):
        tables.append(m.group(0))
        return f"@@TABLE{len(tables)-1}@@"
    html_text = re.sub(r'<table[\s\S]*?</table>', _save_tbl, html_text)
    html_text = re.sub(r'</?(?:p|div|tr|br|h[1-6]|li)[^>]*>', '\n', html_text)
    html_text = re.sub(r'</?(?:span|font|b|strong|i|em|u|o:p|o|a)[^>]*>', '', html_text)
    html_text = re.sub(r'\n{3,}', '\n\n', html_text)
    lines = [ln.strip() for ln in html_text.split('\n')]
    out = [ln for ln in lines if ln]
    text = '\n\n'.join(out)
    for i, t in enumerate(tables):
        text = text.replace(f"@@TABLE{i}@@", t)
    text = re.sub(r'\n{3,}', '\n\n', text)
    return text.strip()


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=1, help="抓取页数")
    ap.add_argument("--db", default="/mnt/data/search.db", help="SQLite 数据库路径")
    ap.add_argument("--script-name", default=SCRIPT_NAME)
    args = ap.parse_args()

    db_path = args.db
    script_name = args.script_name
    if not db_path.startswith("/mnt/"):
        db_path = os.path.expanduser(db_path)

    conn = sqlite3.connect(db_path, timeout=60)
    cur = conn.cursor()
    new_count = 0
    dup_count = 0
    err_count = 0

    for page in range(1, args.pages + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = f"http://www.kmjn.gov.cn/zfxxgk/fdzdgknr/zdlyxxgk/shgysyjsly/hjbh/index_{page}.shtml"
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"[{page}] 请求失败 {url}: {e}")
            err_count += 1
            continue
        items = parse_list(r.text, url)
        print(f"[{page}] 列表 {len(items)} 条 {url}")
        if not items:
            continue
        for title, art_url, pub_date in items:
            try:
                cur.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url=? AND script_name=?", (art_url, script_name))
                if cur.fetchone()[0] > 0:
                    dup_count += 1
                    continue
                dr = requests.get(art_url, headers=HEADERS, timeout=30)
                dr.encoding = "utf-8"
                # 详情页 h1 标题优先(列表标题一般完整, 但以防截断)
                hm = re.search(r'<div class="content">\s*<h1>(.*?)</h1>', dr.text, re.S)
                if hm:
                    ft = re.sub(r'<[^>]+>', '', hm.group(1))
                    ft = clean_title(ft)
                    if ft:
                        title = ft
                content_html, atts = parse_detail(dr.text, art_url)
                if not content_html.strip():
                    print(f"  跳过空正文: {title[:30]}")
                    continue
                content = html_to_text(content_html)
                att_txt = "\n\n".join(atts) if atts else ""
                md5 = hashlib.md5((title + content[:500]).encode("utf-8")).hexdigest()
                summary = content[:500] if content else ""
                cur.execute(
                    "INSERT INTO gov_raw (site_name, source_url, page_url, title, publish_date, date_rank, summary, status, category, visits, content, tags, industry, attachments, group_name, has_table, script_name) "
                    "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
                    (SITE_NAME, art_url, art_url, title, pub_date, 0, summary, "", "", 0, content, "",
                     "other", att_txt, GROUP_NAME, 1 if "<table" in content else 0, script_name),
                )
                # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                conn.commit()
                cur.execute(
                    "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                    (cur.lastrowid, title, SITE_NAME, summary),
                )
                conn.commit()
                new_count += 1
                print(f"  + {pub_date} {title[:40]}")
            except Exception as e:
                err_count += 1
                print(f"  ERR {title[:30]}: {e}")
        time.sleep(0.3)

    conn.close()
    print(f"完成: 新增 {new_count} / 重复 {dup_count} / 错误 {err_count}")


if __name__ == "__main__":
    main()
