#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
njna 南京江北新区 导入脚本 v2 (服务器端)
- 保留 HTML 正文 (不 strip_html)
- site_name 从 jsonl 读取 (环保/公示公告 双栏目)
- 按 URL 先 DELETE 旧行再 INSERT (覆盖 strip_html 旧版)
用法: python3 crawl_njna_xxgk_v2.py --file /path/njna_xxgk_hb.jsonl [--file2 /path/njna_xxgk_gsgg.jsonl]
"""
import json
import sqlite3
import os
import argparse
import re

DB_PATH = "/root/search.db"

CATEGORY = "政府信息公开"
SCRIPT = "crawl_njna_xxgk.py"

def date_rank(date_str):
    """2026-08-14 → 20260814 数字, 便于排序"""
    m = re.match(r"(\d{4})-(\d{2})-(\d{2})", date_str or "")
    if m:
        return int(m.group(1)) * 10000 + int(m.group(2)) * 100 + int(m.group(3))
    return 0

def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--file", action="append", required=True, help="JSONL 文件 (可多次传)")
    args = ap.parse_args()
    files = args.file

    conn = sqlite3.connect(DB_PATH, timeout=290)
    conn.execute("PRAGMA busy_timeout=290000")
    cur = conn.cursor()

    total_added = total_updated = total_skipped = 0
    for path in files:
        added = updated = skipped = 0
        with open(path, encoding="utf-8") as f:
            for line in f:
                line = line.strip()
                if not line:
                    continue
                try:
                    d = json.loads(line)
                except Exception:
                    skipped += 1
                    continue
                url = d.get("url", "")
                title = (d.get("title") or "")[:500]
                date = (d.get("date") or "")[:20]
                content = (d.get("content") or "").strip()
                if not content:
                    skipped += 1
                    continue
                source = (d.get("source") or "")[:100]
                site_name = d.get("site_name") or "南京江北新区管委会-政府信息公开"
                att = d.get("attachments") or []
                att_json = json.dumps(att, ensure_ascii=False) if att else ""
                has_table = 1 if "<table" in content else 0
                dr = date_rank(date)
                summary = re.sub(r"<[^>]+>", " ", content)
                summary = re.sub(r"\s+", " ", summary).strip()[:500] or title
                try:
                    # 覆盖旧行 (旧版 strip_html 或同 URL 旧版)
                    cur.execute("DELETE FROM gov_raw WHERE page_url=?", (url,))
                    cur.execute(
                        """INSERT INTO gov_raw
                           (site_name, source_url, page_url, title, publish_date, date_rank,
                            content, summary, category, script_name, attachments, has_table, status)
                           VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)""",
                        (site_name, url, url, title, date, dr,
                         content, summary, CATEGORY, SCRIPT, att_json, has_table, "1")
                    )
                    if cur.rowcount > 0:
                        added += 1
                    else:
                        updated += 1
                except sqlite3.Error as e:
                    print("  DB err:", e)
                    skipped += 1
                conn.commit()
        print(f"  {os.path.basename(path)}: 新增 {added} / 更新 {updated} / 跳过 {skipped} (site={site_name})")
        total_added += added
        total_updated += updated
        total_skipped += skipped
    conn.close()
    print(f"=== 总计: 新增 {total_added} / 更新 {total_updated} / 跳过 {total_skipped} ===")

if __name__ == "__main__":
    main()
