#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""2026-08-11 荆州正文截断存量修复:
重抓 search.db 中 site_name=荆州市生态环境局 且 content=8000(截断特征) 的详情页,
用完整正文 UPDATE 回 gov_raw (FTS 触发器自动维护 gov_search, id 不变)。
"""
import sqlite3
import sys
import time

sys.path.insert(0, '/root/gov_crawler')
from crawl_jz import get_detail, extract_plain_text

DB = "/mnt/data/search.db"
SITE = "荆州市生态环境局"

conn = sqlite3.connect(DB)
conn.execute("PRAGMA journal_mode=WAL")
rows = conn.execute(
    "SELECT id, page_url, title, publish_date, length(content) FROM gov_raw "
    "WHERE site_name=? AND length(content)=8000 ORDER BY id",
    (SITE,)
).fetchall()
total = len(rows)
print(f"待刷新: {total} 条 (content=8000 截断特征)")

updated = nochange = failed = 0
for i, (rid, url, title, pdate, clen) in enumerate(rows, 1):
    try:
        d = get_detail(url)
        if not d or not d.get('content'):
            failed += 1
            continue
        new_content = d['content']
        new_summary = extract_plain_text(new_content)[:300]
        new_title = (d.get('title') or '').strip() or title
        if len(new_content) <= clen:
            nochange += 1
            continue
        conn.execute(
            "UPDATE gov_raw SET content=?, summary=?, title=? WHERE id=?",
            (new_content, new_summary, new_title, rid)
        )
        conn.commit()
        updated += 1
        if updated <= 5 or updated % 100 == 0:
            print(f"  + {rid} {len(new_content)}字 {new_title[:30]}")
    except Exception as e:
        failed += 1
        print(f"  ERR {rid} {url}: {e}")
    if i % 100 == 0:
        print(f"  ...{i}/{total} (更新{updated} 无变化{nochange} 失败{failed})")
    time.sleep(0.2)

print(f"完成: 更新 {updated} / 无变化 {nochange} / 失败 {failed} / 总 {total}")
conn.close()
