#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""bhna content 重抓修复：对被 v3 截断的记录，用详情页 URL 重新提取 zoom 正文
crawl_bhna.fetch_detail 已用 full_url 正确基准做 clean_zoom_html
"""
import sys, os, time, sqlite3
sys.path.insert(0, '/root/gov_crawler')
import crawl_bhna

DB = "/root/search.db"
SITE_NAME = "港城产业园区-信息公示"

conn = sqlite3.connect(DB, timeout=60)
conn.execute("PRAGMA busy_timeout=30000")

# 找被截断的记录：content 含 bhna.gov.cn 且 URL 结尾是 / (被截断特征)
cur = conn.execute(
    """SELECT id, COALESCE(NULLIF(page_url,''), NULLIF(source_url,'')), title
       FROM gov_raw WHERE site_name=? AND content LIKE '%bhna.gov.cn%'""",
    (SITE_NAME,))
rows = cur.fetchall()
print(f"待重抓: {len(rows)} 条")

ok = skip = fail = 0
for idx, (rid, url, old_title) in enumerate(rows, 1):
    if not url:
        print(f"  [{idx}/{len(rows)}] SKIP 无URL {rid}")
        skip += 1
        continue
    try:
        content, detail_title = crawl_bhna.fetch_detail(url)
    except Exception as e:
        print(f"  [{idx}/{len(rows)}] FAIL {url[-50:]}: {e}")
        fail += 1
        time.sleep(0.3)
        continue
    if not content:
        print(f"  [{idx}/{len(rows)}] SKIP 空正文 {url[-50:]}")
        skip += 1
        time.sleep(0.3)
        continue
    conn.execute("UPDATE gov_raw SET content=? WHERE id=?", (content, rid))
    conn.commit()
    ok += 1
    if idx % 10 == 0:
        print(f"  [{idx}/{len(rows)}] ok={ok} skip={skip} fail={fail}", flush=True)
    time.sleep(0.3)

conn.close()
print(f"\nTOTAL: ok={ok} skip={skip} fail={fail}")
