#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""补刷淄博两站 CUTOFF 边缘残留行 (面包屑/字号干扰), 用修复后的 fetch_detail"""
import sqlite3, sys
sys.path.insert(0, "/root/gov_crawler")

import crawl_zibo_gxqhbj as GX
import crawl_zibo_epb as EPB

DB = "/root/search.db"

def fetch_residual(site_pattern):
    conn = sqlite3.connect(DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    rows = conn.execute(
        """SELECT id, page_url FROM gov_raw
           WHERE site_name LIKE ? AND content LIKE '%sk-breadcrumb%'""",
        (site_pattern,),
    ).fetchall()
    conn.close()
    return rows

def refetch(module, url):
    try:
        content, date, title = module.fetch_detail(url)
        return content, title
    except Exception as e:
        print(f"  ERR {url}: {e}")
        return None, None

total = 0
for module, pattern, name in [(GX, "淄博高新技术产业开发区%", "gxqhbj"), (EPB, "淄博市生态环境局%", "epb")]:
    rows = fetch_residual(pattern)
    print(f"{name}: {len(rows)} 条残留")
    conn = sqlite3.connect(DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    for rid, url in rows:
        content, title = refetch(module, url)
        if not content:
            print(f"  [{rid}] 抓取失败: {url[-50:]}")
            continue
        conn.execute("UPDATE gov_raw SET content=? WHERE id=?", (content, rid))
        conn.commit()
        total += 1
        print(f"  [{rid}] OK {len(content)}B")
    conn.close()
print(f"完成: {total} 条")
