#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""bhna 存量 content 清洗 v3：修正 v1 错误的绝对 URL
v1 用 BASE urljoin 导致 href 缺月份目录 (c104864/<hash>/files -> c104864/<月份>/<hash>/files)
从 source_url 提取月份段修正
"""
import sqlite3, re
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB = "/root/search.db"
SITE_NAME = "港城产业园区-信息公示"

def fix_urls(raw, src_url):
    """修正 content 中的 img/a URL"""
    if not raw or not src_url:
        return raw
    # 从 source_url 提取月份段: c104864/202605/xxx.shtml -> 202605
    m = re.search(r'/c104864/(\d{6})/', src_url)
    if not m:
        return raw
    month = m.group(1)
    # 修正 img src
    def fix_img(mm):
        src = mm.group(0)
        # 已是 http://www.bhna.gov.cn/bhna/c104864/<32hex>/... 且缺月份
        m2 = re.match(r'(http://www\.bhna\.gov\.cn/bhna/c104864/)([0-9a-f]{32}/)', src)
        if m2:
            return f"{m2.group(1)}{month}/{m2.group(2)}"
        return src
    raw = re.sub(r'http://www\.bhna\.gov\.cn/bhna/c104864/[0-9a-f]{32}/[^"\'>\s]*', fix_img, raw)
    return raw

conn = sqlite3.connect(DB, timeout=60)
conn.execute("PRAGMA busy_timeout=30000")
cur = conn.execute(
    "SELECT id, content, page_url, source_url FROM gov_raw WHERE site_name=? AND content != ''",
    (SITE_NAME,))
rows = cur.fetchall()
print(f"bhna 待修正: {len(rows)} 条")

ok = skip = fail = 0
for rid, raw, page_url, src_url in rows:
    base = page_url or src_url
    if not base:
        skip += 1
        continue
    try:
        fixed = fix_urls(raw, src_url)
    except Exception as e:
        print(f"  FAIL {rid}: {e}")
        fail += 1
        continue
    if not fixed or fixed == raw:
        skip += 1
        continue
    conn.execute("UPDATE gov_raw SET content=? WHERE id=?", (fixed, rid))
    conn.commit()
    ok += 1

conn.close()
print(f"\nTOTAL: ok={ok} skip={skip} fail={fail}")
