#!/usr/bin/env python3
"""Fix records with footer chrome elements."""
import subprocess, sqlite3
from bs4 import BeautifulSoup

DB = '/mnt/data/search.db'

def clean(content_html, detail_url):
    soup = BeautifulSoup(content_html, 'html.parser')
    c = soup.select_one('div.m-dttexts.j-fontContent, div#zoom, div.m-dttexts')
    if not c:
        return None
    for tag in c.select('script, style, link, .m-dtcode, .m-btfuns, .m-dtdownload'):
        tag.decompose()
    paras = []
    for child in list(c.children):
        if child.name is None:
            continue
        if child.name == 'table':
            continue
        if child.find('table'):
            continue
        if child.name in ['p', 'div', 'section', 'h1', 'h2', 'h3', 'h4']:
            t = child.get_text(strip=True)
            if t:
                paras.append(t)
    return '\n\n'.join(paras)

conn = sqlite3.connect(DB)
# Find ahlx records with footer chrome (limit to avoid timeout)
rows = conn.execute(
    "SELECT id, source_url FROM gov_raw WHERE source_url LIKE 'https://www.ahlx.gov.cn/Jczwgk/show/%' AND content LIKE '%扫一扫%'"
).fetchall()
print('Found ' + str(len(rows)) + ' records to fix')

fixed = 0
for rid, src_url in rows:
    r = subprocess.run(['curl', '-sS', '-L', '--max-time', '15', '-k', src_url],
                       capture_output=True, timeout=30)
    if r.returncode != 0 or not r.stdout:
        print('  FAIL fetch id=' + str(rid))
        continue
    html = r.stdout.decode('utf-8', errors='replace')
    body = clean(html, src_url)
    if body and len(body) > 5:
        be = body.replace("'", "''")
        conn.execute("UPDATE gov_raw SET content='" + be + "' WHERE id=" + str(rid))
        conn.commit()
        fixed += 1
        print('  FIXED id=' + str(rid) + ': ' + str(len(body)) + ' chars')
    else:
        print('  SKIP id=' + str(rid) + ': still empty after clean')

print('Fixed: ' + str(fixed))
conn.close()
