#!/usr/bin/env python3
"""Re-extract content for all 贵溪市 articles with proper BS4 parsing."""
import sys, os, re, json, time
import requests
from bs4 import BeautifulSoup
import sqlite3

SEARCH_DB = "/root/search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
MARKER = "\x00P\x00"
session = requests.Session()
session.headers.update(HEADERS)

def extract_content(html, title, detail_url):
    """BS4 extraction with paragraph markers and table preservation."""
    soup = BeautifulSoup(html, "html.parser")
    content_div = soup.select_one("div#zoom")
    if not content_div:
        if title:
            return f"[{title}]({detail_url})"
        return ""

    # 1) Save and remove tables
    tables_html = []
    for table in content_div.find_all("table"):
        try:
            tables_html.append(str(table))
        except Exception:
            tables_html.append(table.get_text(separator=" ", strip=True))
        table.decompose()

    # 2) Insert paragraph markers
    for tag in content_div.find_all(["p", "h1", "h2", "h3", "h4", "h5", "h6"]):
        tag.insert(0, MARKER)
        tag.append(MARKER)
    for br in content_div.find_all("br"):
        br.replace_with(MARKER)

    # 3) Unwrap inline tags
    for tag in content_div.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
        tag.unwrap()

    # 4) Extract text with empty separator
    text = content_div.get_text(separator="", strip=True)

    # 5) Clean markers
    text = re.sub(r"\x00P\x00(\s*\x00P\x00)+", "\x00P\x00", text)
    text = text.replace("\x00P\x00", "\n\n")
    text = re.sub(r"\n{3,}", "\n\n", text)
    text = text.strip()

    # 6) Append tables
    for tbl_html in tables_html:
        text += f"\n\n{tbl_html}"

    # 7) Empty fallback
    if not text.strip() or len(text.strip()) < 20:
        if title:
            text = f"[{title}]({detail_url})"

    return text

def main():
    print("=== 贵溪市 内容重新提取 ===", flush=True)
    conn = sqlite3.connect(SEARCH_DB)
    c = conn.cursor()

    # Get all guixi articles
    c.execute("SELECT id, title, page_url FROM gov_raw WHERE page_url LIKE '%guixi.gov.cn%'")
    rows = c.fetchall()
    print(f"共 {len(rows)} 条", flush=True)

    ok, fail = 0, 0
    for idx, (row_id, title, url) in enumerate(rows, 1):
        print(f"  [{idx}/{len(rows)}] {title[:40]}... ", end="", flush=True)
        try:
            r = session.get(url, timeout=20)
            r.encoding = "utf-8"
            content = extract_content(r.text, title or "", url)
            if content:
                c.execute("UPDATE gov_raw SET content=? WHERE id=?", (content, row_id))
                if c.rowcount > 0:
                    ok += 1
                    print(f"OK {len(content)}B", flush=True)
                else:
                    fail += 1
                    print(f"no update", flush=True)
            else:
                fail += 1
                print(f"empty", flush=True)
        except Exception as e:
            fail += 1
            print(f"ERR: {str(e)[:60]}", flush=True)
            time.sleep(2)
        if idx % 10 == 0:
            conn.commit()

    conn.commit()
    conn.close()
    print(f"\n完毕: {ok} 更新, {fail} 失败", flush=True)

if __name__ == "__main__":
    main()
