#!/usr/bin/env python3
"""修复 libo 已入库正文 - 从已存储的HTML中提取div#Zoom并清理底部工具栏"""
import sys, os, sqlite3, re
from bs4 import BeautifulSoup

DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
db = sqlite3.connect(DB)

rows = db.execute(
    "SELECT id, content FROM gov_raw WHERE script_name='crawl_libo.py' AND content != ''"
).fetchall()
print(f"共 {len(rows)} 条需要检查")

fixed = 0
for rid, content in rows:
    if not content:
        continue
    
    soup = BeautifulSoup(content, "html.parser")
    
    # 提取 div#Zoom 中的内容（这是实际正文）
    zoom = soup.select_one("#Zoom")
    if not zoom:
        continue
    
    # 复制 Zoom 内容并清理底部脚本和工具栏
    zoom_html = str(zoom)
    
    # 去掉底部的：视频、分享脚本、【返回顶部】【打印本页】等
    # 找到最后一个实际内容（表格或段落）
    zoom_soup = BeautifulSoup(zoom_html, "html.parser")
    
    # 移除底部的脚本和分享 DIV
    for tag in zoom_soup.find_all(["script", "div"]):
        tag_text = tag.get_text(strip=True)
        if any(x in tag_text for x in ["【返回顶部】", "【打印本页】", "【关闭本页】", "【我要定制】", "【我要推荐】", "上一篇", "下一篇"]):
            tag.decompose()
        elif tag.name == "script":
            tag.decompose()
    
    # 移除顶部的视频脚本 div#sp
    for tag in zoom_soup.find_all(id="sp"):
        tag.decompose()
    
    clean_content = str(zoom_soup)
    
    # 如果清理后内容太短，保留原始 Zoom 内容
    if len(clean_content) < 50:
        clean_content = zoom_html
    
    # 跳过不需要更新的
    if content == clean_content:
        continue
    
    # 保存
    db.execute("UPDATE gov_raw SET content=? WHERE id=?", (clean_content, rid))
    fixed += 1
    if fixed % 100 == 0:
        print(f"  已修复 {fixed} 条...")
        db.commit()

db.commit()
db.close()
print(f"\n完成：修复 {fixed} 条")
