"""Fix specific anyi article with table duplication"""
import requests, sqlite3, re, warnings
from bs4 import BeautifulSoup
warnings.filterwarnings('ignore')

url = "https://anyi.nc.gov.cn/ayxzf/xsthjjgggs2021/202605/0ea881d3a64341b99b9cf8aac83843a7.shtml"
headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": "https://anyi.nc.gov.cn/ayxzf/xsthjjgggs2021/just_list.shtml",
}

r = requests.get(url, headers=headers, timeout=30, verify=False)
r.encoding = 'utf-8'
if r.status_code != 200 or 'error_403' in r.text:
    print(f"BLOCKED: {r.status_code}")
    exit(1)

# Find UCAPCONTENT
m = re.search(r'<UCAPCONTENT>(.*?)</UCAPCONTENT>', r.text, re.DOTALL)
if not m:
    print("No UCAPCONTENT found")
    exit(1)

ucap = m.group(1)

# Strip inline tags
ucap = re.sub(r'</?(?:span|o:p|font|b|strong|em|i|u|s|strike|sub|sup)[^>]*>', '', ucap)

# Extract tables, then remove from text
tables = re.findall(r'<table[\s\S]*?</table>', ucap, re.I)
ucap_no_tables = re.sub(r'<table[\s\S]*?</table>', '', ucap, flags=re.I)

# Get text from paragraphs only (tables removed, no duplication)
soup = BeautifulSoup(ucap_no_tables, 'html.parser')
parts = []
for p in soup.find_all('p'):
    txt = p.get_text(strip=True)
    if txt:
        parts.append(txt)

# Append table HTML
for t_html in tables:
    parts.append(t_html)

new_content = '\n\n'.join(parts)
summary = new_content[:500]

# Update DB
db = sqlite3.connect('/root/search.db', timeout=60)
db.execute("PRAGMA busy_timeout=60000")
db.execute("UPDATE gov_raw SET content = ?, summary = ? WHERE page_url = ?",
           (new_content, summary, url))
db.commit()
db.close()

print(f"✅ Updated: {len(new_content)} chars, {len(tables)} tables")
print(f"First 200 chars: {new_content[:200]}")
print(f"Last 200 chars: {new_content[-200:]}")
