"""Re-crawl remaining WAF-blocked anyi articles with retry"""
import requests, sqlite3, re, time, warnings
from bs4 import BeautifulSoup
warnings.filterwarnings('ignore')

DB_PATH = "/root/search.db"
SITE_NAME = "安义县-环评公示"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://anyi.nc.gov.cn/ayxzf/xsthjjgggs2021/just_list.shtml",
}

def clean_ucap_text(element):
    html = str(element)
    html = re.sub(r'</?(?:span|o:p|font|b|strong|em|i|u|s|strike|sub|sup)[^>]*>', '', html)
    tables = re.findall(r'<table[\s\S]*?</table>', html, re.I)
    html_no_tables = re.sub(r'<table[\s\S]*?</table>', '', html, flags=re.I)
    clean = BeautifulSoup(html_no_tables, 'html.parser')
    parts = []
    for p in clean.find_all('p'):
        txt = p.get_text(strip=True)
        if txt:
            parts.append(txt)
    for t_html in tables:
        parts.append(t_html)
    if not parts:
        txt = clean.get_text('\n\n', strip=True)
        return txt if txt else ""
    return '\n\n'.join(parts)

def extract_content(html):
    soup = BeautifulSoup(html, 'html.parser')
    ucap = soup.find('UCAPCONTENT')
    if ucap:
        return clean_ucap_text(ucap)
    cont = soup.select_one('div.article-content-body, div.article-content')
    if cont:
        ucap2 = cont.find('UCAPCONTENT')
        if ucap2:
            return clean_ucap_text(ucap2)
        return clean_ucap_text(cont)
    return soup.get_text('\n\n', strip=True)

db = sqlite3.connect(DB_PATH)
# Find articles that have no table in content (likely blocked)
urls = [r[0] for r in db.execute(
    "SELECT page_url FROM gov_raw WHERE site_name = ? AND content NOT LIKE '%<table%'",
    (SITE_NAME,)).fetchall()]
print(f"Articles without table: {len(urls)}")

updated = 0
for i, page_url in enumerate(urls):
    for attempt in range(3):  # retry up to 3 times
        try:
            resp = requests.get(page_url, headers=HEADERS, timeout=30, verify=False)
            resp.encoding = 'utf-8'
        except:
            time.sleep(2)
            continue
        
        if resp.status_code == 200 and 'error_403' not in resp.text:
            new_content = extract_content(resp.text)
            if new_content:
                summary = new_content[:500]
                db.execute("UPDATE gov_raw SET content = ?, summary = ? WHERE page_url = ?",
                           (new_content, summary, page_url))
                db.commit()
                updated += 1
                if updated % 10 == 0:
                    print(f"  Updated {updated}...")
            break
        else:
            time.sleep(3 * (attempt + 1))  # exponential backoff
    
    if (i+1) % 20 == 0:
        print(f"  Progress: {i+1}/{len(urls)}, updated: {updated}")

db.close()
print(f"\nDone. Updated: {updated} of {len(urls)}")
