#!/usr/bin/env python3
"""
ahxx.gov.cn - 萧县经济开发区管理委员会 通知公告
List: /public/column/239?type=4&catId=60777821&action=list&nav=3&pageIndex=N
Item: <li class="clearfix"><a class="title" href="/public/239/XXXX.html" title="...">title</a><span class="date">YYYY-MM-DD</span></li>
Detail: <h1 class="newstitle"> + <span class="fbxx">发表时间：<i>2026-06-17 16:04</i></span> + <div class="j-fontContent clearfix">
"""
import requests, sqlite3, re, os, sys, urllib3
from bs4 import BeautifulSoup
import html as html_mod

urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "萧县经开区"
BASE_URL = "https://www.ahxx.gov.cn"
LIST_URL = "/public/column/239?type=4&catId=60777821&action=list&nav=3"
DB_PATH = "/root/search.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}


def extract_clean_text(element):
    """Extract content: plain text paragraphs + HTML tables preserved"""
    html_str = str(element)
    html_str = re.sub(r'<span[^>]*>', '', html_str, flags=re.IGNORECASE)
    html_str = re.sub(r'</span>', '', html_str, flags=re.IGNORECASE)
    html_str = re.sub(r'<font[^>]*>', '', html_str, flags=re.IGNORECASE)
    html_str = re.sub(r'</font>', '', html_str, flags=re.IGNORECASE)
    soup = BeautifulSoup(html_str, 'html.parser')
    tables = []
    for table in soup.find_all('table'):
        idx = len(tables)
        tables.append(str(table))
        table.replace_with(f'__TABLE_{idx}__')
    clean_tables = []
    for t in tables:
        for attr in ['style', 'class', 'width', 'height', 'valign', 'align', 'border', 'cellpadding', 'cellspacing']:
            t = re.sub(rf' {attr}="[^"]*"', '', t)
        t = re.sub(r'<p>\s*', '', t)
        t = re.sub(r'\s*</p>', '', t)
        t = re.sub(r'<span[^>]*>', '', t)
        t = re.sub(r'</span>', '', t)
        t = re.sub(r'<font[^>]*>', '', t)
        t = re.sub(r'</font>', '', t)
        clean_tables.append(t)
    text = str(soup)
    text = re.sub(r'</p>\s*', '\n\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</div>\s*', '\n\n', text, flags=re.IGNORECASE)
    text = re.sub(r'<br\s*/?>\s*', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'<[^>]+>', '', text)
    text = html_mod.unescape(text)
    text = re.sub(r'\n{3,}', '\n\n', text)
    text = re.sub(r'[ \t\u2002\u2003\u2005\u00a0\u3000]+', ' ', text)
    text = re.sub(r'\n ', '\n', text)
    text = re.sub(r' \n', '\n', text)
    text = text.strip()
    for idx, clean_table in enumerate(clean_tables):
        text = text.replace(f'__TABLE_{idx}__', '\n\n' + clean_table + '\n\n')
    text = re.sub(r'\n{3,}', '\n\n', text)
    text = re.sub(r'[ \t\u2002\u2003\u2005\u00a0\u3000]+', ' ', text)
    return text.strip()


def safe_summary(text, max_len=500):
    if len(text) <= max_len:
        return text
    truncated = text[:max_len]
    if '<' in truncated:
        last_open = truncated.rfind('<')
        last_close = truncated.rfind('>')
        if last_open > last_close:
            truncated = truncated[:last_open]
    return truncated


def fetch_list_page(page_index):
    """Fetch a list page, return list of (title, url, date_str)"""
    if page_index == 0:
        url = BASE_URL + LIST_URL
    else:
        url = BASE_URL + LIST_URL + f"&pageIndex={page_index}"
    try:
        r = requests.get(url, headers=HEADERS, verify=False, timeout=30)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'html.parser')
        items = []
        for li in soup.find_all('li', class_='clearfix'):
            a = li.find('a', class_='title', href=True)
            if not a:
                continue
            href = a['href'].strip()
            title = a.get('title', '') or a.get_text(strip=True)
            if not title or not href:
                continue
            if not href.startswith('http'):
                href = BASE_URL + href
            # Date
            date_span = li.find('span', class_='date')
            date_str = date_span.get_text(strip=True) if date_span else ''
            if not re.match(r'\d{4}-\d{2}-\d{2}', date_str):
                date_str = '0000-00-00'
            items.append((title, href, date_str))
        return items
    except Exception as e:
        print(f"  [ERROR] list page {page_index}: {e}", flush=True)
        return []


def fetch_detail(url):
    """Fetch detail page content"""
    try:
        r = requests.get(url, headers=HEADERS, verify=False, timeout=30)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'html.parser')
        for tag in soup(['script', 'style', 'nav', 'footer', 'header', 'aside']):
            tag.decompose()
        content_div = soup.find('div', class_='j-fontContent')
        if not content_div:
            content_div = soup.find('div', class_='xxgk-wzcon') or soup.find('div', class_='wzcon')
        content = extract_clean_text(content_div if content_div else soup)
        return content.strip()
    except Exception as e:
        print(f"  [ERROR] detail {url}: {e}", flush=True)
        return ""


def main():
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    total_new = 0
    total_skip = 0

    for page in range(0, 20):
        items = fetch_list_page(page)
        if not items:
            print(f"List page {page}: empty, stopping", flush=True)
            break
        print(f"List page {page}: {len(items)} items", flush=True)

        for title, url, date_str in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                total_skip += 1
                continue

            content = fetch_detail(url)
            if not content or len(content) < 30:
                print(f"  [SKIP] empty: {url.split('/')[-1][:30]}", flush=True)
                continue

            summary = safe_summary(content)
            date_rank = int(date_str.replace('-', '')) if date_str != '0000-00-00' else 0

            c.execute("""
                INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date, content, summary, date_rank)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (SITE_NAME, f"{BASE_URL}/column/239", url, title, date_str, content, summary, date_rank))
            conn.commit()
            total_new += 1
            print(f"  + {title[:50]}... ({date_str})", flush=True)

    conn.close()
    print(f"\nDone: {total_new} new, {total_skip} skipped")


def incremental():
    """Incremental run - only check first page"""
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    total_new = 0

    items = fetch_list_page(0) or fetch_list_page(1) or []
    if not items:
        items = fetch_list_page(2) or []
    print(f"Incremental: {len(items)} items", flush=True)

    for title, url, date_str in items:
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone():
            continue
        content = fetch_detail(url)
        if not content or len(content) < 30:
            continue
        summary = safe_summary(content)
        date_rank = int(date_str.replace('-', '')) if date_str != '0000-00-00' else 0
        c.execute("""
            INSERT OR IGNORE INTO gov_raw
            (site_name, source_url, page_url, title, publish_date, content, summary, date_rank)
            VALUES (?, ?, ?, ?, ?, ?, ?, ?)
        """, (SITE_NAME, f"{BASE_URL}/column/239", url, title, date_str, content, summary, date_rank))
        conn.commit()
        total_new += 1
        print(f"  + {title[:50]}... ({date_str})", flush=True)

    conn.close()
    print(f"Incremental done: {total_new} new")


if __name__ == "__main__":
    if len(sys.argv) > 1 and sys.argv[1] == '--incremental':
        incremental()
    else:
        main()
