#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
沭阳县人民政府-便民公告 爬虫
UCAP CMS, shtml static pagination
List: /shuyang/bmgg/olist.shtml (p1), olist_N.shtml (N=2..43)
Detail: /shuyang/bmgg/YYYYMM/UUID.shtml
"""

import re, time, os
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup, NavigableString

BASE_URL = "http://www.shuyang.gov.cn"
LIST_PATH = "/shuyang/bmgg/olist.shtml"
TOTAL_PAGES = 43  # from createPageHTML('page_div',43,1,'olist','shtml',636)
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def safe_get(url, timeout=15):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=timeout)
            r.encoding = 'utf-8'
            return r
        except Exception as e:
            if attempt < 2:
                time.sleep(2)
            else:
                print(f"  [WARN] Failed {url}: {e}")
                return None

def extract_ucap_content(ucap_div, page_url):
    """Extract structured content from UCAPCONTENT div."""
    parts = []
    
    for child in ucap_div.children:
        name = getattr(child, 'name', None)
        
        if name == 'table':
            parts.append(str(child))
            parts.append("\n\n")
        elif name in ('p', 'div'):
            # Check for images
            for img in child.find_all('img'):
                src = img.get('src', '')
                if src:
                    full_src = urljoin(page_url, src)
                    alt = img.get('alt', '') or 'image'
                    parts.append(f"![{alt}]({full_src})\n\n")
            
            # Check for tables inside
            inner_table = child.find('table')
            if inner_table:
                parts.append(str(inner_table))
                parts.append("\n\n")
                # Also get any text before/after table
                txt = child.get_text(' ', strip=True)
                if txt:
                    parts.append(txt)
                    parts.append("\n\n")
            else:
                # Strip inline tags
                child_html = str(child)
                child_html = re.sub(
                    r'</?(?:span|b|strong|font|em|i|u|s|sub|sup|small|mark)[^>]*>',
                    '', child_html, flags=re.I
                )
                clean = BeautifulSoup(child_html, 'html.parser')
                txt = clean.get_text(' ', strip=True)
                txt = re.sub(r'\s+', ' ', txt).strip() if txt else ""
                if txt:
                    parts.append(txt)
                    parts.append("\n\n")
        elif name == 'h1':
            txt = child.get_text(' ', strip=True)
            if txt:
                parts.append(f"# {txt}\n\n")
        elif name == 'h2':
            txt = child.get_text(' ', strip=True)
            if txt:
                parts.append(f"## {txt}\n\n")
        elif name == 'a':
            href = child.get('href', '')
            text = child.get_text(' ', strip=True)
            if href and (".doc" in href or ".pdf" in href or ".xls" in href or "upload" in href.lower()):
                parts.append(f"[{text}]({urljoin(page_url, href)})\n\n")
        elif name == 'ul':
            for li in child.find_all('li', recursive=False):
                li_text = li.get_text(' ', strip=True)
                if li_text:
                    parts.append(f"- {li_text}\n")
            parts.append("\n")
        elif name == 'ol':
            for li in child.find_all('li', recursive=False):
                li_text = li.get_text(' ', strip=True)
                if li_text:
                    parts.append(f"  {li_text}\n")
            parts.append("\n")
        elif name is None:
            t = str(child).strip()
            if t and len(t) > 5:
                parts.append(t)
                parts.append("\n\n")
        elif name in ('style', 'script'):
            continue
    
    result = ''.join(parts).strip()
    result = re.sub(r'\n{4,}', '\n\n', result)
    return result

def parse_detail(url):
    r = safe_get(url)
    if not r:
        return None, None, None
    
    soup = BeautifulSoup(r.text, "html.parser")
    
    # Title
    title = ""
    h1 = soup.find("h1", class_="article-title")
    if h1:
        # Remove UCAPTITLE text if present
        title = h1.get_text(strip=True)
    else:
        # Fallback to meta
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta:
            title = meta.get("content", "")
    
    # Date from meta
    date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date:
        d = meta_date.get("content", "")
        m = re.search(r'(\d{4}[-/\.]\d{1,2}[-/\.]\d{1,2})', d)
        if m:
            date = m.group(1).replace("/", "-").replace(".", "-")
    
    if not date:
        pub = soup.find("publishtime")
        if pub:
            m = re.search(r'(\d{4}[-/\.]\d{1,2}[-/\.]\d{1,2})', pub.get_text())
            if m:
                date = m.group(1).replace("/", "-").replace(".", "-")
    
    # Content from UCAPCONTENT
    content = ""
    ucap = soup.find("ucapcontent")
    if ucap:
        content = extract_ucap_content(ucap, url)
    else:
        # Fallback to zoomcon div
        zoom = soup.find("div", id="zoomcon")
        if zoom:
            content = extract_ucap_content(zoom, url)
    
    return title, content, date

def parse_list_page(url):
    r = safe_get(url)
    if not r:
        return []
    
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    
    for a in soup.find_all("a", href=True):
        href = a["href"].strip()
        # Filter: only bmgg article URLs with UUID pattern
        if not re.match(r'/shuyang/bmgg/\d{6}/[a-f0-9]{32}\.shtml', href):
            continue
        
        title = a.get_text(strip=True)
        if len(title) < 5:
            continue
        
        # Normalize URL
        if href.startswith("/"):
            href = urljoin(BASE_URL, href)
        elif not href.startswith("http"):
            href = urljoin(url, href)
        
        # Date from parent list item
        date = ""
        parent = a.find_parent(["li", "dd", "div"])
        if parent:
            m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', parent.get_text())
            if m:
                date = m.group(1)
        
        items.append((title, href, date))
    
    return items

def main():
    import sqlite3
    
    db_paths = ["/root/search.db", "/root/gov_crawler/search.db", os.path.join(SCRIPT_DIR, "search.db")]
    db_path = None
    for p in db_paths:
        if os.path.exists(p):
            db_path = p
            break
    if not db_path:
        db_path = "/root/search.db"
    
    print(f"=== 沭阳县人民政府-便民公告 ===")
    
    all_items = []
    for page in range(1, TOTAL_PAGES + 1):
        if page == 1:
            url = BASE_URL + LIST_PATH
        else:
            url = f"{BASE_URL}/shuyang/bmgg/olist_{page}.shtml"
        
        print(f"  Fetching page {page}/{TOTAL_PAGES}: {url}")
        items = parse_list_page(url)
        if not items:
            print(f"    Empty, stopping")
            break
        print(f"    Found {len(items)} items")
        all_items.extend(items)
        time.sleep(0.2)
    
    print(f"\n  Total: {len(all_items)} items\n")
    
    if not all_items:
        return
    
    # Save debug
    with open("/tmp/shuyang_items.txt", "w") as f:
        for t, h, d in all_items:
            f.write(f"[{d}] {t}\n  {h}\n\n")
    
    db = sqlite3.connect(db_path, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=OFF")
    
    site_name = "沭阳县-便民公告"
    inserted = 0
    skipped = 0
    errors = 0
    
    for idx, (title, href, list_date) in enumerate(all_items):
        existing = db.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,)).fetchone()
        if existing:
            skipped += 1
            continue
        
        detail_title, content, detail_date = parse_detail(href)
        if not content:
            # Check if at least has images
            try:
                r = safe_get(href)
                if r:
                    soup = BeautifulSoup(r.text, "html.parser")
                    ucap = soup.find("ucapcontent")
                    if ucap and ucap.find("img"):
                        content = f'<p><a href="{href}">Image-only article</a></p>'
            except:
                pass
        
        if not content:
            errors += 1
            print(f"  [WARN] Empty: {title[:50]}")
            continue
        
        final_title = detail_title or title
        final_date = detail_date or list_date
        content = content[:500000] if len(content) > 500000 else content
        
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (title, content, page_url, publish_date, site_name) VALUES (?, ?, ?, ?, ?)",
                (final_title, content, href, final_date, site_name)
            )
            db.commit()
            inserted += 1
        except Exception as e:
            print(f"  [DB ERROR] {e}")
            db.rollback()
        
        if (idx + 1) % 50 == 0:
            print(f"  Progress: {idx+1}/{len(all_items)} (inserted={inserted}, skipped={skipped})")
        
        time.sleep(0.2)
    
    # Bulk FTS
    print("\nRebuilding FTS...")

    
    db.close()
    
    print(f"\n========================================")
    print(f"Inserted: {inserted}/{len(all_items)} new")
    print(f"Skipped (already exist): {skipped}")
    print(f"Errors (empty content): {errors}")

if __name__ == "__main__":
    main()
