"""Crawl anyi.nc.gov.cn - 安义县生态环境局环评公示"""
import requests
import sqlite3
import re
import warnings
from bs4 import BeautifulSoup
warnings.filterwarnings('ignore')

BASE = "https://anyi.nc.gov.cn"
DB_PATH = "/root/search.db"
SITE_NAME = "安义县-环评公示"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://anyi.nc.gov.cn/ayxzf/xsthjjgggs2021/just_list.shtml",
}

def extract_content(html):
    """Extract content from UCAPCONTENT or article-content div"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Try UCAPCONTENT first
    ucap = soup.find('UCAPCONTENT')
    if ucap:
        return clean_ucap_text(ucap)
    
    # Fallback to article-content div
    cont = soup.select_one('div.article-content-body, div.article-content')
    if cont:
        ucap2 = cont.find('UCAPCONTENT')
        if ucap2:
            return clean_ucap_text(ucap2)
        return clean_ucap_text(cont)
    
    # Last resort: any content
    return soup.get_text(separator='\n\n', strip=True)


def clean_ucap_text(element):
    """Clean content preserving <p> text and <table> HTML.
    Strip inline span/font tags to prevent date character splitting."""
    html = str(element)
    html = re.sub(r'</?(?:span|o:p|font|b|strong|em|i|u|s|strike|sub|sup)[^>]*>', '', html)
    clean = BeautifulSoup(html, 'html.parser')
    
    parts = []
    for p in clean.find_all('p'):
        txt = p.get_text(strip=True)
        if txt:
            parts.append(txt)
    for t in clean.find_all('table'):
        parts.append(str(t))
    
    if not parts:
        txt = clean.get_text('\n\n', strip=True)
        return txt if txt else ""
    
    return '\n\n'.join(parts)


def crawl():
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA busy_timeout=30000")
    
    total_new = 0
    total_pages = 50  # createPageHTML('page_div',50,...)
    
    for page in range(1, total_pages + 1):
        if page == 1:
            list_url = f"{BASE}/ayxzf/xsthjjgggs2021/just_list.shtml"
        else:
            list_url = f"{BASE}/ayxzf/xsthjjgggs2021/just_list_{page}.shtml"
        
        try:
            resp = requests.get(list_url, headers=HEADERS, timeout=30, verify=False)
            resp.encoding = 'utf-8'
        except Exception as e:
            print(f"Page {page} ERROR: {e}")
            continue
        
        soup = BeautifulSoup(resp.text, 'html.parser')
        
        # Find list items - look for li with a href
        items = []
        for li in soup.find_all('li'):
            a = li.find('a', href=True)
            if not a:
                continue
            href = a['href']
            title = a.get_text(strip=True)
            if not href or not title or len(title) < 5:
                continue
            items.append((href, title))
        
        if not items:
            print(f"Page {page}: no items found")
            continue
        
        print(f"Page {page}: {len(items)} items")
        
        for href, title in items:
            if href.startswith('/'):
                page_url = f"{BASE}{href}"
            elif href.startswith('http'):
                page_url = href
            else:
                page_url = f"{BASE}/ayxzf/xsthjjgggs2021/{href}"
            page_url = page_url.split('?')[0]
            
            # Check exists
            exists = db.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (page_url,)).fetchone()
            if exists:
                print(f"  SKIP: {title[:30]}")
                continue
            
            # Fetch detail
            try:
                dresp = requests.get(page_url, headers=HEADERS, timeout=30, verify=False)
                dresp.encoding = 'utf-8'
            except Exception as e:
                print(f"  FETCH ERR: {title[:30]} - {e}")
                continue
            
            if 'error_403' in dresp.text or 'yunaq' in dresp.text.lower():
                print(f"  WAF BLOCKED: {title[:30]}")
                continue
            
            dsoup = BeautifulSoup(dresp.text, 'html.parser')
            
            # Title from UCAPTITLE
            ucap_title = dsoup.find('UCAPTITLE')
            if ucap_title:
                page_title = ucap_title.get_text(strip=True)
            else:
                h1 = dsoup.select_one('h1.article-title')
                page_title = h1.get_text(strip=True) if h1 else title
            
            # Date - try meta first
            date = ""
            for meta in dsoup.find_all('meta'):
                if meta.get('name') == 'PubDate' and meta.get('content'):
                    d = meta['content'].strip()[:10]
                    if re.match(r'\d{4}[-/]\d{1,2}[-/]\d{1,2}', d):
                        date = d.replace('/', '-')
                        break
            
            if not date:
                # Try from URL path (/202605/...)
                m = re.search(r'/20(\d{2})(\d{2})/', page_url)
                if m:
                    date = f"20{m.group(1)}-{m.group(2)}"
            
            # Content
            content = extract_content(dresp.text)
            if not content:
                content = "(无内容)"
            
            summary = content[:500]
            
            try:
                db.execute(
                    "INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, page_url, page_title, content, date, summary)
                )
                db.commit()
                total_new += 1
                print(f"  [{page}] {page_title[:35]:.35s} | {date}")
            except sqlite3.IntegrityError:
                print(f"  DUP: {page_title[:30]}")
                db.rollback()
            except Exception as e:
                db.rollback()
                print(f"  DB ERR: {page_title[:30]} - {e}")
    
    db.close()
    print(f"\n✅ Done. Total new: {total_new}")

if __name__ == "__main__":
    crawl()
