"""Crawl ddh.hld.gov.cn - 东戴河新区通知公告"""
import requests
import sqlite3
import re
import warnings
from bs4 import BeautifulSoup
warnings.filterwarnings('ignore')

BASE = "https://ddh.hld.gov.cn"
DB_PATH = "/root/search.db"
SITE_NAME = "东戴河新区-通知公告"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def extract_clean_text(soup):
    """Extract content from TRS_Editor div.
    Strip inline <span>/<o:p>/<font> tags so dates aren't split.
    Tables are extracted separately as HTML to avoid duplication."""
    cont = soup.select_one('div.content_con')
    if not cont:
        return ""
    
    editor = cont.select_one('div.TRS_Editor')
    if not editor:
        editor = cont
    
    html = re.sub(r'</?(?:span|o:p|font|b|strong|em|i|u|s|strike|sub|sup)[^>]*>', '', str(editor))
    
    # Extract tables before stripping them from text
    tables = re.findall(r'<table[\s\S]*?</table>', html)
    html_without_tables = re.sub(r'<table[\s\S]*?</table>', '', html)
    
    clean_soup = BeautifulSoup(html_without_tables, 'html.parser')
    text = clean_soup.get_text('\n\n', strip=True)
    
    if not text and not tables:
        return ""
    
    result = text
    if tables:
        result = result + '\n\n' + '\n\n'.join(tables) if result else '\n\n'.join(tables)
    
    return result


def extract_attachments(soup):
    """Extract attachment links"""
    appendix = soup.select_one('div.appendix')
    if not appendix:
        return ""
    
    links = []
    for a in appendix.find_all('a', href=True):
        href = a['href']
        text = a.get_text(strip=True)
        if href and text:
            if href.startswith('.'):
                href = BASE + '/xwzx/tzgg/' + href[2:] if href.startswith('./') else BASE + href
            links.append(f"[{text}]({href})")
    
    return '\n'.join(links)


def crawl():
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA busy_timeout=30000")
    
    total_new = 0
    total_pages = 20
    
    for page in range(1, total_pages + 1):
        if page == 1:
            list_url = f"{BASE}/xwzx/tzgg/"
        else:
            list_url = f"{BASE}/xwzx/tzgg/index_{page}.html"
        
        try:
            resp = requests.get(list_url, headers=HEADERS, timeout=30, verify=False)
            resp.encoding = 'utf-8'
        except Exception as e:
            print(f"Page {page} ERROR: {e}")
            continue
        
        soup = BeautifulSoup(resp.text, 'html.parser')
        
        # Find article items (skip header/nav li items)
        items = []
        for li in soup.find_all('li'):
            a = li.find('a', href=True)
            date_span = li.find('span', class_='date')
            if a and date_span and a.get('href', '').startswith('./20'):
                title = a.get_text(strip=True)
                href = a['href']
                date = date_span.get_text(strip=True)
                if title and len(title) > 5:
                    items.append((href, title, date))
        
        print(f"Page {page}: {len(items)} items")
        if not items:
            continue
        
        for href, title, list_date in items:
            if href.startswith('./'):
                page_url = f"{BASE}/xwzx/tzgg/{href[2:]}"
            else:
                page_url = href if href.startswith('http') else f"{BASE}/{href.lstrip('/')}"
            page_url = page_url.split('?')[0]
            
            # Check exists
            exists = db.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (page_url,)).fetchone()
            if exists:
                print(f"  SKIP: {title[:30]}")
                continue
            
            # Fetch detail
            try:
                dresp = requests.get(page_url, headers=HEADERS, timeout=30, verify=False)
                dresp.encoding = 'utf-8'
                dsoup = BeautifulSoup(dresp.text, 'html.parser')
            except Exception as e:
                print(f"  FETCH ERR: {title[:30]} - {e}")
                continue
            
            # Title
            h1 = dsoup.select_one('div.content h1')
            page_title = h1.get_text(strip=True) if h1 else title
            
            # Date
            date = list_date
            time_span = dsoup.select_one('span.time')
            if time_span:
                m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', time_span.get_text())
                if m:
                    date = m.group(1).replace('/', '-')
            
            # Content
            content = extract_clean_text(dsoup)
            
            # Attachments
            attachments = extract_attachments(dsoup)
            if attachments:
                content = content + '\n\n' + attachments if content else attachments
            
            if not content:
                content = "(无内容)"
            
            summary = content[:500]
            
            try:
                db.execute(
                    "INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, page_url, page_title, content, date, summary)
                )
                db.commit()
                total_new += 1
                print(f"  [{page}] {page_title[:35]:.35s} | {date}")
            except sqlite3.IntegrityError:
                print(f"  DUP: {page_title[:30]}")
                db.rollback()
            except Exception as e:
                db.rollback()
                print(f"  DB ERR: {page_title[:30]} - {e}")
    
    db.close()
    print(f"\n✅ Done. Total new: {total_new}")

if __name__ == "__main__":
    crawl()
