#!/usr/bin/env python3
"""Crawler for 衡东县人民政府 - 公示栏"""
import requests
import re
import sqlite3
import sys
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'http://www.hengdong.gov.cn'
DB_PATH = '/root/search.db'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}
GROUP = '湖南'

def get_detail(detail_url):
    """Get title and content from detail page"""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        print(f"  FAIL fetch: {e}", file=sys.stderr)
        return None, None
    
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # Title from div.finalContent (extract before "来源：")
    fc = soup.find('div', class_='finalContent')
    title = ''
    if fc:
        text = fc.get_text(strip=True)
        # Title is before "来源：" or first line
        m = re.match(r'(.+?)来源：', text)
        if m:
            title = m.group(1).strip()
    
    # Content from div.article
    article = soup.find('div', class_='article')
    if not article:
        print(f"  No article div found", file=sys.stderr)
        return title, None
    
    # Strip inline formatting tags
    for tag in article.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.replace_with(tag.get_text())
    
    # Handle tables - keep as HTML
    tables = []
    def save_table(m):
        idx = len(tables)
        tables.append(m.group(0))
        return f'__TABLE_{idx}__'
    
    content_html = str(article)
    content_html = re.sub(r'<table[^>]*>.*?</table>', save_table, content_html, flags=re.DOTALL)
    
    # Convert to text
    content_soup = BeautifulSoup(content_html, 'html.parser')
    content_text = content_soup.get_text(separator='\n', strip=True)
    
    # Restore tables
    for i, t in enumerate(tables):
        content_text = content_text.replace(f'__TABLE_{i}__', t)
    
    # Clean lines
    lines = content_text.split('\n')
    lines = [l.strip() for l in lines if l.strip()]
    content_text = '\n\n'.join(lines)
    
    return title, content_text

def crawl():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    inserted = 0
    errors = 0
    total_pages = 3  # known: 3 pages (15+15+10 = 40 items)
    items_per_page = 15
    
    for page in range(1, total_pages + 1):
        if page == 1:
            page_url = urljoin(BASE_URL, '/ztzx/yqyd/gsl/index.html')
        else:
            page_url = urljoin(BASE_URL, f'/ztzx/yqyd/gsl/pages/{page}.html')
        
        print(f"Fetching page {page}/{total_pages}...", file=sys.stderr)
        try:
            r = requests.get(page_url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
        except Exception as e:
            print(f"  FAIL page {page}: {e}", file=sys.stderr)
            errors += 1
            continue
        
        soup = BeautifulSoup(r.text, 'html.parser')
        items = soup.find_all('li', class_='nyLine')
        
        if not items:
            print(f"  No items found on page {page}", file=sys.stderr)
            continue
        
        for item in items:
            # Date from span.titleDate
            date_span = item.find('span', class_='titleDate')
            date_str = date_span.get_text(strip=True) if date_span else None
            
            # Title and link
            a_tag = item.find('a', href=True)
            if not a_tag:
                continue
            
            href = a_tag['href']
            if not href.startswith('http'):
                href = urljoin(BASE_URL, href)
            
            list_title = a_tag.get_text(strip=True)
            
            # Check if exists
            exists = c.execute(
                'SELECT id FROM gov_raw WHERE page_url = ?',
                (href,)
            ).fetchone()
            
            if exists:
                print(f"  SKIP (exists): {list_title[:40]}", file=sys.stderr)
                continue
            
            # Fetch detail
            print(f"  Fetch: {list_title[:50]}...", file=sys.stderr)
            title, content = get_detail(href)
            
            if not title:
                title = list_title
            
            if not content:
                content = f'<p><a href="{href}">{title}</a></p>'
            
            try:
                c.execute(
                    'INSERT OR IGNORE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url) VALUES (?, ?, ?, ?, ?, ?)',
                    (href, title, content, date_str, GROUP, href)
                )
                conn.commit()
                inserted += 1
                print(f"  OK: {title[:60]}", file=sys.stderr)
            except Exception as e:
                print(f"  FAIL insert {title[:40]}: {e}", file=sys.stderr)
                conn.rollback()
                errors += 1
    
    conn.close()
    print(f"\nDone. Inserted: {inserted}, Errors: {errors}", file=sys.stderr)
    return inserted, errors

if __name__ == '__main__':
    result = crawl()
    print(f"RESULT:{result[0]}:{result[1]}")
