#!/usr/bin/env python3
"""Crawler for 科尔沁左翼后旗 - 生态环境"""
import requests
import re
import sqlite3
import sys
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'http://www.houqi.gov.cn'
LIST_PATH = '/zwgk/zfxxgk/fdzdgknr/zdlyxxgk/sthj_11832/'
DB_PATH = '/root/search.db'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}
GROUP = '内蒙古'

def extract_date_from_url(url):
    """Extract date from URL like /202606/t20260629_1056926.html"""
    m = re.search(r'/(\d{4})(\d{2})/t(\d{4})(\d{2})(\d{2})_', url)
    if m:
        return f'{m.group(1)}-{m.group(2)}-{m.group(5)}'
    return None

def get_detail(detail_url):
    """Get title and content from detail page"""
    full_url = urljoin(BASE_URL, detail_url)
    try:
        r = requests.get(full_url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        print(f"  FAIL fetch: {e}", file=sys.stderr)
        return None, None
    
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # Title from div.lis_list_part2_title
    title_div = soup.find('div', class_='lis_list_part2_title')
    title = title_div.get_text(strip=True) if title_div else ''
    
    # Content from div.lis_list_part2_content
    content_div = soup.find('div', class_='lis_list_part2_content')
    if not content_div:
        print(f"  No content div found", file=sys.stderr)
        return title, None
    
    # Strip inline formatting tags
    for tag in content_div.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.replace_with(tag.get_text())
    
    # Handle <table> - keep as HTML
    tables = []
    def save_table(m):
        idx = len(tables)
        tables.append(m.group(0))
        return f'__TABLE_{idx}__'
    
    content_html = str(content_div)
    content_html = re.sub(r'<table[^>]*>.*?</table>', save_table, content_html, flags=re.DOTALL)
    
    # Convert to text
    content_soup = BeautifulSoup(content_html, 'html.parser')
    content_text = content_soup.get_text(separator='\n', strip=True)
    
    # Restore tables
    for i, t in enumerate(tables):
        content_text = content_text.replace(f'__TABLE_{i}__', t)
    
    # Clean up lines
    lines = content_text.split('\n')
    lines = [l.strip() for l in lines if l.strip()]
    content_text = '\n\n'.join(lines)
    
    return title, content_text

def crawl():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    inserted = 0
    errors = 0
    
    # Crawl 10 pages (94 records, 10 per page)
    for page in range(0, 10):
        if page == 0:
            page_url = urljoin(BASE_URL, LIST_PATH)
        else:
            page_url = urljoin(BASE_URL, LIST_PATH + f'index_{page}.html')
        
        print(f"Fetching page {page+1}/10...", file=sys.stderr)
        try:
            r = requests.get(page_url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
        except Exception as e:
            print(f"  FAIL page {page}: {e}", file=sys.stderr)
            errors += 1
            continue
        
        soup = BeautifulSoup(r.text, 'html.parser')
        
        # Find all detail links
        seen_links = set()
        links = soup.find_all('a', href=True)
        for a in links:
            href = a['href']
            if '/sthj_11832/' not in href or 'html' not in href:
                continue
            if href in seen_links:
                continue
            seen_links.add(href)
            
            title_text = a.get_text(strip=True)
            if not title_text or len(title_text) < 5:
                continue
            
            # Use full URL or relative URL
            if href.startswith('http'):
                detail_url = href
            else:
                detail_url = href if href.startswith('/') else '/' + href
            detail_full = urljoin(BASE_URL, detail_url)
            
            # Check if exists
            exists = c.execute(
                'SELECT id FROM gov_raw WHERE page_url = ?',
                (detail_url,)
            ).fetchone()
            
            if exists:
                print(f"  SKIP (exists): {title_text[:50]}", file=sys.stderr)
                continue
            
            # Fetch detail
            print(f"  Fetch: {title_text[:50]}...", file=sys.stderr)
            title, content = get_detail(detail_full)
            
            if not title:
                title = title_text
            
            # Extract date from URL
            date_str = extract_date_from_url(detail_url)
            
            if not content:
                content = f'<p><a href="{detail_full}">{title}</a></p>'
            
            try:
                c.execute(
                    'INSERT OR IGNORE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url) VALUES (?, ?, ?, ?, ?, ?)',
                    (detail_url, title, content, date_str, GROUP, detail_full)
                )
                conn.commit()
                inserted += 1
                print(f"  OK: {title[:60]}", file=sys.stderr)
            except Exception as e:
                print(f"  FAIL insert {title[:40]}: {e}", file=sys.stderr)
                conn.rollback()
                errors += 1
    
    conn.close()
    print(f"\nDone. Inserted: {inserted}, Errors: {errors}", file=sys.stderr)
    return inserted, errors

if __name__ == '__main__':
    result = crawl()
    print(f"RESULT:{result[0]}:{result[1]}")
