#!/usr/bin/env python3
"""
crawl_cz_sthjj_tzgg.py — 崇左市生态环境局 - 通知公告
TRS CMS, static HTML pagination, BeautifulSoup parsing
Detail: meta ArticleTitle, TRS_UEDITOR content
"""
import re, sys, json, time, os, urllib.parse, warnings
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests
from bs4 import BeautifulSoup

BASE_URL = "http://sthjj.chongzuo.gov.cn/xxgk/zdgknr/tzgg"
SITE_NAME = "崇左市生态环境局-通知公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

from datetime import datetime, timedelta
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch_list(page_idx):
    """Fetch list page via BeautifulSoup"""
    if page_idx == 0:
        url = f"{BASE_URL}/index.shtml"
    else:
        url = f"{BASE_URL}/index_{page_idx}.shtml"
    
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [ERROR] fetch page {page_idx+1}: {e}")
        return []
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    items = []
    
    for li in soup.select('#morelist li'):
        a_tag = li.find('a', href=True)
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        if not href.endswith('.shtml'):
            continue
        title = a_tag.get('title', '') or a_tag.get_text(strip=True)
        if not title:
            continue
        full_url = urllib.parse.urljoin(BASE_URL + '/', href)
        span = li.find('span')
        date_str = span.get_text(strip=True) if span else ''
        items.append((title, full_url, date_str))
    
    return items

def fetch_detail(url):
    """Fetch detail page"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = 'utf-8'
    except Exception as e:
        return ("", "", "", "")
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    title = ""
    m = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if m and m.get('content'): title = m['content'].strip()
    
    pub_date = ""
    m = soup.find('meta', attrs={'name': 'PubDate'})
    if m and m.get('content'): pub_date = m['content'].strip()[:10]
    
    source = ""
    m = soup.find('meta', attrs={'name': 'ContentSource'})
    if m and m.get('content'): source = m['content'].strip()
    
    content = ""
    for cls in ['TRS_Editor', 'TRS_UEDITOR', 'trs_editor_view']:
        div = soup.find('div', class_=re.compile(cls))
        if div:
            content = str(div)
            break
    if not content:
        div = soup.find('div', class_=re.compile(r'zoom|content|article|maintext', re.I))
        if div:
            content = str(div)
    
    return (content, title, pub_date, source)

def push_to_searchdb(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    cur = conn.cursor()
    new_count = 0
    for title, url, pub_date, source, content in items:
        try:
            cur.execute("""
                INSERT OR IGNORE INTO gov_raw 
                    (site_name, title, page_url, publish_date, source_url, content, category)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (SITE_NAME, title, url, pub_date, source, content, '通知公告'))
            if cur.rowcount > 0:
                new_count += 1
        except Exception as e:
            print(f"  [DB ERROR] {url}: {e}")
    conn.commit()
    conn.close()
    return new_count

def main():
    warnings.filterwarnings("ignore")
    print(f"[{datetime.now().strftime('%H:%M:%S')}] {SITE_NAME}")
    
    all_items = []
    seen_urls = set()
    total_pages = 12
    
    for page_idx in range(total_pages):
        items = fetch_list(page_idx)
        if not items:
            if page_idx > 0:
                print(f"  Page {page_idx+1}: 0 items (stopping)")
                break
            continue
        
        page_items = []
        for title, url, date_str in items:
            if date_str and date_str >= CUTOFF:
                if url not in seen_urls:
                    seen_urls.add(url)
                    page_items.append((title, url, date_str))
        
        all_items.extend(page_items)
        print(f"  Page {page_idx+1}: {len(items)} items, {len(page_items)} within 3yr")
        
        if items and items[-1][2] < CUTOFF:
            print(f"  -> Reached cutoff, stopping")
            break
        
        time.sleep(0.5)
    
    print(f"  Total within 3 years: {len(all_items)}")
    
    if not all_items:
        print("  No items to crawl!")
        return 0
    
    details = []
    with ThreadPoolExecutor(max_workers=5) as executor:
        future_map = {executor.submit(fetch_detail, url): (title, url, date_str)
                      for title, url, date_str in all_items}
        for i, future in enumerate(as_completed(future_map), 1):
            title, url, date_str = future_map[future]
            try:
                content, det_title, pub_date, source = future.result()
                final_title = det_title or title
                final_date = pub_date or date_str
                if content:
                    details.append((final_title, url, final_date, source, content))
                    print(f"  [{i}/{len(all_items)}] {final_title[:50]}... OK ({len(content)}B)")
                else:
                    print(f"  [{i}/{len(all_items)}] {final_title[:50]}... no content")
            except Exception as e:
                print(f"  [{i}/{len(all_items)}] {title[:40]}... ERROR: {e}")
    
    if details:
        new_count = push_to_searchdb(details)
        print(f"\n  === Done === New: {new_count}, Errors: 0")
    else:
        print(f"\n  === Done === No items with content")
    
    return len(details)

if __name__ == '__main__':
    main()
