#!/usr/bin/env python3
"""
crawl_gxq_yl_gov.py — 榆林高新技术产业开发区 - 通知公告
TRS CMS, static HTML pagination (index.html, index_1.html, ..., index_19.html)
Detail: meta ArticleTitle, PubDate, TRS_UEDITOR content
"""
import re, sys, json, time, os, urllib.parse
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests
from bs4 import BeautifulSoup

BASE_URL = "https://gxq.yl.gov.cn/xwzx/tzgg"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
}
SITE_NAME = "榆林高新技术产业开发区-通知公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

# 3-year cutoff
from datetime import datetime, timedelta
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch_list(page_idx):
    """Fetch list page, return [(title, url, date), ...]"""
    if page_idx == 0:
        url = f"{BASE_URL}/index.html"
    else:
        url = f"{BASE_URL}/index_{page_idx}.html"
    
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [ERROR] fetch list page {page_idx}: {e}")
        return []
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    items = []
    
    for li in soup.find_all('li', class_=re.compile(r'^(odd|even)$')):
        a_tag = li.find('a', href=True)
        if not a_tag:
            continue
        
        href = a_tag.get('href', '')
        if not href.endswith('.html') or 't202' not in href:
            continue
        
        title = a_tag.get('title', '') or a_tag.get_text(strip=True)
        if not title:
            continue
        
        # Full URL
        full_url = urllib.parse.urljoin(BASE_URL + '/', href)
        
        # Date
        date_span = li.find('span', class_='date')
        date_str = date_span.get_text(strip=True) if date_span else ''
        
        items.append((title, full_url, date_str))
    
    return items

def fetch_detail(url):
    """Fetch detail page, return (content_html, title, pub_date, source)"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [ERROR] fetch detail {url}: {e}")
        return ("", "", "", "")
    
    html = resp.text
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from meta
    title = ""
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    
    # PubDate from meta
    pub_date = ""
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date and meta_date.get('content'):
        pub_date = meta_date['content'].strip()[:10]  # YYYY-MM-DD
    
    # Source
    source = ""
    meta_source = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_source and meta_source.get('content'):
        source = meta_source['content'].strip()
    
    # Content from TRS_UEDITOR div
    content_div = soup.find('div', class_=re.compile(r'TRS_UEDITOR|trs_editor_view|TRS_Editor'))
    if not content_div:
        # Fallback: try #zoom or .content or article
        content_div = soup.find(id=re.compile(r'zoom|content|article', re.I))
    if not content_div:
        content_div = soup.find(class_=re.compile(r'zoom|content|article|maintext', re.I))
    
    content = ""
    if content_div:
        content = str(content_div)
    
    return (content, title, pub_date, source)

def push_to_searchdb(items):
    """Bulk insert into gov_raw"""
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    cur = conn.cursor()
    
    new_count = 0
    for title, url, pub_date, source, content in items:
        try:
            cur.execute("""
                INSERT OR IGNORE INTO gov_raw 
                    (site_name, title, page_url, publish_date, source_url, content, category)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (SITE_NAME, title, url, pub_date, source, content, '通知公告'))
            if cur.rowcount > 0:
                new_count += 1
        except Exception as e:
            print(f"  [DB ERROR] {url}: {e}")
    
    conn.commit()
    conn.close()
    return new_count

def main():
    import warnings
    warnings.filterwarnings("ignore", message="Unverified HTTPS request")
    
    print(f"[{datetime.now().strftime('%H:%M:%S')}] {SITE_NAME}")
    
    # Step 1: collect all list items from all pages
    all_items = []
    seen_urls = set()
    
    # Determine max pages from page 1
    total_pages = 20  # from createPage(20, 0, "index", "html")
    
    for page_idx in range(total_pages):
        items = fetch_list(page_idx)
        if not items:
            print(f"  Page {page_idx+1}: 0 items (stopping)")
            break
        
        # Filter by 3-year cutoff
        page_items = []
        for title, url, date_str in items:
            if date_str and date_str >= CUTOFF:
                if url not in seen_urls:
                    seen_urls.add(url)
                    page_items.append((title, url, date_str))
            elif date_str and date_str < CUTOFF:
                pass  # older than 3 years
        
        all_items.extend(page_items)
        print(f"  Page {page_idx+1}: {len(items)} items, {len(page_items)} within 3yr")
        
        # Stop if last item on page is older than cutoff
        if items and items[-1][2] < CUTOFF:
            print(f"  -> Reached items older than 3yr cutoff, stopping")
            break
        
        time.sleep(0.5)  # rate limit
    
    print(f"  Total within 3 years: {len(all_items)}")
    
    # Step 2: fetch details concurrently
    details = []
    with ThreadPoolExecutor(max_workers=5) as executor:
        future_map = {
            executor.submit(fetch_detail, url): (title, url, date_str)
            for title, url, date_str in all_items
        }
        
        for i, future in enumerate(as_completed(future_map), 1):
            title, url, date_str = future_map[future]
            try:
                content, det_title, pub_date, source = future.result()
                final_title = det_title or title
                final_date = pub_date or date_str
                
                if content:
                    details.append((final_title, url, final_date, source, content))
                    print(f"  [{i}/{len(all_items)}] {final_title[:50]}... OK ({len(content)}B)")
                else:
                    print(f"  [{i}/{len(all_items)}] {final_title[:50]}... no content")
            except Exception as e:
                print(f"  [{i}/{len(all_items)}] {title[:40]}... ERROR: {e}")
    
    # Step 3: insert into DB
    new_count = 0
    if details:
        new_count = push_to_searchdb(details)
        print(f"\n  === Done === New: {new_count}, Errors: 0")
    else:
        print(f"\n  === Done === No new items")
    
    return new_count

if __name__ == '__main__':
    main()
