#!/usr/bin/env python3
"""Crawl 宜春市生态环境局 - 环保公告 via API
http://sthjj.yichun.gov.cn/ycssthjj/hbgg/pc/list.html
"""
import requests, re, sqlite3, sys, time, hashlib, json
from datetime import datetime, timedelta
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '宜春市生态环境局 - 环保公告'
API_URL = 'https://sthjj.yichun.gov.cn/queryList'
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
PAGE_SIZE = 15

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Content-Type': 'application/json',
}
session = requests.Session()
session.headers.update(HEADERS)

def fetch_page(page, page_size=PAGE_SIZE):
    """Fetch one page from the API"""
    payload = {
        "current": page,
        "pageSize": page_size,
        "channelCode": "hbgg"
    }
    resp = session.post(API_URL, json=payload, timeout=30)
    data = resp.json()
    if 'data' not in data:
        return [], 0
    total = data['data'].get('total', 0)
    results = data['data'].get('results', [])
    items = []
    for r in results:
        s = r['source']
        title = s.get('title', '', timeout=30).strip()
        pub_date = s.get('pubDate', '', timeout=30)[:10] if s.get('pubDate', timeout=30) else ''
        # Get content
        content_obj = s.get('content', {}, timeout=30)
        if isinstance(content_obj, dict):
            content = content_obj.get('content', '')
        elif isinstance(content_obj, str):
            content = content_obj
        else:
            content = ''
        # Get URL
        urls = s.get('urls', {}, timeout=30)
        if isinstance(urls, str):
            try:
                urls = json.loads(urls)
            except:
                urls = {}
        url = ''
        if isinstance(urls, dict):
            url = urls.get('pc', '')
        if url and not url.startswith('http'):
            url = 'http://sthjj.yichun.gov.cn' + url
        
        if title and url:
            items.append({
                'title': title,
                'url': url,
                'date': pub_date,
                'content': content
            })
    return items, total

def main():
    is_incremental = 'incremental' in sys.argv
    print(f'=== {SITE_NAME} ===')
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted, skipped, old_total = 0, 0, 0
    start_time = time.time()
    
    # First, get total count
    _, total = fetch_page(1, 1)
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE if total else 10
    max_pages = 1 if is_incremental else total_pages
    print(f'  Total articles: {total}, pages: {total_pages}, crawling: {max_pages}')
    
    consecutive_old = 0
    min_pages = 5
    
    for pg in range(1, max_pages + 1):
        try:
            items, _ = fetch_page(pg)
        except Exception as e:
            print(f'  [ERROR] Page {pg}: {e}')
            time.sleep(2)
            continue
        
        if not items:
            print(f'  Page {pg}: 0 items (end)')
            break
        
        all_old = all(item['date'] and item['date'] < CUTOFF_DATE for item in items if item['date'])
        if all_old:
            consecutive_old += 1
            if consecutive_old >= 2 and pg >= min_pages:
                print(f'  Page {pg}: all old, stopping')
                break
        else:
            consecutive_old = 0
        
        for item in items:
            if item['date'] and item['date'] < CUTOFF_DATE:
                old_total += 1
                continue
            
            has = c.execute("SELECT 1 FROM gov_raw WHERE page_url=? AND content IS NOT NULL AND content!=''", (item['url'],)).fetchone()
            if has:
                skipped += 1
                continue
            
            content = item['content']
            if content:
                content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
                content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
            
            final_title = item['title']
            final_date = item['date']
            int_id = int(hashlib.md5(item['url'].encode()).hexdigest()[:15], 16) % (2**63)
            date_rank = int(final_date.replace('-', '')) if final_date else 0
            summary = re.sub(r'<[^>]+>', '', content)[:200] if content else final_title
            summary = re.sub(r'\s+', ' ', summary).strip()
            
            try:
                c.execute(
                    'INSERT OR IGNORE INTO gov_raw(id, site_name, source_url, page_url, title, publish_date, content, summary, date_rank) VALUES(?,?,?,?,?,?,?,?,?)',
                    (int_id, SITE_NAME, item['url'], item['url'], final_title, final_date, content, summary, date_rank)
                )
                if c.rowcount > 0:
                    inserted += 1
            except Exception as e:
                print(f'  [DB] {e}')
                skipped += 1
        
        elapsed = time.time() - start_time
        print(f'  Page {pg}/{max_pages}: +{inserted} (skipped {skipped}, >3y {old_total}) [{elapsed:.0f}s]')
    
    conn.commit()
    # FTS rebuild
    try:
        c2 = conn.cursor()
        c2.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
        rows = c2.execute("SELECT id, title, site_name, summary FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
        for r in rows:
            c2.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES(?,?,?,?)", r)
        conn.commit()
        print(f'  FTS: {len(rows)} records rebuilt')
    except Exception as e:
        print(f'  FTS error: {e}')
    conn.close()
    elapsed = time.time() - start_time
    print(f'\nDone ({elapsed:.0f}s). Inserted {inserted}, Skipped {skipped}, Old {old_total}')

if __name__ == '__main__':
    main()
