#!/usr/bin/env python3
"""
Crawler for 荣县人民政府 - 其他法定信息
https://www.rongzhou.gov.cn/rxrmzf/rxzhxzzfjqtfdx/pc/list.html
API: POST /queryList (数融平台)
"""

import json
import urllib.request
import urllib.parse
import sqlite3
import time
import re
import os
import sys

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = 'https://www.rongzhou.gov.cn'
API_URL = BASE_URL + '/queryList'

SITE_NAME = '荣县人民政府-其他法定信息'
CHANNEL_CODE = 'rxzhxzzfjqtfdx'
WEBSITE_CODE = 'rxrmzf'
PAGE_SIZE = 20

# Support CLI args: python3 crawl_rongzhou.py [channel_code] ["site_name"]
if len(sys.argv) > 1 and sys.argv[1].startswith('rx'):
    CHANNEL_CODE = sys.argv[1]
if len(sys.argv) > 2:
    SITE_NAME = sys.argv[2]

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Content-Type': 'application/x-www-form-urlencoded',
    'Referer': 'https://www.rongzhou.gov.cn/rxrmzf/rxzhxzzfjqtfdx/pc/list.html',
}


def fetch_page(current):
    """Fetch a single page of results from the API."""
    data = {
        'current': current,
        'pageSize': PAGE_SIZE,
        'webSiteCode[]': WEBSITE_CODE,
        'channelCode[]': CHANNEL_CODE,
        'sort': 'pubDate',
        'order': 'desc',
    }
    req = urllib.request.Request(API_URL, data=urllib.parse.urlencode(data).encode(), headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        return json.loads(resp.read())
    except Exception as e:
        print(f'[error] API request failed: {e}')
        return None


def save_article(conn, title, content, pub_date, url, source_url):
    """Save or update an article in the database."""
    if not title or not pub_date:
        return False
    # Clean title
    title = title.strip()
    if not title:
        return False
    # Clean content - already HTML
    if not content or content.strip() == '':
        content = ''
    else:
        content = content.strip()
    # Clean date - keep as YYYY-MM-DD HH:MM format
    pub_date = pub_date.strip()[:19] if pub_date else ''
    conn.execute(
        '''INSERT OR REPLACE INTO gov_raw (title, content, publish_date, page_url, source_url, site_name, script_name) VALUES (?, ?, ?, ?, ?, ?, \'crawl_rongzhou.py\')''',
        (title, content, pub_date, url, source_url, SITE_NAME)
    )
    return True


def main():
    start_time = time.time()
    total_saved = 0
    total_articles = 0
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    
    # Fetch first page to get total count
    print('[rongzhou] Fetching page 1...')
    raw = fetch_page(1)
    if not raw:
        print('[rongzhou] Failed to fetch first page')
        conn.close()
        sys.exit(1)
    
    total = raw.get('data', {}).get('total', 0)
    if total == 0:
        print('[rongzhou] No results found')
        conn.close()
        sys.exit(0)
    
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print(f'[rongzhou] Total: {total} articles, {total_pages} pages')
    
    for page in range(1, total_pages + 1):
        if page == 1:
            raw_page = raw
        else:
            print(f'[rongzhou] Fetching page {page}/{total_pages}...')
            raw_page = fetch_page(page)
            time.sleep(0.3)
        
        if not raw_page:
            print(f'[rongzhou] Failed to fetch page {page}')
            continue
        
        results = raw_page.get('data', {}).get('results', [])
        for r in results:
            src = r.get('source', {})
            title = src.get('title', '')
            pub_date = src.get('pubDate', '')
            raw_url = src.get('urls', {})
            content_obj = src.get('content', {})
            
            # Extract URL
            if isinstance(raw_url, dict):
                article_url = raw_url.get('pc', '')
            elif isinstance(raw_url, str):
                try:
                    raw_url_parsed = json.loads(raw_url)
                    article_url = raw_url_parsed.get('pc', '')
                except:
                    article_url = raw_url
            else:
                article_url = str(raw_url) if raw_url else ''
            
            if article_url and not article_url.startswith('http'):
                article_url = BASE_URL + article_url
            
            # Extract content
            if isinstance(content_obj, dict):
                content = content_obj.get('content', '')
            elif isinstance(content_obj, str):
                content = content_obj
            else:
                content = ''
            
            total_articles += 1
            if save_article(conn, title, content, pub_date, article_url, article_url):
                total_saved += 1
        
        conn.commit()
        print(f'[rongzhou] Progress: {total_saved}/{total} saved')
    
    conn.close()
    elapsed = time.time() - start_time
    print(f'[rongzhou] Done: {total_saved}/{total_articles} saved in {elapsed:.1f}s')
    print(f'[rongzhou] Site: {SITE_NAME}')


if __name__ == '__main__':
    main()
