#!/usr/bin/env python3
import os
"""
Crawler for 启东市 - 部门公告公示
http://www.qidong.gov.cn/qdsrmzf/bmgggs/bmgggs.html
TrueCMS + jpage API
"""

import re, sys, time, json, sqlite3
from urllib.request import urlopen, Request
from urllib.parse import urlencode

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE = 'http://www.qidong.gov.cn'
SITE_NAME = '启东市人民政府-部门公告公示'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Referer': 'http://www.qidong.gov.cn/qdsrmzf/bmgggs/bmgggs.html',
}
COLUMN_ID = '0106496c-0949-485f-ba9e-e5c67592bc9a'
API_URL = BASE + '/truecms/messageController/getMessage.do'
GROUP_SIZE = 30  # jpage group size (perPage * groupSize = 10 * 3)


def fetch_list(startrecord, endrecord):
    """Fetch a group of articles from the API."""
    params = {
        'startrecord': startrecord,
        'endrecord': endrecord,
        'perpage': 10,
        'columnId': COLUMN_ID,
    }
    url = f'{API_URL}?{urlencode(params)}&callback=cb'
    req = Request(url, headers=HEADERS)
    resp = urlopen(req, timeout=30)
    body = resp.read().decode('utf-8', errors='replace')
    
    # Parse JSONP
    json_str = body[body.index('{'):body.rindex('}')+1]
    data = json.loads(json_str)
    xml = data.get('result', '')
    
    # Extract records
    records = re.findall(
        r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>\s*<span>(\d{4}-\d{2}-\d{2})</span>',
        xml
    )
    return records


def fetch_content(url):
    """Fetch detail page content."""
    req = Request(url, headers=HEADERS)
    resp = urlopen(req, timeout=30)
    html = resp.read().decode('utf-8', errors='replace')
    
    # Content is in div.content
    c = re.search(r'<div[^>]*class=[\"\']content[\"\']?[^>]*>(.*)', html, re.DOTALL)
    if c:
        rest = c.group(1)
        # Depth counting
        depth = 1
        pos = 0
        for i, ch in enumerate(rest):
            if ch == '<':
                tag_end = rest.find('>', i)
                if tag_end == -1:
                    break
                tag = rest[i+1:tag_end].strip()
                tag_name = tag.split()[0]
                if tag.startswith('/'):
                    depth -= 1
                    if depth == 0:
                        pos = tag_end + 1
                        break
                elif not tag.startswith('!--') and tag_name not in ['br', 'hr', 'img', 'input', 'meta', 'link', '!DOCTYPE'] and not tag.endswith('/'):
                    depth += 1
                i = tag_end
        content = rest[:pos-1].strip() if pos > 0 else rest.strip()
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
        return content.strip()
    return ''


def main():
    start = time.time()
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA busy_timeout=5000")
    
    # First, get total count
    records = fetch_list(1, GROUP_SIZE)
    print(f'[qidong] Total: 3300 records (110 groups of {GROUP_SIZE})')
    
    total_saved = 0
    total_articles = 0
    total_groups = 110  # 3300 / 30
    
    for group in range(total_groups):
        startrec = group * GROUP_SIZE + 1
        endrec = min(startrec + GROUP_SIZE - 1, 3300)
        
        print(f'[qidong] Group {group+1}/{total_groups} (records {startrec}-{endrec})...')
        records = fetch_list(startrec, endrec)
        
        for href, title_raw, date in records:
            title = re.sub(r'<[^>]+>', '', title_raw).strip()
            article_url = BASE + href if href.startswith('/') else href
            
            total_articles += 1
            
            # Fetch detail content
            content = ''
            try:
                content = fetch_content(article_url)
                time.sleep(0.1)
            except Exception as e:
                print(f'  [warn] Detail failed: {e}')
            
            conn.execute(
                'INSERT OR REPLACE INTO gov_raw (title, content, publish_date, page_url, source_url, site_name) VALUES (?, ?, ?, ?, ?, ?)',
                (title, content, date, article_url, article_url, SITE_NAME)
            )
            total_saved += 1
        
        conn.commit()
        print(f'[qidong] Progress: {total_saved}/3300 saved')
        time.sleep(0.3)
    
    elapsed = time.time() - start
    print(f'[qidong] Done: {total_saved}/{total_articles} saved in {elapsed:.1f}s')
    conn.close()


if __name__ == '__main__':
    main()
