#!/usr/bin/env python3
"""
库伦旗人民政府 - 生态环境（重点领域信息）
API: /nmsearch/openSearch/getZwgkList?siteId=14
生态环境 channel ID: 11463
"""
import sys
import re
import requests
import sqlite3
from datetime import datetime, timedelta
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = 'http://www.kulun.gov.cn/nmsearch/openSearch/getZwgkList'
SITE_NAME = '库伦旗-生态环境'
CHANNEL_ID = 11463  # 生态环境

def get_conn():
    conn = sqlite3.connect(DB_PATH)
    conn.execute('PRAGMA journal_mode=WAL')
    return conn

def extract_date(text):
    m = re.search(r'(\d{4}-\d{2}-\d{2})', str(text))
    return m.group(1) if m else None

def is_recent(date_str, years=3):
    if not date_str: return False
    try:
        dt = datetime.strptime(date_str, '%Y-%m-%d')
        return dt >= (datetime.now() - timedelta(days=365*years))
    except: return False

def format_content(text):
    """Convert plain text to HTML with paragraph tags"""
    if not text or '<p>' in text:
        return text
    # Split on Chinese numbered sections: 一、二、三、... （1）（2）... 1. 2. etc.
    parts = re.split(r'(?=[一二三四五六七八九十]+[、.．])', text)
    if len(parts) <= 1:
        # Try splitting on consecutive Chinese chars followed by new patterns
        parts = re.split(r'(?<=。)(?=[一二三四五六七八九十]+)', text)
    if len(parts) <= 1:
        # Fallback: wrap the whole thing in a single <p>
        return '<p>' + text.strip() + '</p>'
    return ''.join(f'<p>{p.strip()}</p>' for p in parts if p.strip())

def fetch_page(page_num, page_size=10):
    params = {'pageNum': page_num, 'pageSize': page_size, 'officeNo': '',
              'cdesc': '', 'sort': 'time', 'suitability': 1, 'position': 1,
              'keywords': '', 'siteId': 14}
    try:
        resp = requests.get(BASE_URL, params=params, timeout=15)
        if resp.status_code != 200: return None
        data = resp.json()
        return data['data'] if data.get('state') == 200 else None
    except Exception as e:
        print(f'  Error: {e}'); return None

def crawl(incremental=False):
    conn = get_conn()
    cursor = conn.cursor()
    max_pages = 1 if incremental else 5
    total_new = total_skip = total_out = total_old = 0

    for page in range(1, max_pages + 1):
        print(f'Page {page}/{max_pages}...')
        result = fetch_page(page)
        if not result: print('  No data, stopping'); break
        items = result.get('data', [])
        print(f'  Got {len(items)} items')
        if not items: break

        for item in items:
            if CHANNEL_ID not in item.get('docchannel', []):
                total_out += 1; continue
            title = (item.get('gk_doctitle') or item.get('title', '')).strip()
            source_url = item.get('docpuburl', '').strip()
            content = format_content(item.get('gk_doccontent') or '')
            date_str = extract_date(item.get('docpubtime', ''))
            if not title or not source_url:
                total_skip += 1; continue
            if date_str and not is_recent(date_str):
                total_old += 1; continue
            summary = re.sub(r'<[^>]+>', '', content)[:200] if content else ''
            cursor.execute("""
                INSERT OR IGNORE INTO gov_raw
                (source_url, page_url, title, summary, content, site_name, publish_date)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (source_url, source_url, title, summary, content, SITE_NAME, date_str or ''))
            if cursor.rowcount > 0: total_new += 1
        conn.commit()
    conn.close()
    print(f'\nDone. New: {total_new}, Skipped: {total_skip}, Not-生态环境: {total_out}, Old: {total_old}')

if __name__ == '__main__':
    crawl(incremental='incremental' in sys.argv)
