#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""crawl_dsq.py — 大石桥市人民政府-环评公示 (EWB system)"""

import os, re, json, sqlite3, time, sys, random
from datetime import datetime, timedelta
import urllib.request, urllib.error

DB_PATH = os.environ.get('SEARCH_DB', '/root/search.db')
SITE_NAME = '大石桥市人民政府-环评公示'
API_URL = 'http://www.dsq.gov.cn/EWB_YK_Mid/rest/lightfrontaction/getgovinfolist'
BASE = 'http://www.dsq.gov.cn'
SITE_GUID = 'f1c580dd-d7d2-4148-b90e-b08b2ed43c3a'
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
PAGE_SIZE = 50
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36',
    'Content-Type': 'application/json',
    'Referer': 'http://www.dsq.gov.cn/dynamic/zw/openlist.html?categorynum=003002004004',
}


def fetch_api(page_index):
    """Fetch one page of API data"""
    data = {
        'token': '',
        'params': {
            'deptcode': '',
            'categorynum': '003002004004',
            'pageIndex': page_index,
            'pageSize': PAGE_SIZE,
            'siteGuid': SITE_GUID,
        }
    }
    req = urllib.request.Request(API_URL, data=json.dumps(data).encode(), headers=HEADERS)
    for attempt in range(3):
        try:
            resp = urllib.request.urlopen(req, timeout=15)
            return json.loads(resp.read())
        except Exception as e:
            if attempt < 2:
                time.sleep(2)
            else:
                print(f'  [WARN] API page {page_index} failed: {e}')
                return None


def fetch_detail(url):
    """Fetch detail page HTML"""
    full_url = urllib.parse.urljoin(BASE, url) if url.startswith('/') else url
    req = urllib.request.Request(full_url, headers={
        'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    })
    for attempt in range(3):
        try:
            resp = urllib.request.urlopen(req, timeout=15)
            html = resp.read().decode('utf-8', errors='replace')
            return html
        except Exception as e:
            if attempt < 2:
                time.sleep(1)
            else:
                print(f'  [WARN] Detail fetch failed: {url} - {e}')
                return None


def parse_detail(html):
    """Extract title and content from detail HTML"""
    # Title from h3#ivs_title
    title = ''
    m = re.search(r'<h3\s+id="ivs_title"[^>]*>(.*?)</h3>', html, re.DOTALL)
    if m:
        title = m.group(1).strip()
    
    # Content from div#ivs_content
    content = ''
    m = re.search(r'<div\s+class="ewb-article-content"\s+id="ivs_content"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not m:
        m = re.search(r'<div\s+id="ivs_content"[^>]*>(.*?)</div>\s*<div[^>]*class="ewb-source"', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
        # Clean up
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
        content = re.sub(r'<o:p>\s*</o:p>', '', content)
        content = re.sub(r'<o:p/>', '', content)
        content = re.sub(r'<p[^>]*>\s*(<br\s*/?>\s*)*</p>', '', content)
    
    return title, content.strip()


def main():
    print(f'[{datetime.now().strftime("%H:%M:%S")}] {SITE_NAME}')
    
    # Step 1: Get first page to know total
    result = fetch_api(0)
    if not result:
        print('[ERROR] API failed')
        return
    total = result['custom']['total']
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print(f'Total: {total} records, {total_pages} pages (pageSize={PAGE_SIZE})')
    
    # Step 2: Collect all items from all pages
    all_items = result['custom']['data']  # page 0 items
    for p in range(1, total_pages):
        r = fetch_api(p)
        if r and 'data' in r['custom']:
            all_items.extend(r['custom']['data'])
        time.sleep(random.uniform(0.3, 0.5))
    
    print(f'Collected {len(all_items)} items')
    
    # Filter by 3-year cutoff
    recent = [item for item in all_items if item.get('infodate', '') >= CUTOFF]
    old = len(all_items) - len(recent)
    print(f'Within 3 years: {len(recent)} (filtered: {old})')
    
    # Step 3: Fetch details and insert
    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    new_count = skip_count = error_count = 0
    
    for i, item in enumerate(recent, 1):
        title = item['title'].strip()
        pub_date = item['infodate']
        page_url = urllib.parse.urljoin(BASE, item['infourl'])
        
        print(f'  [{i}/{len(recent)}] {title[:30]}...', end=' ')
        
        # Check duplicate
        if c.execute('SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?',
                     (page_url, SITE_NAME)).fetchone():
            print('DUPLICATE')
            skip_count += 1
            continue
        
        # Fetch detail
        html = fetch_detail(item['infourl'])
        if not html:
            print('FETCH ERR')
            error_count += 1
            continue
        
        detail_title, content = parse_detail(html)
        final_title = detail_title or title
        if not content:
            print('WARN (empty content)')
        
        try:
            c.execute('''INSERT OR IGNORE INTO gov_raw
                (site_name, title, page_url, publish_date, source_url, content)
                VALUES (?,?,?,?,?,?)''',
                (SITE_NAME, final_title, page_url, pub_date, page_url, content))
            if c.rowcount:
                new_count += 1
                print(f'OK ({len(content)}B)')
            else:
                print('DUP(insert)')
                skip_count += 1
        except Exception as e:
            print(f'ERR: {e}')
            error_count += 1
        
        conn.commit()
        time.sleep(random.uniform(0.3, 0.5))
    
    conn.close()
    print(f'\n=== Done ===')
    print(f'New: {new_count}, Skipped: {skip_count}, Errors: {error_count}')


if __name__ == '__main__':
    main()
