#!/usr/bin/env python3
"""crawl_baiyin_ssthjj.py - 白银市生态环境局-建设项目环境影响评价信息 (JPAAS API)"""
import sys, os, requests, re, json
from datetime import datetime, timedelta

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime('%Y-%m-%d')
SITE_NAME = '白银市生态环境局-建设项目环评'
BASE_URL = 'https://www.baiyin.gov.cn'
API_URL = BASE_URL + '/api-gateway/jpaas-publish-server/front/page/build/unit'
API_PARAMS = {
    'parseType': 'bulidstatic',
    'webId': '95a12467959f44d29b0148749a6e0daf',
    'tplSetId': '718c52a1506547739d141ce0ed891fd3',
    'pageType': 'column',
    'tagId': '政府信息公开指南ls',
    'editType': 'null',
    'pageId': 'a5d7178f3839440c85b058f34789516d',
}
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/120.0'}

def get_list(page):
    try:
        params = dict(API_PARAMS)
        params['paramJson'] = json.dumps({"pageNo": page, "pageSize": 20})
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=(5, 20), verify=False)
        data = r.json()
        html = data['data']['html']
        items = re.findall(
            r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>',
            html, re.DOTALL
        )
        count_m = re.search(r'count="(\d+)"', html)
        total = int(count_m.group(1)) if count_m else len(items)
        return items, total
    except Exception as e:
        print(f"  API error page {page}: {e}", flush=True)
        return [], 0

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=(5, 20), verify=False)
        r.encoding = 'utf-8'
        html = r.text
    except:
        return '', '', ''
    title = ''
    m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m:
        title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    if not title:
        m = re.search(r'<title>([^<]*)</title>', html)
        if m:
            title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
            title = re.sub(r'\s*[-_|_].*$', '', title).strip()
    pub_date = ''
    m = re.search(r'日期[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if m:
        pub_date = m.group(1)
    if not pub_date:
        m = re.search(r'(\d{4}-\d{2}-\d{2})', html[:8000])
        if m:
            pub_date = m.group(1)
    content = ''
    idx = html.find('<div id="zoom"')
    if idx < 0:
        idx = html.find('<div class="zwxxgk_ndbgwz"')
    if idx < 0:
        idx = html.find('<div class="article"')
    if idx > 0:
        start = html.find('>', idx) + 1
        depth = 1
        pos = start
        while pos < len(html) and depth > 0:
            if html[pos:pos+4] == '<div':
                depth += 1
                pos += 4
            elif html[pos:pos+6] == '</div>':
                depth -= 1
                pos += 6
            else:
                pos += 1
        content = html[start:pos-6].strip()
    if not content:
        m = re.search(r'<div class="content"[^>]*>(.*?)</div>', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
    return title, content, pub_date

def run(incremental=False):
    records = []
    seen = set()
    page = 1
    total = 1
    max_pages = 1 if incremental else 60
    while page <= max_pages:
        items, total = get_list(page)
        if not items:
            break
        print(f"  Page {page}: {len(items)} items (total {total})")
        for href, title_html in items:
            title = re.sub(r'<[^>]+>', '', title_html).strip()
            if len(title) < 4:
                continue
            if href.startswith('/'):
                full_url = BASE_URL + href
            elif href.startswith('http'):
                full_url = href
            else:
                full_url = BASE_URL + '/' + href
            if full_url in seen:
                continue
            seen.add(full_url)
            t, c, d = fetch_detail(full_url)
            if not t or not c.strip():
                continue
            if d and d < CUTOFF:
                continue
            records.append({
                'title': t,
                'url': full_url,
                'pub_date': d,
                'site_name': SITE_NAME,
                'content': c,
                'summary': '',
            })
        page += 1
        if page > (total + 9) // 10:
            break
    print(f"  Total: {len(records)} items")
    valid = [r for r in records if r['content'].strip()]
    if valid:
        push_to_searchdb(valid, "baiyin_ssthjj")
    return len(valid)

if __name__ == '__main__':
    cnt = run(incremental='--incremental' in sys.argv)
    print(f"Done: {cnt} records")
