#!/usr/bin/env python3
"""crawl_baiyin_kfq.py - 白银高新区管委会-通知公告 (JPAAS API, 主站bmzq栏目)"""
import sys, os, requests, re, json
from datetime import datetime, timedelta

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime('%Y-%m-%d')
SITE_NAME = '白银高新区-通知公告'
BASE_URL = 'https://www.baiyin.gov.cn'
API_URL = BASE_URL + '/api-gateway/jpaas-publish-server/front/page/build/unit'
API_PARAMS = {
    'parseType': 'bulidstatic',
    'webId': '95a12467959f44d29b0148749a6e0daf',
    'tplSetId': '718c52a1506547739d141ce0ed891fd3',
    'pageType': 'column',
    'tagId': '信息列表',
    'editType': 'null',
    'pageId': 'f20bc9adf0a24dbfa1bd713b8ef5872f',
}
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/120.0'}

def get_list(page):
    try:
        params = dict(API_PARAMS)
        params['paramJson'] = json.dumps({"pageNo": page, "pageSize": 20})
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=(5, 20), verify=False)
        data = r.json()
        html = data['data']['html']
        items = re.findall(
            r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>',
            html, re.DOTALL
        )
        count_m = re.search(r'count="(\d+)"', html)
        total = int(count_m.group(1)) if count_m else len(items)
        return items, total
    except Exception as e:
        print(f"  API error page {page}: {e}", flush=True)
        return [], 0

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=(5, 20), verify=False)
        r.encoding = 'utf-8'
        html = r.text
    except:
        return '', '', ''
    title = ''
    m = re.search(r'<title>([^<]*)</title>', html)
    if m:
        title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
        title = re.sub(r'\s*[-_|_].*$', '', title).strip()
    pub_date = ''
    # 优先取正文"发布时间" (正确日期), fallback meta PubDate / 页面首日期
    m = re.search(r'发布时间[：:]\s*(?:<[^>]+>\s*)?(\d{4}-\d{2}-\d{2})', html)
    if m:
        pub_date = m.group(1)
    if not pub_date:
        m = re.search(r'name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
        if m:
            pub_date = m.group(1)
    if not pub_date:
        m = re.search(r'(\d{4}-\d{2}-\d{2})', html[:6000])
        if m:
            pub_date = m.group(1)
    content = ''
    m = re.search(r'<div class="content"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    return title, content, pub_date

def run(incremental=False):
    records = []
    seen = set()
    page = 1
    total = 1
    max_pages = 1 if incremental else 60
    while page <= max_pages:
        items, total = get_list(page)
        if not items:
            break
        print(f"  Page {page}: {len(items)} items (total {total})")
        for href, title_html in items:
            title = re.sub(r'<[^>]+>', '', title_html).strip()
            if len(title) < 4:
                continue
            if href.startswith('/'):
                full_url = BASE_URL + href
            elif href.startswith('http'):
                full_url = href
            else:
                full_url = BASE_URL + '/' + href
            if full_url in seen:
                continue
            seen.add(full_url)
            t, c, d = fetch_detail(full_url)
            if not t or not c.strip():
                continue
            if d and d < CUTOFF:
                continue
            records.append({
                'title': t,
                'url': full_url,
                'pub_date': d,
                'site_name': SITE_NAME,
                'content': c,
                'summary': '',
            })
        page += 1
        if page > (total + 9) // 10:
            break
    print(f"  Total: {len(records)} items")
    valid = [r for r in records if r['content'].strip()]
    if valid:
        push_to_searchdb(valid, "baiyin_kfq")
    return len(valid)

if __name__ == '__main__':
    cnt = run(incremental='--incremental' in sys.argv)
    print(f"Done: {cnt} records")
