#!/usr/bin/env python3
"""crawl_funing.py - 富宁县人民政府-通知公告"""
import sys, os, json, re, time, requests, warnings
from datetime import datetime, timedelta

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

warnings.filterwarnings('ignore', category=requests.packages.urllib3.exceptions.InsecureRequestWarning)

CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime('%Y-%m-%d')
SITE_NAME = '富宁县人民政府-通知公告'
API_URL = 'https://www.ynfn.gov.cn/queryList'
API_HEADERS = {'Content-Type': 'application/json'}
PAGE_SIZE = 15

def run(incremental=False):
    records = []
    for pn in range(1, 50):
        payload = {
            "channelCode": ["ynfnxtzgg100"],
            "webSiteCode": ["fnxrmzf"],
            "current": pn,
            "pageSize": PAGE_SIZE
        }
        try:
            r = requests.post(API_URL, json=payload, headers=API_HEADERS, timeout=15, verify=False)
            data = r.json()
            items = data.get('data', {}).get('results', [])
        except Exception as e:
            print(f"  API error page {pn}: {e}")
            break
        if not items:
            break
        for item in items:
            src = item.get('source', {})
            pub_date = src.get('pubDate', '')[:10]
            if pub_date < CUTOFF:
                continue
            title = src.get('title', '').strip()
            urls_str = src.get('urls', '{}')
            detail_url = ''
            try:
                urls_obj = json.loads(urls_str)
                detail_url = 'https://www.ynfn.gov.cn' + urls_obj.get('pc', '')
            except:
                pass
            content = src.get('content', {}).get('content', '')
            records.append({
                'title': title,
                'url': detail_url,
                'pub_date': pub_date,
                'site_name': SITE_NAME,
                'content': content,
                'summary': '',
            })
        print(f"  Page {pn}: {len(items)} items -> {sum(1 for i in items if i.get('source',{}).get('pubDate','')[:10] >= CUTOFF)} in 3y")
        if incremental:
            break
        # Early exit: all items past 3y cutoff
        all_past = all(i.get('source',{}).get('pubDate','')[:10] < CUTOFF for i in items)
        if all_past:
            break
        time.sleep(0.3)
    print(f"  Total: {len(records)} items")
    if records:
        push_to_searchdb(records, "funing_tzgg")
    return len(records)

if __name__ == '__main__':
    inc = '--incremental' in sys.argv
    cnt = run(incremental=inc)
    print(f"Done: {cnt} records")
