#!/usr/bin/env python3
"""crawl_bzzjcs_tzgg.py - 滨州中介超市-通知公告 (bzzjcs.cn)

站点: http://bzzjcs.cn/tzgg/list (滨州中介超市网站, Vue SPA)
CMS: JEECG 低代码平台, 前端 SPA + REST API
列表API: GET http://222.134.12.11:8094/jeecg-zjcs/news/apiNews/list
  params: catid=1537596722527481858 (通知公告栏目), column=dtime, order=desc, pageNo, pageSize=20
  返回: result.records[] (id/title/dtime/catid/...), result.total=609 (31页)
详情API: GET http://222.134.12.11:8094/jeecg-zjcs/news/apiNews/queryById?id={id}
  返回: result.body (HTML正文), result.title, result.dtime
注意: 列表接口 body=None, 正文必须走 queryById; 详情接口需要直接 GET(无token)
"""
import sys, os, requests, argparse, json
from datetime import datetime, timedelta

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime('%Y-%m-%d')
SITE_NAME = '滨州中介超市-通知公告'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/120.0',
           'Referer': 'http://bzzjcs.cn/tzgg/list'}
API_BASE = 'http://222.134.12.11:8094/jeecg-zjcs/'
CATID = '1537596722527481858'
PAGE_SIZE = 20
TOTAL = 609  # 实测 total, 运行时以接口返回为准

def fetch_list(page_idx):
    url = API_BASE + 'news/apiNews/list'
    params = {'catid': CATID, 'column': 'dtime', 'order': 'desc',
              'pageNo': page_idx, 'pageSize': PAGE_SIZE}
    try:
        r = requests.get(url, params=params, timeout=(8, 20), headers=HEADERS, verify=False)
        d = r.json()
        records = d.get('result', {}).get('records', [])
    except Exception as e:
        print(f"    [WARN] list page {page_idx} error: {e}")
        return []
    results = []
    for rec in records:
        title = (rec.get('title') or '').strip()
        dtime = rec.get('dtime') or ''
        pub_date = dtime[:10] if dtime else ''
        results.append({
            'id': rec.get('id', ''),
            'url': f"http://bzzjcs.cn/tzgg/detail?id={rec.get('id', '')}",
            'pub_date': pub_date,
            'title': title,
        })
    return results

def fetch_detail(rec):
    """rec: {id,url,pub_date,title} -> (title, content)"""
    url = API_BASE + 'news/apiNews/queryById'
    try:
        r = requests.get(url, params={'id': rec['id']}, timeout=(8, 20),
                         headers=HEADERS, verify=False)
        d = r.json()
        result = d.get('result') or {}
    except Exception:
        return rec['title'], ''
    title = (result.get('title') or rec['title']).strip()
    body = result.get('body') or ''
    if not body.strip():
        # 列表接口的 body 字段兜底
        return title, ''
    return title, body

def run(max_pages=10):
    records = []
    seen = set()
    page = 1
    total_pages = 999
    while page <= min(max_pages, total_pages):
        items = fetch_list(page)
        if not items:
            break
        # 从接口 total 计算总页数 (仅第一页)
        if page == 1:
            try:
                r = requests.get(API_BASE + 'news/apiNews/list',
                                 params={'catid': CATID, 'column': 'dtime', 'order': 'desc',
                                         'pageNo': 1, 'pageSize': PAGE_SIZE},
                                 timeout=(8, 20), headers=HEADERS, verify=False)
                total = r.json().get('result', {}).get('total', 0)
                total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
                print(f"  total={total} 共{total_pages}页")
            except Exception:
                pass
        print(f"  Page {page}: {len(items)} items")
        for it in items:
            if it['pub_date'] and it['pub_date'] < CUTOFF:
                continue
            if it['url'] in seen:
                continue
            seen.add(it['url'])
            title, content = fetch_detail(it)
            if not title:
                title = it['title']
            if not content.strip():
                continue
            records.append({
                'title': title, 'url': it['url'], 'source_url': it['url'],
                'pub_date': it['pub_date'],
                'site_name': SITE_NAME, 'content': content, 'summary': '',
            })
        page += 1
    print(f"  Total: {len(records)} items")
    valid = [r for r in records if r['content'].strip()]
    if valid:
        push_to_searchdb(valid, "bzzjcs_tzgg")
    return len(valid)

if __name__ == '__main__':
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=5)
    args = ap.parse_args()
    cnt = run(max_pages=args.pages)
    print(f"Done: {cnt} records")
