#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
兰考县政府-公示公告 独立爬虫
站点: www.lankao.gov.cn (JPAAS 大明CMS, JS渲染列表)
栏目: /kfslkxwz/tzggx/pc/list.html (公示公告)
API: POST /queryList  (webSiteCode[]=kfslkxwz&channelCode[]=tzggx)
注意: 列表 API 直接返回完整正文(content.content) + 附件(articleFiles)
"""
import re, os, sys, time, json
import requests, warnings
warnings.filterwarnings('ignore')
from datetime import datetime, timedelta

_MAX_PG = None
for _a in sys.argv[1:]:
    if _a.startswith("--pages="):
        try:
            _MAX_PG = int(_a.split("=", 1)[1])
        except Exception:
            pass
    elif _a.isdigit():
        _MAX_PG = int(_a)
if _MAX_PG is not None:
    print(f'[AutoPg] max_pages={_MAX_PG}')

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "兰考县-公示公告"
DOMAIN = "www.lankao.gov.cn"
BASE_URL = f"http://{DOMAIN}"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36"}

API_URL = f"{BASE_URL}/queryList"


def fetch_list(page, page_size=15):
    """POST /queryList 拉列表 (数组形式参数)"""
    data = {
        "current": page,
        "pageSize": page_size,
        "webSiteCode[]": "kfslkxwz",
        "channelCode[]": "tzggx",
    }
    for retry in range(3):
        try:
            r = requests.post(API_URL, data=data, headers=HEADERS, timeout=20)
            if r.status_code == 200:
                d = r.json()
                results = d.get("data", {}).get("results", []) or []
                total = d.get("data", {}).get("total", 0)
                items = []
                for res in results:
                    s = res.get("source", {})
                    title = (s.get("title", timeout=30) or "").strip()
                    pub = (s.get("pubDate", timeout=30) or "")[:10]
                    content = s.get("content", {}, timeout=30).get("content", "") or ""
                    # 规范化 <p style=...> -> <p> (search_app 严格匹配闭合 <p>)
                    content = re.sub(r'<p[^>]*>', '<p>', content)
                    # 附件: articleFiles -> <p><a href="domain+url">fileName</a></p>
                    files = s.get("articleFiles", timeout=30) or "[]"
                    try:
                        farr = json.loads(files) if isinstance(files, str) else files
                    except Exception:
                        farr = []
                    att_blocks = []
                    for f in farr or []:
                        fn = (f.get("fileName") or "").strip()
                        fu = (f.get("url") or "").strip()
                        fd = (f.get("domainName") or "").strip()
                        if fn and fu:
                            href = fu if fu.startswith("http") else (fd + fu if fd else BASE_URL + fu)
                            att_blocks.append(f'<p><a href="{href}">{fn}</a></p>')
                    if att_blocks:
                        content = content.rstrip() + "\n" + "\n".join(att_blocks)
                    # 详情页 URL (source_url 去重用)
                    try:
                        urls = json.loads(s.get("urls", timeout=30) or "{}")
                        pc_url = urls.get("pc", "")
                    except Exception:
                        pc_url = ""
                    detail_url = (BASE_URL + pc_url) if pc_url else (BASE_URL + f"/kfslkxwz/tzggx/pc/content/content_{s.get('id', timeout=30)}.html")
                    items.append({
                        "site_name": SITE_NAME,
                        "title": title,
                        "pub_date": pub,
                        "content": content,
                        "source_url": detail_url,
                        "url": detail_url,
                        "id": s.get("id", "", timeout=30),
                    })
                return items, total
            elif r.status_code in (403, 412):
                print(f"    [WARN] 第{page}页 HTTP {r.status_code} (WAF?), 重试{retry+1}")
                time.sleep(3)
            else:
                print(f"    [WARN] 第{page}页 HTTP {r.status_code}")
        except Exception as e:
            print(f"    [WARN] 第{page}页异常: {e}")
        time.sleep(1)
    return [], 0


def run(max_pages=200):
    print(f"\n{'='*50}")
    print(f"🚀 {SITE_NAME}")
    print(f"{'='*50}")

    items, total = fetch_list(1)
    if not items:
        print("⚠ API无返回")
        return
    print(f"  共{total}条, 15条/页 = {(total+14)//15}页")

    total_pages = min(max_pages, (total + 14) // 15)
    all_items = list(items)
    for pg in range(2, min(total_pages, _MAX_PG or total_pages) + 1):
        its, _ = fetch_list(pg)
        if not its:
            print(f"  第{pg}页: 空 -> 结束")
            break
        all_items.extend(its)
        dates = [it["pub_date"] for it in its if it["pub_date"]]
        print(f"  第{pg}页: {len(its)} 条 ({dates[-1] if dates else '?'} ~ {dates[0] if dates else '?'})")
        if dates and max(dates) < CUTOFF:
            print(f"  该页已全部早于截断日{CUTOFF}，停止翻页")
            break
        time.sleep(0.2)

    print(f"\n📊 API共 {len(all_items)} 条")
    # CUTOFF 过滤
    kept = [it for it in all_items if not it["pub_date"] or it["pub_date"] >= CUTOFF]
    print(f"  窗口内(>= {CUTOFF}): {len(kept)} 条")

    # 质量检查
    empty = [it for it in kept if not (it["content"] or "").strip()]
    if empty:
        print(f"  ⚠ 正文为空 {len(empty)} 条:")
        for it in empty[:10]:
            print(f"      {it['title'][:50]}")
    if kept:
        no_p = sum(1 for it in kept if '<p>' not in it["content"] and '<table' not in it["content"])
        if no_p:
            print(f"  ⚠ 无<p>/<table>段落: {no_p} 条")

    # 入库
    sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
    from crawler_lib import push_to_searchdb
    push_to_searchdb(kept, batch_label=SITE_NAME)


if __name__ == "__main__":
    run(max_pages=200)
