#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
鸡西梨树区-政府信息公开 爬虫 (黑龙江政府站群 /common/search JSON API)
站点: jixilishu.gov.cn
栏目: /lsq/b3a02c4392c747c2bd19b698c9a72787/zfxxgk.shtml (政府信息公开, 含环评公示)
列表+正文: GET /common/search/{channelId}?_isJson=true&_pageSize=20&page=N
  → data.results[]: title/url/publishedTimeStr/contentHtml(完整正文)
  → total=15, 单页
"""
import re, os, sys, time
import requests, warnings
warnings.filterwarnings('ignore')
from datetime import datetime, timedelta

_MAX_PG = None
for _a in sys.argv[1:]:
    if _a.startswith("--pages="):
        try:
            _MAX_PG = int(_a.split("=", 1)[1])
        except Exception:
            pass
    elif _a.isdigit():
        _MAX_PG = int(_a)
if _MAX_PG is not None:
    print(f'[AutoPg] max_pages={_MAX_PG}')

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "鸡西梨树区-政府信息公开"
BASE_URL = "https://jixilishu.gov.cn"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0",
    "Referer": f"{BASE_URL}/lsq/b3a02c4392c747c2bd19b698c9a72787/zfxxgk.shtml",
}

CHANNEL_ID = "d1eed7d7bd9f4a2582338a4a67a6194c"
LIST_URL = f"{BASE_URL}/common/search/{CHANNEL_ID}?_isAgg=false&_isJson=true&_pageSize=20&_template=index&_rangeTimeGte=&_channelName=&page="


def fetch_list_page(page):
    try:
        r = requests.get(f"{LIST_URL}{page}", headers=HEADERS, timeout=20, verify=False)
        if r.status_code != 200:
            print(f"  [WARN] {page} HTTP {r.status_code}")
            return []
        j = r.json()
        items = []
        for d in j.get("data", {}).get("results", []):
            url = d.get("url", "")
            title = (d.get("title") or "").strip()
            date = (d.get("publishedTimeStr") or "")[:10]
            ch = d.get("contentHtml", "") or ""
            if not url or not title:
                continue
            # ⚠️ 外链(微信/jixi.gov.cn等)非本栏目内容 → 跳过
            if re.match(r'https?://', url) and 'jixilishu.gov.cn' not in url:
                continue
            items.append({"url": url, "title": title, "date": date, "html": ch})
        return items
    except Exception as e:
        print(f"  [WARN] {page} err: {e}")
        return []


def clean_content(html, page_url):
    """API contentHtml → 规范化正文: 保留 <p>, 图片/附件绝对化, 剥 script/style"""
    c = html
    for m in re.finditer(r'<img[^>]*src="([^"]+)"', c):
        h = m.group(1)
        if h.startswith("/"):
            c = c.replace(h, BASE_URL + h)
        elif not h.startswith("http"):
            c = c.replace(h, page_url.rsplit("/", 1)[0] + "/" + h)
    for m in re.finditer(r'<a[^>]*href="([^"]+)"', c):
        h = m.group(1)
        if h.startswith("/"):
            c = c.replace(h, BASE_URL + h)
        elif not h.startswith("http") and re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)', h, re.I):
            c = c.replace(h, page_url.rsplit("/", 1)[0] + "/" + h)
    for m in re.finditer(r'<iframe[^>]*src="([^"]+\.pdf)"[^>]*>', c, re.I):
        h = m.group(1)
        if h.startswith("/"):
            h = BASE_URL + h
        elif not h.startswith("http"):
            h = page_url.rsplit("/", 1)[0] + "/" + h
        c = c.replace(m.group(0), f'<p><a href="{h}">PDF原文</a></p>')
    c = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', c, flags=re.DOTALL|re.I)
    c = re.sub(r'<p[^>]*>', '<p>', c)
    text = re.sub(r'<[^>]+>', '', c).strip()
    if text or '<a href=' in c or '<img' in c:
        return c.strip()
    return ""


def run(max_pages=5):
    print(f"\n{'='*50}")
    print(f"🚀 {SITE_NAME}")
    print(f"{'='*50}")

    all_items = []
    for pg in range(1, min(max_pages, _MAX_PG or max_pages) + 1):
        its = fetch_list_page(pg)
        if not its:
            print(f"  第{pg}页: 空 -> 结束")
            break
        dates = [it["date"] for it in its if it["date"]]
        print(f"  第{pg}页: {len(its)} 条 ({min(dates) if dates else '?'} ~ {max(dates) if dates else '?'})")
        all_items.extend(its)
        if dates and max(dates) < CUTOFF:
            print(f"  该页已全部早于截断日{CUTOFF}，停止翻页")
            break
        time.sleep(0.2)

    print(f"\n📊 列表共 {len(all_items)} 条")
    kept = [it for it in all_items if not it["date"] or it["date"] >= CUTOFF]
    print(f"  窗口内(>= {CUTOFF}): {len(kept)} 条")

    out = []
    err = 0
    for i, it in enumerate(kept, 1):
        url = it["url"]
        if not url.startswith("http"):
            url = BASE_URL + url
        content = clean_content(it["html"], url)
        text_len = len(re.sub(r'<[^>]+>', '', content).strip()) if content else 0
        if text_len < 5 and '<img' not in (content or ''):
            print(f"  ⚠ 空正文: {it['title'][:40]}")
            err += 1
            continue
        out.append({
            "site_name": SITE_NAME,
            "title": it["title"],
            "pub_date": it["date"],
            "content": content,
            "source_url": url,
            "url": url,
        })
        if i % 20 == 0:
            print(f"  ...{i}/{len(kept)}")
        time.sleep(0.05)

    print(f"\n📦 待入库: {len(out)}, 失败: {err}")
    sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
    from crawler_lib import push_to_searchdb
    push_to_searchdb(out, batch_label=SITE_NAME)


if __name__ == "__main__":
    run(max_pages=5)
