#!/usr/bin/env python3
"""
绍兴市上虞区人民政府 - 建设项目环境影响评价信息公示 爬虫
CMS: JCMS (浙江政务 jpaas-publish-server)
列表: GET api-gateway/jpaas-publish-server/front/page/build/unit
      (pageId=1229857239, 分页用 paramJson={"pageNo":N,"pageSize":"30"})
详情: curl 直连, 正文在 div.article, 完整标题在 h2
用法: python3 crawl_shangyu.py [--pages N]  (默认 5 页 = 150 条)
"""
import os, sys, json, re, time, sqlite3, requests
from datetime import datetime, date
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "绍兴上虞区政府-环评公示"
API = "https://www.shangyu.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
BASE = "https://www.shangyu.gov.cn"
WEB_ID = "2328"
PAGE_ID = "1229857239"
TAG_ID = "当前栏目列表1a"
TPL_SET_ID = "9nuI0IJgTDxmXUM0arEPb"
CUTOFF_DATE = date(2023, 6, 1)
MAX_PAGES_DEFAULT = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
    "Referer": f"https://www.shangyu.gov.cn/col/col{PAGE_ID}/index.html?key",
}


def fetch_list_page(pn):
    """Fetch one list page from the JCMS build/unit API. Returns list of dicts."""
    params = {
        "webId": WEB_ID,
        "pageId": PAGE_ID,
        "parseType": "bulidstatic",
        "pageType": "column",
        "tagId": TAG_ID,
        "tplSetId": TPL_SET_ID,
    }
    if pn > 1:
        params["paramJson"] = json.dumps({"pageNo": pn, "pageSize": "30"}, ensure_ascii=False)
    r = requests.get(API, params=params, headers=HEADERS, timeout=30)
    r.raise_for_status()
    d = r.json()
    if not d.get("success"):
        print(f"  [API] page {pn} not success: {d.get('message')}")
        return []
    soup = BeautifulSoup(d["data"]["html"], "lxml")
    items = []
    for tr in soup.find_all("tr"):
        a = tr.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href:
            continue
        if not href.startswith("http"):
            href = BASE + href
        # 完整标题优先取 title 属性（列表可见文本可能被截断带省略号）
        title = (a.get("title") or a.get_text(strip=True)).strip()
        pub_date = ""
        for td in tr.find_all("td"):
            m = re.search(r"(\d{4}-\d{2}-\d{2})", td.get_text(strip=True))
            if m:
                pub_date = m.group(1)
                break
        if href and title:
            items.append({"url": href, "title": title, "pub_date": pub_date})
    return items


def fetch_detail(url):
    """Fetch detail page: return (full_title, content_html)."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")
        h2 = soup.find("h2")
        title = h2.get_text(strip=True) if h2 else ""
        art = soup.find("div", class_="article")
        if not art:
            art = soup.find("div", class_="content")
        if art:
            # 去掉面包屑/位置导航
            for nav in art.find_all(class_=["position", "warp", "crumb"]):
                nav.decompose()
            return title, str(art)
        return title, ""
    except Exception:
        return "", ""


def main():
    max_pages = MAX_PAGES_DEFAULT
    args = sys.argv[1:]
    for i, a in enumerate(args):
        if a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
        elif a.startswith("--pages="):
            max_pages = int(a.split("=")[1])

    print(f"[shangyu] SITE={SITE_NAME} pages={max_pages} db={DB_PATH}", flush=True)

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    # ---- 收集列表 ----
    items, seen = [], set()
    for pn in range(1, max_pages + 1):
        try:
            page_items = fetch_list_page(pn)
        except Exception as e:
            print(f"  [list] page {pn} error: {e}", flush=True)
            break
        if not page_items:
            break
        new = 0
        for it in page_items:
            if it["url"] in seen:
                continue
            seen.add(it["url"])
            items.append(it)
            new += 1
        print(f"  Page {pn}: +{new} links (total {len(items)})", flush=True)
        # 整页都早于截止日期 → 停止
        if page_items and all(
            it["pub_date"] and datetime.strptime(it["pub_date"], "%Y-%m-%d").date() < CUTOFF_DATE
            for it in page_items if it["pub_date"]
        ):
            print("  All items older than cutoff, stop", flush=True)
            break
        time.sleep(0.5)
    print(f"[shangyu] Total collected: {len(items)}", flush=True)

    # ---- 抓详情入库 ----
    inserted = skipped = empty = 0
    for i, it in enumerate(items):
        c.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (it["url"],))
        if c.fetchone():
            skipped += 1
            continue
        title, content = fetch_detail(it["url"])
        if not title:
            title = it["title"]
        if not content:
            empty += 1
            print(f"  [detail] empty content: {it['url'][-60:]}", flush=True)
            continue  # 不写空正文, 下次运行重试
        c.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date, content, summary)
               VALUES (?, ?, ?, ?, ?, ?, ?)""",
            (SITE_NAME, it["url"], it["url"], title, it["pub_date"], content, title),
        )
        if c.rowcount:
            inserted += 1
        time.sleep(0.3)
        if (i + 1) % 20 == 0:
            conn.commit()
            print(f"  [{i+1}/{len(items)}] +{inserted} dup={skipped} empty={empty}", flush=True)
    conn.commit()
    conn.close()
    print(f"[shangyu] Done. +{inserted} new, {skipped} dup, {empty} empty-skip", flush=True)


if __name__ == "__main__":
    main()
