#!/usr/bin/env python3
"""
义乌市人民政府 - 建设项目环境影响评价信息公示 爬虫
CMS: JCMS (浙江政务 jpaas-publish-server)
列表: GET api-gateway/jpaas-publish-server/front/page/build/unit
      (webId=3549, pageId=1229856974, tagId=政府信息公开列表, 翻页用 paramJson={"pageNo":N,"pageSize":"15"})
详情: curl 直连, 正文在 div.main_section (嵌套), 附件链接绝对化
用法: python3 crawl_yw_jhgh.py [--pages N]  (默认 5 页 = 75 条)
"""
import os, sys, json, re, time, sqlite3, requests
from datetime import datetime, date
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "yw.gov.cn-环境影响评价信息公示"
API = "https://www.yw.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
BASE = "https://www.yw.gov.cn"
WEB_ID = "3549"
PAGE_ID = "1229856974"
TAG_ID = "政府信息公开列表"
TPL_SET_ID = "zfHuDn7pjzPV0dB1wLhu9"
CUTOFF_DATE = date(2023, 8, 15)
MAX_PAGES_DEFAULT = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36",
    "Referer": f"https://www.yw.gov.cn/col/col{PAGE_ID}/index.html",
}


def fetch_list_page(pn):
    """Fetch one list page from the JCMS build/unit API. Returns list of dicts."""
    params = {
        "parseType": "bulidstatic",
        "webId": WEB_ID,
        "pageId": PAGE_ID,
        "tagId": TAG_ID,
        "tplSetId": TPL_SET_ID,
        "pageType": "column",
    }
    if pn > 1:
        params["paramJson"] = json.dumps({"pageNo": pn, "pageSize": "15"}, ensure_ascii=False)
    r = requests.get(API, params=params, headers=HEADERS, timeout=30)
    r.raise_for_status()
    d = r.json()
    if not d.get("success"):
        print(f"  [API] page {pn} not success: {d.get('message')}", file=sys.stderr)
        return []
    soup = BeautifulSoup(d["data"]["html"], "lxml")
    items = []
    for li in soup.find_all("li"):
        a = li.find("a", class_="bt_link")
        if not a:
            continue
        href = a.get("href", "")
        if not href:
            continue
        if not href.startswith("http"):
            href = BASE + href
        # 标题在 a > p 文本 (title 属性为空)
        p = a.find("p")
        title = p.get_text(strip=True) if p else a.get_text(strip=True)
        span = a.find("span")
        pub_date = span.get_text(strip=True) if span else ""
        if href and title:
            items.append({"url": href, "title": title, "pub_date": pub_date})
    return items


def fetch_detail(url):
    """Fetch detail page: return (title, content_html)."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")
        h2 = soup.find("h2")
        title = h2.get_text(strip=True) if h2 else ""
        # 正文容器（嵌套 main_section，取最后一个）
        arts = soup.find_all("div", class_="main_section")
        if not arts:
            return title, ""
        art = arts[-1]
        # 去掉信息区（标题/时间/访问统计等）
        for tag in art.find_all(class_=["main_title", "artic_kopen", "artic_tother", "linke", "yybb", "artic_key"]):
            tag.decompose()
        # 附件/图片链接绝对化
        for a in art.find_all("a", href=True):
            href = a["href"]
            if href and not href.startswith("http"):
                a["href"] = BASE + href
        for img in art.find_all("img"):
            src = img.get("src", "")
            if src and not src.startswith("http"):
                img["src"] = BASE + src
        # 去掉底部二维码/脚本
        for tag in art.find_all(["script", "style"]):
            tag.decompose()
        for tag in art.find_all(id=["ewm", "ewm2", "append"]):
            tag.decompose()
        return title, str(art)
    except Exception as e:
        print(f"  [detail] {url} ERR {e}", file=sys.stderr)
        return "", ""


def main():
    max_pages = MAX_PAGES_DEFAULT
    args = sys.argv[1:]
    for i, a in enumerate(args):
        if a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
        elif a.startswith("--pages="):
            max_pages = int(a.split("=")[1])

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    all_new = 0
    all_skip = 0
    reached_cutoff = False

    for page in range(1, max_pages + 1):
        items = fetch_list_page(page)
        if not items:
            print(f"  第{page}页: 空 -> 结束", file=sys.stderr)
            break
        print(f"  第{page}页: {len(items)} 条", file=sys.stderr)
        for it in items:
            pub = it["pub_date"]
            if pub and pub < CUTOFF_DATE.strftime("%Y-%m-%d"):
                reached_cutoff = True
                print(f"  已达截断日 {CUTOFF_DATE} (当前 {pub})，停止", file=sys.stderr)
                break
            url = it["url"]
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if c.fetchone():
                all_skip += 1
                continue
            detail_title, content = fetch_detail(url)
            if not content or len(content) < 20:
                print(f"    ! 空正文: {it['title'][:40]}", file=sys.stderr)
                all_skip += 1
                continue
            title = detail_title or it["title"]
            pub_date = pub or ""
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, source_url, status) "
                "VALUES (?, ?, ?, ?, ?, ?, ?)",
                (SITE_NAME, url, title[:500], content, pub_date[:20], url, "done"),
            )
            if c.rowcount > 0:
                all_new += 1
                print(f"    + {title[:50]} ({pub_date}) body:{len(content)}B", file=sys.stderr)
            else:
                all_skip += 1
            conn.commit()
            time.sleep(0.3)
        if reached_cutoff:
            break

    print(f"\n[{SITE_NAME}] 完成! 新增 {all_new} 条, 跳过 {all_skip} 条", file=sys.stderr)

    if all_new > 0:
        c.execute("""INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary)
                      SELECT rowid, title, site_name, substr(content,1,500) FROM gov_raw
                      WHERE site_name=? AND rowid NOT IN (SELECT rowid FROM gov_search)""", (SITE_NAME,))
        conn.commit()
    conn.close()


if __name__ == "__main__":
    main()
