#!/usr/bin/env python3
"""
长兴县人民政府 - 信息公开双栏目
URL: https://www.zjcx.gov.cn/col/col1229518369/index.html
JPAAS信息公开系统，API筛选
  --col=huanping (默认): 环境影响评价 (xxgkId=II001-007-003, number=GG001)
  --col=gggs        : 公告公示 (xxgkId=N001)
"""
import sys, os, re, json, time, sqlite3
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

COLUMNS = {
    "huanping": {
        "site": "长兴县-环境影响评价",
        "column": "环境影响评价",
        "xxgkId": "II001-007-003",
    },
    "gggs": {
        "site": "长兴县-公告公示",
        "column": "公告公示",
        "xxgkId": "N001",
    },
}

SITE = COLUMNS["huanping"]["site"]
COLUMN = COLUMNS["huanping"]["column"]
SEARCH = {"xxgkId": COLUMNS["huanping"]["xxgkId"], "xxgkType": "xxgk_combination", "nodeId": "330522000000", "className": ""}
PROVINCE = "浙江"
BASE_URL = "https://www.zjcx.gov.cn"
API_URL = f"{BASE_URL}/api-gateway/jpaas-publish-server/front/page/build/unit"
API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "3645",
    "pageId": "1229518369",
    "pageType": "column",
    "tagId": "组配分类list",
    "tplSetId": "QIrUapMnq9Avhahnnyp8M",
}
PAGE_SIZE = 15

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)
session.verify = False


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def fetch_list(page_no, page_size=PAGE_SIZE):
    params = {**API_PARAMS}
    search_str = json.dumps(SEARCH, ensure_ascii=False)
    param_json = json.dumps({"pageNo": page_no, "pageSize": page_size, "search": search_str}, ensure_ascii=False)
    params["paramJson"] = param_json
    r = session.get(API_URL, params=params, timeout=60)
    data = r.json()
    html = data["data"]["html"]

    items = []
    for li in re.finditer(r'<li[^>]*class="cf"[^>]*>(.*?)</li>', html, re.DOTALL):
        li_html = li.group(1)
        href_m = re.search(r'href="([^"]+)"', li_html)
        title_m = re.search(r'title="([^"]*)"', li_html)
        date_m = re.search(r'<span class="fr">(\d{4}-\d{2}-\d{2})', li_html)
        if href_m:
            href = href_m.group(1)
            full_url = urljoin(BASE_URL, href)
            title = title_m.group(1).strip() if title_m else ""
            date_str = date_m.group(1) if date_m else ""
            items.append({"url": full_url, "title": title, "date": date_str})

    total = 0
    for m in re.finditer(r'count="(\d+)"', html):
        total = int(m.group(1))
    return items, total


def clean_content_html(zw):
    """从 div.zhengw 提取干净 HTML：去 script/style/Word残留，附件绝对化，表格保留"""
    # 附件链接绝对化
    for a in zw.find_all("a", href=True):
        h = a["href"]
        if h.startswith("/") or h.startswith("./") or h.startswith("../"):
            a["href"] = urljoin(BASE_URL, h)
        elif not h.startswith("http"):
            a["href"] = urljoin(BASE_URL, h)
    # 图片绝对化
    for img in zw.find_all("img"):
        src = img.get("src") or ""
        if src.startswith("/") or (src and not src.startswith("http")):
            img["src"] = urljoin(BASE_URL, src)
    # 移除 script/style
    for s in zw.find_all(["script", "style"]):
        s.decompose()
    # 清理 Word 残留 span/font 包装（保留文本，去样式）
    for sp in zw.find_all(["span", "font"]):
        sp.unwrap()
    # 清理空标签
    for tag in zw.find_all(["p", "div", "span", "font"]):
        if not tag.get_text(strip=True) and not tag.find(["img", "table", "a"]):
            tag.decompose()
    # 清理 href 为空的 a（避免残留空链接）
    for a in zw.find_all("a"):
        if not a.get("href"):
            a.unwrap()
    # 附件收集（清理前 href 已绝对化）
    attachments = []
    for a in zw.find_all("a", href=True):
        h = a["href"]
        if "document/download" in h or re.search(r'\.(docx?|xlsx?|pdf)$', h, re.I):
            attachments.append(h)
    for script_removed in []:
        pass
    return str(zw), attachments


def fetch_detail(detail_url, list_title):
    try:
        r = session.get(detail_url, timeout=60)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        return list_title, "", list_title, [], str(e)

    # 标题
    full_title = list_title
    h1 = soup.find("h1")
    if h1:
        t = h1.get_text(strip=True)
        if t:
            full_title = t
    if not full_title or full_title == list_title:
        mt = soup.find("meta", attrs={"name": "ArticleTitle"})
        if mt and mt.get("content"):
            full_title = mt["content"].strip()

    # 日期
    date_str = list_title
    mt = soup.find("meta", attrs={"name": "PubDate"})
    if mt and mt.get("content"):
        date_str = mt["content"][:10]

    # 正文 div.zhengw -> 干净 HTML
    content = ""
    attachments = []
    zw = soup.find("div", class_="zhengw")
    if zw:
        content, attachments = clean_content_html(zw)

    if not content and attachments:
        content = f"[本公告为附件格式，共{len(attachments)}个附件]"

    return full_title, content, date_str, ",".join(attachments), None


def import_to_db(record):
    try:
        db = sqlite3.connect(DB_PATH, timeout=30)
        title = (record.get("title") or "")[:500]
        page_url = (record.get("page_url") or "")[:1000]
        content = record.get("content") or ""
        publish_date = (record.get("publish_date") or "")[:20]
        site_name = (record.get("site_name") or "unknown")[:100]
        attachments_str = json.dumps(record.get("attachments") or [], ensure_ascii=False)

        old = db.execute("SELECT rowid FROM gov_raw WHERE page_url = ?", (page_url,)).fetchall()
        for (rid,) in old:
            db.execute("DELETE FROM gov_search WHERE rowid = ?", (rid,))
        db.execute(
            "INSERT OR REPLACE INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?, ?, ?, ?, ?, ?, 'synced', ?)",
            (title, page_url, content, publish_date, site_name, page_url, attachments_str)
        )
        new_rowid = db.execute("SELECT last_insert_rowid()").fetchone()[0]
        summary = content[:500] if content else title[:500]
        db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                   (new_rowid, title, site_name, summary))
        db.commit()
        db.close()
        log(f"  ✅ {title[:30]}")
        return True
    except Exception as e:
        log(f"  [ERR] DB import failed: {e}")
        return False


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--col", choices=list(COLUMNS.keys()), default="huanping",
                        help="栏目: huanping=环境影响评价, gggs=公告公示")
    parser.add_argument("--pages", type=int, default=5,
                        help="爬取前N页（每页15条，默认5页=75条）")
    parser.add_argument("--total", type=int, default=0,
                        help="覆盖总条数（用于 --pages 计算，默认自动）")
    args = parser.parse_args()

    global SITE, COLUMN, SEARCH
    col = COLUMNS[args.col]
    SITE = col["site"]
    COLUMN = col["column"]
    SEARCH = {"xxgkId": col["xxgkId"], "xxgkType": "xxgk_combination", "nodeId": "330522000000", "className": ""}

    # 先取第一页确认总数
    first_items, total = fetch_list(1)
    if args.total:
        total = args.total
    if not total:
        log("[ERROR] 无法获取总条数")
        return
    total_pages = min(args.pages, (total + PAGE_SIZE - 1) // PAGE_SIZE)
    total_needed = min(args.pages * PAGE_SIZE, total)
    count = 0
    seen = set()

    log(f"📋 {SITE} 栏目={COLUMN} 总数={total} 本次计划={total_needed}")

    for page in range(1, total_pages + 1):
        log(f"📄 列表[page={page}]")
        items, total_now = fetch_list(page)
        if not items:
            log(f"  → 无更多数据")
            break
        log(f"  → {len(items)} 条 (总计{total_now})")

        for idx, item in enumerate(items, 1):
            url = item["url"]
            if url in seen:
                continue
            seen.add(url)
            log(f"  ({idx}/{len(items)}) {url.split('/')[-1][:40]}")
            full_title, content, detail_date, attachments_str, err = fetch_detail(url, item["title"])

            record = {
                "title": full_title,
                "page_url": url,
                "publish_date": detail_date or item["date"],
                "content": content,
                "attachments": attachments_str.split(",") if attachments_str else [],
                "site_name": SITE,
                "column": COLUMN,
                "province": PROVINCE,
            }
            ok = import_to_db(record)
            if ok:
                count += 1
            time.sleep(0.5)

        if count >= total_needed:
            break

    log(f"\n✅ {SITE} 爬取完成，共入库 {count} 条")


if __name__ == "__main__":
    main()
