#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""兴化市-乡镇公告 (JPAAS API, 当前栏目列表)"""

import requests, re, sqlite3, os, json
from datetime import datetime, timedelta

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/120.0.0.0"}
BASE = "http://www.xinghua.gov.cn"
API = BASE + "/api-gateway/jpaas-publish-server/front/page/build/unit"

CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
SITE = "兴化市-乡镇公告"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "6f2146312e324f9e98099f52002f10eb",
    "pageType": "column",
    "tagId": "当前栏目列表",
    "pageId": "SWOGYwcQw0SUbN8jJCM9V",
    "tplSetId": "c074ff6fcd38439684741941c539cfce",
}

def get_list_items(page_no):
    param_json = {"pageNo": page_no, "pageSize": 15}
    params = {**API_PARAMS, "paramJson": json.dumps(param_json)}
    r = requests.get(API, params=params, headers=HEADERS, timeout=30)
    data = r.json()
    if not data.get("success"):
        return []
    html = data["data"]["html"]
    items = re.findall(
        r'<a href="([^"]+)"[^>]*title="([^"]*)"[^>]*>([^<]*)</a>.*?<span[^>]*>(\d{4}-\d{2}-\d{2})</span>',
        html, re.DOTALL
    )
    result = []
    for url_path, title_attr, title_text, date in items:
        url = BASE + url_path if url_path.startswith("/") else url_path
        title = title_text.strip() or title_attr.strip()
        result.append({"title": title, "url": url, "date": date})
    return result

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except:
        return "", "正文为空"

    # Title from h1
    title = ""
    m = re.search(r"<h1[^>]*>\s*([^<]+?)\s*</h1>", r.text)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r"<title>([^<]+)</title>", r.text)
        if m:
            t = m.group(1).strip()
            if t and "欢迎" not in t and "404" not in t:
                title = t

    # Content from div.wzzw-article
    content = "正文为空"
    m = re.search(r'<div class="wzzw-article"[^>]*>(.*?)</div>', r.text, re.DOTALL)
    if m:
        c = m.group(1).strip()
        if len(c) > 50:
            content = c

    if content != "正文为空":
        content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL)
        content = re.sub(r"<iframe[^>]*>.*?</iframe>", "", content, flags=re.DOTALL)
        content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL)
        content = content.strip()
        if not content:
            content = "正文为空"

    title = re.sub(r"<[^>]+>", "", title).strip()
    return title, content

def main():
    pages = int(os.environ.get("PAGES", "5"))

    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")

    total_new = 0
    total_skipped = 0

    for page_no in range(1, pages + 1):
        items = get_list_items(page_no)
        if not items:
            print(f"Page {page_no}: 0 items, stopping")
            break
        print(f"Page {page_no}: {len(items)} items")

        for item in items:
            url = item["url"]
            title = item["title"]
            daytime = item["date"]

            if not title or not url:
                continue
            if daytime < CUTOFF:
                total_skipped += 1
                continue

            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if cur.fetchone():
                total_skipped += 1
                continue

            try:
                detail_title, content = get_detail(url)
                if content == "正文为空":
                    total_skipped += 1
                    continue

                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                    (SITE, url, detail_title or title, daytime, content, int(daytime.replace("-", "")), "政府公告")
                )
                if cur.rowcount > 0:
                    total_new += 1
            except Exception as e:
                print(f"  ERR: {title[:30]} - {str(e)[:60]}")
                total_skipped += 1

        conn.commit()

    conn.close()
    print(f"\n结果: {total_new} 新增, {total_skipped} 跳过")

if __name__ == "__main__":
    main()
