#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""库车市-公示公告 (博山CMS, 内嵌JSON数据, channelId=1121)"""
import re, os, urllib.request, ssl, sqlite3, json
from datetime import datetime, timedelta

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "库车市-公示公告"
BASE = "https://www.xjkc.gov.cn"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch(url, timeout=30):
    req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"})
    resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
    return resp.read().decode("utf-8", errors="ignore")

def get_all_items():
    """Extract all items from the embedded JSON in the first page"""
    html = fetch(BASE + "/gsgg/index.html")
    m = re.search(r"var dataList=\[(.*?)\];", html, re.DOTALL)
    if not m:
        return []
    raw = "[" + m.group(1) + "]"
    data = json.loads(raw)
    items = []
    for page_data in data:
        if isinstance(page_data, dict) and "infolist" in page_data:
            for item in page_data["infolist"]:
                items.append(item)
    return items

def fetch_detail(url):
    html = fetch(url)

    # Title from h1
    title = ""
    m = re.search(r"<h1[^>]*>\s*([^<]+?)\s*</h1>", html)
    if m: title = m.group(1).strip()
    if not title:
        m = re.search(r"<title>([^<]+)</title>", html)
        if m: title = m.group(1).strip()

    # Content from main_51231 content
    content = "正文为空"
    m = re.search(r"<div class=\"main_51231 content\"[^>]*>(.*?)</div>", html, re.DOTALL)
    if m:
        c = m.group(1).strip()
        if len(c) > 50:
            content = c

    if content != "正文为空":
        content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL)
        content = re.sub(r"<iframe[^>]*>.*?</iframe>", "", content, flags=re.DOTALL)
        content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL)
        content = content.strip()
        if not content:
            content = "正文为空"

    title = re.sub(r"<[^>]+>", "", title).strip()
    return title, content

def main():
    pages = int(os.environ.get("PAGES", "5"))
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")

    items = get_all_items()
    print("Total items from JSON: {}".format(len(items)))

    total_new = 0
    total_skipped = 0

    for item in items[:pages*15] if pages else items:
        title = item.get("title", "").strip()
        url = item.get("url", "")
        daytime = item.get("daytime", "")

        if not title or not url:
            continue
        if daytime < CUTOFF:
            total_skipped += 1
            continue

        cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if cur.fetchone():
            total_skipped += 1
            continue

        try:
            detail_title, content = fetch_detail(url)
            if content == "正文为空":
                total_skipped += 1
                continue

            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, url, detail_title or title, daytime, content, int(daytime.replace("-", "")), "政府公告")
            )
            if cur.rowcount > 0:
                total_new += 1
        except Exception as e:
            print("  ERR: {} - {}".format(title[:30], str(e)[:60]))
            total_skipped += 1

    conn.commit()
    conn.close()
    print("\n结果: {} 新增, {} 跳过".format(total_new, total_skipped))

if __name__ == "__main__":
    main()
