#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""库车市-生态环境 (博山CMS, 内嵌JSON数据)"""
import re, os, urllib.request, ssl, sqlite3, json
from datetime import datetime, timedelta

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "库车市-生态环境"
BASE = "https://www.xjkc.gov.cn"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch(url, timeout=30):
    req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"})
    resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
    return resp.read().decode("utf-8", errors="ignore")

def get_all_items():
    """Extract all items from the embedded JSON in the first page"""
    html = fetch(BASE + "/zwgk/zdlyxxgk/hjbh/index.html")
    m = re.search(r'var dataList=\[(.*?)\];', html, re.DOTALL)
    if not m:
        return []
    raw = "[" + m.group(1) + "]"
    data = json.loads(raw)
    items = []
    for page_data in data:
        if isinstance(page_data, dict) and "infolist" in page_data:
            for item in page_data["infolist"]:
                items.append(item)
    return items

def fetch_detail(url):
    html = fetch(url)
    
    # Title from h1
    title = ""
    m = re.search(r'<h1[^>]*>\s*([^<]+?)\s*</h1>', html)
    if m: title = m.group(1).strip()
    if not title:
        m = re.search(r'<title>([^<]+)</title>', html)
        if m: title = m.group(1).strip()
    
    # Content from main_51231 content
    content = "正文为空"
    m = re.search(r'<div class="main_51231 content"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if m:
        c = m.group(1).strip()
        if len(c) > 50:
            content = c
    
    if content != "正文为空":
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
        content = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content, flags=re.DOTALL)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
        content = content.strip()
        if not content:
            content = "正文为空"
    
    title = re.sub(r'<[^>]+>', '', title).strip()
    return title, content

def main():
    conn = sqlite3.connect(SEARCH_DB)
    cur = conn.cursor()
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    
    items = get_all_items()
    print("Total items from JSON: {}".format(len(items)))
    
    total_new = 0
    total_skipped = 0
    
    for item in items:
        title = item.get("title", "").strip()
        url = item.get("url", "")
        daytime = item.get("daytime", "")
        
        if not title or not url:
            continue
        if daytime < CUTOFF:
            total_skipped += 1
            continue
        
        cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if cur.fetchone():
            total_skipped += 1
            continue
        
        try:
            detail_title, content = fetch_detail(url)
            if content == "正文为空":
                total_skipped += 1
                continue
            
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, url, detail_title or title, daytime, content, int(daytime.replace("-", "")), "生态环境")
            )
            if cur.rowcount > 0:
                total_new += 1
        except Exception as e:
            print("  ERR: {} - {}".format(title[:30], str(e)[:60]))
            total_skipped += 1
    
    conn.commit()
    conn.close()
    print("\n结果: {} 新增, {} 跳过".format(total_new, total_skipped))

if __name__ == "__main__":
    main()
