#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""乌兰察布察哈尔高新区-通知公告 (tzgg) - uses curl for API"""
import json, subprocess, re, os, sys
from datetime import datetime, timedelta

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "乌兰察布察哈尔高新区-通知公告"
BASE = "https://che.wulanchabu.gov.cn"
API = BASE + "/rmt/tmpl/api/site/467045ff03994199b783c61da198b941/channelid/5b95810da38f4254b57e6d45e4db895d"
PARAMS = "5186ec152b6fdbf8f9925254821388e35f569179f80020ffd8a62f429078cef3261c7fef0b91d53fa73a3c6163670aea718fbf0304bac398d71847f294dca445cbec7fd627bdfb3fdca3f1eee07c5564bc35faadc6215e8761f16b528cff19678232098cc8d525003f03c4556dd2c379ff438b917bbbccd91e2ac3c1f256efe701ecd379aae27643ae4ed1342128b7eeb4d5878b8cba9868df5ac4ac63ebd6e8d34f4fcc60ae1cce16d06efef6b1b3ee96239a410222a06b02dcab0b8bf1069cacbc31875c748b7ab014547e9731b1fd057da8f919dd244d53714a5a8a185be89faa83f9f9e1581a8a1bbae5f9b8576f43cec72bcb6ab99b53a0e053830a993026175cb57f5554ec136ae9e6f4e092aa75f35a73a6b61116ecea4c9e56a73ac474804d9955c2a677f599be3a347289c2"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def curl(url, data=None):
    cmd = ["curl", "-s", "--max-time", "30"]
    if data:
        cmd += ["-H", "Content-Type: application/json;charset=utf-8"]
        cmd += ["-H", "User-Agent: Mozilla/5.0"]
        cmd += ["-d", data]
    else:
        cmd += ["-H", "User-Agent: Mozilla/5.0"]
    cmd.append(url)
    result = subprocess.run(cmd, capture_output=True, text=True, timeout=35)
    return result.stdout

def fetch_list(page=1):
    body = json.dumps({"currpage": page, "pagesize": 20, "params": PARAMS})
    raw = curl(API, body)
    if not raw:
        return []
    try:
        data = json.loads(raw)
        return data.get("data", [])
    except:
        return []

def fetch_detail(docno):
    url = f"{BASE}/tzgg/{docno}.html"
    html = curl(url)
    
    m = re.search(r'<h1[^>]*>([^<]+)</h1>', html)
    title = m.group(1).strip() if m else ""
    if not title:
        m = re.search(r'<title>([^<]+)</title>', html)
        title = m.group(1).strip() if m else ""
        title = re.sub(r'\s*-\s*乌兰察布察哈尔高新技术开发区\s*$', '', title)
    
    content = "正文为空"
    for selector in ["article_content", "article-con", "content", "text", "article"]:
        m = re.search(r'class="' + selector + '"[^>]*>(.*?)</div>', html, re.DOTALL)
        if m:
            c = m.group(1).strip()
            if len(c) > 50:
                content = c
                break
    
    if content == "正文为空":
        for id_name in ["content", "article", "maintext", "zoom"]:
            m = re.search(r'id="' + id_name + '"[^>]*>(.*?)</div>', html, re.DOTALL)
            if m:
                c = m.group(1).strip()
                if len(c) > 50:
                    content = c
                    break
    
    if content != "正文为空":
        content = re.sub(r'<div[^>]*style="[^"]*display:\s*none[^"]*"[^>]*>.*?</div>', '', content, flags=re.DOTALL)
        content = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content, flags=re.DOTALL)
        content = content.strip()
        if not content: content = "正文为空"
    
    date = ""
    m = re.search(r'(?:发布时间|发布日期)[：:]\s*(\d{4}[-/]\d{2}[-/]\d{2})', html)
    if m: date = m.group(1).replace("/", "-")
    if not date:
        m = re.search(r'(\d{4}-\d{2}-\d{2})\s*\d{2}:\d{2}', html)
        if m: date = m.group(1)
    
    return title, date, content

def main():
    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    conn.execute("PRAGMA journal_mode=WAL")
    
    items_p1 = fetch_list(1)
    total_count = 256
    total_pages = (total_count + 19) // 20
    print(f"Total pages: {total_pages}")
    total_new = 0
    total_skipped = 0
    
    for page in range(1, total_pages + 1):
        items = fetch_list(page)
        if not items:
            print(f"  Page {page}: empty response, stopping")
            break
        
        page_new = 0
        for item in items:
            title = item.get("title", "").strip()
            docno = item.get("docno", "")
            ts = item.get("inputTime", 0)
            if ts:
                date = datetime.fromtimestamp(ts / 1000).strftime("%Y-%m-%d")
            else:
                date = ""
            
            if not title or not docno:
                continue
            if date and date < CUTOFF:
                total_skipped += 1
                continue
            
            page_url = f"{BASE}/tzgg/{docno}.html"
            
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,))
            if cur.fetchone():
                total_skipped += 1
                continue
            
            try:
                detail_title, pub_date, content = fetch_detail(docno)
                if content == "正文为空" or not content:
                    total_skipped += 1
                    continue
                
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                    (SITE_NAME, page_url, detail_title or title, pub_date or date, content, int(date.replace("-","")), "通知公告")
                )
                if cur.rowcount > 0:
                    total_new += 1
                    page_new += 1
                    print(f"  P{page}[{total_new}] {title[:35]} | {date}")
                else:
                    total_skipped += 1
            except Exception as e:
                print(f"  P{page}[ERR] {title[:30]}: {e}")
                total_skipped += 1
        
        # Commit after each page
        conn.commit()
        print(f"  Page {page} done: {page_new} new (total {total_new})")
    
    conn.close()
    print(f"\n结果: {total_new} 新增, {total_skipped} 跳过")

if __name__ == "__main__":
    main()
