#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""仪征市化学工业园区-通知公告 (Hanweb CMS, API须--resolve)"""
import re, os, subprocess, json, sqlite3, urllib.parse
from datetime import datetime, timedelta

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "仪征化工园区-通知公告"
BASE = "https://yizheng.yangzhou.gov.cn"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
RESOLVE = "yizheng.yangzhou.gov.cn:443:123.6.81.35"

def curl(url, timeout=15):
    cmd = ["curl", "-sL", "--max-time", str(timeout), "--resolve", RESOLVE, url, "-H", "User-Agent: Mozilla/5.0"]
    r = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout+5)
    return r.stdout

def fetch_list(page_no=1):
    params = {
        "parseType": "bulidstatic",
        "webId": "y6CgeNe9yhPDWezyQsTp8",
        "tplSetId": "I1p2W2qXfR0Sxtl3xljQj",
        "pageType": "column",
        "tagId": "列表一list",
        "editType": "null",
        "pageId": "8sF38ay0nwtQcANIwKF8z",
        "paramJson": json.dumps({"pageNo": page_no, "pageSize": 15}, ensure_ascii=False)
    }
    qs = "&".join("{}={}".format(k, urllib.parse.quote(str(v))) for k, v in params.items())
    url = BASE + "/api-gateway/jpaas-publish-server/front/page/build/unit?" + qs
    resp = curl(url)
    try:
        d = json.loads(resp)
        html = d["data"]["html"]
    except:
        return []
    
    items = []
    for m in re.finditer(
        r'<a[^>]*href="([^"]+)"[^>]*target="_blank">([^<]+)</a>\s*<span>(\d{4}-\d{2}-\d{2})',
        html
    ):
        items.append((m.group(2).strip(), BASE + m.group(1), m.group(3)))
    
    # Get total pages from first page
    if page_no == 1:
        cm = re.search(r'count="(\d+)"', html)
        if cm:
            total = int(cm.group(1))
            return items, (total + 14) // 15, total
    
    return items, None, None

def fetch_detail(url):
    html = curl(url, timeout=20)
    
    title = ""
    m = re.search(r'<title>([^<]+)</title>', html)
    if m: title = m.group(1).strip()
    
    content = "正文为空"
    m = re.search(r'<div class="bt-content zoom[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if m:
        c = m.group(1).strip()
        if len(c) > 50:
            content = c
    
    if content != "正文为空":
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
        content = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content, flags=re.DOTALL)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
        content = content.strip()
        if not content: content = "正文为空"
    
    date = ""
    m = re.search(r'PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if m: date = m.group(1)
    
    title = re.sub(r'<[^>]+>', '', title).strip()
    return title, date, content

def main():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    
    items, total_pages, total_count = fetch_list(1)
    if total_pages is None:
        print("Failed to get page 1")
        return
    
    print("Total: {} records, {} pages".format(total_count, total_pages))
    all_items = items
    
    for page in range(2, total_pages + 1):
        more, _, _ = fetch_list(page) or ([], None, None)
        all_items.extend(more)
    
    print("Total items: {}".format(len(all_items)))
    
    total_new = 0
    total_skipped = 0
    
    for title, url, date in all_items:
        if date < CUTOFF:
            total_skipped += 1
            continue
        
        cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if cur.fetchone():
            total_skipped += 1
            continue
        
        try:
            detail_title, pub_date, content = fetch_detail(url)
            if content == "正文为空":
                total_skipped += 1
                continue
            
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, url, detail_title or title, pub_date or date, content, int(date.replace("-", "")), "通知公告")
            )
            if cur.rowcount > 0:
                total_new += 1
        except Exception as e:
            print("  ERR: {} - {}".format(title[:30], str(e)[:60]))
            total_skipped += 1
    
    conn.commit()
    conn.close()
    print("\n结果: {} 新增, {} 跳过".format(total_new, total_skipped))

if __name__ == "__main__":
    main()
