#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""鹤壁宝山经济技术开发区-通知公告 (汉武CMS API)"""
import json, urllib.request, re, os, ssl
from datetime import datetime, timedelta

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "鹤壁宝山经开区-通知公告"
BASE = "https://bsq.hebi.gov.cn"
API = BASE + "/api-gateway/jpaas-publish-server/front/page/build/unit"
API_PARAMS = "?parseType=bulidstatic&webId=cd99a5a2bcc74862841e2b1ed4d8aec0&tplSetId=7ff27a02566e44adbca78834be229727&pageType=column&tagId=%E6%A0%8F%E7%9B%AElist&editType=null&pageId=62916237c54f4d679c7bc7495fca63c3"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch_list(page=1):
    url = API + API_PARAMS
    if page > 1:
        url += "&page=" + str(page)
    req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
    resp = urllib.request.urlopen(req, timeout=30, context=ctx)
    data = json.loads(resp.read())
    html = data["data"]["html"]
    items = re.findall(r'<li class="clearfix">(.*?)</li>', html, re.DOTALL)
    result = []
    for item in items:
        a = re.search(r'<a href="([^"]+)"[^>]*title="([^"]*)"', item)
        s = re.search(r'<span>([^<]+)</span>', item)
        if a:
            result.append({
                "url": a.group(1),
                "title": a.group(2),
                "date": s.group(1) if s else ""
            })
    return result

def fetch_detail(url):
    full_url = url if url.startswith("http") else BASE + url
    req = urllib.request.Request(full_url, headers={"User-Agent": "Mozilla/5.0"})
    resp = urllib.request.urlopen(req, timeout=30, context=ctx)
    html = resp.read().decode("utf-8", errors="ignore")
    
    # Title
    m = re.search(r'<h1[^>]*class="title"[^>]*>([^<]+)</h1>', html)
    title = m.group(1).strip() if m else ""
    if not title:
        m = re.search(r'<title>([^<]+)</title>', html)
        title = m.group(1).strip() if m else ""
    
    # Content
    content = "正文为空"
    m = re.search(r'class="zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        c = m.group(1).strip()
        if len(c) > 50:
            content = c
    
    if content == "正文为空":
        m = re.search(r'class="main content"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
        if m:
            c = m.group(1).strip()
            if len(c) > 50:
                content = c
    
    if content != "正文为空":
        content = re.sub(r'<div[^>]*style="[^"]*display:\s*none[^"]*"[^>]*>.*?</div>', '', content, flags=re.DOTALL)
        content = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content, flags=re.DOTALL)
        # Remove script tags
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
        content = content.strip()
        if not content:
            content = "正文为空"
    
    # Date from meta PubDate
    date = ""
    m = re.search(r'PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if m:
        date = m.group(1)
    else:
        m = re.search(r'(\d{4}年\d{1,2}月\d{1,2}日)', html)
        if m:
            d = m.group(1)
            d = d.replace("年", "-").replace("月", "-").replace("日", "")
            date = d
    
    return title, date, content

def main():
    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    conn.execute("PRAGMA journal_mode=WAL")
    
    total_new = 0
    total_skipped = 0
    page = 1
    
    while True:
        items = fetch_list(page)
        if not items:
            break
        
        page_new = 0
        for item in items:
            date = item["date"]
            if date < CUTOFF:
                total_skipped += 1
                continue
            
            page_url = item["url"] if item["url"].startswith("http") else BASE + item["url"]
            
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,))
            if cur.fetchone():
                total_skipped += 1
                continue
            
            try:
                title, pub_date, content = fetch_detail(item["url"])
                if content == "正文为空":
                    total_skipped += 1
                    continue
                
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                    (SITE_NAME, page_url, title or item["title"], pub_date or date, content, int(date.replace("-","")), "通知公告")
                )
                if cur.rowcount > 0:
                    total_new += 1
                    page_new += 1
                    print(f"  [{total_new}] {item['title'][:40]} | {date}")
                else:
                    total_skipped += 1
            except Exception as e:
                print(f"  [ERR] {item['title'][:30]}: {e}")
                total_skipped += 1
        
        conn.commit()
        print(f"  Page {page} done: {page_new} new (total {total_new})")
        
        # If page 1 had fewer items than typical page size, no more pages
        if len(items) < 12:
            break
        page += 1
        
        # Safety limit
        if page > 30:
            break
    
    conn.close()
    print(f"\n结果: {total_new} 新增, {total_skipped} 跳过")

if __name__ == "__main__":
    main()
