#!/usr/bin/env python3
import os
"""八师石河子市-主动公开内容 (dept=030, cat=160)"""
import json, urllib.request, urllib.parse, re, os, sys, ssl
from datetime import datetime, timedelta

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "八师石河子市-主动公开内容"
BASE_URL = "https://www.shz.gov.cn"
API_URL = "https://www.shz.gov.cn/EpointWebBuilder/rest/govenopen/getgovinfolist"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch_list(deptcode, catenum, page=0, size=20):
    params = json.dumps({"deptcode":deptcode,"categorynum":catenum,"title":"","pageIndex":page,"pageSize":size}, ensure_ascii=False)
    data = urllib.parse.urlencode({"params": params}).encode()
    req = urllib.request.Request(API_URL, data=data, headers={"Content-Type":"application/x-www-form-urlencoded","User-Agent":"Mozilla/5.0"})
    resp = urllib.request.urlopen(req, timeout=30, context=ctx)
    return json.loads(resp.read())

def fetch_detail(infourl):
    url = BASE_URL + infourl
    req = urllib.request.Request(url, headers={"User-Agent":"Mozilla/5.0"})
    resp = urllib.request.urlopen(req, timeout=30, context=ctx)
    html = resp.read().decode("utf-8", errors="ignore")
    m = re.search(r'class="article-cont">(.*?)</div>\s*(?:<!--|</div>)', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
        # Clean up but preserve basic HTML tags
        content = re.sub(r'<div[^>]*style="[^"]*display:\s*none[^"]*"[^>]*>.*?</div>', '', content, flags=re.DOTALL)
        content = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content, flags=re.DOTALL)
        content = content.strip()
        if content:
            return content
    # Fallback: extract all text from article area
    m = re.search(r'class="article-cont">(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
        # Remove hidden divs
        content = re.sub(r'<div[^>]*style="[^"]*display:\s*none[^"]*"[^>]*>.*?</div>', '', content, flags=re.DOTALL)
        content = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content, flags=re.DOTALL)
        content = content.strip()
        return content if content else "正文为空"
    return "正文为空"

def main():
    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    
    total_new = 0
    total_skipped = 0
    total_pages = 0
    page = 0
    
    while True:
        try:
            result = fetch_list("030", "160", page)
        except Exception as e:
            print(f"[ERROR] page {page}: {e}")
            break
        
        items = result["custom"]["data"]
        if not items:
            break
        
        total_pages += 1
        for item in items:
            title = item.get("title", "").strip()
            date = item.get("infodate", "").strip()
            infourl = item.get("infourl", "").strip()
            
            if not title or not infourl:
                continue
            
            # Filter by 3 years
            if date and date < CUTOFF:
                total_skipped += 1
                continue
            
            page_url = BASE_URL + infourl
            
            # Check if already exists
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,))
            if cur.fetchone():
                total_skipped += 1
                continue
            
            # Fetch detail
            content = fetch_detail(infourl)
            
            if content == "正文为空":
                total_skipped += 1
                continue
            
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, page_url, title, date, content, int(date.replace("-","")) if date else 0, "主动公开内容")
            )
            if cur.rowcount > 0:
                total_new += 1
                print(f"  [{total_new}] {title[:40]} | {date}")
            else:
                total_skipped += 1
        
        page += 1
        # Safety limit
        if page > 50:
            break
    
    conn.commit()
    conn.close()
    
    print(f"\n结果: {total_new} 新增, {total_skipped} 跳过, {total_pages} 页")

if __name__ == "__main__":
    main()
