#!/usr/bin/env python3
import os
"""
万载县人民政府 - 建设项目环境影响评价
https://www.wanzai.gov.cn/wzxrmzf/jsxmhjyxpj/pc/list.html
政府信息公开平台 (数融) - API: POST /searchManuscript
channelId = 1996822173703577600
"""
import sys, os, re, time, json, sqlite3
from datetime import datetime, timezone, timedelta
import requests, urllib3
urllib3.disable_warnings()

SITE_NAME = "万载环评"
API_URL = "http://www.wanzai.gov.cn/searchManuscript"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
PAGE_SIZE = 15

HEADERS = {
    "Content-Type": "application/json",
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/125.0.0.0",
    "Referer": "http://www.wanzai.gov.cn/wzxrmzf/jsxmhjyxpj/pc/list.html",
}

def fetch_api(page):
    body = {
        "current": page,
        "pageSize": PAGE_SIZE,
        "channelTreeIds": ["1996822173703577600"],
        "title": "",
        "Contenthtml": "",
        "isSearchHighlight": False
    }
    try:
        r = requests.post(API_URL, json=body, headers=HEADERS, timeout=20)
        return r.json().get('data', {})
    except Exception as e:
        print(f"\n❌ API请求失败 page={page}: {e}", flush=True)
        return None

def insert_to_db(items):
    if not items:
        return
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, skip = 0, 0
    for item in items:
        try:
            db.execute("INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, summary, content, status, category, tags) "
                "VALUES (?,?,?,?,?,?,?,?,?,?)", (
                item.get("site_name","")[:200], item.get("url",""),
                item.get("url",""), (item.get("title") or "")[:500],
                (item.get("pub_date") or "")[:10],
                (item.get("summary") or "")[:500],
                item.get("content",""), "active", "",
                (item.get("tags") or "")[:100],
            ))
            if db.total_changes > 0: ok += 1
            else: skip += 1
        except: skip += 1
    db.commit()
    site_name = items[0].get("site_name", "")
    db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary FROM gov_raw r "
        "WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?", (site_name,))
    db.commit()
    db.close()
    print(f"\n💾 入库: 新增{ok}, 跳过{skip}")

def main(incremental=False):
    t0 = time.time()
    print(f"\n{SITE_NAME}\n{'='*40}")
    
    page_limit = 1 if incremental else MAX_PAGES
    all_items = []
    total = 0
    
    for page in range(1, page_limit + 1):
        print(f"\n📄 第{page}页...", end=" ", flush=True)
        data = fetch_api(page)
        if not data:
            print("❌ 无响应")
            break
        
        results = data.get('results', [])
        if not results:
            print("空结果")
            break
        
        total = data.get('total', 0)
        print(f"📊 {len(results)} 条 (共{total}条)")
        
        for item in results:
            title = item.get('title', '').strip()
            pub_date = (item.get('pubDate') or '')[:10]
            url_obj = item.get('urls', {})
            # urls 是 JSON 字符串, 不是dict
            if isinstance(url_obj, str):
                try:
                    url_obj = json.loads(url_obj)
                except:
                    url_obj = {}
            url = url_obj.get('pc', '') if isinstance(url_obj, dict) else ''
            
            if not title or not url:
                continue
            if pub_date and pub_date < CUTOFF:
                continue
            
            # 完整URL
            full_url = f"http://www.wanzai.gov.cn{url}" if url.startswith('/') else url
            
            # Content 已在API返回中，HTML格式（有<p>标签）
            c = item.get('content', {})
            content_html = c.get('content', '') if isinstance(c, dict) else ''
            content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.S|re.I)
            content_html = re.sub(r'<style[^>]*>.*?</style>', '', content_html, flags=re.S|re.I)
            
            summary = re.sub(r'<[^>]+>', '', content_html)[:200].strip()
            
            all_items.append({"site_name": SITE_NAME, "title": title,
                "url": full_url, "content": content_html, "summary": summary,
                "pub_date": pub_date, "tags": SITE_NAME})
        
        # 判断是否有下一页
        if page * PAGE_SIZE >= total:
            break
        time.sleep(0.5)
    
    print(f"\n📊 列表总计: {len(all_items)} 条 (已过滤超3年)")
    
    if all_items:
        insert_to_db(all_items)
    print(f"\n✅ 完成! 共 {len(all_items)} 条\n⏱ {time.time()-t0:.1f}s")

if __name__ == "__main__":
    incremental = len(sys.argv) > 1 and sys.argv[1] in ("1", "--incremental")
    main(incremental=incremental)
