#!/usr/bin/env python3
"""
独山子区 - 生态环境信息公开
https://www.dsz.gov.cn/dsz/sthj/zfxxgk.shtml
政府信息公开平台 - API直取搜索接口: /search/{channelId}?page=N&_pageSize=15
channelId = dc418fb236024fbf81277f159a17ada5
"""
import sys, os, re, time, json, sqlite3
from datetime import datetime, timezone, timedelta
import requests, urllib3
import os
from bs4 import BeautifulSoup
urllib3.disable_warnings()

SITE_NAME = "独山子生态环境"
LIST_URL = "https://www.dsz.gov.cn/search/{channelId}?page={page}&_pageSize=15&_isAgg=true&_isJson=true&_template=index&_rangeTimeGte=&_channelName="
CHANNEL_ID = "dc418fb236024fbf81277f159a17ada5"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5  # 用户策略: 最多5页
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/125.0.0.0"}

def fetch_api(page):
    url = LIST_URL.format(channelId=CHANNEL_ID, page=page)
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        return r.json()
    except Exception as e:
        print(f"\n❌ API请求失败 page={page}: {e}", flush=True)
        return None

def fetch_detail(url):
    """抓详情页正文 —— API 的 content 字段是【平铺无分段】的，
    网页 #zoomcon 里才是带 <p> 分段的原文（实测 10 个 <p> → 5 段）。
    失败返回 None，调用方回落 API content。
    """
    try:
        r = requests.get(url, headers=HEADERS, timeout=25, verify=False)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        box = soup.find(id="zoomcon")
        if not box:
            return None
        parts = []
        for p in box.find_all("p"):
            t = p.get_text(strip=True)
            t = re.sub(r'[\ue000-\uf8ff]', '', t).strip()       # 图标字体私用区
            if not t:
                continue
            if re.match(r'^(索引号|有效性|发文机关|文号|成文日期|发布日期|关键字)[：:]', t):
                continue
            if re.match(r'^(日期|来源|作者|访问量)[：:]', t) and len(t) < 60:
                continue
            if p.find("br"):                                       # <br> 还原换行
                t = "\n".join(s for s in p.get_text("\n", strip=True).split("\n") if s.strip())
            parts.append(t)
        return "\n\n".join(parts) if parts else None
    except Exception as e:
        print(f"   ⚠️ 详情页失败 {url}: {e}")
        return None


def insert_to_db(items):
    if not items:
        return
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, skip = 0, 0
    for item in items:
        try:
            db.execute("INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, summary, content, status, category, tags) "
                "VALUES (?,?,?,?,?,?,?,?,?,?)", (
                item.get("site_name","")[:200], item.get("url",""),
                item.get("url",""), (item.get("title") or "")[:500],
                (item.get("pub_date") or "")[:10],
                (item.get("summary") or "")[:500],
                item.get("content",""), "active", "",
                (item.get("tags") or "")[:100],
            ))
            if db.total_changes > 0: ok += 1
            else: skip += 1
        except: skip += 1
    db.commit()
    site_name = items[0].get("site_name", "")
    db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary FROM gov_raw r "
        "WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?", (site_name,))
    db.commit()
    db.close()
    print(f"\n💾 入库: 新增{ok}, 跳过{skip}")

def main(incremental=False):
    t0 = time.time()
    print(f"\n{SITE_NAME}\n{'='*40}")
    
    page_limit = 1 if incremental else MAX_PAGES
    total_records = 0
    
    for page in range(1, page_limit + 1):
        print(f"\n📄 第{page}页...", end=" ", flush=True)
        data = fetch_api(page)
        if not data or not data.get('data'):
            print("❌ 无数据")
            break
        
        d = data['data']
        results = d.get('results', [])
        if not results:
            print("空结果")
            break
        
        print(f"📊 {len(results)} 条 (共{d.get('total',0)}条)")
        total_records = d.get('total', 0)
        
        # 入库
        items = []
        for item in results:
            title = item.get('title', '').strip()
            url = item.get('url', '').strip()
            pub_date = (item.get('publishedTimeStr') or '')[:10]
            content = item.get('content', '') or ''
            
            if not title or not url:
                continue
            if pub_date and pub_date < CUTOFF:
                continue
            
            # 清理Content - 已有HTML内容
            content_clean = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.S|re.I)
            content_clean = re.sub(r'<style[^>]*>.*?</style>', '', content_clean, flags=re.S|re.I)
            # HTML 实体解码（API 里残留 &ensp; 等）
            try:
                from html import unescape as _unesc
                content_clean = _unesc(content_clean)
                content_clean = re.sub(r'[\u00a0\u2000-\u200a\u3000]+', ' ', content_clean)
            except Exception:
                pass

            # ★ 2026-09-17 修：API content 无分段 → 改抓详情页 #zoomcon 的 <p> 分段原文
            detail = fetch_detail(url)
            if detail and len(detail) > len(content_clean) * 0.4:
                content_clean = detail
            else:
                content_clean = content_clean.strip()

            summary = re.sub(r'<[^>]+>', '', content_clean)[:200].strip()
            
            # 修复URL: API返回http://, 实际支持https://
            if url.startswith('http://'):
                url = 'https://' + url[7:]
            
            items.append({"site_name": SITE_NAME, "title": title,
                "url": url, "content": content_clean, "summary": summary,
                "pub_date": pub_date, "tags": SITE_NAME})
        
        if items:
            insert_to_db(items)
        
        # 判断是否还有下一页
        page_size = 15
        if page * page_size >= total_records:
            break
        
        time.sleep(0.5)
    
    print(f"\n✅ 完成! 共扫描 {min(page_limit, -(-total_records // 15))} 页\n⏱ {time.time()-t0:.1f}s")

if __name__ == "__main__":
    incremental = len(sys.argv) > 1 and sys.argv[1] in ("1", "--incremental")
    main(incremental=incremental)
