#!/usr/bin/env python3
"""
济南市生态环境局 - 已批准项目公告
https://jnepb.jinan.gov.cn/col/col115241/index.html

JPaaS CMS 系统，API 翻页 + 标准详情页
"""
import sys, os, re, time, json
sys.path.insert(0, '/root/gov_crawler')
from crawler_lib import fetch_page, clean_html
import sqlite3
import os

SITE_NAME = "济南市生态环境局"
BASE = "https://jnepb.jinan.gov.cn"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
THREE_YEARS_AGO = "2023-06-01"

# JPaaS API 参数（从页面脚本标签中提取）
API_URL = f"{BASE}/api-gateway/jpaas-publish-server/front/page/build/unit"
API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "21",
    "tplSetId": "pMB9pkaz19HYYbLAOTZtY",
    "pageType": "column",
    "tagId": "信息列表",
    "editType": "null",
    "pageId": "115241",
}
TOTAL_PAGES = 122  # 共122页
PAGE_SIZE = 15

def fetch_list_page(page_no):
    """调用JPaaS API获取第N页列表HTML"""
    params = dict(API_PARAMS)
    params["pageNo"] = str(page_no)
    try:
        import requests
        r = requests.get(API_URL, params=params, timeout=30,
            headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"})
        if r.status_code == 200:
            data = r.json()
            if data.get("success") and data.get("data", {}).get("html"):
                return data["data"]["html"]
        return None
    except Exception as e:
        print(f"    ⚠ API请求异常: {e}")
        return None

def parse_list(html):
    """从API返回的HTML中提取 (title, url, date_str)"""
    items = []
    # <li><a href="/col/col115241/art/2026/art_xxx.html" title="完整标题" target="_blank">截断标题</a><span>[YYYY-MM-DD]</span></li>
    pattern = re.compile(
        r'<li>.*?<a\s+href="(/[^"]+)"\s+title="([^"]+)"[^>]*>.*?</a>\s*<span[^>]*>\[(\d{4}-\d{2}-\d{2})\].*?</li>',
        re.DOTALL
    )
    for m in pattern.finditer(html):
        href = m.group(1).strip()
        title = m.group(2).strip()
        date_str = m.group(3)
        if href and title:
            items.append((title, href, date_str))
    return items

def fetch_detail(url):
    """获取详情页的正文内容"""
    full_url = BASE + url if url.startswith('/') else url
    html = fetch_page(full_url, encoding='utf-8')
    if not html:
        return None, None, None

    # 发布日期
    pub_date = ""
    m = re.search(r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if m:
        pub_date = m.group(1)

    # 正文 — <div id="zoom">
    content = ""
    m = re.search(r'<div\s+id="zoom"[^>]*>(.*?)</div>\s*', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    if not content:
        # fallback: 正文区域
        m = re.search(r'<div\s+id="zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
        if m:
            content = m.group(1).strip()

    if content:
        # 移除无用标签
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
        # 保留结构化 HTML，仅清理 style/class
        content = clean_html(content)

        # 提取附件链接
        attachments = []
        for att_href, att_text in re.findall(
            r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>\s*([^<]+)\s*</a>',
            html, re.IGNORECASE
        ):
            att_full = BASE + att_href if att_href.startswith('/') else att_href
            attachments.append(f'<a href="{att_full}" target="_blank">{att_text}</a>')
        # API-gateway 下载链接
        for dl_href, dl_text in re.findall(
            r'<a[^>]*href="(/api-gateway[^"]*download[^"]*)"[^>]*>\s*([^<]+)\s*</a>', html, re.IGNORECASE
        ):
            att_full = BASE + dl_href if dl_href.startswith('/') else dl_href
            attachments.append(f'<a href="{att_full}" target="_blank">{dl_text}</a>')

        if attachments:
            content += '\n<p><strong>附件：</strong></p>\n' + '\n'.join(f'<p>{a}</p>' for a in attachments)

    # 标题
    title = ""
    m = re.search(r'<div\s+class="news-title"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not m:
        m = re.search(r'<div\s+class="article-title"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        title = re.sub(r'<[^>]+>', '', m.group(1)).strip()

    return title or None, content or None, pub_date or None


def insert_to_db(items):
    if not items:
        print("  ⏭ 无数据")
        return
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, skip = 0, 0
    for item in items:
        try:
            db.execute("INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, summary, content, status, category, tags) "
                "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", (
                item.get("site_name","")[:200],
                item.get("url",""),
                item.get("url",""),
                (item.get("title") or "")[:500],
                (item.get("pub_date") or "")[:10],
                (item.get("summary") or "")[:500],
                item.get("content",""),
                "active", "", (item.get("tags") or "")[:100],
            ))
            if db.total_changes > 0:
                ok += 1
            else:
                skip += 1
        except:
            skip += 1
    db.commit()
    db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary FROM gov_raw r "
        "WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?", (SITE_NAME,))
    db.commit()
    db.close()
    print(f"  💾 入库: 新增{ok}, 跳过{skip}, FTS已同步")

def main():
    print(f"\n{'='*50}")
    print(f"🏠 {SITE_NAME} - 已批准项目公告")
    print(f"{'='*50}")

    all_items = []
    seen_urls = set()

    # 只爬最近3页（已批准项目公告通常最相关的在最新几页）
    # 如需全量，改 range(1, TOTAL_PAGES+1)
    MAX_PAGES = 3  # 增量用，每天只跑前3页
    if len(sys.argv) > 1 and sys.argv[1] == "full":
        MAX_PAGES = TOTAL_PAGES
        print("📋 全量模式：122页")

    for page_no in range(1, MAX_PAGES + 1):
        print(f"  📄 第 {page_no}/{MAX_PAGES} 页...", end=" ", flush=True)
        html = fetch_list_page(page_no)
        if not html:
            print("❌ 无响应")
            break
        items = parse_list(html)
        if not items:
            print("0 条（可能已到底）")
            break
        print(f"✅ {len(items)} 条")

        for title, href, date_str in items:
            full_url = BASE + href if href.startswith('/') else href
            if full_url in seen_urls:
                continue
            seen_urls.add(full_url)

            if date_str and date_str < THREE_YEARS_AGO:
                continue

            print(f"    {title[:55]}...", end=" ", flush=True)
            dt, content, pub_date = fetch_detail(href)
            if dt:
                title = dt
            if pub_date:
                date_str = pub_date

            summary = re.sub(r'<[^>]+>', ' ', content or '').strip()[:300]
            summary = re.sub(r'\s+', ' ', summary)

            all_items.append({
                "site_name": SITE_NAME,
                "title": title,
                "url": full_url,
                "content": content or "",
                "pub_date": date_str or "",
                "summary": summary,
                "tags": "环评审批",
            })
            print(f"✅ {date_str}")
            time.sleep(0.3)

        if len(items) < PAGE_SIZE:
            print("    (末页)")
            break
        time.sleep(0.5)

    if all_items:
        insert_to_db(all_items)
    print(f"\n✅ 完成! 共 {len(all_items)} 条")


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
