#!/usr/bin/env python3
"""
泰州医药高新区（高港区） — 通知公告
Hanweb CMS AJAX (unitbuild API)
https://www.cmc.gov.cn/xwzx/tzgg/index.html
"""
import json
import re
import os
import sys
import requests
import sqlite3
from datetime import datetime, timedelta
from urllib.parse import urljoin

SITE_NAME = "泰州医药高新区（高港区）人民政府"
BASE_URL = "https://www.cmc.gov.cn"
COLUMN_URL = BASE_URL + "/xwzx/tzgg/index.html"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
CUTOFF_DATE = "2023-06-18"

# 通过--resolve绕过DNS解析问题
RESOLVE = "www.cmc.gov.cn:443:218.90.228.239"

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

DATE_RANK_BASE = 20260601  # 用于date_rank字段

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "application/json, text/plain, */*",
    "Referer": COLUMN_URL,
}

# 基础API参数（从页面提取）
BASE_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "5f26120115824722ba2229d0ceaa65b2",
    "tplSetId": "158d93f47443484c9ef19d54c1ab9ab2",
    "pageType": "column",
    "tagId": "信息列表",
    "editType": "null",
    "pageId": "dE67Ga7Sg0xeYjn7E3pZ5",
}


def api_get(page_no, page_size=15, timeout=15):
    """Call the Hanweb API for a specific page."""
    params = dict(BASE_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": page_size}, ensure_ascii=False)
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=timeout)
        r.raise_for_status()
        data = r.json()
        if data.get("success"):
            return data["data"]["html"]
        else:
            print(f"  API error (page {page_no}): {data.get('message', 'unknown')}")
            return None
    except Exception as e:
        print(f"  Request failed (page {page_no}): {e}")
        return None


def parse_list(html):
    """Parse list HTML to extract items (title, url, date)."""
    items = []
    # Find each <li class="clearfix"> block
    li_pattern = re.compile(r'<li[^>]*class="clearfix"[^>]*>(.*?)</li>', re.DOTALL)
    for li in li_pattern.finditer(html):
        li_html = li.group(1)
        title_m = re.search(r'title="([^"]*)"', li_html)
        href_m = re.search(r'href="([^"]+)"', li_html)
        date_m = re.search(r'\[(\d{4}-\d{2}-\d{2})\]', li_html)
        if title_m and href_m and date_m:
            items.append({
                "title": title_m.group(1),
                "url": urljoin(BASE_URL, href_m.group(1)),
                "date": date_m.group(1),
            })
    return items


def fetch_detail(url, timeout=15):
    """Fetch detail page and extract title, date, content."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=timeout)
        r.raise_for_status()
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        return {"title": "", "date": "", "content": "", "error": str(e)}

    # Title from <title>
    title_m = re.search(r'<title>([^<]*)</title>', html)
    title = title_m.group(1).strip() if title_m else ""

    # Date from <meta PubDate>
    date_m = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    date = date_m.group(1) if date_m else ""

    # Content from div#zoom
    content = ""
    zoom_m = re.search(r'<div\s+id="zoom"[^>]*>(.*?)</div>\s*</div>\s*</div>', html, re.DOTALL)
    if zoom_m:
        content = zoom_m.group(1).strip()
    else:
        # Fallback: find div#zoom more broadly
        zoom_m = re.search(r'<div\s+id="zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
        if zoom_m:
            content = zoom_m.group(1).strip()

    if not content:
        return {"title": title, "date": date, "content": "", "error": "empty content"}

    return {"title": title, "date": date, "content": content}


def insert_db(items):
    """Insert items into search.db."""
    if not items:
        print("  无数据插入")
        return 0
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    inserted = 0
    for item in items:
        try:
            # summary: strip tags for FTS search
            clean_text = re.sub(r'<[^>]+>', '', item["content"]).strip()
            # date_rank: 用于排序
            try:
                dt = datetime.strptime(item["date"], "%Y-%m-%d")
                date_rank = int(dt.strftime("%Y%m%d"))
            except:
                date_rank = 0
            c.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (title, page_url, site_name, summary, content, publish_date)
                   VALUES (?, ?, ?, ?, ?, ?)""",
                (item["title"], item["url"], SITE_NAME, clean_text[:500],
                 item["content"], item["date"]),
            )
            if c.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  DB insert error: {e}")
    conn.commit()
    conn.close()
    return inserted


def rebuild_fts():
    """Rebuild FTS index."""
    try:
        conn = sqlite3.connect(DB_PATH)
        c = conn.cursor()
        c.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
        conn.commit()
        conn.close()
        print("  FTS 重建完成")
    except Exception as e:
        print(f"  FTS 重建失败: {e}")


def main():
    print(f"=== 开始爬取 {SITE_NAME} - 通知公告 ===")
    print(f"截止日期: {CUTOFF_DATE}")

    # Step 1: Get first page to find total count
    html = api_get(1)
    if not html:
        print("  ❌ 无法获取第一页数据")
        return

    # Extract total count
    count_m = re.search(r'count="(\d+)"', html)
    total_count = int(count_m.group(1)) if count_m else 0
    page_size = 15
    total_pages = (total_count + page_size - 1) // page_size
    print(f"  总数: {total_count}, 总页数: {total_pages}")

    # Parse first page items
    all_items = parse_list(html)
    last_date = all_items[-1]["date"] if all_items else ""

    # Step 2: Paginate through remaining pages
    for page_no in range(2, total_pages + 1):
        html = api_get(page_no)
        if not html:
            continue
        items = parse_list(html)
        if items:
            all_items.extend(items)
            last_date = items[-1]["date"]
            # Check if we've passed the cutoff
            if last_date < CUTOFF_DATE:
                # This page might have mixed dates; continue to be safe
                pass
        # Progress every 10 pages
        if page_no % 10 == 0:
            print(f"  已扫描 {page_no}/{total_pages} 页, 累计 {len(all_items)} 条")

    print(f"  列表扫描完成: {len(all_items)} 条")

    # Step 3: Filter by date (last 3 years)
    filtered = [item for item in all_items if item["date"] >= CUTOFF_DATE]
    print(f"  3年内: {len(filtered)} 条")

    # Step 4: Fetch detail pages
    success = []
    skipped_empty = 0
    skipped_error = 0
    for i, item in enumerate(filtered):
        detail = fetch_detail(item["url"])
        if detail.get("error"):
            if "empty content" in detail["error"]:
                skipped_empty += 1
            else:
                skipped_error += 1
            if (i + 1) % 50 == 0:
                print(f"  详情页进度: {i+1}/{len(filtered)} (空正文: {skipped_empty}, 错误: {skipped_error})")
            continue
        success.append({
            "title": detail["title"] or item["title"],
            "url": item["url"],
            "date": detail["date"] or item["date"],
            "content": detail["content"],
        })
        if (i + 1) % 50 == 0:
            print(f"  详情页进度: {i+1}/{len(filtered)} (成功: {len(success)}, 空正文: {skipped_empty})")

    print(f"  详情页完成: 成功 {len(success)}, 空正文跳过 {skipped_empty}, 错误 {skipped_error}")

    # Step 5: Insert into DB
    inserted = insert_db(success)
    print(f"  入库: {inserted} 条")

    # Step 6: Rebuild FTS
    rebuild_fts()

    print(f"=== 完成: 共 {len(success)} 条成功, 新增 {inserted} 条 ===")


if __name__ == "__main__":
    main()
