#!/usr/bin/env python3
"""福建三钢集团-环境信息公开 爬虫
数据来源：www.fjsg.com.cn 列表页JSON内嵌（含完整正文）
"""
import json
import re
import os
import sys
import requests

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "福建三钢集团-环境信息公开"
BASE_URL = "https://www.fjsg.com.cn"
LIST_URL = f"{BASE_URL}/sgweb/lists?topicId=116"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}


def extract_json_data(html):
    """从HTML中提取 var cl = [...] JSON 数据"""
    m = re.search(r'var cl = (\[.*?\])\s*;?\s*\n', html, re.DOTALL)
    if m:
        raw_json = m.group(1)
        # Some unicode escapes might need cleanup
        return json.loads(raw_json)
    return []


def parse_page(page=1):
    """获取并解析单页数据，返回 [(cont_id, title, content, issue_date), ...]"""
    url = LIST_URL if page == 1 else f"{LIST_URL}&page={page}"
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
    except Exception as e:
        print(f"  ❌ 第{page}页请求失败: {e}")
        return [], 0

    # Extract total count
    tt = 0
    m = re.search(r'var tt = (\d+)', resp.text)
    if m:
        tt = int(m.group(1))

    items = extract_json_data(resp.text)
    results = []
    for item in items:
        cont_id = item.get("cont_id", "")
        title = item.get("title", "")
        content = item.get("content", "")
        issue_date = item.get("issue_date", "")
        results.append((cont_id, title, content, issue_date))

    return results, tt


def crawl():
    conn = None
    try:
        import sqlite3
        conn = sqlite3.connect(SEARCH_DB, timeout=60)
        c = conn.cursor()
        c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            title TEXT,
            content TEXT,
            source_url TEXT UNIQUE,
            site_name TEXT,
            publish_date TEXT
        )''')

        # Get first page to determine total
        first_items, total_items = parse_page(1)
        if total_items == 0:
            print("❌ 无法获取总条数")
            return

        page_size = 9
        total_pages = (total_items + page_size - 1) // page_size
        print(f"总条数: {total_items}, 总页数: {total_pages}")

        # Collect all items across pages
        all_items = {}  # cont_id -> (title, content, issue_date)
        seen_ids = set()

        for page in range(1, total_pages + 1):
            items, _ = parse_page(page) if page > 1 else (first_items, total_items)
            for cont_id, title, content, issue_date in items:
                if cont_id in seen_ids:
                    continue  # Skip duplicates across pages
                seen_ids.add(cont_id)
                detail_url = f"{BASE_URL}/sgweb/detail?contId={cont_id}"
                all_items[cont_id] = (title, content, issue_date, detail_url)
            print(f"第{page}页: {len(items)}条 (累积 {len(all_items)}/{total_items})")

        print(f"\n共获取 {len(all_items)} 条列表项")

        # Now insert into DB
        total_new = 0
        total_skip = 0
        sorted_ids = sorted(all_items.keys(), reverse=True)  # newest first

        for cont_id in sorted_ids:
            title, content, issue_date_str, detail_url = all_items[cont_id]

            # Check if already exists
            c.execute("SELECT id FROM gov_raw WHERE source_url = ?", (detail_url,))
            if c.fetchone():
                total_skip += 1
                continue

            # Parse publish date
            publish_date = ""
            if issue_date_str:
                m = re.match(r'(\d{4}-\d{2}-\d{2})', issue_date_str)
                if m:
                    publish_date = m.group(1)

            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                    (title, content, detail_url, SITE_NAME, publish_date)
                )
                conn.commit()
                total_new += 1
                content_len = len(content) if content else 0
                status = " ⚠️ 内容过短" if content_len < 200 else ""
                print(f"  ✅ {title[:40]}... ({publish_date}){status}")
            except Exception as e:
                print(f"  ❌ 入库失败 {title[:30]}: {e}")

        print(f"\n爬取完成！新增: {total_new}, 跳过(已存在): {total_skip}")
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
        total = c.fetchone()[0]
        print(f"  共 {total} 条")

    finally:
        if conn:
            conn.close()


if __name__ == "__main__":
    crawl()
