#!/usr/bin/env python3
"""淮安市-生态环境局信息公开 爬虫
API: POST /articleCommonController/lists.do
"""
import requests
import re
import os
import json
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "淮安市生态环境局-信息公开"
BASE_URL = "https://www.huaian.gov.cn"
API_URL = f"{BASE_URL}/articleCommonController/lists.do"
API_DATA = {
    "topic": "78",
    "pagesize": 15,
    "rdeptid": "0000000064a8f16d0164ad1d9d730006",
}
TOTAL = 163

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Content-Type": "application/x-www-form-urlencoded",
    "Referer": f"{BASE_URL}/cmsweb/zwgk/sj/index.html?type=2&rdeptid=0000000064a8f16d0164ad1d9d730006&topic=78",
}


def fetch_api(page):
    """调用POST API获取列表"""
    data = {**API_DATA, "page": str(page)}
    try:
        r = requests.post(API_URL, data=data, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        resp = r.json()
        if resp.get("status") == "1" and resp.get("value", {}).get("list"):
            return resp["value"]["list"]
        return []
    except Exception as e:
        print(f"  ❌ API失败: {e}")
        return []


def fetch_detail(url):
    """获取详情页"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        return None


def parse_detail(html):
    """解析详情页"""
    soup = BeautifulSoup(html, "html.parser")

    # Title from meta
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"]
    if not title:
        bt = soup.find("div", class_="bt")
        if bt:
            title = bt.get_text(strip=True)

    # Date from meta
    publish_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        m = re.search(r'(\d{4}-\d{2}-\d{2})', meta_date["content"])
        if m:
            publish_date = m.group(1)

    # Content from <div class="wz3">
    content = ""
    wz3 = soup.find("div", class_="wz3")
    if wz3:
        content = str(wz3)

    return title, content, publish_date


def crawl():
    conn = None
    try:
        import sqlite3
        conn = sqlite3.connect(SEARCH_DB)
        c = conn.cursor()
        c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            title TEXT,
            content TEXT,
            source_url TEXT UNIQUE,
            site_name TEXT,
            publish_date TEXT
        )''')

        total_new = 0
        total_skipped = 0
        total_pages = (TOTAL + 14) // 15  # 11

        for page in range(1, total_pages + 1):
            items = fetch_api(page)
            if not items:
                print(f"第{page}/{total_pages}页: 无数据")
                continue
            print(f"第{page}/{total_pages}页: {len(items)}条")

            for item in items:
                title = item.get("title", "")
                release_time = item.get("releaseTime", "")
                domain = item.get("domain", "")
                path = item.get("path", "")
                if not path:
                    continue
                if domain and not domain.startswith("http"):
                    domain = "http://" + domain
                detail_url = domain.rstrip("/") + "/" + path.lstrip("/")

                # Date from API
                publish_date = ""
                m = re.search(r'(\d{4}-\d{2}-\d{2})', release_time)
                if m:
                    publish_date = m.group(1)

                # Check if exists
                c.execute("SELECT id FROM gov_raw WHERE source_url = ?", (detail_url,))
                if c.fetchone():
                    total_skipped += 1
                    continue

                detail_html = fetch_detail(detail_url)
                if not detail_html:
                    print(f"  ❌ 详情失败: {title[:30]}...")
                    continue

                _, content, detail_date = parse_detail(detail_html)
                if detail_date:
                    publish_date = detail_date

                try:
                    c.execute(
                        "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                        (title, content, detail_url, SITE_NAME, publish_date or "")
                    )
                    conn.commit()
                    total_new += 1
                    content_len = len(content) if content else 0
                    status = " ⚠️ 内容过短" if content_len < 200 else ""
                    print(f"  ✅ {title[:35]}... ({publish_date}){status}")
                except Exception as e:
                    print(f"  ❌ 入库失败: {e}")

        print(f"\n爬取完成！新增: {total_new}, 跳过(已存在): {total_skipped}")
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
        total = c.fetchone()[0]
        print(f"  共 {total} 条")

    finally:
        if conn:
            conn.close()


if __name__ == "__main__":
    crawl()
