#!/usr/bin/env python3
"""爬取 yw.gov.cn - 建设项目环评"""
import requests
from bs4 import BeautifulSoup
import sqlite3
import os
import re
import json
from datetime import datetime

BASE_URL = "https://www.yw.gov.cn"
SITE_NAME = "yw.gov.cn-建设项目环评"
CUTOFF_DATE = "2023-06-16"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
COL_ID = "1229137186"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

def fetch_list():
    """从 JPAAS API 获取文章列表"""
    url = f"{BASE_URL}/api-gateway/jpaas-publish-server/front/page/build/unit"
    params = {
        "parseType": "bulidstatic", "webId": "3549", "pageId": COL_ID,
        "tagId": "\u653f\u5e9c\u4fe1\u606f\u516c\u5f00\u5217\u8868",
        "tplSetId": "zfHuDn7pjzPV0dB1wLhu9",
        "pageType": "column", "rows": "15"
    }
    resp = requests.get(url, params=params, headers=HEADERS, timeout=30)
    data = resp.json()
    html = data["data"]["html"]
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("li"):
        a = li.find("a", class_="bt_link")
        if not a:
            continue
        href = a.get("href", "")
        if href and not href.startswith("http"):
            href = BASE_URL + href
        title = a.get("title", "").strip()
        span = li.find("span")
        pub_date = span.get_text(strip=True) if span else ""
        items.append({"title": title, "url": href, "publish_date": pub_date})
    return items

def get_detail(url):
    """获取详情页的正文HTML"""
    resp = requests.get(url, headers=HEADERS, timeout=30)
    resp.encoding = "utf-8"
    soup = BeautifulSoup(resp.text, "html.parser")
    
    # 正文区域
    content_parts = []
    content_div = soup.select_one("div.main_section")
    if content_div:
        # 排除前面的信息区
        for tag in content_div.select("ul.artic_key, div.main_title, div.artic_kopen, div.artic_tother, div.linke, div.yybb"):
            tag.decompose()
        # 补全附件链接
        for a in content_div.find_all("a"):
            href = a.get("href", "")
            if href and not href.startswith("http"):
                a["href"] = BASE_URL + href
        content_parts.append(str(content_div))
    
    if content_parts:
        return "".join(content_parts)
    return ""

def crawl():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    c.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            title TEXT,
            page_url TEXT UNIQUE,
            content TEXT,
            publish_date TEXT,
            site_name TEXT DEFAULT '',
            crawl_time TEXT DEFAULT (datetime('now', '+8 hours')),
            similar TEXT
        )
    """)
    conn.commit()
    
    print("获取文章列表...")
    items = fetch_list()
    print(f"共 {len(items)} 条")
    
    total_added = 0
    total_skipped = 0
    
    for item in items:
        pub = item["publish_date"]
        if pub and pub < CUTOFF_DATE:
            print(f"  截止日期 {CUTOFF_DATE} (当前 {pub})，跳过")
            break
        
        if not item["url"]:
            total_skipped += 1
            continue
        
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
        if c.fetchone():
            total_skipped += 1
            continue
        
        content = get_detail(item["url"])
        if not content:
            print(f"    警告: 空正文 {item['title'][:30]}")
            total_skipped += 1
            continue
        
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (title, page_url, content, publish_date, site_name) VALUES (?, ?, ?, ?, ?)",
            (item["title"], item["url"], content, pub, SITE_NAME)
        )
        if c.rowcount > 0:
            total_added += 1
        conn.commit()
    
    conn.close()
    print(f"完成: 新增 {total_added} 条, 跳过 {total_skipped} 条")

if __name__ == "__main__":
    crawl()
