#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# 江山市 - 建设项目环境影响评价信息公示
# CMS: Hanweb JPAAS, API-driven list
# List: JPAAS API, 15 items total
# Content: div.text on detail page

import requests
import re
import json
import sqlite3
from bs4 import BeautifulSoup
from datetime import datetime

API_URL = "https://www.jiangshan.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
REFERER = "https://www.jiangshan.gov.cn/col/col1229857236/index.html"
BASE_DOMAIN = "https://www.jiangshan.gov.cn"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": REFERER
}

DB_PATH = "/root/search.db"

def get_pages():
    """Get all article URLs from the JPAAS API"""
    articles = []
    
    params = {
        "parseType": "bulidstatic",
        "webId": "1839",
        "tplSetId": "DdaSIvZkD2fqQMzw19yIN",
        "pageType": "column",
        "tagId": "文章列表",
        "editType": "null",
        "pageId": "1229857236"
    }
    
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, verify=False, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            print(f"  [ERR] API returned {r.status_code}")
            return articles
        
        data = r.json()
        if not data.get("success"):
            print(f"  [ERR] API: {data.get('message')}")
            return articles
        
        html = data["data"]["html"]
        soup = BeautifulSoup(html, "html.parser")
        
        # Find items in the HTML
        for li in soup.find_all("li"):
            a = li.find("a")
            span = li.find("span")
            if a and span:
                href = a.get("href", "")
                title = a.get("title", "") or a.get_text(strip=True)
                date_str = span.get_text(strip=True)[:10]
                
                if href and title and date_str:
                    if href.startswith("/"):
                        full_url = BASE_DOMAIN + href
                    else:
                        full_url = href
                    
                    articles.append({
                        "url": full_url,
                        "title": title,
                        "date": date_str
                    })
        
        print(f"  [OK] API: {len(articles)} items")
        
    except Exception as e:
        print(f"  [ERR] API: {e}")
    
    return articles

def get_detail(url):
    """Fetch detail page and extract content and date"""
    try:
        r = requests.get(url, headers=HEADERS, verify=False, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200 or len(r.text) < 1000:
            return None, None
        
        soup = BeautifulSoup(r.text, "html.parser")
        
        # Content: div.article_text (primary), fallback div.content
        content = None
        for cls in ["article_text", "content"]:
            div = soup.find("div", class_=cls)
            if div:
                inner = "".join(str(tag) for tag in div.contents).strip()
                text_only = re.sub(r'<[^>]+>', '', inner).strip()
                if len(text_only) >= 50:
                    content = inner
                    break
        
        if not content:
            return None, None
        
        # Date from PubDate meta
        pub_date = ""
        meta = soup.find("meta", attrs={"name": "PubDate"})
        if meta and meta.get("content"):
            pub_date = meta["content"].strip()[:10]
        
        return content, pub_date
    
    except Exception as e:
        print(f"  [ERR] {url[:80]}: {e}")
        return None, None

def main():
    print("=" * 60)
    print("江山市 - 建设项目环评公示 爬虫")
    print(f"启动时间: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
    print("=" * 60)
    
    print("\n[1/3] 获取列表...")
    articles = get_pages()
    print(f"  共找到 {len(articles)} 条记录")
    
    if not articles:
        return
    
    print("\n[2/3] 检查数据库去重...")
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    
    c.execute("SELECT page_url FROM gov_raw")
    existing = set(row[0] for row in c.fetchall())
    
    new_articles = [a for a in articles if a["url"] not in existing]
    print(f"  新增: {len(new_articles)}/{len(articles)}")
    
    if not new_articles:
        print("  没有新记录")
        conn.close()
        return
    
    print("\n[3/3] 抓取详情页...")
    imported = 0
    skipped = 0
    
    for i, article in enumerate(new_articles, 1):
        title_short = article["title"][:45].replace("\n", " ")
        print(f"  [{i}/{len(new_articles)}] {title_short}...", end=" ", flush=True)
        
        content, pub_date = get_detail(article["url"])
        
        if not content:
            print("✗ 无正文")
            skipped += 1
            continue
        
        date = pub_date or article["date"]
        
        text_only = re.sub(r'<[^>]+>', '', content).strip()
        summary = text_only[:200]
        
        try:
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                (title, content, page_url, summary, site_name, publish_date, category)
                VALUES (?, ?, ?, ?, ?, ?, ?)""",
                (article["title"], content, article["url"], summary,
                 "江山市", date, "环评公示"))
            if c.rowcount > 0:
                conn.commit()
                imported += 1
                print(f"✓ ({len(text_only)}字)")
            else:
                print("• 已存在")
        except Exception as e:
            print(f"✗ DB错误: {e}")
            conn.rollback()
    
    conn.close()
    print(f"\n{'=' * 60}")
    print(f"导入完成: {imported} 条 (跳过 {skipped} 条)")
    print(f"{'=' * 60}")

if __name__ == "__main__":
    main()
