#!/usr/bin/env python3
"""
szsnews.com (石嘴山新闻网) 通知公告爬虫 — 环评公示
CMS: 天马CMS (Tianma, SPA+Vue)
API: GET /api/msite/web/index?sheet_id=1 → 40条列表
     POST /api/msite/Subpage/details → 详情HTML
"""

import requests, json, sys, os, re, time
from datetime import datetime

BASE = "https://www.szsnews.com"
LIST_API = f"{BASE}/api/msite/web/index?sheet_id=1"
DETAIL_API = f"{BASE}/api/msite/Subpage/details"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{BASE}/",
}
COMPONENT = "tenma01zx_eqlil"
INSTANCE_ID = "143"

CRAWLER_DIR = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, CRAWLER_DIR)
from crawler_lib import push_to_searchdb

MAX_PAGES = int(sys.argv[sys.argv.index("--pages") + 1]) if "--pages" in sys.argv else 5

def fetch_list(page=1):
    """Fetch notification list (no pagination needed, returns 40 items)"""
    try:
        r = requests.get(LIST_API, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        data = r.json()
        if data.get("code") != 200:
            print(f"[ERROR] List API code={data.get('code')}")
            return []
        modules = data.get("data", {}).get("modules", [])
        for mod in modules:
            cols = mod.get("columns", {})
            for col_list in cols.values():
                for col in col_list:
                    return col.get("list", [])
        return []
    except Exception as e:
        print(f"[ERROR] fetch_list: {e}")
        return []

def fetch_detail(article_id):
    """Fetch article detail via POST"""
    try:
        r = requests.post(DETAIL_API, json={
            "id": article_id,
            "component": COMPONENT,
            "instance_id": INSTANCE_ID
        }, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        data = r.json()
        if data.get("code") != 200:
            print(f"[ERROR] Detail API code={data.get('code')} for article {article_id}")
            return None
        return data.get("data", {})
    except Exception as e:
        print(f"[ERROR] fetch_detail({article_id}): {e}")
        return None

def clean_html(html_content):
    """Clean inline styles and strip HTML tags for plain text, preserve paragraphs"""
    if not html_content:
        return ""
    # Remove <style> blocks
    html_content = re.sub(r'<style[^>]*>.*?</style>', '', html_content, flags=re.DOTALL)
    # Replace </p><p> with double newline for paragraph breaks
    html_content = re.sub(r'</p>\s*<p[^>]*>', '\n\n', html_content)
    # Replace <br> with newline
    html_content = re.sub(r'<br\s*/?>', '\n', html_content)
    # Remove remaining HTML tags
    text = re.sub(r'<[^>]+>', '', html_content)
    # Decode HTML entities
    text = text.replace('&nbsp;', ' ').replace('&amp;', '&').replace('&lt;', '<').replace('&gt;', '>').replace('&quot;', '"')
    # Normalize whitespace
    text = re.sub(r'[ \t]+', ' ', text)
    text = re.sub(r'\n{3,}', '\n\n', text)
    text = text.strip()
    return text

def main():
    items = fetch_list()
    if not items:
        print("[ERROR] No items fetched")
        sys.exit(1)
    
    print(f"[INFO] List fetched: {len(items)} items")
    
    # Limit to pages
    if MAX_PAGES > 0:
        items = items[:MAX_PAGES * 10]

    batch = []
    
    for idx, item in enumerate(items):
        article_id = item.get("article_id")
        title = item.get("title", "").strip()
        publish_ts = item.get("publish_time", 0)
        publish_date = datetime.fromtimestamp(publish_ts).strftime("%Y-%m-%d") if publish_ts else ""
        source = item.get("source", "")
        url = f"{BASE}/article/{article_id}"
        
        if not article_id:
            continue
        
        # Fetch detail for content
        detail = fetch_detail(article_id)
        if detail is None:
            continue
        
        # Get content
        raw_html = detail.get("content", "")
        content = clean_html(raw_html)
        
        # Attach embed if content < 500 chars
        if len(content) < 500 and raw_html:
            content += "\n\n[嵌入内容]\n" + raw_html[:2000]
        
        # Source from detail if list doesn't have it
        if not source:
            source = detail.get("source", "石嘴山市新闻传媒中心")
        
        batch.append({
            "url": url,
            "title": title,
            "content": content,
            "pub_date": publish_date,
            "site_name": "石嘴山新闻网-通知公告",
            "source_url": url,
        })
        
        print(f"  [{idx+1}] {title[:40]}...")
        time.sleep(0.3)
    
    if batch:
        result = push_to_searchdb(batch)
        imported = result.get("imported", 0) if isinstance(result, dict) else len(batch)
        print(f"\n[DONE] 完成: {len(batch)} 条处理")
    else:
        print("[DONE] 无数据")

if __name__ == "__main__":
    main()
