#!/usr/bin/env python3
"""Crawler for 乌兰察布市生态环境局 - 受理公示 (slgs13)
API: POST /rmt/tmpl/api/site/{siteId}/channelid/{channelId}
Content-Type: application/json
Body: {"currpage": N, "pagesize": 9, "params": "..."}
Detail: <div id="content"> with full HTML + PDF attachments
"""
import requests
import json
import re
import sqlite3
import time
import random
import os
from datetime import datetime
from concurrent.futures import ThreadPoolExecutor, as_completed

SITE_NAME = "乌兰察布市生态环境局-受理公示"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
API_URL = "https://sthjj.wulanchabu.gov.cn/rmt/tmpl/api/site/34f332e2d1d142309c4b49e54103c4c6/channelid/37c36539d83443ff95b94a6dba9c9552"
PARAMS = "5186ec152b6fdbf8f9925254821388e33f1bf26f04b05fd746e5575a4111485e17dfbdc2e13666401fa60e98f783528276699fa580e601be852d2feac106fe9e0fff607a5aed63d1003c194190192c80b4299bdcb70d22818e92317f41b240666b09a4cbad3eeb99aac8bd71796d1895a0bb14c482c642ae7d2ce1d41492fb7501ecd379aae27643ae4ed1342128b7ee8d4d00a342fbdc88038447bd54b40880c2193091eb0d5614b7b03e9e766cacfd5a5ab8639a392eb94a37b75ebd810216752dbf729663b703e62ae4d766adb9ebab1dd89d2f63d6d53584676662f91b09c5316b1ea39084a892a9eac56a398b322b914e8e79a9f2636420aa3a72140ffed044742f155dfdf851ec44992af5fe68"
PER_PAGE = 9
HEADERS = {
    "Content-Type": "application/json;charset=utf-8",
    "X-Requested-With": "XMLHttpRequest",
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
DELAY = (0.3, 0.5)

def log(msg):
    print(f"[{time.strftime('%H:%M:%S')}] {msg}", flush=True)


def crawl_list_page(page_num):
    """Fetch one page of list data from API."""
    payload = {"currpage": page_num, "pagesize": PER_PAGE, "params": PARAMS}
    try:
        r = requests.post(API_URL, json=payload, headers=HEADERS, timeout=30)
        data = r.json()
    except Exception as e:
        log(f"  API error page {page_num}: {e}")
        return [], 0
    
    if data.get("errcode") != 0:
        log(f"  API err page {page_num}: {data.get('errmsg')}")
        return [], 0
    
    items = data.get("data", [])
    return items, data.get("total", 0)


def crawl_detail(url):
    """Crawl a detail page for content."""
    try:
        r = requests.get(url, headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        }, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        log(f"  ERROR fetch {url[:60]}: {e}")
        return "", "", "", ""
    
    html = r.text
    
    title_m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    date_m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
    src_m = re.search(r'<meta name="ContentSource" content="([^"]*)"', html)
    
    title = title_m.group(1).strip() if title_m else ""
    pub_date = date_m.group(1).strip()[:10] if date_m else ""
    source = src_m.group(1).strip() if src_m else ""
    
    # Extract #content
    m = re.search(r'id="content"', html)
    if not m:
        return title, pub_date, source, ""
    
    gt_pos = html.find(">", m.start())
    if gt_pos == -1:
        return title, pub_date, source, ""
    
    depth = 1
    i = gt_pos + 1
    while i < len(html) and depth > 0:
        if html[i:i+6] == "</div>":
            depth -= 1
            i += 6
        elif html[i:i+4] == "<div" and i + 4 < len(html) and html[i+4] in (' ', '>', '\n', '\t', '\r', "'", '"'):
            depth += 1
            i += 4
        else:
            i += 1
    
    content = html[m.start():i+6]
    return title, pub_date, source, content


def save_to_db(items):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    cur = conn.cursor()
    
    inserted = 0
    for item in items:
        try:
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw 
                   (site_name, title, page_url, publish_date, source_url, content)
                   VALUES (?, ?, ?, ?, ?, ?)""",
                (SITE_NAME, item["title"], item["url"], item["date"], SITE_NAME, item["content"]),
            )
            if cur.rowcount > 0:
                inserted += 1
        except Exception as e:
            log(f"  ERROR insert: {e}")
    
    conn.commit()
    conn.close()
    return inserted


def main():
    log(f"Starting crawl for {SITE_NAME}")
    log(f"DB: {DB_PATH}")
    
    # Step 1: Get total from first page HTML
    log("=== Phase 1: Get list (API) ===")
    try:
        r = requests.get("https://sthjj.wulanchabu.gov.cn/slgs13/", headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        }, timeout=30)
        m = re.search(r'totalcount:\s*(\d+)', r.text)
        total = int(m.group(1)) if m else 0
    except:
        total = 0
    
    # 3-year cutoff (2023-06-17)
    cutoff_ts = datetime(2023, 6, 17).timestamp() * 1000
    total_pages = (total + PER_PAGE - 1) // PER_PAGE if total > 0 else 68
    log(f"Total: {total} items, {total_pages} pages max, cutoff: 2023-06-17")
    
    all_articles = []
    pg = 1
    while pg <= total_pages:
        if pg > 1:
            time.sleep(random.uniform(*DELAY))
        items, _ = crawl_list_page(pg)
        if not items:
            log(f"  Page {pg}: empty, stopping")
            break
        
        # Check if this page is past the cutoff
        last_ts = items[-1].get("inputTime", 0)
        if last_ts < cutoff_ts:
            # Filter items within range
            for item in items:
                if item.get("inputTime", 0) >= cutoff_ts:
                    all_articles.append(item)
            log(f"  Page {pg}: reached cutoff ({len(items)} items, {len([i for i in items if i.get('inputTime',0) >= cutoff_ts])} kept)")
            break
        
        all_articles.extend(items)
        if pg % 10 == 0 or pg == 1:
            log(f"  Page {pg}: collected {len(all_articles)} items")
        pg += 1
    
    log(f"Total articles from API: {len(all_articles)}")
    
    # Step 2: Crawl detail pages with thread pool
    log("\n=== Phase 2: Crawl detail pages ===")
    enriched = []
    
    def fetch_one(art):
        docno = art.get("docno", "")
        url = f"https://sthjj.wulanchabu.gov.cn/slgs13/{docno}.html"
        try:
            title, pub_date, source, content = crawl_detail(url)
            if not pub_date and art.get("inputTime"):
                ts = art["inputTime"] / 1000
                pub_date = datetime.fromtimestamp(ts).strftime("%Y-%m-%d")
            return {
                "url": url,
                "title": title or art.get("title", ""),
                "date": pub_date,
                "source": source or "乌兰察布市生态环境局",
                "content": content or "",
            }
        except Exception as e:
            log(f"  ERROR {docno}: {e}")
            return {
                "url": url,
                "title": art.get("title", ""),
                "date": "",
                "source": "",
                "content": "",
            }
    
    with ThreadPoolExecutor(max_workers=5) as executor:
        futures = [executor.submit(fetch_one, art) for art in all_articles]
        for i, future in enumerate(as_completed(futures)):
            enriched.append(future.result())
            if (i + 1) % 30 == 0 or i == 0:
                log(f"  Detail {i+1}/{len(all_articles)}")
    
    # Step 3: Save
    log("\n=== Phase 3: Save to DB ===")
    saved = save_to_db(enriched)
    
    log(f"\n=== SUMMARY ===")
    log(f"Total articles: {len(enriched)}")
    log(f"Saved: {saved} new items")
    log(f"Done!")


if __name__ == "__main__":
    main()
