#!/usr/bin/env python3
"""Crawler for 乌兰察布市生态环境局 - 拟审批公示 (nscgs13)
API: POST with params (same as slgs13 but different channelId + params suffix)
"""
import requests, json, re, sqlite3, time, random, os
from datetime import datetime

SITE_NAME = "乌兰察布市生态环境局-拟审批公示"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
API_URL = "https://sthjj.wulanchabu.gov.cn/rmt/tmpl/api/site/34f332e2d1d142309c4b49e54103c4c6/channelid/237e2e83a084462896ac7f34171ecef9"
PARAMS = "5186ec152b6fdbf8f9925254821388e33f1bf26f04b05fd746e5575a4111485e17dfbdc2e13666401fa60e98f783528276699fa580e601be852d2feac106fe9e0fff607a5aed63d1003c194190192c80b4299bdcb70d22818e92317f41b240666b09a4cbad3eeb99aac8bd71796d1895a0bb14c482c642ae7d2ce1d41492fb7501ecd379aae27643ae4ed1342128b7ee8d4d00a342fbdc88038447bd54b40880c2193091eb0d5614b7b03e9e766cacfd5a5ab8639a392eb94a37b75ebd810216752dbf729663b703e62ae4d766adb9ebab1dd89d2f63d6d53584676662f91b09626c99ed7866d90c8638424482519697bcb4115dec5ba85ac0f09cfd0b0e87ff425ece3d1768ce6ce9a3742147a06a75"
PER_PAGE = 9
HEADERS = {"Content-Type": "application/json;charset=utf-8", "X-Requested-With": "XMLHttpRequest", "User-Agent": "Mozilla/5.0"}
DELAY = (0.3, 0.5)

def log(msg):
    print(f"[{time.strftime('%H:%M:%S')}] {msg}", flush=True)

def crawl_list_page(page_num):
    payload = {"currpage": page_num, "pagesize": PER_PAGE, "params": PARAMS}
    try:
        r = requests.post(API_URL, json=payload, headers=HEADERS, timeout=30)
        items = r.json().get("data", [])
        return items
    except Exception as e:
        log(f"  API error page {page_num}: {e}")
        return []

def crawl_detail(url):
    try:
        r = requests.get(url, headers={"User-Agent": "Mozilla/5.0"}, timeout=30)
        r.encoding = "utf-8"
    except:
        return "", "", "", ""
    html = r.text
    title = (re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html) or [None, ""]).group(1)
    pub_date = (re.search(r'<meta name="PubDate" content="([^"]*)"', html) or [None, ""]).group(1)[:10] or ""
    src = (re.search(r'<meta name="ContentSource" content="([^"]*)"', html) or [None, ""]).group(1) or ""
    m = re.search(r'id="content"', html)
    content = ""
    if m:
        gt = html.find(">", m.start())
        if gt > 0:
            depth, i = 1, gt + 1
            while i < len(html) and depth > 0:
                if html[i:i+6] == "</div>": depth -= 1; i += 6
                elif html[i:i+4] == "<div" and html[i+4] in (' ', '>', '\n', '\t', '\r', "'", '"'): depth += 1; i += 4
                else: i += 1
            content = html[m.start():i+6]
    return title or "", pub_date, src or "", content

def main():
    log(f"Starting crawl for {SITE_NAME}")
    log(f"DB: {DB_PATH}")
    
    # Get total from page
    try:
        r = requests.get("https://sthjj.wulanchabu.gov.cn/nscgs13/", headers={"User-Agent": "Mozilla/5.0"}, timeout=30)
        m = re.search(r'totalcount:\s*(\d+)', r.text)
        total = int(m.group(1)) if m else 0
    except:
        total = 0
    
    cutoff_ts = datetime(2023, 6, 17).timestamp() * 1000
    total_pages = (total + PER_PAGE - 1) // PER_PAGE if total > 0 else 60
    log(f"Total: {total} items, {total_pages} pages max")
    
    all_articles = []
    pg = 1
    import sys as _SYS
    _MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
    if _MAX_PG is not None:
        print('[AutoPg] max_pages=' + str(_MAX_PG))
    # END AUTO PAGES
    while pg <= total_pages:
        if _MAX_PG and pg >= _MAX_PG: break
        if pg > 1:
            time.sleep(random.uniform(*DELAY))
        items = crawl_list_page(pg)
        if not items:
            break
        
        last_ts = items[-1].get("inputTime", 0)
        if last_ts < cutoff_ts:
            kept = [i for i in items if i.get("inputTime", 0) >= cutoff_ts]
            all_articles.extend(kept)
            log(f"  Page {pg}: cutoff ({len(items)} items, {len(kept)} kept)")
            break
        
        all_articles.extend(items)
        if pg % 10 == 0 or pg == 1:
            log(f"  Page {pg}: {len(all_articles)} items")
        pg += 1
    
    log(f"Total: {len(all_articles)} articles")
    
    # Phase 2: Detail pages
    log("\n=== Detail pages ===")
    from concurrent.futures import ThreadPoolExecutor, as_completed
        
    def fetch_one(art):
        docno = art.get("docno", "")
        url = f"https://sthjj.wulanchabu.gov.cn/nscgs13/{docno}.html"
        try:
            title, pub_date, source, content = crawl_detail(url)
            if not pub_date and art.get("inputTime"):
                pub_date = datetime.fromtimestamp(art["inputTime"]/1000).strftime("%Y-%m-%d")
            return {"url": url, "title": title or art.get("title", ""), "date": pub_date, "source": source or SITE_NAME, "content": content or ""}
        except Exception as e:
            log(f"  ERROR {docno}: {e}")
            return {"url": url, "title": art.get("title", ""), "date": "", "source": SITE_NAME, "content": ""}
    
    enriched = []
    with ThreadPoolExecutor(max_workers=5) as ex:
        futures = [ex.submit(fetch_one, art) for art in all_articles]
        for i, f in enumerate(as_completed(futures)):
            enriched.append(f.result())
            if (i+1) % 30 == 0 or i == 0:
                log(f"  {i+1}/{len(all_articles)}")
    
    # Save
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    cur = conn.cursor()
    saved = 0
    for item in enriched:
        cur.execute("""INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content) VALUES (?,?,?,?,?,?)""",
            (SITE_NAME, item["title"], item["url"], item["date"], SITE_NAME, item["content"]))
        if cur.rowcount > 0: saved += 1
    conn.commit(); conn.close()
    
    log(f"\n=== SUMMARY ===")
    log(f"Articles: {len(enriched)}, Saved: {saved}")
    log(f"Done!")

if __name__ == "__main__":
    main()
