#!/usr/bin/env python3
"""
土默特右旗人民政府 - 通知公告 爬虫
CMS: TRS WCM 静态分页
列表: index.html → index_{N-1}.html  (10条/页, ~45页)
详情: /ywdt/tzgg/YYYYMM/t2026XXXX.html

运行: python3 crawl_tmtyq.py [--full]
  --full: 全量 (所有45页)
  不加: 仅第1页 (日跑增量)
"""

import sys, re, time, os, json, sqlite3
from concurrent.futures import ThreadPoolExecutor, as_completed
import urllib.request, urllib.error

DB_PATH = os.environ.get("SEARCH_DB", "/root/search.db")
SITE_NAME = "土默特右旗人民政府-通知公告"
SOURCE = "土默特右旗人民政府"

BASE_URL = "http://www.tmtyq.gov.cn"
LIST_URL = BASE_URL + "/ywdt/tzgg/"

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
MAX_PAGES = 50
NUM_THREADS = 5

def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)

def get_cutoff():
    from datetime import datetime, timezone, timedelta
    return (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")

def save_to_db(articles):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    saved = skipped = 0
    for art in articles:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content) VALUES (?,?,?,?,?,?)",
                (SITE_NAME, art["title"], art["page_url"], art["publish_date"], SOURCE, art["content"])
            )
            if c.rowcount > 0: saved += 1
            else: skipped += 1
        except Exception as e:
            log(f"DB写入错误: {e}")
    conn.commit()
    conn.close()
    return saved, skipped

def fetch_page(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=15)
        return resp.read().decode()
    except Exception as e:
        log(f"请求失败 {url[:60]}: {e}")
        return None

def extract_list_items(html_text):
    """从列表页提取文章信息"""
    items = []
    pattern = re.compile(r'<li>.*?<a href="([^"]+)"[^>]*>\s*<span>\s*([^<]+?)\s*</span>\s*<span>\s*(\d{4}-\d{2}-\d{2})\s*</span>', re.DOTALL)
    for m in pattern.finditer(html_text):
        href = m.group(1)
        title = m.group(2).strip()
        pub_date = m.group(3)
        # Normalize URL
        if href.startswith("./"):
            href = href[2:]
        full_url = LIST_URL + href if not href.startswith("http") else href
        items.append((full_url, title, pub_date))
    return items

def fetch_article_list(max_pages):
    """获取所有文章元信息"""
    cutoff = get_cutoff()
    log(f"截止日期: {cutoff}")
    
    all_articles = []
    
    for pg in range(0, max_pages):
        if pg == 0:
            url = LIST_URL + "index.html"
        else:
            url = LIST_URL + f"index_{pg}.html"
        
        log(f"列表页 {pg+1}: {url}")
        html = fetch_page(url)
        if not html:
            break
        
        items = extract_list_items(html)
        log(f"  找到 {len(items)} 条")
        
        if not items:
            log(f"  页 {pg+1} 无数据，终止")
            break
        
        # Filter by date
        valid = [(u, t, d) for u, t, d in items if d >= cutoff]
        log(f"  近3年: {len(valid)}/{len(items)}")
        
        for u, t, d in valid:
            all_articles.append({"title": t, "url": u, "pub_date": d})
        
        # Check stop condition
        dates = [d for _, _, d in items]
        if dates and max(dates) < cutoff:
            log(f"  本页最新日期 {max(dates)} < {cutoff}，终止")
            break
        
        time.sleep(0.3)
    
    # Deduplicate by URL
    seen = set()
    unique = []
    for a in all_articles:
        if a["url"] not in seen:
            seen.add(a["url"])
            unique.append(a)
    
    log(f"共收集 {len(unique)} 条")
    return unique

def fetch_detail(url):
    """抓取详情页"""
    html = fetch_page(url)
    if not html:
        return None
    
    # Title
    tm = re.search(r'<h1>(.*?)</h1>', html, re.DOTALL)
    title = re.sub(r'<[^>]+>', '', tm.group(1)).strip() if tm else ""
    
    # Date
    dm = re.search(r'发文时间:\s*(\d{4}-\d{2}-\d{2})', html)
    pub_date = dm.group(1) if dm else ""
    
    # Content
    cm = re.search(r'<div[^>]*class="trs_editor_view[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    content_html = ""
    if cm:
        raw = cm.group(1)
        raw = re.sub(r'<script[^>]*>.*?</script>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<style[^>]*>.*?</style>', '', raw, flags=re.DOTALL)
        content_html = raw.strip()
    
    return {
        "title": title,
        "page_url": url,
        "publish_date": pub_date,
        "source_url": SOURCE,
        "content": content_html or title,
    }

def run(full=False):
    max_pages = MAX_PAGES if full else 1
    log(f"开始爬取 (full={'是' if full else '否'}, pages={max_pages})")
    
    # Phase 1: List pages
    articles = fetch_article_list(max_pages)
    
    if not articles:
        log("无文章可爬")
        return {"total": 0, "saved": 0, "skipped": 0}
    
    # Phase 2: Details
    log(f"爬取详情 ({len(articles)} 条)")
    results = []
    with ThreadPoolExecutor(max_workers=NUM_THREADS) as executor:
        futures = {executor.submit(fetch_detail, a["url"]): a for a in articles}
        for i, future in enumerate(as_completed(futures)):
            art = futures[future]
            try:
                result = future.result()
                if result:
                    results.append(result)
                    log(f"详情 [{i+1}/{len(articles)}]: {art['title'][:40]}... ✓")
                else:
                    log(f"详情 [{i+1}/{len(articles)}]: {art['title'][:40]}... ❌")
            except Exception as e:
                log(f"详情 [{i+1}/{len(articles)}]: {art['title'][:30]}... ❌ {e}")
    
    saved, skipped = save_to_db(results)
    log(f"完成: total={len(articles)}, saved={saved}, skipped={skipped}")
    return {"total": len(articles), "saved": saved, "skipped": skipped}

if __name__ == "__main__":
    full = "--full" in sys.argv
    res = run(full=full)
    print(json.dumps(res, ensure_ascii=False))
