#!/usr/bin/env python3
"""
沿滩区人民政府 - 环境保护 爬虫
CMS: 数融CMS portal-saas
列表: POST /queryList JSON API (875条, 50条/页)
详情: 静态HTML, curl可用
"""

import sys, re, time, os, json, requests, sqlite3
from datetime import datetime, timezone, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = os.environ.get("SEARCH_DB", "/root/search.db")
SITE_NAME = "沿滩区人民政府-环境保护"
SOURCE = "沿滩区人民政府"

API_URL = "https://www.zgyt.gov.cn/queryList"
BASE_URL = "https://www.zgyt.gov.cn"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Content-Type": "application/json",
    "Origin": "https://www.zgyt.gov.cn",
    "Referer": "https://www.zgyt.gov.cn/ytqrmzf/hjbh2226/pc/list.html",
    "X-Requested-With": "XMLHttpRequest",
}
DETAIL_HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
PAGE_SIZE = 50
MAX_WORKERS = 8

def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)

def get_cutoff():
    return (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")

def save_to_db(articles):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    saved = skipped = 0
    for art in articles:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content) VALUES (?,?,?,?,?,?)",
                (SITE_NAME, art["title"], art["page_url"], art["publish_date"], SOURCE, art["content"])
            )
            if c.rowcount > 0: saved += 1
            else: skipped += 1
        except Exception as e:
            log(f"DB写入错误: {e}")
    conn.commit()
    conn.close()
    return saved, skipped

def fetch_article_list():
    """通过API获取所有文章元信息"""
    cutoff = get_cutoff()
    log(f"截止日期: {cutoff}")
    
    all_articles = []
    page_no = 1
    
    while True:
        payload = {
            "channelCode": ["hjbh2226"],
            "pageSize": PAGE_SIZE,
            "current": page_no,
            "notReturnContent": True,
        }
        
        try:
            r = requests.post(API_URL, json=payload, headers=HEADERS, timeout=15)
            r.raise_for_status()
            data = r.json()
        except Exception as e:
            log(f"API请求失败 (page {page_no}): {e}")
            break
        
        results = data.get("data", {}).get("results", [])
        total = data.get("data", {}).get("total", 0)
        
        log(f"API page {page_no}: {len(results)} results (total={total})")
        
        if not results:
            break
        
        page_articles = []
        for item in results:
            source = item.get("source", {})
            title = source.get("title", "").strip()
            pub_date = source.get("pubDate", "")
            if pub_date:
                pub_date = pub_date.split(" ")[0]  # "2026-01-12 10:35" -> "2026-01-12"
            
            if not title or not pub_date:
                continue
            
            if pub_date < cutoff:
                # Page is sorted by date DESC, so once we hit old dates in a page,
                # subsequent pages will be even older
                log(f"  当前页检出超限日期 {pub_date} < {cutoff}")
                page_articles.append({
                    "title": title,
                    "url": None,
                    "pub_date": pub_date,
                })
                continue
            
            # Get detail URL from urls field
            urls = source.get("urls", "")
            detail_url = ""
            if urls:
                try:
                    url_data = json.loads(urls) if isinstance(urls, str) else urls
                    detail_url = url_data.get("pc", "")
                    if detail_url and not detail_url.startswith("http"):
                        detail_url = BASE_URL + detail_url
                except:
                    pass
            
            page_articles.append({
                "title": title,
                "url": detail_url,
                "pub_date": pub_date,
            })
        
        # Filter - keep only within date range
        valid = [a for a in page_articles if a["pub_date"] >= cutoff and a["url"]]
        all_articles.extend(valid)
        
        # Check if we hit articles older than cutoff
        if any(a["pub_date"] < cutoff for a in page_articles):
            log(f"  命中超限日期，终止翻页")
            break
        
        # Check if we've got all pages
        if page_no * PAGE_SIZE >= total:
            break
        
        page_no += 1
        time.sleep(0.3)
    
    log(f"API共收集 {len(all_articles)} 条 (近3年)")
    return all_articles

def fetch_detail(title, detail_url, pub_date):
    """抓取详情页内容"""
    try:
        r = requests.get(detail_url, headers=DETAIL_HEADERS, timeout=15)
        r.raise_for_status()
        html = r.text
    except Exception as e:
        log(f"  ❌ 请求失败 {detail_url[:60]}: {e}")
        return None
    
    # Extract title from page
    tm = re.search(r'<div[^>]*class="article-title"[^>]*>(.*?)</div>', html, re.DOTALL)
    page_title = re.sub(r'<[^>]+>', '', tm.group(1)).strip() if tm else title
    
    # Extract content
    cm = re.search(r'<div[^>]*class="article-content"[^>]*>(.*?)</div>\s*</div>\s*</div>', html, re.DOTALL)
    content_html = ""
    if cm:
        raw = cm.group(1)
        raw = re.sub(r'<script[^>]*>.*?</script>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<style[^>]*>.*?</style>', '', raw, flags=re.DOTALL)
        # Trim at non-content markers
        for marker in ['<!-- 分享', '<!--分享', '<div class="article-extend', '<div class="article-ewm']:
            idx = raw.find(marker)
            if idx > 0:
                raw = raw[:idx]
        content_html = raw.strip()
    
    # Build attachment links from JS-embedded data in the page
    # Only include files whose path contains this article's ID
    article_id = detail_url.split('/')[-2] if '/content/' in detail_url else ''
    attachments = []
    file_pattern = r'"fileName":"([^"]+?\.(?:pdf|docx?|xlsx?))","totalSize":(\d+),"filePath":"([^"]+)"'
    for fm in re.finditer(file_pattern, html):
        fn = fm.group(1)
        fs = int(fm.group(2))
        fp = fm.group(3)
        # Skip files from other articles (path must contain this article's ID)
        if article_id and article_id not in fp:
            continue
        if not fp.startswith('http'):
            fp = 'https://www.zgyt.gov.cn' + fp
        if fs >= 1048576:
            size_str = f'{fs/1048576:.1f} MB'
        elif fs >= 1024:
            size_str = f'{fs/1024:.0f} KB'
        else:
            size_str = f'{fs} B'
        attachments.append(f'<p><a href="{fp}" target="_blank" rel="noopener">📎 {fn}</a> ({size_str})</p>')
    
    if attachments:
        # Deduplicate by URL
        seen_urls = set()
        unique_attachments = []
        for a in attachments:
            url_start = a.find('href="') + 6
            url_end = a.find('"', url_start)
            url = a[url_start:url_end] if url_start > 5 and url_end > 0 else ''
            if url and url not in seen_urls:
                seen_urls.add(url)
                unique_attachments.append(a)
            elif not url:
                unique_attachments.append(a)
        content_html += '\n<!-- 附件列表 -->\n<div class="article-attachments">\n<h3>附件下载</h3>\n' + '\n'.join(unique_attachments) + '\n</div>'
    
    return {
        "title": page_title,
        "page_url": detail_url,
        "publish_date": pub_date,
        "source_url": SOURCE,
        "content": content_html or title,
    }

def run(full=False):
    log(f"开始爬取 (full={'是' if full else '否'})")
    
    # Phase 1: Get article list from API
    articles = fetch_article_list()
    
    if not articles:
        log("无文章可爬")
        return {"total": 0, "saved": 0, "skipped": 0}
    
    if not full:
        articles = articles[:15]  # Only first page for daily run
        log(f"增量模式: 仅爬 {len(articles)} 条")
    
    # Phase 2: Fetch detail pages in parallel
    log(f"开始爬取详情 ({len(articles)} 条, {MAX_WORKERS} 线程)")
    
    results = []
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        futures = {
            executor.submit(fetch_detail, a["title"], a["url"], a["pub_date"]): a
            for a in articles
        }
        for i, future in enumerate(as_completed(futures)):
            art = futures[future]
            try:
                result = future.result()
                if result:
                    results.append(result)
                    log(f"详情 [{i+1}/{len(articles)}]: {art['title'][:40]}... ✓")
                else:
                    log(f"详情 [{i+1}/{len(articles)}]: {art['title'][:40]}... ❌ 失败")
            except Exception as e:
                log(f"详情 [{i+1}/{len(articles)}]: {art['title'][:30]}... ❌ {e}")
    
    # Save to DB
    saved, skipped = save_to_db(results)
    log(f"完成: total={len(articles)}, saved={saved}, skipped={skipped}")
    return {"total": len(articles), "saved": saved, "skipped": skipped}

if __name__ == "__main__":
    full = "--full" in sys.argv
    res = run(full=full)
    print(json.dumps(res, ensure_ascii=False))
