#!/usr/bin/env python3
import os
"""
蚌埠经济开发区管理委员会 — 建设项目环境影响评价 爬虫
AJAX JSON API + 政府信息公开详情页
"""
import requests, sqlite3, re, os, sys, time, json
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE = "https://jjkfq.bengbu.gov.cn"
API_URL = BASE + "/zfxxgk/site/label/8888"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "X-Requested-With": "XMLHttpRequest",
    "Referer": BASE + "/zfxxgk/public/column/29661?type=4&catId=7560121&action=list",
}
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

session = requests.Session()
session.headers.update(HEADERS)


def fetch_list(page=1):
    """通过AJAX API获取列表数据"""
    params = {
        "labelName": "publicInfoList",
        "siteId": "6795621",
        "organId": "29661",
        "pageSize": "20",
        "pageIndex": str(page),
        "isDate": "true",
        "dateFormat": "yyyy-MM-dd",
        "length": "50",
        "type": "4",
        "action": "list",
        "result": "",
        "isJson": "true",
        "isSetValue": "true",
        "catId": "7560121",
    }
    try:
        r = session.get(API_URL, params=params, timeout=20)
        data = r.json()
        return data.get("data", []), data.get("total", 0), data.get("pageCount", 0)
    except Exception as e:
        print(f"  [ERROR] API page {page}: {e}")
        return [], 0, 0


def parse_detail(url):
    """解析详情页，返回 (title, pub_date, content_html)"""
    try:
        r = session.get(url, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Detail {url}: {e}")
        return None, None, None
    
    html = r.text
    
    # Title from h1
    m = re.search(r'<h1[^>]*class="gk_title"[^>]*>(.*?)</h1>', html, re.DOTALL)
    title = m.group(1).strip() if m else ""
    
    # Date from published info
    m = re.search(r'发布时间：(\d{4}-\d{2}-\d{2})', html)
    pub_date = m.group(1).strip() if m else ""
    
    # Content: <div id="zoom" class="j-fontContent gkwz_contnet">
    m = re.search(r'<div\s+id="zoom"[^>]*>', html, re.DOTALL)
    content = ""
    if m:
        start = m.end()
        # Find matching closing div using depth counting
        depth = 1
        i = start
        while i < len(html) and depth > 0:
            if html[i:i+4] == "<div" and html[i+4:i+5] not in ">/":
                depth += 1
                i += 4
            elif html[i:i+6] == "</div>":
                depth -= 1
                i += 6
            else:
                i += 1
        content = html[start:i-6].strip() if depth == 0 else ""
    
    return title, pub_date, content


def save_to_db(items_data):
    """批量写入search.db"""
    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    inserted = 0
    updated = 0
    for url, title, pub_date, content in items_data:
        try:
            c.execute("""
                INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, source_url, category, site_name, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, 'crawl_bengbu.py')
            """, (
                url,
                title,
                content,
                pub_date or "",
                url,
                "建设项目环境影响评价",
                "蚌埠经济开发区",
            ))
            if c.rowcount == 1:
                inserted += 1
            else:
                updated += 1
        except Exception as e:
            print(f"  [ERROR] DB insert: {e}")
    conn.commit()
    conn.close()
    return inserted, updated


def main():
    print(f"=== 蚌埠经济开发区 — 建设项目环境影响评价 爬虫 ===")
    print(f"截止日期: {CUTOFF_DATE}")
    
    all_items = []
    seen_links = set()
    
    # 循环翻页
    page = 1
    while True:
        print(f"\n[Page {page}] Fetching API...")
        items, total, page_count = fetch_list(page)
        if not items:
            print(f"  -> No items, stopping")
            break
        
        new_count = 0
        stop = False
        for item in items:
            link = item.get("link", "")
            if not link or link in seen_links:
                continue
            seen_links.add(link)
            
            title = item.get("title", "").strip()
            pub_date = item.get("publishDate", "")[:10] if item.get("publishDate") else ""
            
            if pub_date < CUTOFF_DATE:
                stop = True
                break
            
            all_items.append((link, title, pub_date))
            new_count += 1
        
        print(f"  -> {new_count} new items, total: {len(all_items)}, total={total}, pages={page_count}")
        
        if stop or page >= page_count:
            print(f"  -> {'Reached cutoff date' if stop else 'Last page'}, stopping")
            break
        
        page += 1
        time.sleep(0.5)
    
    print(f"\n=== Total list items: {len(all_items)} ===")
    
    # 爬详情页
    detail_data = []
    for i, (url, title, date_str) in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {url.split('/')[-1]}...", end=" ")
        det_title, det_date, content = parse_detail(url)
        if det_title:
            title = det_title
        if det_date:
            date_str = det_date
        detail_data.append((url, title, date_str, content))
        print(f"len={len(content) if content else 0}", end="")
        if not content:
            print(" [WARN empty]", end="")
        print()
        time.sleep(0.3)
    
    # 写入DB
    ins, upd = save_to_db(detail_data)
    print(f"\n=== DB write complete: {ins} inserted, {upd} updated ===")
    print(f"Total in this run: {len(detail_data)}")


if __name__ == "__main__":
    main()
