#!/usr/bin/env python3
"""乐清市-建设项目环境影响评价信息公示 crawler"""
import os, sys, re, requests, json
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

SITE_NAME = "乐清市-建设项目环境影响评价信息公示"
DATE_THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

API_URL = "https://www.yueqing.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
API_PARAMS = {
    "webId": "1828",
    "pageId": "1229870564",
    "parseType": "bulidstatic",
    "pageType": "column",
    "tagId": "重点领域乐清政府标题",
    "tplSetId": "CRHIRmIPXDNfEZc5zbFzp",
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "Referer": "https://www.yueqing.gov.cn/col/col1229870564/index.html",
}
import urllib3
urllib3.disable_warnings()

def fetch_list(page_no):
    params = API_PARAMS.copy()
    params['paramJson'] = json.dumps({"pageNo": page_no, "pageSize": 13})
    r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
    data = r.json()
    html = data.get('data', {}).get('html', '')
    if not html:
        return [], 0
    
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for li in soup.select('.page-content li'):
        a = li.find('a')
        span = li.find('span')
        if a:
            href = a.get('href', '')
            if href.startswith('/'):
                href = 'https://www.yueqing.gov.cn' + href
            title = a.get_text(strip=True)
            date_str = span.get_text(strip=True) if span else ''
            items.append({'title': title, 'url': href, 'date': date_str})
    
    # Get total count
    pagination = soup.find('div', class_='pagination')
    count = int(pagination.get('count', '0')) if pagination else 0
    
    return items, count

def fetch_detail(url):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # Full title
    full_title = ''
    if soup.title:
        t = soup.title.get_text(strip=True)
        t = re.sub(r'\s*[-–—]\s*乐清市[^<]*$', '', t).strip()
        full_title = t
    
    # Content in .wapdetail
    content = ''
    for cls in ['wapdetail', 'content', 'article-content']:
        div = soup.find('div', class_=cls)
        if div and len(div.get_text(strip=True)) > 100:
            content = str(div)
            break
    
    if not content:
        # Try #zoom or #c
        for sel in ['#zoom', '#c', 'table#c']:
            el = soup.select_one(sel)
            if el and len(el.get_text(strip=True)) > 100:
                content = str(el)
                break
    
    return content, full_title

def main():
    import sqlite3
    
    # Get total count from page 1
    items, total = fetch_list(1)
    if total == 0:
        total = len(items)
    
    total_pages = (total + 12) // 13  # 13 per page
    print(f"共 {total} 条, {total_pages} 页")
    
    all_items = list(items)
    for page in range(2, total_pages + 1):
        more_items, _ = fetch_list(page)
        all_items.extend(more_items)
        print(f"Page {page}/{total_pages}: +{len(more_items)} (累计 {len(all_items)})")
    
    # Filter by date
    to_fetch = [it for it in all_items if it['date'] >= DATE_THRESHOLD]
    print(f"\n近3年: {len(to_fetch)}/{len(all_items)} 条")
    
    # Fetch details
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        site_name TEXT, page_url TEXT UNIQUE, title TEXT,
        content TEXT, publish_date TEXT, summary TEXT,
        created_at TEXT DEFAULT (datetime('now','localtime'))
    )''')
    
    for i, item in enumerate(to_fetch, 1):
        try:
            content, full_title = fetch_detail(item['url'])
        except Exception as e:
            print(f"  Error fetching {item['url'][-40:]}: {e}")
            continue
        
        if content:
            title_to_use = full_title if full_title else item['title']
            try:
                conn.execute(
                    'INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?,?,?,?,?,?)',
                    (SITE_NAME, item['url'], title_to_use, content, item['date'], SITE_NAME)
                )
                conn.commit()
            except Exception as e:
                print(f"  DB error: {e}")
        if i % 5 == 0:
            print(f"  详情 {i}/{len(to_fetch)}...")
    
    total_in_db = conn.execute('SELECT COUNT(*) FROM gov_raw WHERE site_name=?', (SITE_NAME,)).fetchone()[0]
    print(f"\n=== 完成 ===")
    print(f"已入库: {total_in_db}")
    conn.close()

if __name__ == '__main__':
    main()
