#!/usr/bin/env python3
"""
泰州市生态环境局 - 项目环评公示 爬虫
URL: https://hbj.taizhou.gov.cn/ztzl/jsxm/xmhpgs/index.html
CMS: Hanweb JPAAS
API: /api-gateway/jpaas-publish-server/front/page/build/unit
"""
import json, re, urllib.request, urllib.parse, sys, os, time

BASE_URL = "https://hbj.taizhou.gov.cn"

# ── DB ──
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "泰州市生态环境局-项目环评公示"

def ensure_db():
    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY,
            title TEXT,
            content TEXT,
            publish_date TEXT,
            page_url TEXT UNIQUE,
            source_url TEXT,
            site_name TEXT DEFAULT '',
            category TEXT DEFAULT ''
        )
    """)
    conn.execute("CREATE INDEX IF NOT EXISTS idx_site_name ON gov_raw(site_name)")
    conn.commit()
    return conn

def save_item(conn, title, content, publish_date, url):
    conn.execute(
        "INSERT OR IGNORE INTO gov_raw (title, content, publish_date, page_url, source_url, site_name) VALUES (?, ?, ?, ?, ?, ?)",
        (title, content, publish_date, url, BASE_URL, SITE_NAME)
    )
    conn.commit()

# ── API ──
API_PARAMS = {
    'parseType': 'bulidstatic',
    'webId': 'e4f0e8a733e34ed28f0620789e66a128',
    'tplSetId': '6a37c2a62cce48538e78db1205a34cc6',
    'pageType': 'column',
    'tagId': '当前栏目信息列表',
    'editType': 'null',
    'pageId': '477be6a07a6d446e938718552dfcec37',
}

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Referer': BASE_URL + '/ztzl/jsxm/xmhpgs/index.html',
}

def fetch_list(page_no):
    """获取列表页数据"""
    params = API_PARAMS.copy()
    params['paramJson'] = json.dumps({"pageNo": page_no, "pageSize": 15}, ensure_ascii=False)
    url = BASE_URL + '/api-gateway/jpaas-publish-server/front/page/build/unit?' + urllib.parse.urlencode(params)
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    data = json.loads(resp.read())
    html = data['data']['html']
    
    # 提取标题、链接、日期
    items = []
    pattern = re.compile(
        r'<a[^>]*href="([^"]+)"[^>]*>([^<]+)</a>\s*<span>(\d{4}-\d{2}-\d{2})</span>'
    )
    for m in pattern.finditer(html):
        url_path = m.group(1)
        title = m.group(2).strip()
        date = m.group(3)
        full_url = BASE_URL + url_path if not url_path.startswith('http') else url_path
        items.append((title, date, full_url))
    
    # 获取总数和分页信息
    count_m = re.search(r'count="(\d+)"', html)
    total = int(count_m.group(1)) if count_m else len(items)
    
    return items, total

def fetch_detail(url):
    """获取详情页内容"""
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    html = resp.read().decode('utf-8', errors='replace')
    
    # 标题
    title_m = re.search(r'<div class="wzzw-title">(.*?)</div>', html, re.DOTALL)
    title = title_m.group(1).strip() if title_m else ''
    
    # 日期
    date_m = re.search(r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    date = date_m.group(1) if date_m else ''
    
    # 正文
    content_m = re.search(
        r'<div class="wzzw-article">(.*?)</div>\s*<!--\s*扫一扫\s*-->',
        html, re.DOTALL
    )
    content = content_m.group(1).strip() if content_m else ''
    
    if not content:
        content_m = re.search(r'<div class="wzzw-article">(.*?)</div>', html, re.DOTALL)
        content = content_m.group(1).strip() if content_m else ''
    
    return title, date, content

def main():
    conn = ensure_db()
    total_items = 0
    
    # 先获取第一页，知道总数
    first_items, total = fetch_list(1)
    total_pages = (total + 14) // 15  # 向上取整
    print(f"Total items: {total}, pages: {total_pages}")
    
    # 处理所有页面
    all_page_items = []
    for page_no in range(1, total_pages + 1):
        if page_no == 1:
            items = first_items
        else:
            items, _ = fetch_list(page_no)
        
        print(f"Page {page_no}: {len(items)} items")
        all_page_items.extend(items)
        time.sleep(0.3)
    
    print(f"\nTotal items from list: {len(all_page_items)}")
    
    # 爬取详情
    for idx, (title, date, url) in enumerate(all_page_items, 1):
        try:
            det_title, det_date, content = fetch_detail(url)
            use_title = title or det_title
            use_date = date or det_date
            
            # 验证
            text = re.sub(r'<[^>]+>', '', content).strip() if content else ''
            content_ok = len(text) >= 50
            save_item(conn, use_title, content, use_date, url)
            status = "OK" if content_ok else "SHORT"
            print(f"  [{idx}/{len(all_page_items)}] {status} {use_date} {use_title[:50]}")
        except Exception as e:
            print(f"  [{idx}/{len(all_page_items)}] ERROR {url}: {e}")
        
        time.sleep(0.5)
    
    # 统计
    cur = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=? AND content IS NOT NULL AND length(content)>50", (SITE_NAME,))
    ok_count = cur.fetchone()[0]
    total_count = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
    
    print(f"\n{'='*50}")
    print(f"完成！总计: {total_count} 条, 有内容: {ok_count} 条")
    
    conn.close()

if __name__ == '__main__':
    main()
