#!/usr/bin/env python3
"""
晋江市人民政府 - 城市管理局信息公开 爬虫
URL: https://www.jinjiang.gov.cn/xxgk/zfxxgkzl/bmzfxxgk/csglj/zfxxgkml/
API: POST https://www.jinjiang.gov.cn/ssp/search/api/v2/external
CHNLID: 38251
"""

import sys, json, time, requests, sqlite3, datetime
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.jinjiang.gov.cn/xxgk/zfxxgkzl/bmzfxxgk/csglj/zfxxgkml/"
API_URL = "https://www.jinjiang.gov.cn/ssp/search/api/v2/external"
SITE_NAME = "jinjiang.gov.cn-城市管理局-信息公开"
CHNLID = "38251"
SEARCH_DB = "/root/temp_search.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Content-Type": "application/json",
    "Referer": "https://www.jinjiang.gov.cn/",
}

def fetch_list_page(session, page, page_size=100):
    payload = {
        "siteId": "000000008fd72c27018fdd69fb8b0002",
        "apiName": "WCMDOCAPI",
        "pageSize": page_size,
        "relevanceField": "",
        "relevanceType": "all",
        "sortField": "docorderpri desc,docreltime desc",
        "matchMode": "smart",
        "aggregateField": "publishyear",
        "isCollapse": "Y",
        "page": page,
        "filter": {
            "docstatus": [{"eq": 10}],
            "chnlid": [{"in": [CHNLID]}],
            "publishyear": []
        }
    }
    resp = session.post(API_URL, json=payload, timeout=30)
    if resp.status_code != 200:
        return [], 0
    data = resp.json()
    if not data.get("data"):
        return [], 0
    result = data["data"].get("result", [])
    total = int(data["data"].get("total", 0))
    return result, total

def fetch_detail(url, session):
    try:
        resp = session.get(url, timeout=30)
        if resp.status_code != 200:
            return "", None, None
        soup = BeautifulSoup(resp.text, 'html.parser')
        content_div = soup.select_one('div.TRS_Editor')
        content = str(content_div) if content_div else ""
        meta_title = soup.find('meta', attrs={"name": "ArticleTitle"})
        title = meta_title['content'].strip() if meta_title and meta_title.get('content') else None
        meta_date = soup.find('meta', attrs={"name": "PubDate"})
        date = meta_date['content'].strip() if meta_date and meta_date.get('content') else None
        return content, title, date
    except Exception as e:
        return "", None, None

def main():
    print(f"===== 晋江市-城市管理局-信息公开 爬虫 =====")
    session = requests.Session()
    session.headers.update(HEADERS)
    
    # Step 1: API获取列表
    print("\n--- 第1步: 获取列表 ---")
    page1, total = fetch_list_page(session, 1, 100)
    if not page1:
        print("API无数据")
        return
    print(f"总记录: {total}")
    
    all_results = page1
    print(f"  [OK] 第1页: {len(page1)} 条")
    total_pages = (total + 99) // 100
    for p in range(2, total_pages + 1):
        results, _ = fetch_list_page(session, p, 100)
        if not results:
            break
        all_results.extend(results)
        print(f"  [OK] 第{p}页: {len(results)} 条")
        time.sleep(0.3)
    print(f"共获取: {len(all_results)} 条")
    
    # Step 2: 获取详情
    print(f"\n--- 第2步: 获取详情 ---")
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, content TEXT, source_url TEXT UNIQUE,
        site_name TEXT, publish_date TEXT,
        crawl_time TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    
    inserted = 0
    short_count = 0
    for idx, item in enumerate(all_results):
        title = item.get("doctitle", "")
        ts = int(item.get("docreltime", 0)) / 1000
        date = datetime.datetime.fromtimestamp(ts).strftime("%Y-%m-%d") if ts > 0 else ""
        detail_url = item.get("docpuburl", "")
        if not detail_url:
            continue

        print(f"  [{idx+1}/{len(all_results)}] {date} {title[:40]}...", end=" ")
        sys.stdout.flush()
        
        content, det_title, det_date = fetch_detail(detail_url, session)
        if not content or len(content.strip()) < 200:
            content = content or ""
            print(f"SHORT({len(content.strip())})")
            short_count += 1
        else:
            print(f"OK({len(content.strip())})")
        
        final_title = det_title or title
        final_date = det_date or date
        
        try:
            c.execute("INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                      (final_title, content, detail_url, SITE_NAME, final_date))
            if c.rowcount > 0:
                inserted += 1
        except Exception as e:
            pass
    
    conn.commit()
    conn.close()
    print(f"\n{'='*50}")
    print(f"完成！总计: {inserted} 条入库")
    print(f"有内容: {len(all_results)-short_count} 条, SHORT: {short_count} 条")
    print(f"{'='*50}")

if __name__ == "__main__":
    main()
