#!/usr/bin/env python3
"""
淮安市洪泽区人民政府 - 政府信息公开
http://www.hongze.gov.cn/cmsweb/zwgk/hz/index.html
CMS: cmsweb (articCommonController API)
"""
import os, sys, re, json, requests, sqlite3
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "淮安市洪泽区人民政府-信息公开"
BASE_URL = "http://www.hongze.gov.cn"
LIST_API = BASE_URL + "/articleCommonController/lists.do"
DETAIL_API = BASE_URL + "/articleCommonController/findArticleById.do"
THREADS = 10
THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
TOTAL_PAGES = 84  # pages 1-83 all within 3yr, page 84 partially

POST_DATA = {
    "topic": "4648",
    "pagesize": 15,
    "rdeptid": "2c94939266a4efa70166a5122426004a",
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
}

import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

def fetch_list(page):
    """获取列表页"""
    data = {**POST_DATA, "page": page}
    try:
        r = requests.post(LIST_API, data=data, headers=HEADERS, timeout=20)
        d = r.json()
        val = d.get("value", {})
        items = val.get("list", [])
        return items
    except Exception as e:
        print(f"  [!] 列表页 {page} 异常: {e}", file=sys.stderr)
        return []

def parse_items(api_items):
    """解析API返回的列表数据"""
    result = []
    for item in api_items:
        title = item.get("title", "").strip()
        pub_date = item.get("releaseTime", "")[:10]
        item_id = item.get("id", "")
        if not title or not item_id:
            continue
        result.append({
            "title": title,
            "date": pub_date,
            "id": item_id,
        })
    return result

def fetch_detail(item_id):
    """获取详情"""
    try:
        r = requests.post(DETAIL_API, data={"id": item_id}, headers=HEADERS, timeout=15)
        d = r.json()
        val = d.get("value")
        if not val:
            return None
        if isinstance(val, str):
            return None
        title = val.get("title", "").strip()
        pub_date = val.get("releaseTime", "")[:10] if val.get("releaseTime") else ""
        html_content = val.get("htmlContent", "")
        if not html_content or len(html_content.strip()) < 30:
            return None
        return {"title": title, "date": pub_date, "content": html_content}
    except Exception as e:
        print(f"  [!] 详情失败: {item_id} - {e}", file=sys.stderr)
        return None

def store_item(item):
    try:
        conn = sqlite3.connect(DB_PATH)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (title, publish_date, site_name, page_url, content, summary)
            VALUES (?, ?, ?, ?, ?, ?)
        """, (
            item["title"], item["date"], SITE_NAME, item["url"],
            item.get("content", ""), item.get("summary", "")
        ))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected > 0
    except Exception as e:
        print(f"  [!] DB写入失败: {e}", file=sys.stderr)
        return False

def main():
    print(f"[*] 站点: {SITE_NAME}")
    print(f"[*] 时间阈值: {THRESHOLD}")
    print(f"[*] 总页数: {TOTAL_PAGES}")

    # 1. 列表页
    all_items = []
    for page in range(1, TOTAL_PAGES + 1):
        api_items = fetch_list(page)
        items = parse_items(api_items)
        # 过滤3年内
        items = [it for it in items if it["date"] >= THRESHOLD]
        all_items.extend(items)
        if page % 10 == 0:
            print(f"  [→] 列表页 {page}/{TOTAL_PAGES}: 累计 {len(all_items)} 条")

    print(f"\n[*] 共获取 {len(all_items)} 条 (3年内)")

    if not all_items:
        print("[*] 无需新增")
        return

    # 2. 去重
    try:
        conn = sqlite3.connect(DB_PATH)
        existing = set()
        for it in all_items:
            c = conn.cursor()
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (BASE_URL + f"/article?id={it['id']}", SITE_NAME))
            if c.fetchone():
                existing.add(it["id"])
        conn.close()
    except:
        existing = set()

    to_crawl = [it for it in all_items if it["id"] not in existing]
    print(f"[*] 需爬详情: {len(to_crawl)} 条 (已有 {len(all_items)-len(to_crawl)} 条跳过)")

    if not to_crawl:
        print("[*] 无需新增")
        return

    # 3. 并发详情
    new_count = 0
    skip_count = 0

    with ThreadPoolExecutor(max_workers=THREADS) as executor:
        fut_map = {executor.submit(fetch_detail, it["id"]): it for it in to_crawl}
        for fut in as_completed(fut_map):
            it = fut_map[fut]
            try:
                detail = fut.result()
                if detail:
                    it["content"] = detail["content"]
                    it["title"] = detail.get("title", it["title"])
                    it["date"] = detail.get("date", it["date"])
                    it["url"] = BASE_URL + f"/article?id={it['id']}"
                    if store_item(it):
                        new_count += 1
                        if new_count % 40 == 0:
                            print(f"  [✓] #{new_count} {it['title'][:35]}")
                else:
                    skip_count += 1
            except Exception as e:
                print(f"  [!] 异常: {e}")
                skip_count += 1

    print(f"\n{'='*50}")
    print(f"  新增: {new_count}")
    print(f"  跳过: {skip_count}")
    print(f"  总计: {len(to_crawl)}")
    print(f"{'='*50}")

if __name__ == "__main__":
    main()
