#!/usr/bin/env python3
"""
河北正润环境科技有限公司 - 项目公示
http://www.zhengrunhuanjing.com/#/home/dynamic?idx=0
企业环境信息披露平台，JSON API，含完整正文
"""
import requests
import re
import sys
import os
import json
import time
from datetime import datetime, timedelta

API_URL = "http://www.zhengrunhuanjing.com/zrweb/app/videoList"
SITE_NAME = "zhengrun_xmgs"

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
})

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")


def fetch_page(cat_id, page, limit=20):
    """获取列表页数据"""
    try:
        r = session.get(API_URL, params={
            "catId": cat_id,
            "page": page,
            "limit": limit
        }, timeout=30)
        r.encoding = "utf-8"
        data = r.json()
        if data.get("code") == 0 and "page" in data:
            pg = data["page"]
            return pg.get("list", []), pg.get("totalCount", 0)
    except Exception as e:
        print(f"  API page {page} fail: {e}")
    return [], 0


def insert_item(item, conn):
    """插入到数据库"""
    c = conn.cursor()

    title = item.get("title", "")
    content = item.get("content", "")
    date_str = item.get("contentDate", "") or item.get("addTime", "")[:10]
    video_id = item.get("videoId", "")
    source_url = f"http://www.zhengrunhuanjing.com/#/home/dynamic?idx=0&videoId={video_id}"

    text_content = re.sub(r"<[^>]+>", "", content).strip() if content else ""

    if not content or len(text_content) < 50:
        print(f"  Short: {title[:30]}... ({len(text_content)} chars)")

    try:
        c.execute("""
            INSERT OR IGNORE INTO gov_raw
            (title, page_url, source_url, content, publish_date, site_name)
            VALUES (?, ?, ?, ?, ?, ?)
        """, (title, source_url, source_url, content, date_str, SITE_NAME))
        affected = c.rowcount
        conn.commit()
        if affected > 0:
            print(f"  OK {title[:30]}...")
        return affected
    except Exception as e:
        print(f"  Insert fail: {e}")
        return 0


def crawl(days_back=365):
    import sqlite3
    now = datetime.now()
    cutoff = now - timedelta(days=days_back)

    conn = sqlite3.connect(SEARCH_DB, timeout=60)

    # catId=0 = 项目公示
    cat_id = 0
    limit = 20

    results, total = fetch_page(cat_id, 1, limit)
    if not results:
        print("No data!")
        return

    total_pages = (total + limit - 1) // limit
    print(f"Total: {total} records, {total_pages} pages")

    all_items = []

    # Page 1
    for r in results:
        all_items.append(r)

    # Remaining pages
    for page in range(2, total_pages + 1):
        print(f"Fetching page {page}/{total_pages}...")
        results, _ = fetch_page(cat_id, page, limit)
        all_items.extend(results)
        time.sleep(0.3)

    print(f"\nTotal items: {len(all_items)}")
    total_ok = 0

    for i, item in enumerate(all_items):
        print(f"[{i+1}/{len(all_items)}] {item.get('title','')[:30]}...")
        affected = insert_item(item, conn)
        if affected > 0:
            total_ok += 1

    conn.close()
    print(f"\nDone: {total_ok} new, {len(all_items) - total_ok} skipped/dup")
    return total_ok


if __name__ == "__main__":
    days = int(sys.argv[1]) if len(sys.argv) > 1 else 365
    crawl(days_back=days)
