#!/usr/bin/env python3
"""
韶关市生态环境局南雄分局 - 建设项目环境影响评价信息(分类8932)
https://www.gdnx.gov.cn/sgnxsthjj/gkmlpt/index#8932
广东政府信息公开平台(gkmlpt) - 搜索JSONP API
取全量数据后过滤classify_main=8932
"""
import sys, os, json, time
import urllib.request
import sqlite3

SITE_NAME = "gdnx_sthjj_eia"
API_TMPL = "https://search.gd.gov.cn/jsonp/site/751227?callback=jsonp&pageSize=20&pageNo={}"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 40
CAT_TARGET = 8932


def fetch_page(page):
    url = API_TMPL.format(page)
    req = urllib.request.Request(url, headers=HEADERS)
    r = urllib.request.urlopen(req, timeout=30).read().decode()
    return json.loads(r[6:-1])


def main():
    total_new = 0
    total_skip = 0
    total_err = 0
    cat_count = 0

    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=30000")

    for page in range(1, MAX_PAGES + 1):
        print(f"Page {page}")
        try:
            data = fetch_page(page)
        except Exception as e:
            print(f"  Fetch error: {e}")
            break

        results = data.get("results", [])
        if not results:
            print("  No results, done")
            break

        for item in results:
            if item.get("classify_main") != CAT_TARGET:
                continue
            cat_count += 1
            title = item.get("title", "").strip()
            content = item.get("content", "")
            pub_date = item.get("pub_time", "")[:10]
            url = item.get("url", "")
            if not title or not url:
                continue
            try:
                c = conn.execute(
                    """INSERT OR IGNORE INTO gov_raw
                       (title, page_url, source_url, content, publish_date, site_name)
                       VALUES (?, ?, ?, ?, ?, ?)""",
                    (title, url, url, content, pub_date, SITE_NAME)
                )
                conn.commit()
                affected = c.rowcount
                if affected > 0:
                    total_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"  DB error: {e}")
                total_err += 1

        print(f"  results:{len(results)} cat8932:{sum(1 for r in results if r.get('classify_main')==CAT_TARGET)} total_new:{total_new}")

        if len(results) < 20:
            print("  Last page reached")
            break

        time.sleep(0.3)

    print(f"\nDone: new={total_new} skip={total_skip} err={total_err} (total cat8932 items found: {cat_count})")
    conn.close()


if __name__ == "__main__":
    main()
