#!/usr/bin/env python3
import os
"""宜春市水利局—通知公告爬虫
API: POST /queryList (数融接口, content already embedded in response)
URL: http://slj.yichun.gov.cn/ycsslj/tzgg/pc/list.html
"""

import requests, sqlite3, json
from datetime import datetime, timedelta

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "X-Requested-With": "XMLHttpRequest",
    "Content-Type": "application/x-www-form-urlencoded"
}
API = "https://slj.yichun.gov.cn/queryList"
PAGE_SIZE = 15
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
SITE = "宜春市水利局"

def fetch_page(page):
    data = {
        "current": page,
        "pageSize": PAGE_SIZE,
        "channelId[]": "1996819342217027584",
        "webSiteCode[]": "ycsslj"
    }
    r = requests.post(API, headers=HEADERS, data=data, timeout=30)
    return r.json()

def main():
    today = datetime.now().strftime("%Y-%m-%d")
    count = 0
    conn = sqlite3.connect(os.getenv("SEARCH_DB", "/root/search.db"), timeout=60)
    c = conn.cursor()

    # get first page to know total
    res = fetch_page(1)
    total = res.get("data", {}).get("total", 0)
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print("Total articles: %d, pages: %d" % (total, total_pages))

    for page in range(1, total_pages + 1):
        res = fetch_page(page)
        results = res.get("data", {}).get("results", [])
        if not results:
            break
        for item in results:
            src = item.get("source", {})
            pub_date = src.get("pubDate", "")[:10]
            if pub_date < CUTOFF:
                continue
            title = src.get("title", "")
            cc = src.get("content", {})
            body = cc.get("content", "") if cc else ""
            urls_json = src.get("urls", "{}")
            try:
                urls = json.loads(urls_json)
            except:
                urls = {}
            detail_url = urls.get("pc", "")
            print("  [%s] %s..." % (pub_date, title[:45]), end="", flush=True)
            if not body:
                print("无正文")
                continue
            c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary, category, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_ycslj.py')""",
                (detail_url, title, body, pub_date, SITE, "http://slj.yichun.gov.cn" + detail_url, "", ""))
            count += 1
            print("OK")
        print("  --- page %d/%d done ---" % (page, total_pages))

    conn.commit()
    conn.close()
    print("\n\U00002705 宜春市水利局：共入库 %d 条" % count)

if __name__ == "__main__":
    main()
