#!/usr/bin/env python3
import os
# -*- coding: utf-8 -*-
# Crawler for Yidu.gov.cn (宜都市) - 公示公告栏目
import requests, json, re, sqlite3, time, sys
from datetime import datetime, timedelta

API_BASE = "https://xxgkapi.yichang.gov.cn/show"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
SITE_NAME = "宜都市"
FULL = "--full" in sys.argv

JSONP_RE = re.compile(r'cb\(([\s\S]+)\)')

def parse_jsonp(text):
    m = JSONP_RE.search(text)
    if m:
        return json.loads(m.group(1))
    return None

def fetch_list(page):
    url = API_BASE + "/lists?jsoncallback=cb"
    url += "&areaid=8&webid=998&recommend=9&page=%d&pagenums=20" % page
    url += "&_=%d" % int(time.time() * 1000)
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        if r.status_code == 200:
            return parse_jsonp(r.text)
    except:
        pass
    return None

def fetch_detail(n_id):
    url = API_BASE + "/detail?jsoncallback=cb&areaid=8&id=%s" % n_id
    url += "&_=%d" % int(time.time() * 1000)
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        if r.status_code == 200:
            data = parse_jsonp(r.text)
            if data and len(data) > 0:
                return data[0]
    except:
        pass
    return None

def run():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = total_skip = total_before = total_err = 0
    pages = 12 if FULL else 3

    for pg in range(1, pages + 1):
        data = fetch_list(pg)
        if not data or not data.get("lists"):
            print("Page %d: no data" % pg)
            if pg == 1:
                print("API failed, aborting")
                return
            break

        articles = data["lists"]
        for art in articles:
            n_id = art["n_id"]
            title = art.get("title", "")
            pub_date = art.get("vc_inputtime", "")[:10]

            if pub_date < CUTOFF:
                total_before += 1
                continue

            # Build page URL
            page_url = "https://www.yidu.gov.cn/zfxxgk/show.html?id=%s" % n_id

            c.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (page_url,))
            if c.fetchone():
                total_skip += 1
                continue

            detail = fetch_detail(n_id)
            if not detail:
                total_err += 1
                continue

            content = detail.get("content", "")
            if not content:
                total_skip += 1
                continue

            det_title = detail.get("title", "") or title
            det_date = detail.get("vc_inputtime", "")[:10] or pub_date

            c.execute("INSERT OR IGNORE INTO gov_raw "
                      "(site_name, source_url, page_url, title, publish_date, content, summary) "
                      "VALUES (?,?,?,?,?,?,?)",
                      (SITE_NAME, page_url, page_url, det_title,
                       det_date, content, det_title))
            total_new += 1

        print("Page %d: +%d new, %d skip, %d pre-cutoff, %d err" %
              (pg, total_new, total_skip, total_before, total_err))
        time.sleep(0.3)

        if total_before > 40:
            print("All remaining before cutoff, stopping")
            break

    conn.commit()
    conn.close()
    print("Done: %d new, %d skip, %d before cutoff, %d err" %
          (total_new, total_skip, total_before, total_err))

if __name__ == "__main__":
    run()
