#!/usr/bin/env python3
import os
"""Crawl 石河子市人民政府 - 通知公告"""
import sys, re, json, time, sqlite3
from datetime import datetime, timedelta
from urllib.request import Request, urlopen
import urllib.parse

BASE = "https://www.shz.gov.cn"
API_URL = BASE + "/EpointWebBuilder/rest/frontAppNotNeedLoginAction/getPageInfoList"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "石河子-通知公告"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
SITE_GUID = "7eb5f7f1-9041-43ad-8e13-8fcb82ea831a"
CATEGORY = "002002"

def fetch_page(page_index):
    inner = {"siteGuid": SITE_GUID, "categoryNum": CATEGORY,
             "pageIndex": page_index, "controlname": "subpagelist"}
    data = urllib.parse.urlencode({"params": json.dumps(inner)}).encode()
    req = Request(API_URL, data=data, headers={"User-Agent": "Mozilla/5.0"})
    try:
        with urlopen(req, timeout=15) as resp:
            return json.loads(resp.read())["custom"]
    except Exception as e:
        print(f"  API error page {page_index}: {e}")
        return None

def fetch(url, retries=3):
    for i in range(retries):
        try:
            with urlopen(url, timeout=15) as resp:
                return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
    return ""

def extract_detail(html):
    title = ""
    m = re.search(r'<meta name="ArticleTitle"[^>]*content="([^"]*)"', html)
    if m: title = m.group(1).strip()
    date = ""
    m = re.search(r'<meta name="PubDate"[^>]*content="([^"]*)"', html)
    if m: date = m.group(1).strip()[:10]
    content = ""
    m = re.search(r'<div[^>]*class="article-info"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m: content = m.group(1).strip()
    return title, date, content

def main():
    db = sqlite3.connect(DB_PATH, timeout=60)
    total_ins = 0
    total_proc = 0

    for pi in range(0, 200):
        data = fetch_page(pi)
        if not data:
            break
        items = data.get("infodata", [])
        if not items:
            break

        # Check if all past cutoff
        all_past = True
        for item in items:
            if item["infodate"] >= CUTOFF:
                all_past = False
                break
        if all_past:
            print(f"Page {pi}: all past cutoff, stopping")
            break

        for item in items:
            date_val = item["infodate"]
            if date_val < CUTOFF:
                continue

            # Skip external URLs (WeChat, etc.)
            if item["infourl"].startswith("http"):
                total_proc += 1
                continue

            detail_url = BASE + item["infourl"]
            exists = db.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (detail_url,)).fetchone()
            if exists:
                total_proc += 1
                continue

            html = fetch(detail_url)
            if not html:
                print(f"  Skip {item['title'][:30]}... (fetch fail)")
                total_proc += 1
                continue

            title, pub_date, content = extract_detail(html)
            if not title:
                title = item["title"]
            if not pub_date:
                pub_date = date_val
            if not content:
                content = f"<p>{title}</p>"

            summary = re.sub(r'<[^>]+>', "", content)[:200].strip()
            uuid = item["infoid"]
            record_id = hash(uuid) % (2**31)

            try:
                db.execute(
                    "INSERT OR REPLACE INTO gov_raw (id, title, content, publish_date, source_url, page_url, site_name, summary, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_shz.py')",
                    (record_id, title, content, pub_date, detail_url, detail_url, SITE_NAME, summary)
                )
                total_ins += 1
            except Exception as e:
                print(f"  DB Error: {e}")

            time.sleep(0.3)
            total_proc += 1

        db.commit()
        print(f"  Page {pi}: {total_ins} ins, {total_proc} proc")

    db.close()
    print(f"\nDone! {total_ins} new records inserted")

if __name__ == "__main__":
    main()
