#!/usr/bin/env python3
"""
crawl_shz_gov.py -- Shihezi city government info crawler (EpointWebBuilder)
"""

import sys, re, time, os, json, urllib.request, urllib.parse
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

def extract_plain_text(html_text):
    if not html_text:
        return ""
    soup = BeautifulSoup(html_text, "html.parser")
    for tag in soup(["script", "style"]):
        tag.decompose()
    text = soup.get_text(separator=" ", strip=True)
    text = re.sub(r"\s+", " ", text)
    return text[:500]

API_URL = "https://www.shz.gov.cn/EpointWebBuilder/rest/govenopen/govopeninfolist"
DEPT_NUM = "030"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "八师石河子市-政府信息公开"
PAGE_SIZE = 15
MAX_PAGES = 5

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
stats = {"new": 0, "skip": 0, "errors": 0}


def fetch_list(page):
    data = {
        "catenum": "",
        "deptnum": DEPT_NUM,
        "pageindex": str(page),
        "pagesize": str(PAGE_SIZE),
        "identifier": "",
        "title": ""
    }
    body = urllib.parse.urlencode(data).encode()
    req = urllib.request.Request(API_URL, data=body,
        headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
            "Content-Type": "application/x-www-form-urlencoded",
            "Referer": "https://www.shz.gov.cn/government_dept.html?id=030"
        })
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        j = json.loads(resp.read().decode("utf-8"))
        return j.get("custom", {}).get("list", []), j.get("custom", {}).get("total", 0)
    except Exception as e:
        print(f"  [ERROR] API error: {e}")
        return [], 0


def fetch_detail(infourl, infotype):
    if infotype == "Link":
        return ""  # External link, no content
    if infotype == "Attach":
        return ""  # Attachment, no detail page
    # News type
    url = "https://www.shz.gov.cn/govxxgk" + infourl
    try:
        req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode("utf-8")
        soup = BeautifulSoup(html, "html.parser")
        content_div = soup.find("div", class_="article-cont")
        if content_div:
            return str(content_div)
        return ""
    except Exception:
        return ""


def store_item(title, url, content, date_str):
    import sqlite3
    try:
        conn = sqlite3.connect(DB_PATH)
        c = conn.cursor()
        summary = extract_plain_text(content)
        sql = "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, content, source_url, summary, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)"
        c.execute(sql,
                  (SITE_NAME, title, url, date_str, content, url, summary, "shz_gov"))
        if c.rowcount > 0:
            stats["new"] += 1
        else:
            stats["skip"] += 1
        conn.commit()
        conn.close()
    except Exception as e:
        stats["errors"] += 1


def crawl(full=False, test_n=None):
    print("=" * 60)
    print("Shihezi city government info crawler")
    print("=" * 60)

    page_count = MAX_PAGES
    print(f"Total pages: {page_count}")

    seen = 0
    for page in range(1, page_count + 1):
        print(f"\n--- Page {page}/{page_count}")
        items, total = fetch_list(page)
        if not items:
            print("  [ERROR] No items returned")
            break

        print(f"  API total: {total}, Items this page: {len(items)}")

        for idx, item in enumerate(items):
            title = item.get("title", "")
            date_str = item.get("infodate", "")[:10]
            infourl = item.get("infourl", "")
            infotype = item.get("infotype", "News")

            if not title or not infourl:
                continue

            if date_str and date_str < THREE_YEARS_AGO:
                print(f"  [SKIP] Too old: {date_str} {title[:30]}")
                stats["skip"] += 1
                continue

            # Build detail URL
            if infotype == "News":
                url = "https://www.shz.gov.cn/govxxgk" + infourl
            else:
                url = infourl if infourl.startswith("http") else "https://www.shz.gov.cn" + infourl

            print(f"  [{idx+1}] {title[:40]} | {date_str} | {infotype}")

            if test_n and seen >= test_n:
                print(f"\n  Test mode: {test_n} items done")
                return

            content = ""
            if infotype == "News":
                content = fetch_detail(infourl, infotype)

            store_item(title, url, content, date_str)
            seen += 1
            time.sleep(0.5)

    print(f"\nDone! New: {stats['new']}, Skip: {stats['skip']}, Errors: {stats['errors']}")


if __name__ == '__main__':
    full = '--full' in sys.argv
    test_n = None
    for arg in sys.argv:
        if arg.startswith('--test='):
            test_n = int(arg.split('=')[1])
    crawl(full=full, test_n=test_n)
