#!/usr/bin/env python3
"""
crawl_shz_hjbh.py -- Shihezi Environmental Protection column crawler (EpointWebBuilder)
"""

import sys, re, time, os, json, urllib.request, urllib.parse
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

def extract_plain_text(html_text):
    if not html_text:
        return ""
    soup = BeautifulSoup(html_text, "html.parser")
    for tag in soup(["script", "style"]):
        tag.decompose()
    text = soup.get_text(separator=" ", strip=True)
    text = re.sub(r"\s+", " ", text)
    return text[:500]

API_URL = "https://www.shz.gov.cn/EpointWebBuilder/rest/govenopen/getgovinfolist"
SITE_GUID = "7eb5f7f1-9041-43ad-8e13-8fcb82ea831a"
DEPT_CODE = "013"
CATEGORY_NUM = "263002"
PAGE_SIZE = 15
MAX_PAGES = 5
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "八师石河子市-环境保护"

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
stats = {"new": 0, "skip": 0, "errors": 0}


def fetch_list(page):
    params = json.dumps({
        "deptcode": DEPT_CODE,
        "categorynum": CATEGORY_NUM,
        "title": "",
        "pageIndex": page,
        "pageSize": PAGE_SIZE,
        "siteGuid": SITE_GUID
    })
    data = {"params": params}
    body = urllib.parse.urlencode(data).encode()
    req = urllib.request.Request(API_URL, data=body,
        headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
            "Content-Type": "application/x-www-form-urlencoded",
            "Referer": "https://www.shz.gov.cn/government_bulletin.html"
        })
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        j = json.loads(resp.read().decode("utf-8"))
        items = j.get("custom", {}).get("data", [])
        total = j.get("custom", {}).get("total", 0)
        return items, total
    except Exception as e:
        print(f"  [ERROR] API: {e}")
        return [], 0


def fetch_detail(infourl):
    url = "https://www.shz.gov.cn" + infourl
    try:
        req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode("utf-8")
        soup = BeautifulSoup(html, "html.parser")
        content_div = soup.find("div", class_="article-cont")
        if content_div:
            return str(content_div)
        return ""
    except Exception:
        return ""


def store_item(title, url, content, date_str):
    import sqlite3
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        summary = extract_plain_text(content)
        sql = "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, content, source_url, summary, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)"
        c.execute(sql,
                  (SITE_NAME, title, url, date_str, content, url, summary, "shz_hjbh"))
        if c.rowcount > 0:
            stats["new"] += 1
        else:
            stats["skip"] += 1
        conn.commit()
        conn.close()
    except Exception as e:
        stats["errors"] += 1


def crawl(full=False, test_n=None):
    print("=" * 60)
    print("Shihezi Environmental Protection crawler")
    print("=" * 60)

    page_count = MAX_PAGES
    print(f"Total pages: {page_count}")

    seen = 0
    for page in range(1, page_count + 1):
        print(f"\n--- Page {page}/{page_count}")
        items, total = fetch_list(page)
        if not items:
            print("  [ERROR] No items returned")
            break

        print(f"  Total: {total}, Page items: {len(items)}")

        for idx, item in enumerate(items):
            title = item.get("title", "")
            date_str = item.get("infodate", "")[:10]
            infourl = item.get("infourl", "")

            if not title or not infourl:
                continue

            if date_str and date_str < THREE_YEARS_AGO:
                print(f"  [SKIP] Too old: {date_str} {title[:30]}")
                stats["skip"] += 1
                continue

            url = "https://www.shz.gov.cn" + infourl

            print(f"  [{idx+1}] {title[:45]} | {date_str}")

            if test_n and seen >= test_n:
                print(f"\n  Test mode: {test_n} items done")
                return

            content = fetch_detail(infourl)
            store_item(title, url, content, date_str)
            seen += 1
            time.sleep(0.5)

    print(f"\nDone! New: {stats['new']}, Skip: {stats['skip']}, Errors: {stats['errors']}")


if __name__ == '__main__':
    full = '--full' in sys.argv
    test_n = None
    for arg in sys.argv:
        if arg.startswith('--test='):
            test_n = int(arg.split('=')[1])
    crawl(full=full, test_n=test_n)
