#!/usr/bin/env python3
"""
乐安县人民政府 — 环境保护
Hanweb JPAAS system, list via POST search.jsp, detail via /art/... with div#zoom
"""
import requests
import re
import sys
import os
import sqlite3
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "乐安县-环境保护"
BASE_URL = "http://www.jxlean.gov.cn"
SEARCH_URL = BASE_URL + "/module/xxgk/search.jsp"
LIST_URL = BASE_URL + "/col/col27559/index.html?number=B00004B00007B00020"
DAYS = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else 365

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": LIST_URL,
}

session = requests.Session()
session.headers.update(HEADERS)


def parse_list_page(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.find("ul", class_="zfxxgk_zdgkc")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a")
        b = li.find("b")
        if a and b:
            href = a.get("href", "")
            if href and not href.startswith("http"):
                href = BASE_URL + href
            title = a.get("title", a.get_text(strip=True))
            date_str = b.get_text(strip=True)
            items.append({"page_url": href, "title": title, "publish_date": date_str})
    return items


def get_total_pages(html):
    match = re.search(r'name="nTotalCount"[^>]*value="(\d+)"', html)
    if match:
        total = int(match.group(1))
        return (total + 17) // 18
    return 1


def fetch_detail(page_url):
    try:
        r = session.get(page_url, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")

        title_el = soup.find("p", id="title")
        title = title_el.get_text(strip=True) if title_el else ""

        zoom = soup.find("div", id="zoom")
        content = ""
        if zoom:
            for tag in zoom.find_all(["script", "style"]):
                tag.decompose()
            content = str(zoom)

        return title, content
    except Exception as e:
        print(f"  [ERROR] fetch detail failed: {e}", flush=True)
        return "", ""


def main():
    print(f"[{SITE_NAME}] Starting crawl (days={DAYS})", flush=True)

    conn = sqlite3.connect(SEARCH_DB)
    c = conn.cursor()

    # Get existing page_urls for skip check
    c.execute("SELECT page_url FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
    existing = set(row[0] for row in c.fetchall())

    params = {
        "infotypeId": "B00004B00007B00020",
        "jdid": "3",
        "divid": "div448",
    }

    params["currpage"] = "1"
    r = session.post(SEARCH_URL, data=params, timeout=15)
    r.encoding = "utf-8"

    total_pages = get_total_pages(r.text)
    print(f"  Total pages: {total_pages}", flush=True)

    all_list_items = []

    for page in range(1, total_pages + 1):
        if page > 1:
            params["currpage"] = str(page)
            r = session.post(SEARCH_URL, data=params, timeout=15)
            r.encoding = "utf-8"
        items = parse_list_page(r.text)
        print(f"  Page {page}/{total_pages}: {len(items)} items", flush=True)
        all_list_items.extend(items)

    print(f"  Total items in list: {len(all_list_items)}", flush=True)

    new_count = 0
    for i, item in enumerate(all_list_items):
        if item["page_url"] in existing:
            continue

        title, content = fetch_detail(item["page_url"])
        if title:
            item["title"] = title
        item["content"] = content

        c.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date, content)
               VALUES (?, ?, ?, ?, ?, ?)""",
            (SITE_NAME, LIST_URL, item["page_url"], item["title"],
             item["publish_date"], item["content"])
        )
        conn.commit()
        new_count += 1

        if (i + 1) % 10 == 0:
            print(f"  Progress: {i+1}/{len(all_list_items)} processed, {new_count} new", flush=True)

    conn.close()
    print(f"  New items added: {new_count}", flush=True)

    conn2 = sqlite3.connect(SEARCH_DB)
    total = conn2.execute(
        "SELECT COUNT(*) FROM gov_raw WHERE site_name = ?", (SITE_NAME,)
    ).fetchone()[0]
    conn2.close()
    print(f"  Total in DB for {SITE_NAME}: {total}", flush=True)
    print(f"[{SITE_NAME}] Done!", flush=True)


if __name__ == "__main__":
    main()
