#!/usr/bin/env python3
"""
陵川县人民政府-部门工作（环境保护相关）
列表页: http://xxgk.lczf.gov.cn/lczxdw/lcsthjj/fdzdgknr/gzdt/
详情页: div#Zoom
单页20条，无分页
"""
import requests
import os
import sys
import sqlite3
import re
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "陵川县-部门工作"
LIST_URL = "http://xxgk.lczf.gov.cn/lczxdw/lcsthjj/fdzdgknr/gzdt/"
DAYS = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else 365

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

session = requests.Session()
session.headers.update(HEADERS)


def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.find("ul", class_="list-items-box-inner")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "").strip()
        title = a.get_text(strip=True)
        if not title:
            title = a.get("title", "")
        span = li.find("span")
        date_str = span.get_text(strip=True) if span else ""
        if href and not href.startswith("http"):
            href = urljoin(LIST_URL, href)
        if title and href:
            items.append({"page_url": href, "title": title, "publish_date": date_str})
    return items


def fetch_detail(page_url):
    try:
        r = session.get(page_url, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        zoom = soup.find("div", id="Zoom")
        content = ""
        if zoom:
            for tag in zoom.find_all(["script", "style"]):
                tag.decompose()
            content = str(zoom)
        return content
    except Exception as e:
        print(f"  [ERROR] {e}", flush=True)
        return ""


def main():
    print(f"[{SITE_NAME}] Starting crawl (days={DAYS})", flush=True)

    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    c = conn.cursor()

    c.execute("SELECT page_url FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
    existing = set(row[0] for row in c.fetchall())

    r = session.get(LIST_URL, timeout=15)
    r.encoding = "utf-8"
    items = parse_list(r.text)
    print(f"  Found {len(items)} items", flush=True)

    new_count = 0
    for i, item in enumerate(items):
        if item["page_url"] in existing:
            continue
        content = fetch_detail(item["page_url"])
        item["content"] = content
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, content) VALUES (?, ?, ?, ?, ?, ?)",
            (SITE_NAME, LIST_URL, item["page_url"], item["title"], item["publish_date"], item["content"])
        )
        conn.commit()
        new_count += 1

    conn.close()
    total = sqlite3.connect(SEARCH_DB, timeout=60).execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
    print(f"  New: {new_count}, Total in DB: {total}", flush=True)
    print(f"[{SITE_NAME}] Done!", flush=True)


if __name__ == "__main__":
    main()
