#!/usr/bin/env python3
"""蓬溪县人民政府 - 生态环境（公示公告等）"""
import requests, re, sqlite3, os, sys
from bs4 import BeautifulSoup
from datetime import datetime

BASE = "https://www.pengxi.gov.cn"
LIST_URL = "/gongkai/kuozhan/10482.html"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
DB = "/root/search.db"
site_name = "蓬溪县-环评审批"

seen_urls = set()
count = 0
max_pages = 5

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, verify=False, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        content_div = (
            soup.select_one("div.msg-content")
            or soup.select_one("div.xqing-web-box")
        )
        content = ""
        attachments = []
        if content_div:
            content = str(content_div)
        else:
            body = soup.find("body")
            if body:
                content = str(body)
        for a in soup.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", re.I)):
            href = a.get("href", "")
            if href:
                if href.startswith("http"):
                    attachments.append(href)
                else:
                    attachments.append(BASE + href)
        return content.strip(), attachments
    except Exception as e:
        print(f"  [WARN] detail error: {e}")
        return "", []

conn = sqlite3.connect(DB, timeout=60)
c = conn.cursor()
for page in range(max_pages):
    page_url = f"{BASE}{LIST_URL}" if page == 0 else f"{BASE}{LIST_URL}?page={page+1}"
    print(f"\n=== Page {page+1}: {page_url} ===")
    try:
        r = requests.get(page_url, headers=HEADERS, verify=False, timeout=15)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] page request failed: {e}")
        continue

    soup = BeautifulSoup(r.text, "html.parser")
    items = soup.select("ul.content-list li")
    if not items:
        print(f"  No items found")
        continue

    for li in items:
        a = li.find("a")
        span = li.find("span")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get("title") or a.get_text(strip=True)
        pub_date = span.get_text(strip=True)
        url = BASE + href if href.startswith("/") else href
        if url in seen_urls:
            continue
        seen_urls.add(url)

        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone():
            print(f"  [SKIP] {title[:40]}... (exists)")
            continue

        content, attachments = get_detail(url)
        if not content or len(content) < 50:
            print(f"  [SKIP] {title[:40]}... (empty content)")
            continue

        summary = re.sub(r"<[^>]+>", "", content)
        summary = re.sub(r"\s+", " ", summary).strip()[:200]

        c.execute(
            "INSERT INTO gov_raw (title, summary, content, page_url, publish_date, category, site_name) VALUES (?,?,?,?,?,?,?)",
            (title.strip(), summary, content, url, pub_date, "环评审批", site_name),
        )
        conn.commit()
        count += 1
        print(f"  [{count}] {title[:50]} ({pub_date})")

conn.close()
print(f"\n===== DONE: {count} new records =====")
