#!/usr/bin/env python3
"""潍坊环保网 - 通知公告（环评验收公示）"""
import requests, re, sqlite3, os, sys
from bs4 import BeautifulSoup
from datetime import datetime

BASE = "http://www.weifanghb.com"
LIST_URL = "/news_147.html"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0",
}
DB = "/root/search.db"
site_name = "潍坊环保-通知公告"

seen_urls = set()
count = 0
max_pages = 5

def resolve_url(href):
    if href.startswith("http"):
        return href
    if href.startswith("/"):
        return BASE + href
    return BASE + "/" + href

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        content_div = (soup.select_one("div.content")
                       or soup.select_one("div.nscontent")
                       or soup.select_one("div.contMain"))
        content = ""
        if content_div:
            content = str(content_div)
        else:
            body = soup.find("body")
            if body:
                content = str(body)

        # Extract date from detail page
        pub_date = ""
        page_text = soup.get_text()
        m = re.search(r'时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', page_text)
        if m:
            pub_date = m.group(1)
        # Attachments
        attachments = []
        for a in soup.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", re.I)):
            href = a.get("href", "")
            if href:
                attachments.append(resolve_url(href))
        return content.strip(), pub_date, attachments
    except Exception as e:
        print(f"  [WARN] detail error: {e}")
        return "", "", []

conn = sqlite3.connect(DB, timeout=60)
c = conn.cursor()
# Determine total pages from page 1
total_pages = max_pages
r = requests.get(BASE + LIST_URL, headers=HEADERS, timeout=15)
r.encoding = "utf-8"
soup = BeautifulSoup(r.text, "html.parser")
all_page_links = soup.find_all("a", href=re.compile(r"/news_147_\d+\.html"))
for a in all_page_links:
    m = re.search(r'news_147_(\d+)\.html', a.get("href",""))
    if m:
        pn = int(m.group(1))
        if pn > total_pages:
            total_pages = pn
print(f"Total pages found: {total_pages}, crawling top {min(total_pages, max_pages)}")

for page in range(min(total_pages, max_pages)):
    page_url = f"{BASE}{LIST_URL}" if page == 0 else f"{BASE}/news_147_{page+1}.html"
    if page > 0:
        r = requests.get(page_url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")

    items = soup.find_all("li")
    for li in items:
        a_tag = li.select_one("a.title")
        if not a_tag:
            continue
        href = a_tag.get("href", "")
        title = a_tag.get_text(strip=True)
        if not href or "news_view" not in href:
            continue
        url = resolve_url(href)
        if url in seen_urls:
            continue
        seen_urls.add(url)

        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone():
            print(f"  [SKIP] {title[:40]}... (exists)")
            continue

        content, pub_date, attachments = get_detail(url)
        if not content or len(content) < 50:
            print(f"  [SKIP] {title[:40]}... (empty content)")
            continue

        summary = re.sub(r"<[^>]+>", "", content)
        summary = re.sub(r"\s+", " ", summary).strip()[:200]

        c.execute(
            "INSERT INTO gov_raw (title, summary, content, page_url, publish_date, category, site_name) VALUES (?,?,?,?,?,?,?)",
            (title.strip(), summary, content, url, pub_date, "环评验收", site_name),
        )
        conn.commit()
        count += 1
        print(f"  [{count}] {title[:50]} ({pub_date})")

conn.close()
print(f"\n===== DONE: {count} new records =====")
