#!/usr/bin/env python3
"""爬虫：河津市人民政府 — 环境保护领域·通知公示
www.sxhj.gov.cn/zfxxgk/fdzdgknr/hjbhly/tzgs/
"""

import os
import sys
import re
import time
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
BASE_URL = "http://www.sxhj.gov.cn/zfxxgk/fdzdgknr/hjbhly/tzgs/"
SITE_NAME = "河津市环境保护通知公示"
MAX_PAGES = 3  # pageCount=15, pageSize=20 → ~300条

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def get_page_url(page):
    """第1页=index.shtml, 第N页(N>=2)=index_{N}.shtml (1-based)"""
    if page == 1:
        return BASE_URL
    return f"{BASE_URL}index_{page}.shtml"

def extract_detail(detail_url):
    """提取详情页：标题、日期、来源、正文、附件"""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        return None, None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")

    # 标题
    h1 = soup.select_one(".mainCont h1")
    title = h1.get_text(strip=True) if h1 else ""

    # 日期 + 来源
    pub_date = ""
    source_text = ""
    info_p = soup.select_one(".mainCont .explain")
    if info_p:
        text = info_p.get_text(strip=True)
        # 发布日期：2026-07-08 10:54
        m = re.search(r"发布[日期：:]\s*(\d{4}-\d{2}-\d{2})", text)
        if m:
            pub_date = m.group(1)
        # 来源： XXX
        m2 = re.search(r"来源[：:]\s*(.+?)(?:\s*浏览次数|$)", text)
        if m2:
            source_text = m2.group(1).strip()

    # 正文
    zoom = soup.select_one("#Zoom")
    if not zoom:
        return title, pub_date, source_text, "", ""

    # 附件
    attachments = []
    for a_tag in zoom.find_all("a"):
        href = a_tag.get("href", "")
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href.lower()):
            full_url = urljoin(detail_url, href)
            text = a_tag.get_text(strip=True)
            attachments.append(f"[{text}]({full_url})")

    # 提取正文段落和表格
    parts = []
    has_table = bool(zoom.find("table"))

    for tag in zoom.find_all(["p", "table"]):
        if tag.name == "p" and tag.find_parent("table"):
            continue
        if tag.name == "table":
            if has_table:
                parts.append(str(tag))
            continue
        text = tag.get_text(strip=True)
        text = re.sub(r"\n+", "", text)
        if text:
            parts.append(text)

    if not parts:
        text = re.sub(r"\n+", "", zoom.get_text(strip=True))
        if text:
            parts.append(text)

    content = "\n\n".join(parts)
    attachments_str = "\n".join(attachments) if attachments else ""

    # 图片内容回退
    if not content.strip() and not attachments:
        imgs = zoom.find_all("img")
        if imgs:
            img_parts = []
            for img in imgs:
                src = img.get("src", "")
                alt = img.get("alt", "")
                if src:
                    full_src = urljoin(detail_url, src)
                    img_parts.append(f"![{alt}]({full_src})")
            if img_parts:
                content = f"[{title}]({detail_url})\n\n" + "\n\n".join(img_parts)
    
    if not content.strip() and attachments:
        content = f"[{title}]({detail_url})\n\n{attachments_str}"

    return title, pub_date, source_text, content, attachments_str

def crawl():
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    cur = conn.cursor()

    total = 0
    for page in range(1, MAX_PAGES + 1):
        url = get_page_url(page)
        print(f"  第{page}页: {url}")

        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERROR] {e}")
            time.sleep(2)
            continue

        soup = BeautifulSoup(r.text, "html.parser")
        ul = soup.select_one("ul.pd20.clearfix")
        if not ul:
            print(f"  [WARN] 未找到列表ul")
            items = []
        else:
            items = []
            for li in ul.find_all("li"):
                a_tag = li.find("a")
                span = li.find("span")
                if not a_tag or not span:
                    continue
                href = a_tag.get("href", "")
                title = a_tag.get("title", "") or a_tag.get_text(strip=True)
                pub_date = span.get_text(strip=True)
                if not href:
                    continue
                detail_url = urljoin(url, href)
                items.append((title, pub_date, detail_url))

        if not items:
            print(f"  第{page}页无数据，终止")
            break

        for title, list_date, detail_url in items:
            print(f"    提取: {title[:40]}...")
            full_title, pub_date, source_text, content, attachments_str = extract_detail(detail_url)

            if not full_title:
                full_title = title
            if not pub_date:
                pub_date = list_date

            date_rank = 0
            if pub_date:
                try:
                    date_rank = 0 - int(pub_date.replace("-", "") + "0000")
                except ValueError:
                    date_rank = 0

            try:
                cur.execute("""
                    INSERT OR REPLACE INTO gov_raw
                    (page_url, site_name, title, publish_date, content, summary, attachments, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                """, (
                    detail_url,
                    SITE_NAME,
                    full_title,
                    pub_date,
                    content,
                    source_text,
                    attachments_str,
                    date_rank
                ))
                total += 1
            except Exception as e:
                print(f"    [DB ERROR] {e}")

            time.sleep(0.3)

        conn.commit()
        print(f"  第{page}页完成，累计{total}条")
        time.sleep(0.5)

    conn.close()
    print(f"\n总计: {total}条")

if __name__ == "__main__":
    crawl()
