#!/usr/bin/env python3
"""爬虫：遂溪县人民政府 — 公告公示
www.suixi.gov.cn/gggs/gggs/
"""

import os
import re
import time
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
BASE_URL = "http://www.suixi.gov.cn/gggs/gggs/"
SITE_NAME = "遂溪县公告公示"
MAX_PAGES = 3  # 50页×20条≈1000条

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def get_page_url(page):
    if page == 1:
        return BASE_URL
    return f"{BASE_URL}index_{page}.html"

def extract_detail(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception:
        return None, None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")
    title = ""
    h2 = soup.select_one("article.articleCon h2.title")
    if h2:
        title = h2.get_text(strip=True)

    pub_date = ""
    source_text = ""
    prop = soup.select_one("article.articleCon .property span")
    if prop:
        text = prop.get_text(strip=True)
        m = re.search(r"时间[：:]\s*(\d{4}-\d{2}-\d{2})", text)
        if m:
            pub_date = m.group(1)
        m2 = re.search(r"来源[：:]\s*([^\s|]+)", text)
        if m2:
            source_text = m2.group(1).strip()

    con = soup.select_one("article.articleCon .conTxt")
    if not con:
        return title, pub_date, source_text, "", ""

    # 附件
    attachments = []
    for a_tag in con.find_all("a"):
        href = a_tag.get("href", "")
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href.lower()):
            text = a_tag.get_text(strip=True) or a_tag.get("alt", "")
            attachments.append(f"[{text}]({href})")

    # 提取正文
    parts = []
    has_table = bool(con.find("table"))
    for tag in con.find_all(["p", "table"]):
        if tag.name == "p" and tag.find_parent("table"):
            continue
        if tag.name == "table":
            if has_table:
                parts.append(str(tag))
            continue
        text = tag.get_text(strip=True)
        text = re.sub(r"\n+", "", text)
        if text:
            parts.append(text)

    content = "\n\n".join(parts)
    attachments_str = "\n".join(attachments) if attachments else ""

    # 图片内容回退
    if not content.strip() and not attachments:
        imgs = con.find_all("img")
        if imgs:
            img_parts = []
            for img in imgs:
                src = img.get("src", "")
                alt = img.get("alt", "")
                if src:
                    img_parts.append(f"![{alt}]({src})")
            if img_parts:
                content = f"[{title}]({detail_url})\n\n" + "\n\n".join(img_parts)

    if not content.strip() and attachments:
        content = f"[{title}]({detail_url})\n\n{attachments_str}"

    return title, pub_date, source_text, content, attachments_str

def crawl():
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    cur = conn.cursor()
    total = 0

    for page in range(1, MAX_PAGES + 1):
        url = get_page_url(page)
        print(f"  第{page}页: {url}")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERROR] {e}")
            time.sleep(2)
            continue

        soup = BeautifulSoup(r.text, "html.parser")
        ul = soup.select_one("ul.newsList")
        if not ul:
            print(f"  [WARN] 未找到列表ul")
            break

        items = []
        for li in ul.find_all("li"):
            a_tag = li.find("a")
            span = li.find("span")
            if not a_tag or not span:
                continue
            href = a_tag.get("href", "")
            title = a_tag.get_text(strip=True)
            list_date = span.get_text(strip=True).strip("[]").strip()
            if not href:
                continue
            detail_url = href if href.startswith("http") else urljoin(url, href)
            items.append((title, list_date, detail_url))

        if not items:
            print(f"  第{page}页无数据，终止")
            break

        for title, list_date, detail_url in items:
            print(f"    提取: {title[:40]}...")
            full_title, pub_date, source_text, content, attachments_str = extract_detail(detail_url)
            if not full_title:
                full_title = title
            if not pub_date:
                pub_date = list_date

            date_rank = 0
            if pub_date:
                try:
                    date_rank = 0 - int(pub_date.replace("-", "") + "0000")
                except ValueError:
                    date_rank = 0

            try:
                cur.execute("""
                    INSERT OR REPLACE INTO gov_raw
                    (page_url, site_name, title, publish_date, content, summary, attachments, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                """, (detail_url, SITE_NAME, full_title, pub_date, content, source_text, attachments_str, date_rank))
                total += 1
            except Exception as e:
                print(f"    [DB ERROR] {e}")

            time.sleep(0.3)

        conn.commit()
        print(f"  第{page}页完成，累计{total}条")
        time.sleep(0.5)

    conn.close()
    print(f"\n总计: {total}条")

if __name__ == "__main__":
    crawl()
