#!/usr/bin/env python3
"""crawl_szyq.py - 宿州市埇桥区人民政府 公示公告"""
import requests
from bs4 import BeautifulSoup
import sqlite3
import os
import re
import sys
import time

BASE_URL = "https://www.szyq.gov.cn"
COLUMN_ID = "22954469"
LIST_URL = f"{BASE_URL}/content/column/{COLUMN_ID}?pageIndex={{page}}&pageSize=20"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 20
DELAY = 1.5  # WAF rate limit

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "宿州市埇桥区人民政府"
COLUMN_NAME = "公示公告"

session = requests.Session()
session.headers.update(HEADERS)


def get_soup(url):
    try:
        r = session.get(url, timeout=TIMEOUT)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [WARN] 请求失败: {e}")
        return None


def extract_detail(url):
    soup = get_soup(url)
    if not soup:
        return "", "", ""

    # Title
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()

    # Publish date
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # Content
    content_div = soup.find("div", class_="j-fontContent")
    if not content_div:
        content_div = soup.find("div", class_="newscontnet")
    if not content_div:
        content_div = soup.find("div", class_="minh500")

    if content_div:
        # Remove scripts/styles
        for tag in content_div.find_all(["script", "style"]):
            tag.decompose()

        # Process attachments first - find file links
        attachments_list = []
        for a in content_div.find_all("a"):
            href = a.get("href", "")
            if not href:
                continue
            has_file_icon = bool(a.find_previous("img", src=re.compile(r"/assets/images/files2/")))
            if has_file_icon:
                if not href.startswith("http"):
                    href = BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href
                a_text = a.get_text(strip=True)
                if not a_text:
                    a_text = f"附件.{href.split('.')[-1]}"
                markdown_link = f"[{a_text}]({href})"
                new_tag = BeautifulSoup(f"<span>{markdown_link}</span>", "html.parser")
                a.replace_with(new_tag)
                attachments_list.append(markdown_link)

        # Process remaining <a> tags for external links
        for a in content_div.find_all("a"):
            href = a.get("href", "")
            if not href or href.startswith("#"):
                a.unwrap()
                continue
            if not href.startswith("http"):
                if href.startswith("/"):
                    href = BASE_URL + href
                else:
                    href = BASE_URL + "/" + href
            a_text = a.get_text(strip=True)
            if a_text:
                markdown_link = f"[{a_text}]({href})"
                new_tag = BeautifulSoup(f"<span>{markdown_link}</span>", "html.parser")
                a.replace_with(new_tag)
            else:
                a.unwrap()

        # Process images
        for img in content_div.find_all("img"):
            src = img.get("src", "")
            alt = img.get("alt", "")
            if src and not src.startswith("http"):
                src = BASE_URL + src if src.startswith("/") else BASE_URL + "/" + src
            img_tag = f"![{alt}]({src})"
            new_tag = BeautifulSoup(f"<span>{img_tag}</span>", "html.parser")
            img.replace_with(new_tag)

        # Extract content in document order, skip p tags inside tables
        paragraphs = []
        for tag in content_div.find_all(["p", "table"]):
            if tag.name == "table":
                paragraphs.append(str(tag))
            elif tag.name == "p":
                if tag.find_parent("table"):
                    continue
                text = tag.get_text("\n", strip=True)
                if not text or text in ("\xa0", ""):
                    continue
                text = text.replace("\xa0", " ").strip()
                if text:
                    paragraphs.append(text)

        content = "\n\n".join(paragraphs)
    else:
        content = ""

    return title, pub_date, content


def crawl():
    page = 1
    total_added = 0

    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()

    while True:
        url = LIST_URL.format(page=page)
        print(f"  第{page}页...", end=" ", flush=True)
        soup = get_soup(url)
        if not soup:
            print("FAIL")
            break

        items = soup.find_all("li", class_=re.compile(r"^(odd|even)$"))
        if not items:
            print("空列表，结束")
            break

        print(f"{len(items)}条")
        for item in items:
            a_tag = item.find("a")
            if not a_tag:
                continue
            link = a_tag.get("href", "").strip()
            if not link:
                continue
            if not link.startswith("http"):
                link = BASE_URL + link if link.startswith("/") else BASE_URL + "/" + link
            # Skip external links
            if BASE_URL not in link:
                continue
            list_title = a_tag.get("title", "").strip()
            if not list_title:
                list_title = a_tag.get_text(strip=True)

            # Check if already exists
            c.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (link,))
            if c.fetchone():
                continue

            # Extract detail
            time.sleep(DELAY)
            title, pub_date, content = extract_detail(link)
            if not title:
                title = list_title
            if not content:
                content = f"[{title}]({link})"
                print(f"    ⚠ {title} - 正文为空，回退为链接")

            # Insert
            try:
                c.execute(
                    """INSERT OR REPLACE INTO gov_raw
                    (page_url, title, site_name, publish_date, content, date_rank, summary)
                    VALUES (?,?,?,?,?,?,?)""",
                    (link, title, SITE_NAME, pub_date, content, pub_date.replace("-", "") if pub_date else "0", title[:200]),
                )
                total_added += 1
            except Exception as e:
                print(f"    ERROR: {e}")

        # Check if last page
        if len(items) < 20:
            break
        page += 1
        time.sleep(DELAY)

    conn.commit()
    conn.close()
    print(f"\n完成: 新增 {total_added} 条")
    return total_added


if __name__ == "__main__":
    print(f"站点: {SITE_NAME} - {COLUMN_NAME}")
    added = crawl()
    print(f"总计新增: {added}")
