#!/usr/bin/env python3
"""马尔康市人民政府（maerkang.gov.cn）通知公告爬虫"""
import requests
import re
import sqlite3
import sys
import time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.maerkang.gov.cn"
LIST_URL = BASE_URL + "/maerkang/c100053/nav_list.shtml"
LIST_PAGES = 5
DB_PATH = "/root/search.db"
SITE_NAME = "马尔康市通知公告"
SLEEP = 1.5

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"[WARN] 请求失败: {url} - {e}")
        return None


def parse_list_items(soup):
    items = []
    ul = soup.select_one("div.nav_list_list_container > ul")
    if not ul:
        return items
    for li in ul.find_all("li", recursive=False):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a.get("href", "").strip()
        title = a.get("title", "") or ""
        span = li.find("span")
        date_str = span.get_text(strip=True) if span else ""
        if href and title:
            full_url = urljoin(BASE_URL, href)
            items.append((full_url, title, date_str))
    return items


def parse_detail(soup, url):
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title:
        title = meta_title.get("content", "").strip()
    if not title:
        h1 = soup.select_one("div.detail > h1")
        if h1:
            title = h1.get_text(strip=True)

    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date:
        m = re.search(r"\d{4}-\d{2}-\d{2}", meta_date.get("content", ""))
        if m:
            pub_date = m.group()

    content_parts = []
    container = soup.find("div", class_="common_detail")
    if not container:
        return title, pub_date, "", "[]"

    for child in container.find_all(["p", "table", "img"], recursive=True):
        if child.name == "p":
            if child.find_parent("table"):
                continue
            # Skip <p> that contains block-level children (nested <p> wrappers)
            if child.find(["p", "table", "ul", "ol", "div"]):
                continue
            text = child.get_text(strip=True)
            if text:
                content_parts.append(text)
        elif child.name == "table":
            md = html_table_to_markdown(child)
            if md:
                content_parts.append(md)
        elif child.name == "img":
            src = child.get("src", "").strip()
            alt = child.get("alt", "").strip()
            if src and not src.endswith(".gif") and "fileTypeImages" not in src:
                full_src = urljoin(BASE_URL, src)
                content_parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")

    content = "\n\n".join(content_parts)

    # Attachments
    attachments = []
    for a_tag in container.find_all("a", href=re.compile(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|txt)$", re.I)):
        href = a_tag.get("href", "").strip()
        title_att = a_tag.get_text(strip=True) or href.split("/")[-1]
        if href:
            full_url = urljoin(BASE_URL, href)
            attachments.append({"url": full_url, "title": title_att})

    return title, pub_date, content, str(attachments)


def html_table_to_markdown(table):
    rows = table.find_all("tr")
    if not rows:
        return ""
    md_parts = []
    table_rows = []
    has_title = False
    for tr in rows:
        cells = tr.find_all(["th", "td"])
        if len(cells) == 1 and cells[0].get("colspan", "") and int(cells[0].get("colspan", 1)) > 1:
            title_text = cells[0].get_text(strip=True)
            if title_text:
                md_parts.append(title_text)
                has_title = True
            continue
        row_data = [cell.get_text(strip=True) for cell in cells]
        if all(c == "" for c in row_data):
            continue
        table_rows.append("| " + " | ".join(row_data) + " |")
    if not table_rows:
        return "\n".join(md_parts)
    if has_title:
        md_parts.append("")
    md_parts.append(table_rows[0])
    sep = "| " + " | ".join(["---"] * (len(table_rows[0].split("|")) - 2)) + " |"
    md_parts.append(sep)
    md_parts.extend(table_rows[1:])
    return "\n".join(md_parts)


def get_page_url(page):
    if page == 1:
        return LIST_URL
    return f"{BASE_URL}/maerkang/c100053/nav_list_{page}.shtml"


def main():
    pages = LIST_PAGES
    if len(sys.argv) > 2 and sys.argv[1] == "--pages":
        pages = int(sys.argv[2])

    total_new = 0
    for page in range(1, pages + 1):
        url = get_page_url(page)
        print(f"[列表页] 第{page}页: {url}")
        soup = get_soup(url)
        if not soup:
            continue
        items = parse_list_items(soup)
        if not items:
            print("  无数据，可能已到末页")
            break
        print(f"  找到 {len(items)} 条")

        for item_url, list_title, list_date in items:
            print(f"  → {list_title[:50]}...")

            conn = sqlite3.connect(DB_PATH)
            c = conn.cursor()
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (item_url, SITE_NAME))
            if c.fetchone():
                conn.close()
                print("    已存在，跳过")
                continue
            conn.close()

            time.sleep(SLEEP)
            detail_soup = get_soup(item_url)
            if not detail_soup:
                continue

            detail_title, pub_date, content, attachments = parse_detail(detail_soup, item_url)
            date = pub_date or list_date
            use_title = detail_title or list_title

            if not content:
                print("    正文为空，跳过")
                continue

            conn = sqlite3.connect(DB_PATH)
            c = conn.cursor()
            try:
                summary = content[:200]
                has_table = 1 if "| ---" in content else 0
                c.execute("""INSERT OR REPLACE INTO gov_raw
                    (page_url, title, publish_date, site_name, content, attachments, summary, has_table)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
                    (item_url, use_title, date, SITE_NAME, content, attachments, summary, has_table))
                conn.commit()
                print(f"    入库成功 (表格={has_table})")
                total_new += 1
            except Exception as e:
                print(f"    入库失败: {e}")
            finally:
                conn.close()

    print(f"\n✅ {SITE_NAME} 爬取完成！共入库 {total_new} 条")


if __name__ == "__main__":
    main()
