#!/usr/bin/env python3
"""孝南区-通知公告爬虫
http://www.xiaonan.gov.cn/tzgg/index.jhtml
JEECMS, 20条/页, index_N.jhtml分页
详情: div.contitle标题, div.conxx日期/来源, div.r-rest正文
"""
import requests
import sqlite3
import json
import os
import re
import argparse
from urllib.parse import urljoin
from bs4 import BeautifulSoup

BASE_URL = "http://www.xiaonan.gov.cn"
LIST_URL = "http://www.xiaonan.gov.cn/tzgg/index.jhtml"

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "孝南区-通知公告"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def get_soup(url, timeout=30):
    resp = requests.get(url, headers=HEADERS, timeout=timeout)
    resp.encoding = "utf-8"
    return BeautifulSoup(resp.text, "html.parser")


def parse_list_page(soup):
    """div.inside > ul > li > a + span"""
    items = []
    inside = soup.find("div", class_="inside")
    if not inside:
        return items
    ul = inside.find("ul")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href or href.startswith("#") or href.startswith("javascript"):
            continue
        full_url = urljoin(LIST_URL, href)
        title = a.get_text(strip=True)
        span = li.find("span")
        pub_date = span.get_text(strip=True) if span else ""
        items.append((full_url, title, pub_date))
    return items


def extract_detail(soup, url):
    title = ""
    pub_date = ""
    source = ""
    content = ""
    attachments = []

    # Title - div.contitle
    title_div = soup.select_one("div.contitle")
    if title_div:
        title = title_div.get_text(strip=True)

    # Date/Source - div.conxx > span
    conxx = soup.select_one("div.conxx")
    if conxx:
        txt = conxx.get_text(strip=True)
        m = re.search(r"发布时间[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})", txt)
        if m:
            pub_date = m.group(1)
        m = re.search(r"信息来源[：:]\s*(.+?)(?:\s|$)", txt)
        if m:
            source = m.group(1).strip()

    # Content - div.r-rest
    article = soup.select_one("div.r-rest")
    if not article:
        article = soup.select_one("div.insidexw_con")

    if article:
        # Attachments
        for a_tag in article.find_all("a"):
            a_href = a_tag.get("href", "")
            if a_href.endswith((".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar")):
                if not a_href.startswith("http"):
                    a_href = urljoin(BASE_URL, a_href)
                attachments.append({
                    "name": a_tag.get_text(strip=True) or os.path.basename(a_href),
                    "url": a_href,
                })

        # Build content preserving tables and paragraphs
        text_parts = []
        for child in article.children:
            if not hasattr(child, "name") or not child.name:
                continue
            if child.name in ("script", "style"):
                continue
            if child.name == "table":
                text_parts.append(str(child))
                text_parts.append("")
            elif child.name in ("p", "div"):
                inner_table = child.find("table")
                if inner_table:
                    text_parts.append(str(inner_table))
                    text_parts.append("")
                    continue
                txt = child.get_text(strip=True)
                if txt:
                    text_parts.append(txt)
            elif child.name in ("br",):
                text_parts.append("")

        content = "\n\n".join(text_parts)

        # If content is empty but has images, embed as markdown
        if len(content.strip()) < 20:
            imgs = article.find_all("img")
            if imgs:
                img_lines = []
                for img in imgs:
                    src = img.get("src", "")
                    if src and not src.startswith("http"):
                        src = urljoin(BASE_URL, src)
                    alt = img.get("alt", "") or img.get("title", "") or ""
                    if src:
                        img_lines.append("![%s](%s)" % (alt, src))
                content = '<p><a href="%s">%s</a></p>\n\n%s' % (url, title, "\n".join(img_lines)) if title else "\n".join(img_lines)

    return title, pub_date, source, content, attachments


def crawl(max_pages=None):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()

    # Build page list
    pages = []
    pages.append(LIST_URL)
    if max_pages is None or max_pages > 1:
        limit = max_pages if max_pages else 99
        for n in range(2, limit + 1):
            pages.append("http://www.xiaonan.gov.cn/tzgg/index_%d.jhtml" % n)

    all_items = []
    for page_url in pages:
        try:
            soup = get_soup(page_url)
            items = parse_list_page(soup)
            if not items:
                print("[列表] %s -> 0 条（可能已无更多页）" % page_url)
                break
            print("[列表] %s -> %d 条" % (page_url, len(items)))
            all_items.extend(items)
        except Exception as e:
            print("[列表] %s -> 异常: %s" % (page_url, str(e)))
            break

    print("\n列表共 %d 条，开始抓详情..." % len(all_items))
    new_count = 0
    error_count = 0
    skip_count = 0

    for idx, (page_url, list_title, list_date) in enumerate(all_items, 1):
        cur.execute(
            "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
            (page_url, SITE_NAME),
        )
        if cur.fetchone():
            skip_count += 1
            print("  [%d/%d] 跳过: %s..." % (idx, len(all_items), list_title[:30]))
            continue

        print("  [%d/%d] %s..." % (idx, len(all_items), list_title[:40]))
        try:
            soup = get_soup(page_url)
            title, pub_date, source, content, attachments = extract_detail(soup, page_url)
            if not title:
                title = list_title
            if not pub_date:
                pub_date = list_date

            att_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""

            cur.execute(
                """INSERT OR REPLACE INTO gov_raw (page_url, site_name, title, publish_date, source_url, content, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, 'crawl_xiaonan.py')""",
                (page_url, SITE_NAME, title, pub_date, page_url, content, att_json),
            )
            conn.commit()
            new_count += 1
            print("    -> 新增 (%d chars)" % len(content))
        except Exception as e:
            error_count += 1
            print("    -> 异常: %s" % str(e))

    conn.close()
    return new_count, skip_count, len(all_items), error_count


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None)
    args = parser.parse_args()

    new, skip, total, errors = crawl(args.max_pages)
    print("\n完成: 新增 %d, 跳过 %d, 列表 %d, 异常 %d" % (new, skip, total, errors))
