#!/usr/bin/env python3
"""
运城市生态环境局 - 通知公告
https://sthjj.yuncheng.gov.cn/xwzx/tzgg/
政府网站 - 自定义CMS
"""
import re
import sys
import time
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "运城市生态环境局-通知公告"
CATEGORY = "山西"
BASE_URL = "https://sthjj.yuncheng.gov.cn/xwzx/tzgg/"
MAX_PAGES = 4  # 共4页，61条
PAGE_SIZE = 20

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://sthjj.yuncheng.gov.cn/",
}

INSERT_SQL = """
INSERT OR IGNORE INTO gov_raw 
    (site_name, page_url, title, publish_date, summary, content, category, attachments)
    VALUES (?, ?, ?, ?, '', ?, ?, ?)
"""


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def extract_list_page(page_num):
    """Fetch a list page and extract article URLs + titles + dates."""
    if page_num == 1:
        url = f"{BASE_URL}index.shtml"
    else:
        url = f"{BASE_URL}index_{page_num}.shtml"

    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Failed to fetch page {page_num}: {e}", flush=True)
        return []

    if r.status_code != 200:
        print(f"  [ERROR] HTTP {r.status_code} for page {page_num}", flush=True)
        return []

    soup = BeautifulSoup(r.text, "html.parser")
    items = []

    # Find the list area
    list_div = soup.select_one("div.tab_list")
    if not list_div:
        list_div = soup

    for li in list_div.find_all("li"):
        a_tag = li.find("a")
        if not a_tag:
            continue

        href = a_tag.get("href", "").strip()
        if not href or href.startswith("javascript") or 'gov.cn' in href:
            continue

        # Title from title attribute (full)
        title = a_tag.get("title", "").strip()
        if not title:
            title = a_tag.get_text(strip=True)

        full_url = urljoin(BASE_URL, href)

        # Extract date
        li_text = li.get_text(strip=True)
        date_match = re.search(r'(20\d{2}[-/.]\d{2}[-/.]\d{2})', li_text)
        date_str = ""
        if date_match:
            date_str = date_match.group(1).replace("/", "-")

        items.append({
            "title": title,
            "url": full_url,
            "date": date_str,
            "is_pdf": href.lower().endswith('.pdf'),
        })

    print(f"  Page {page_num}: {len(items)} items", flush=True)
    return items


def extract_detail(url):
    """Fetch a detail page and extract full content."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Failed to fetch {url}: {e}", flush=True)
        return None

    if r.status_code != 200:
        print(f"  [ERROR] HTTP {r.status_code} for {url}", flush=True)
        return None

    soup = BeautifulSoup(r.text, "html.parser")
    
    # Get title
    title = ""
    # Prefer meta ArticleTitle (cleanest)
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    
    # Fall back to <title> tag with suffix cleanup
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            t = title_tag.get_text(strip=True)
            t = re.sub(r'[-–—].*?(网站|页面|运城市).*$', '', t).strip()
            if t:
                title = t

    # Get date from the content text "发表时间：YYYY-MM-DD HH:MM"
    page_text = soup.get_text()
    pub_date = ""
    date_match = re.search(r'发表时间[：:]\s*(20\d{2}[-/]\d{2}[-/]\d{2})', page_text)
    if date_match:
        pub_date = date_match.group(1)

    # Get content from info_content_mid
    content_div = soup.select_one("div.info_content_mid")
    if not content_div:
        content_div = soup.select_one("div.info_area_box")
    if not content_div:
        content_div = soup.select_one("div.mainBox")

    content_text = ""
    attachments_list = []

    if content_div:
        # Build content preserving tables as Markdown-style text tables
        content_html_str = str(content_div)
        
        # Replace tables with placeholders
        all_tables = content_div.find_all("table")
        for ti, table in enumerate(all_tables):
            marker = f"\n__TABLE_{ti}__\n"
            content_html_str = content_html_str.replace(str(table), marker, 1)
        
        # Get clean text (non-table parts)
        clean_soup = BeautifulSoup(content_html_str, "html.parser")
        clean_text = body_text(clean_soup)
        
        # Reconstruct: interleave table renderings with surrounding text
        parts = []
        segments = clean_text.split("\n")
        table_idx = 0
        for seg in segments:
            seg = seg.strip()
            if not seg:
                continue
            if seg.startswith("__TABLE_") and seg.endswith("__"):
                ti = table_idx
                table_idx += 1
                if ti < len(all_tables):
                    table = all_tables[ti]
                    rows = table.find_all("tr")
                    if rows:
                        num_cols = 0
                        for row in rows:
                            cells = row.find_all(["td", "th"])
                            num_cols = max(num_cols, len(cells))
                        
                        if num_cols == 2:
                            # Key-value table
                            col_width = 16
                            table_lines = []
                            for row in rows:
                                cells = row.find_all(["td", "th"])
                                if len(cells) >= 2:
                                    label = cells[0].get_text(" ", strip=True)
                                    value = body_text(cells[1])
                                    if label:
                                        table_lines.append(f"{label.ljust(col_width)} | {value}")
                                    else:
                                        table_lines.append(f"{'':{col_width}} | {value}")
                                elif len(cells) == 1:
                                    table_lines.append(cells[0].get_text(" ", strip=True))
                            if table_lines:
                                parts.append("\n" + "\n".join(table_lines) + "\n")
                        else:
                            # Multi-column data table: Markdown table
                            col_widths = [0] * num_cols
                            for row in rows:
                                cells = row.find_all(["td", "th"])
                                for i, cell in enumerate(cells):
                                    cell_text = cell.get_text(" ", strip=True)
                                    col_widths[i] = max(col_widths[i], len(cell_text))
                            col_widths = [max(6, min(w + 2, 40)) for w in col_widths]
                            table_lines = []
                            for ri, row in enumerate(rows):
                                cells = row.find_all(["td", "th"])
                                cell_texts = []
                                for ci in range(num_cols):
                                    txt = cells[ci].get_text(" ", strip=True) if ci < len(cells) else ""
                                    cell_texts.append(txt.ljust(col_widths[ci]))
                                sep = "| "
                                table_lines.append(sep + sep.join(cell_texts) + " |")
                                if ri == 0:
                                    sep_line = []
                                    for w in col_widths:
                                        sep_line.append("-" * w)
                                    table_lines.append("|-" + "-|-".join(sep_line) + "-|")
                            if table_lines:
                                parts.append("\n" + "\n".join(table_lines) + "\n")
            else:
                parts.append(seg)
        
        content_text = "\n".join(parts)

        # Extract file attachments
        for a_tag in content_div.find_all("a", href=True):
            href = a_tag["href"]
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|rar|zip)$', href, re.IGNORECASE):
                att_title = a_tag.get_text(strip=True) or href.split("/")[-1]
                full_url = urljoin(url, href)
                attachments_list.append({"name": att_title, "url": full_url})

    # Handle empty content
    if len(content_text.strip()) < 20:
        content_text = f"[{title or url.split('/')[-1]}]({url})\n（本文为PDF附件）"

    return {
        "title": title,
        "content": content_text,
        "pub_date": pub_date,
        "attachments": json.dumps(attachments_list, ensure_ascii=False) if attachments_list else "",
    }


def crawl(test_mode=False, max_pages=MAX_PAGES):
    """Main crawl function."""
    print(f"[{SITE_NAME}] Starting crawl, max_pages={max_pages}", flush=True)
    conn = init_db()
    cur = conn.cursor()

    total_inserted = 0
    total_skipped = 0
    pages_crawled = 0

    for page_num in range(1, max_pages + 1):
        items = extract_list_page(page_num)
        if not items:
            print(f"  Page {page_num} empty, stopping", flush=True)
            break

        pages_crawled += 1
        for item in items:
            title = item["title"]
            url = item["url"]
            date = item["date"]

            # Check if already exists
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                total_skipped += 1
                continue

            # For PDF items, no detail page to fetch
            if item["is_pdf"]:
                content = f'<p><a href="{url}">{title}</a></p>\n（PDF附件）'
                attachments = json.dumps([{"name": title, "url": url}], ensure_ascii=False)
                pub_date = date
            else:
                detail = extract_detail(url)
                if detail is None:
                    total_skipped += 1
                    continue

                # Use detail title if more complete
                if detail["title"] and len(detail["title"]) > len(title):
                    title = detail["title"]

                content = detail["content"]
                pub_date = date or detail["pub_date"]
                attachments = detail["attachments"]

            cur.execute(INSERT_SQL, (
                SITE_NAME, url, title, pub_date,
                content, CATEGORY, attachments,
            ))
            total_inserted += 1

            if test_mode and total_inserted >= 10:
                break

        conn.commit()

        if test_mode and total_inserted >= 10:
            print(f"  Test mode: stopping after {total_inserted} inserts", flush=True)
            break

        time.sleep(0.5)

    conn.close()
    print(f"[{SITE_NAME}] Done. Inserted={total_inserted}, Skipped={total_skipped}, Pages={pages_crawled}", flush=True)
    return total_inserted


if __name__ == "__main__":
    test_mode = "--test" in sys.argv
    mp = 3 if test_mode else MAX_PAGES
    if "--max-pages" in sys.argv:
        idx = sys.argv.index("--max-pages")
        if idx + 1 < len(sys.argv):
            mp = int(sys.argv[idx + 1])
    crawl(test_mode=test_mode, max_pages=mp)
