#!/usr/bin/env python3
"""
永修县人民政府 - 部门乡镇栏目汇总
https://www.yongxiu.gov.cn/bmxzlmhz/
CMS: TRS (江西省政务统一平台)
列表: div.tzlx > ul.comlist2 > li > span(date) + a[title]
分页: index.html -> index_{N}.html, 50页共约500条
详情: meta[ArticleTitle]标题 + meta[PubDate]日期 + div#content > p#wznr 正文
适用: 服务器端运行 (使用 push_to_searchdb)
运行: python3 crawl_yongxiu_bmxz.py --pages 5   (初始5页)
      python3 crawl_yongxiu_bmxz.py --pages 1   (日跑增量)
"""

import sys, re, json
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup, Tag

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb

BASE_URL = "https://www.yongxiu.gov.cn"
LIST_PATH = "/bmxzlmhz/index.html"
SITE_NAME = "永修县人民政府-部门乡镇栏目汇总"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://www.yongxiu.gov.cn/bmxzlmhz/",
}
session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def get_page_url(n):
    """Get page URL: index.html -> index_1.html -> index_49.html"""
    if n <= 1:
        return urljoin(BASE_URL, LIST_PATH)
    base = LIST_PATH.replace(".html", "")
    return urljoin(BASE_URL, f"{base}_{n-1}.html")


def parse_list(html, base_url):
    """Parse list page, return list of (title, url, date_str)"""
    soup = BeautifulSoup(html, "html.parser")
    items = []

    # Multiple sections: div.tzlx > ul.comlist2 > li
    for ul in soup.find_all("ul", class_="comlist2"):
        for li in ul.find_all("li"):
            a_tag = li.find("a")
            if not a_tag:
                continue
            # title: from a[title] attribute (complete), fallback to text
            title = (a_tag.get("title") or "").strip()
            if not title:
                title = a_tag.get_text(strip=True)
            if not title:
                continue

            href = a_tag.get("href", "").strip()
            if not href:
                continue
            # Resolve relative URLs like ./202406/t20240605_6570600.html
            url = urljoin(base_url, href)

            # date: from span
            span = li.find("span")
            date_str = span.get_text(strip=True) if span else ""

            items.append((title, url, date_str))
    return items


def _extract_content_from_node(url, node, out):
    """
    Recursively extract content from a BeautifulSoup node into the `out` list.
    Handles: p, div (wrapper → recurse into children), table→Markdown,
    img→![alt](src), a→[text](url), ul/ol, text nodes.
    """
    if isinstance(node, str):
        t = node.strip()
        if t:
            out.append(t)
        return
    if not isinstance(node, Tag):
        return
    name = node.name.lower()

    if name in ("p", "h1", "h2", "h3", "h4", "h5", "h6"):
        # Extract all text/images/links within, preserving inline order
        parts = []
        for child in node.children:
            if isinstance(child, str):
                t = child.strip()
                if t:
                    parts.append(t)
            elif isinstance(child, Tag):
                cn = child.name.lower()
                if cn == "br":
                    parts.append("\n")
                elif cn == "img":
                    src = child.get("src", "")
                    alt = child.get("alt", "")
                    if src:
                        parts.append(f"![{alt}]({urljoin(url, src)})")
                elif cn == "a":
                    lt = child.get_text(strip=True)
                    hr = child.get("href", "")
                    parts.append(f"[{lt}]({urljoin(url, hr)})" if hr else lt)
                elif cn == "span":
                    parts.append(child.get_text(strip=True))
                elif cn == "o:p":
                    continue
                else:
                    # Unknown inline element — extract text
                    t = child.get_text(strip=True)
                    if t:
                        parts.append(t)
        txt = re.sub(r"[ \t]+", " ", "".join(parts)).strip()
        if txt:
            out.append(txt)

    elif node.name == 'table':
        tbl_html = html_table_to_html(node, url)
        if tbl_html:
            out.append(tbl_html)
    elif name == "div":
        # Recurse into children (wrapper div like ue_table)
        for child in node.children:
            _extract_content_from_node(url, child, out)

    elif name in ("ul", "ol"):
        for li in node.find_all("li", recursive=False):
            txt = body_text(li).strip()
            if txt:
                out.append("- " + txt)

    elif name == "br":
        # Standalone <br> — skip
        pass

    else:
        # Unknown block element — try to get its text
        t = body_text(node).strip()
        if t:
            out.append(t)


def parse_detail(url, html):
    """Parse detail page, return (title, date_str, content, attachments)"""
    soup = BeautifulSoup(html, "html.parser")

    # Title from meta ArticleTitle (complete, no ellipsis)
    title = ""
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()
    if not title:
        h1 = soup.select_one("div.title h1")
        if h1:
            title = h1.get_text(strip=True)

    # Date from meta PubDate
    date_str = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        date_str = md["content"].strip()

    # Content container
    content_container = soup.select_one("div#content")
    if not content_container:
        content_container = soup.select_one("div.xl-text")
    if not content_container:
        content_container = soup.select_one("#content")

    content_parts = []

    if content_container:
        # Find TRS editor view div (the real content body)
        trs_div = content_container.find("div", class_=re.compile(r"trs_editor|TRS_UEDITOR"))
        if not trs_div:
            trs_div = content_container

        # Recursively extract all content in document order
        for child in list(trs_div.children):
            _extract_content_from_node(url, child, content_parts)

        # Attachments: find <a> with file extensions
        seen_urls = set()
        for a_tag in trs_div.find_all("a"):
            href = a_tag.get("href", "").strip()
            if not href:
                continue
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar|txt)(\?|$)', href, re.I):
                full_url = urljoin(url, href)
                if full_url not in seen_urls:
                    seen_urls.add(full_url)
                    name = a_tag.get_text(strip=True) or "附件"
                    content_parts.append(f"[{name}]({full_url})")

    content = "\n\n".join(content_parts)
    return title, date_str, content


def main():
    # Parse --pages arg
    max_pages = 5
    for i, arg in enumerate(sys.argv):
        if arg == "--pages" and i + 1 < len(sys.argv):
            max_pages = int(sys.argv[i + 1])

    print(f"\n{'=' * 50}")
    print(f"{SITE_NAME} - 爬虫, pages={max_pages}")
    print(f"{'=' * 50}")

    all_items = []

    for page in range(1, max_pages + 1):
        url = get_page_url(page)
        print(f"\n  [Page {page}] {url}")
        try:
            resp = session.get(url, timeout=20)
            resp.encoding = "utf-8"
            if resp.status_code != 200:
                print(f"  HTTP {resp.status_code}, 跳过")
                continue
        except Exception as e:
            print(f"  请求失败: {e}")
            continue

        items = parse_list(resp.text, url)
        print(f"  Found {len(items)} items")
        all_items.extend(items)

        # Check if empty (no more pages)
        if not items:
            print("  [停止] 无更多内容")
            break

    if not all_items:
        print("\n[EMPTY] 未找到任何内容")
        return

    print(f"\n{'=' * 50}")
    print(f"共获取 {len(all_items)} 条列表项，开始爬取详情...")
    print(f"{'=' * 50}")

    records = []
    for idx, (title, url, date_str) in enumerate(all_items, 1):
        print(f"\n  [{idx}/{len(all_items)}] {title[:50]}...")
        try:
            resp = session.get(url, timeout=20)
            resp.encoding = "utf-8"
            if resp.status_code != 200:
                print(f"    HTTP {resp.status_code}, 跳过")
                continue
        except Exception as e:
            print(f"    请求失败: {e}")
            continue

        detail_title, detail_date, content = parse_detail(url, resp.text)

        # Use detail title if available (complete, from meta)
        final_title = detail_title or title
        final_date = detail_date or date_str

        print(f"    -> 标题: {final_title[:40]}... 日期: {final_date} 正文: {len(content)}")

        records.append({
            "title": final_title,
            "source_url": url,
            "url": url,
            "content": content,
            "publish_date": final_date,
            "site_name": SITE_NAME,
        })

    if records:
        print(f"\n[*] 入库 {len(records)} 条...")
        push_to_searchdb(records, SITE_NAME)
        print(f"  DONE: {len(records)}")
    else:
        print("\n[EMPTY] 无记录入库")


if __name__ == "__main__":
    main()
