#!/usr/bin/env python3
"""crawl_jzth.py - 太和区人民政府 通知公告 (VSB9 CMS)"""
import requests, sys, json, re, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime

BASE = "http://www.jzth.gov.cn/xxzx/"
LIST_URL = urljoin(BASE, "tzgg.htm")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
}

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def extract_content(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        return f"<p>[请求失败: {e}]</p>", ""
    soup = BeautifulSoup(r.text, "html.parser")
    
    # Title - from detail page div.title
    title_tag = soup.select_one("div.newShow div.title")
    detail_title = title_tag.get_text(strip=True) if title_tag else ""
    
    # Content - VSB9: div#vsb_content > div.v_news_content
    content_div = soup.select_one("div#vsb_content div.v_news_content")
    if not content_div:
        content_div = soup.select_one("div#vsb_content")
    if not content_div:
        # fallback: any .content > div
        content_div = soup.select_one("div.content > div")
    
    # Also find iframes/PDF attachments embedded directly
    pdf_urls = []
    if content_div:
        for iframe in content_div.find_all("iframe"):
            src = iframe.get("src", "")
            if ".pdf" in src.lower():
                pdf_urls.append(urljoin(detail_url, src))
    
    if content_div:
        parts = []
        for child in content_div.children:
            if child.name == "table":
                parts.append(str(child))
            elif child.name == "p":
                txt = body_text(child)
                if txt:
                    parts.append(txt)
            elif child.name == "ul":
                for li in child.find_all("li", recursive=False):
                    txt = li.get_text(" ", strip=True)
                    if txt:
                        parts.append(txt)
            elif child.name is None:
                txt = str(child).strip()
                if txt:
                    parts.append(txt)
        content = "\n\n".join(parts) if parts else body_text(content_div)
    else:
        content = ""
    
    # Strip HTML tags from content for clean text
    content = re.sub(r'<[^>]+>', '', content).strip()
    
    # Attachments (link-based)
    atts = []
    for a in soup.select("a[href]"):
        href = a["href"]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            txt = a.get_text(strip=True) or "附件"
            full_url = urljoin(detail_url, href)
            atts.append(f"[{txt}]({full_url})")
    
    if atts:
        content += "\n\n" + "\n".join(atts)
    
    # If content is empty/short and has PDF iframe, embed title+URL for searchability
    if (not content or len(content.strip()) < 20) and pdf_urls:
        content = f'<p><a href="{detail_url}">{detail_title}</a></p>'
        # Also add PDF links
        for pu in pdf_urls:
            content += f'\n\n<p><a href="{pu}">PDF文件</a></p>'
    elif not content or len(content.strip()) < 20:
        # Still empty - embed original URL
        content = f'<p><a href="{detail_url}">{detail_title}</a></p>'
    
    # Clean title prefix/suffix
    title = detail_title
    if not title:
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta:
            title = meta.get("content", "")
    title = re.sub(r'[-–—]太和区人民政府\s*$', '', title).strip()
    title = re.sub(r'^太和区人民政府[-–—]\s*', '', title).strip()
    
    return content, title

def crawl_page(page_url, progress_label):
    r = requests.get(page_url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "html.parser")
    
    items = []
    for li in soup.select("ul.newsList > li"):
        a = li.find("a")
        span = li.find("span", class_="fr")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get("title", "") or a.get_text(strip=True)
        date_str = span.get_text(strip=True)
        detail_url = urljoin(page_url, href)
        
        # Get content & detail title
        content, detail_title = extract_content(detail_url)
        if detail_title:
            final_title = detail_title
        else:
            final_title = title
        
        item = {
            "title": final_title,
            "date": date_str,
            "url": detail_url,
            "content": content
        }
        items.append(item)
        print(f"  [{progress_label}] {date_str} {final_title[:50]}...", flush=True)
    
    return items

def main(limit_pages=None):
    total_pages = 22
    if limit_pages:
        total_pages = min(limit_pages, total_pages)
    
    # Determine pagination
    # Page 1 = tzgg.htm
    # Page N (2..22) = tzgg/{23-N}.htm
    
    all_items = []
    for p in range(1, total_pages + 1):
        if p == 1:
            url = LIST_URL
        else:
            url = urljoin(BASE, f"tzgg/{total_pages + 1 - p}.htm")
        
        print(f"\n--- Page {p}/{total_pages}: {url}", flush=True)
        items = crawl_page(url, f"{p}/{total_pages}")
        all_items.extend(items)
        time.sleep(0.5)
    
    # Output as JSON lines for DB insert
    result = {
        "total": len(all_items),
        "items": all_items
    }
    print(f"\n总计: {len(all_items)} 条", flush=True)
    print("JSON_BEGIN:" + json.dumps(result, ensure_ascii=False) + ":JSON_END", flush=True)
    return result

if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--limit", type=int, help="仅爬前N页")
    args = parser.parse_args()
    main(limit_pages=args.limit)
