#!/usr/bin/env python3
"""crawl_jzth.py - 太和区人民政府 通知公告 (VSB9 CMS)"""
import requests, sys, json, re, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime

BASE = "http://www.jzth.gov.cn/xxzx/"
LIST_URL = urljoin(BASE, "tzgg.htm")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
}

def extract_content(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        return f"<p>[请求失败: {e}]</p>", ""
    soup = BeautifulSoup(r.text, "html.parser")
    
    # Title - from detail page div.title
    title_tag = soup.select_one("div.newShow div.title")
    detail_title = title_tag.get_text(strip=True) if title_tag else ""
    
    # Content - VSB9: div#vsb_content > div.v_news_content
    content_div = soup.select_one("div#vsb_content div.v_news_content")
    if not content_div:
        content_div = soup.select_one("div#vsb_content")
    if not content_div:
        # fallback: any .content > div
        content_div = soup.select_one("div.content > div")
    
    # Also find iframes/PDF attachments embedded directly
    pdf_urls = []
    if content_div:
        for iframe in content_div.find_all("iframe"):
            src = iframe.get("src", "")
            if ".pdf" in src.lower():
                pdf_urls.append(urljoin(detail_url, src))
    
    if content_div:
        parts = []
        for child in content_div.children:
            if child.name == "table":
                parts.append(str(child))
            elif child.name == "p":
                txt = child.get_text("\n", strip=True)
                if txt:
                    parts.append(txt)
            elif child.name == "ul":
                for li in child.find_all("li", recursive=False):
                    txt = li.get_text(" ", strip=True)
                    if txt:
                        parts.append(txt)
            elif child.name is None:
                txt = str(child).strip()
                if txt:
                    parts.append(txt)
        content = "\n\n".join(parts) if parts else content_div.get_text("\n", strip=True)
    else:
        content = ""
    
    # Strip HTML tags from content for clean text
    content = re.sub(r'<[^>]+>', '', content).strip()
    
    # Attachments (link-based)
    atts = []
    for a in soup.select("a[href]"):
        href = a["href"]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            txt = a.get_text(strip=True) or "附件"
            full_url = urljoin(detail_url, href)
            atts.append(f"[{txt}]({full_url})")
    
    if atts:
        content += "\n\n" + "\n".join(atts)
    
    # If content is empty/short and has PDF iframe, embed title+URL for searchability
    if (not content or len(content.strip()) < 20) and pdf_urls:
        content = f"[{detail_title}]({detail_url})"
        # Also add PDF links
        for pu in pdf_urls:
            content += f"\n\n[PDF文件]({pu})"
    elif not content or len(content.strip()) < 20:
        # Still empty - embed original URL
        content = f"[{detail_title}]({detail_url})"
    
    # Clean title prefix/suffix
    title = detail_title
    if not title:
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta:
            title = meta.get("content", "")
    title = re.sub(r'[-–—]太和区人民政府\s*$', '', title).strip()
    title = re.sub(r'^太和区人民政府[-–—]\s*', '', title).strip()
    
    return content, title

def crawl_page(page_url, progress_label):
    r = requests.get(page_url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "html.parser")
    
    items = []
    for li in soup.select("ul.newsList > li"):
        a = li.find("a")
        span = li.find("span", class_="fr")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get("title", "") or a.get_text(strip=True)
        date_str = span.get_text(strip=True)
        detail_url = urljoin(page_url, href)
        
        # Get content & detail title
        content, detail_title = extract_content(detail_url)
        if detail_title:
            final_title = detail_title
        else:
            final_title = title
        
        item = {
            "title": final_title,
            "date": date_str,
            "url": detail_url,
            "content": content
        }
        items.append(item)
        print(f"  [{progress_label}] {date_str} {final_title[:50]}...", flush=True)
    
    return items

def main(limit_pages=None):
    total_pages = 22
    if limit_pages:
        total_pages = min(limit_pages, total_pages)
    
    # Determine pagination
    # Page 1 = tzgg.htm
    # Page N (2..22) = tzgg/{23-N}.htm
    
    all_items = []
    for p in range(1, total_pages + 1):
        if p == 1:
            url = LIST_URL
        else:
            url = urljoin(BASE, f"tzgg/{total_pages + 1 - p}.htm")
        
        print(f"\n--- Page {p}/{total_pages}: {url}", flush=True)
        items = crawl_page(url, f"{p}/{total_pages}")
        all_items.extend(items)
        time.sleep(0.5)
    
    # Output as JSON lines for DB insert
    result = {
        "total": len(all_items),
        "items": all_items
    }
    print(f"\n总计: {len(all_items)} 条", flush=True)
    print("JSON_BEGIN:" + json.dumps(result, ensure_ascii=False) + ":JSON_END", flush=True)
    return result

if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--limit", type=int, help="仅爬前N页")
    args = parser.parse_args()
    main(limit_pages=args.limit)
