#!/usr/bin/env python3
"""crawl_jishan.py - 稷山经济技术开发区-公告公示 (Custom CMS)"""
import requests, json, re, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE = "http://www.jishan.gov.cn/jsjjjskfq/gggs/"
LIST_PAGE = urljoin(BASE, "index.shtml")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
}

def extract_content(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        return f"[请求失败: {e}]", ""
    soup = BeautifulSoup(r.text, "html.parser")
    
    # Title from h2
    h2 = soup.select_one("div.info.left h2")
    detail_title = h2.get_text(strip=True) if h2 else ""
    
    # Content from div#Zoom (inside div.info)
    zoom = soup.select_one("div#Zoom")
    if zoom:
        # Find PDF iframes
        pdf_urls = []
        for iframe in zoom.find_all("iframe"):
            src = iframe.get("src", "")
            if ".pdf" in src.lower():
                pdf_urls.append(urljoin(detail_url, src))
        
        parts = []
        for child in zoom.children:
            if child.name == "table":
                # Strip excessive whitespace from HTML table to avoid markdown rendering issues
                table_html = re.sub(r'\s+', ' ', str(child)).strip()
                parts.append(table_html)
            elif child.name == "p":
                # Use empty separator to avoid <span> line-breaks splitting dates like "2026" → "202\n6"
                txt = child.get_text(separator="", strip=True)
                if txt:
                    parts.append(txt)
            elif child.name == "ul" or child.name == "ol":
                txt = child.get_text(separator="", strip=True)
                if txt:
                    parts.append(txt)
            elif child.name is None:
                txt = str(child).strip()
                if txt:
                    parts.append(txt)
        content = "\n\n".join(parts) if parts else zoom.get_text(separator="", strip=True)
    else:
        content = ""
    
    # Attachments
    atts = []
    for a in soup.select("a[href]"):
        href = a["href"]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            txt = a.get_text(strip=True) or "附件"
            full_url = urljoin(detail_url, href)
            atts.append(f"[{txt}]({full_url})")
    if atts:
        content += "\n\n" + "\n".join(atts)
    
    # If empty/short content, embed title+URL for searchability
    if not content or len(content.strip()) < 20:
        content = f'<p><a href="{detail_url}">{detail_title}</a></p>'
        if zoom:
            # Check for images (e.g. jpg plans/tables)
            for img in zoom.find_all("img"):
                src = img.get("src", "")
                if src:
                    content += f'\n\n<p><a href="{urljoin(detail_url, src)}">查看图片</a></p>'
            for iframe in zoom.find_all("iframe"):
                src = iframe.get("src", "")
                if src:
                    content += f'\n\n<p><a href="{urljoin(detail_url, src)}">嵌入式文件</a></p>'
    
    title = detail_title.strip() if detail_title else ""
    return content, title

def main(limit_pages=None):
    total_pages = 5
    if limit_pages:
        total_pages = min(limit_pages, total_pages)
    
    all_items = []
    for p in range(1, total_pages + 1):
        if p == 1:
            url = LIST_PAGE
        else:
            url = urljoin(BASE, f"index_{p}.shtml")
        
        print(f"\n--- Page {p}/{total_pages}: {url}", flush=True)
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        
        count = 0
        for li in soup.select("div.list ul > li"):
            a = li.find("a")
            span = li.find("span", class_="date")
            if not a or not span:
                continue
            href = a.get("href", "")
            title = a.get_text(strip=True) or a.get("title", "")
            date_str = span.get_text(strip=True)
            
            detail_url = urljoin(url, href)
            content, detail_title = extract_content(detail_url)
            final_title = detail_title or title
            
            item = {
                "title": final_title,
                "date": date_str[:10],
                "url": detail_url,
                "content": content
            }
            all_items.append(item)
            count += 1
            print(f"  [{p}/{total_pages}] {date_str} {final_title[:50]}...", flush=True)
        
        print(f"  页{p}: {count} 条", flush=True)
        time.sleep(0.5)
    
    result = {"total": len(all_items), "items": all_items}
    print(f"\n总计: {len(all_items)} 条", flush=True)
    print("JSON_BEGIN:" + json.dumps(result, ensure_ascii=False) + ":JSON_END", flush=True)
    return result

if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--limit", type=int, help="仅爬前N页")
    args = parser.parse_args()
    main(limit_pages=args.limit)
