#!/usr/bin/env python3
"""crawl_hequ.py - 河曲县人民政府-通知公告 (TRS CMS)"""
import requests, json, re, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin



import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
BASE = "http://www.hequ.gov.cn/zwyw/tzgg/"
LIST_URL = BASE
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
}

TOTAL_PAGES = 78

def extract_content(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        return f"[请求失败: {e}]", ""
    soup = BeautifulSoup(r.text, "html.parser")
    
    # Title from page
    title_tag = soup.find("meta", attrs={"name": "ArticleTitle"})
    detail_title = title_tag.get("content", "") if title_tag else ""
    if not detail_title:
        h = soup.select_one("div.article-con h1, h2")
        if h:
            detail_title = h.get_text(strip=True)
    
    # Content from TRS_Editor
    zoom = soup.select_one("div.TRS_Editor")
    if not zoom:
        zoom = soup.select_one("div.article-con")
    if not zoom:
        zoom = soup.select_one("div.contents")
    
    if zoom:
        pdf_iframes = []
        for iframe in zoom.find_all("iframe"):
            src = iframe.get("src", "")
            if ".pdf" in src.lower():
                pdf_iframes.append(urljoin(detail_url, src))
        
        parts = []
        for child in zoom.children:
            if child.name == "table":
                parts.append(re.sub(r'\s+', ' ', str(child)).strip())
            elif child.name == "p":
                txt = child.get_text(separator="", strip=True)
                if txt:
                    parts.append(txt)
            elif child.name == "div":
                txt = child.get_text(separator="", strip=True)
                if txt:
                    parts.append(txt)
            elif child.name in ("ul", "ol"):
                txt = child.get_text(separator="", strip=True)
                if txt:
                    parts.append(txt)
            elif child.name is None:
                txt = str(child).strip()
                if txt:
                    parts.append(txt)
        content = "\n\n".join(parts) if parts else zoom.get_text(separator="", strip=True)
    else:
        content = ""
    
    # Attachments
    atts = []
    for a in soup.select("p a[href]"):
        href = a["href"]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            txt = a.get_text(strip=True) or "附件"
            full_url = urljoin(detail_url, href)
            atts.append(f"[{txt}]({full_url})")
    if atts:
        content += "\n\n" + "\n".join(atts)
    
    # Fallback for empty content
    if not content or len(content.strip()) < 20:
        content = f"[{detail_title or '查看原文'}]({detail_url})"
    
    title = detail_title.strip() if detail_title else ""
    return content, title

def main(limit_pages=None):
    total = TOTAL_PAGES
    if limit_pages:
        total = min(limit_pages, total)
    
    all_items = []
    for p in range(1, min(total, _MAX_PG or total)+1):
        if p == 1:
            url = LIST_URL
        else:
            url = urljoin(BASE, f"index_{p-1}.html")
        
        print(f"\n--- Page {p}/{total}: {url}", flush=True)
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        
        count = 0
        for li in soup.select("div.aon-con ul > li"):
            a = li.find("a")
            span = li.find("span")
            if not a or not span:
                continue
            href = a.get("href", "")
            title = a.get("title", "") or a.get_text(strip=True)
            date_str = span.get_text(strip=True)
            
            detail_url = urljoin(url, href)
            content, detail_title = extract_content(detail_url)
            final_title = detail_title or title
            
            item = {
                "title": final_title,
                "date": date_str[:10],
                "url": detail_url,
                "content": content
            }
            all_items.append(item)
            count += 1
            print(f"  [{p}/{total}] {date_str} {final_title[:45]}...", flush=True)
        
        print(f"  页{p}: {count} 条", flush=True)
        time.sleep(0.5)
    
    result = {"total": len(all_items), "items": all_items}
    print(f"\n总计: {len(all_items)} 条", flush=True)
    print("JSON_BEGIN:" + json.dumps(result, ensure_ascii=False) + ":JSON_END", flush=True)
    return result

if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--limit", type=int, help="仅爬前N页")
    args = parser.parse_args()
    main(limit_pages=args.limit)
