#!/usr/bin/env python3
"""
白云鄂博矿区 - 生态环境信息公开
分页: index.html (第1页), index_1.html (第2页)...index_8.html (第9页)
列表: module_list > li > a > span (标题 + 日期)
详情页: ./202607/t20260701_933633.html 格式
"""

import sys, os, re, json, requests, time
from bs4 import BeautifulSoup

BASE_URL = "http://www.byeb.gov.cn/zwgk/fdzdgknr/zdlyxxgk/sthj/"
DOMAIN = "www.byeb.gov.cn"

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

FULL = len(sys.argv) > 1 and sys.argv[1] == 'full'
MAX_PAGES = 5 if FULL else 1

def get_list_page(page_num):
    """获取列表页HTML。page1=index.html, page2=index_1.html..."""
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"{BASE_URL}index_{page_num-1}.html"
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [WARN] 列表页 {page_num} 请求失败: {e}", file=sys.stderr)
        return ""

def parse_list(html):
    """解析 module_list 下的文章条目"""
    items = []
    for m in re.finditer(r'<a href="\.([^"]+)"[^>]*>\s*<span>([^<]+)</span>\s*<span>([^<]*)</span>', html):
        href = m.group(1).strip()
        title = m.group(2).strip()
        date_str = m.group(3).strip()
        full_url = f"http://{DOMAIN}/zwgk/fdzdgknr/zdlyxxgk/sthj{href}"
        items.append({
            "title": title,
            "url": full_url,
            "date": date_str,
        })
    return items

def get_content(url):
    """获取详情页正文和完整标题"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        r.raise_for_status()
    except Exception as e:
        return "", "", "", []
    
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # 完整标题: 从 <title> 或 h1 取
    full_title = ""
    title_tag = soup.find('title')
    if title_tag:
        t = title_tag.get_text(strip=True)
        # 去掉 "_白云鄂博区人民政府" 等后缀
        for sep in ['_白云', '_', ' - ']:
            idx = t.find(sep)
            if idx > 0:
                t = t[:idx]
                break
        full_title = t
    if not full_title:
        h1 = soup.select_one('.cont_items .title, .title h1, h1, .article_title')
        if h1:
            full_title = h1.get_text(strip=True)
    
    # 正文: .cont_items > .time_source(日期) + .cont(text) + .trs_editor_view(备用)
    content_div = (
        soup.select_one('.trs_editor_view')
        or soup.select_one('.cont')
        or soup.select_one('.cont_items')
        or soup.select_one('.main_box')
    )
    
    content_html = ""
    pub_date = ""
    attachments = []
    if content_div:
        # 日期: .time_source 或 div.source/info等
        time_source = soup.select_one('.time_source')
        if time_source:
            ts = time_source.get_text(strip=True)
            m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', ts)
            if m: pub_date = m.group(1)
        
        if not pub_date:
            for s in ['div.article-source', 'div.source', 'div.info', 'div.times', 'div.pub-date']:
                src = soup.select_one(s)
                if src:
                    pub_date = re.sub(r'\s+', ' ', src.get_text(strip=True))
                    break
        
        # 附件: .annex 区域（附件专用容器）
        annex_div = soup.select_one('.annex')
        if annex_div:
            # 从 annex 提取附件链接
            for a_tag in annex_div.find_all('a', href=True):
                h = a_tag['href']
                if not h.startswith('http'):
                    h = f"http://{DOMAIN}{h}" if h.startswith('/') else f"{url.rsplit('/', 1)[0]}/{h}"
                attachments.append({"url": h, "name": a_tag.get_text(strip=True) or os.path.basename(h)})
                a_tag.replace_with(f"[{a_tag.get_text(strip=True) or os.path.basename(h)}]({h})")
        else:
            # 无annex分区时从正文找附件链接
            for a_tag in content_div.find_all('a', href=True):
                h = a_tag['href']
                if h.endswith(('.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar')):
                    if not h.startswith('http'):
                        h = f"http://{DOMAIN}{h}" if h.startswith('/') else f"{url.rsplit('/', 1)[0]}/{h}"
                    attachments.append({"url": h, "name": a_tag.get_text(strip=True) or os.path.basename(h)})
                    a_tag.replace_with(f"[{a_tag.get_text(strip=True) or os.path.basename(h)}]({h})")
        
        for tag in content_div.find_all(['span', 'font', 'b', 'strong', 'i', 'em', 'u']):
            tag.unwrap()
        for p in content_div.find_all('p'):
            p.insert(0, '\n\n')
            p.unwrap()
        
        # 正文用空连接符——靠手动插入的 \n\n 分段
        content_html = content_div.get_text(separator='', strip=False)
        content_html = re.sub(r'\n{3,}', '\n\n', content_html).strip()
    
    if not content_html or len(content_html) < 20:
        content_html = f'<p><a href="{url}">{full_title or "无正文"}</a></p>'
    
    return content_html, pub_date, full_title, attachments

def main():
    all_items = []
    
    for page in range(1, MAX_PAGES + 1):
        print(f"正在爬取列表第 {page} 页...", file=sys.stderr)
        html = get_list_page(page)
        if not html:
            continue
        items = parse_list(html)
        print(f"  第 {page} 页: 找到 {len(items)} 条", file=sys.stderr)
        
        for item in items:
            print(f"  处理: {item['title'][:40]}...", file=sys.stderr)
            content, pub_date, full_title, attachments = get_content(item['url'])
            date = item['date'] or pub_date
            # 用详情页完整的标题替代列表页截断的标题
            use_title = full_title or item['title']
            
            all_items.append({
                "title": use_title,
                "url": item['url'],
                "date": date,
                "content": content,
                "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
            })
        
        time.sleep(0.3)
    
    for item in all_items:
        print(json.dumps(item, ensure_ascii=False))

if __name__ == '__main__':
    main()
