#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_kdl_sthj.py - 昆都仑区-生态环境
CMS: 包头市昆都仑区人民政府网站 (内蒙古)
列表: index.html (第1页), index_{N-1}.html (第N页)
详情: /{YYYYMM}/t{date}_{id}.html (相对于栏目目录)
内容: div.inner > 正文区 (TRS_UEDITOR 或 table)
标题: div.title (完整)
日期: div.info
"""
import requests
import re
import sys
import os

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
import urllib.parse

SITE_NAME = "昆都仑区-生态环境"
BASE_URL = "https://www.kdl.gov.cn"
LIST_DIR = "/zfxxgk/fdzdgknr/zdlygk/sthj"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_list_page(html):
    from bs4 import BeautifulSoup as BS
    soup = BS(html, 'html.parser')
    items = []
    for li in soup.select('ul.module_list li'):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        spans = a.find_all('span')
        title = spans[0].get_text(strip=True) if spans else a.get_text(strip=True)
        date_str = spans[1].get_text(strip=True) if len(spans) >= 2 else ''
        if not href or not title:
            continue
        # make absolute URL
        if href.startswith('./'):
            full_url = f"{BASE_URL}{LIST_DIR}/{href[2:]}"
        elif href.startswith('/'):
            full_url = BASE_URL + href
        elif href.startswith('http'):
            full_url = href
        else:
            full_url = f"{BASE_URL}{LIST_DIR}/{href}"
        items.append({
            'title': title.strip(),
            'url': full_url,
            'date': date_str.strip()
        })
    return items

def parse_detail_page(html):
    from bs4 import BeautifulSoup as BS
    soup = BS(html, 'html.parser')
    
    # 标题
    title_tag = soup.select_one('div.title')
    title = title_tag.get_text(strip=True) if title_tag else ''
    
    # 日期
    date_str = ''
    info_tag = soup.select_one('div.info, div.info_item')
    if info_tag:
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', info_tag.get_text())
        if m:
            date_str = m.group(1)
    
    # 正文 - inner div 内所有内容
    inner = soup.select_one('div.inner')
    body_parts = []
    attachments = []
    
    if inner:
        top = inner.select_one('div.top')
        if top:
            # 取 top 之后的所有兄弟元素
            for sib in top.find_next_siblings():
                if sib.name == 'div' and 'annex' in (sib.get('class') or []):
                    # 附件区域
                    for a in sib.find_all('a'):
                        href = a.get('href', '')
                        if re.search(r'\.(docx?|pdf|xlsx?|rar|zip)$', href, re.I):
                            attach_name = a.get_text(strip=True) or os.path.basename(href)
                            if href.startswith('/'):
                                full_url = BASE_URL + href
                            elif href.startswith('http'):
                                full_url = href
                            elif href.startswith('./'):
                                full_url = f"{BASE_URL}{LIST_DIR}/{href[2:]}"
                            else:
                                full_url = f"{BASE_URL}{LIST_DIR}/{href.lstrip('/')}"
                            attachments.append({'name': attach_name, 'url': full_url})
                    continue
                
                if sib.name == 'p':
                    p_text = body_text(sib)
                    if p_text:
                        body_parts.append(p_text)
                elif sib.name == 'table':
                    md = html_table_to_html(sib)
                    if md:
                        body_parts.append(md)
                elif sib.find('table'):
                    for table in sib.find_all('table'):
                        md = html_table_to_html(table)
                        if md:
                            body_parts.append(md)
                elif sib.name in ['div', 'section']:
                    # TRS_UEDITOR or other div containers
                    for p in sib.find_all('p'):
                        p_text = body_text(p)
                        if p_text:
                            body_parts.append(p_text)
                    for table in sib.find_all('table'):
                        md = html_table_to_html(table)
                        if md:
                            body_parts.append(md)
    
    # 如果没有找到正文（inner为空），尝试TRS_UEDITOR
    if not body_parts:
        editor = soup.select_one('div.trs_editor_view, div.TRS_UEDITOR')
        if editor:
            for p in editor.find_all('p'):
                p_text = body_text(p)
                if p_text:
                    body_parts.append(p_text)
            for table in editor.find_all('table'):
                md = html_table_to_html(table)
                if md:
                    body_parts.append(md)
    
    # 如果还没有内容，尝试整个inner去top后的剩余
    if not body_parts and inner:
        for sib in top.find_next_siblings() if top else inner.children:
            txt = body_text(sib) if hasattr(sib, 'get_text') else ''
            if txt:
                body_parts.append(txt)
    
    body = '\n\n'.join(body_parts)
    if attachments:
        if body:
            body += '\n\n'
        body += '\n'.join(['[附件：{}]({})'.format(a['name'], a['url']) for a in attachments])
    
    attach_str = ','.join(a['url'] for a in attachments)
    
    return title, date_str, body, attach_str

def main():
    import argparse
    parser = argparse.ArgumentParser(description='昆都仑区-生态环境爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数')
    args = parser.parse_args()
    
    all_list_items = []
    for page in range(1, args.pages + 1):
        if page == 1:
            url = f"{BASE_URL}{LIST_DIR}/index.html"
        else:
            url = f"{BASE_URL}{LIST_DIR}/index_{page-1}.html"
        print(f"  列表页 {page}/{args.pages}: {url}")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
        except Exception as e:
            print(f"    FAIL: {e}")
            continue
        items = parse_list_page(r.text)
        if not items:
            print(f"    → 0条，停止")
            break
        print(f"    → {len(items)}条")
        all_list_items.extend(items)
    
    print(f"\n共获取 {len(all_list_items)} 条列表项")
    
    results = []
    for i, item in enumerate(all_list_items):
        print(f"  [{i+1}/{len(all_list_items)}] ", end="")
        try:
            r = requests.get(item['url'], headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
        except Exception as e:
            print(f"  FAIL (fetch): {e}")
            continue
        
        title, date, body, attach = parse_detail_page(r.text)
        
        if not title:
            title = item['title']
        
        if not body or len(body.strip()) < 5:
            print(f"  跳过(空内容): {title[:30]}")
            continue
        
        print(f"{title[:50]}...")
        
        results.append({
            "title": title,
            "url": item['url'],
            "source_url": item['url'],
            "content": body,
            "pub_date": date or item['date'],
            "site_name": SITE_NAME,
            "attachments": attach
        })
        print(f"    正文: {len(body)}字 | 日期: {date or item['date']}")
    
    if not results:
        print("未获取到数据")
        return
    
    print(f"\n入库 {len(results)} 条...")
    push_to_searchdb(results, SITE_NAME)
    print(f"完成！共入库 {len(results)} 条")

if __name__ == '__main__':
    main()
