#!/usr/bin/env python3
"""平罗县环境保护爬虫"""
import sys, re, os, json
sys.path.insert(0, '/root')
sys.path.insert(0, '/root/gov_crawler')
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

BASE_URL = "https://www.pingluo.gov.cn/xxgk/zdlyxxgk/hjbh"
SITE_NAME = "平罗县环境保护"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
session = requests.Session()
session.headers.update(HEADERS)


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def clean_title(title):
    """清洗标题中的多余空白"""
    if not title:
        return ""
    return re.sub(r'\s+', ' ', title).strip()


def parse_list_page(url):
    """解析列表页，返回 (items, page_count)"""
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        print(f"  [ERROR] 请求失败: {e}")
        return [], 0
    if r.status_code != 200:
        print(f"  [ERROR] HTTP {r.status_code}")
        return [], 0
    html = r.text
    soup = BeautifulSoup(html, 'html.parser')

    items = []
    for li in soup.select('.zfxxgk_zdgkc li'):
        a_tag = li.find('a')
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        title = clean_title(a_tag.get('title') or a_tag.get_text(strip=True))
        if not href or not title or href.startswith('javascript'):
            continue
        date_tag = li.find('b')
        date_str = date_tag.get_text(strip=True) if date_tag else ''
        items.append({
            'url': urljoin(url, href),
            'title': title,
            'date': date_str,
        })

    # 获取总页数
    page_count = 1
    match = re.search(r'createPageHTML\((\d+),', html)
    if match:
        page_count = int(match.group(1))

    print(f"  list: {len(items)} items on this page (total pages: {page_count})")
    return items, page_count


def parse_detail(url):
    """解析详情页，返回 (content, attachments_list)"""
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        print(f"    [ERROR] 请求详情页失败: {e}")
        return "", []
    if r.status_code != 200:
        return "", []

    html = r.text
    soup = BeautifulSoup(html, 'html.parser')

    # 正文 -> mainTextBox > mainText#zoomcon > .view.TRS_UEDITOR
    content_parts = []
    attachments = []

    main_text = soup.find('div', class_='mainTextBox')
    if main_text:
        content_div = main_text.find('div', id='zoomcon') or main_text.find('div', class_='mainText')
        if content_div:
            view_div = content_div.find('div', class_=re.compile(r'TRS_UEDITOR|view'))
            if not view_div:
                view_div = content_div
            # 提取所有 p, table, img
            for el in view_div.find_all(['p', 'table', 'img']):
                if el.name == 'p':
                    # 跳过表格内的 p
                    if el.find_parent('table'):
                        continue
                    txt = re.sub(r'\s+', ' ', el.get_text(' ', strip=True))
                    # 清除中文字符之间的空格 (span碎片拼接造成)
                    txt = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])', '', txt)
                    if txt:
                        content_parts.append(txt)
                elif el.name == 'table':
                    tbl_html = html_table_to_html(el, url)
                    if tbl_html:
                        content_parts.append(tbl_html)
                elif el.name == 'img':
                    src = el.get('src', '')
                    alt = el.get('alt', '图片')
                    if src and not src.startswith('data:'):
                        full_src = urljoin(url, src)
                        content_parts.append(f'![{alt}]({full_src})')

    # 提取附件 (PDF链接)
    for a_tag in main_text.find_all('a', href=True) if main_text else soup.find_all('a', href=True):
        href = a_tag['href']
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            full_url = urljoin(url, href)
            link_text = clean_title(a_tag.get_text(strip=True))
            att_name = link_text or os.path.basename(href) or href.split('/')[-1]
            attachments.append({'name': att_name, 'url': full_url})

    content = '\n\n'.join(content_parts)

    # 内嵌附件链接到正文
    if attachments:
        if content and len(content) > 20:
            content += '\n\n'
        content += '\n'.join(
            f'[{att["name"]}]({att["url"]})' for att in attachments
        )

    return content, attachments


def main():
    pages_to_fetch = 5  # 前5页
    all_items = []

    print(f"=== {SITE_NAME} ===")
    for page in range(pages_to_fetch):
        if page == 0:
            url = f"{BASE_URL}/index.html"
        else:
            url = f"{BASE_URL}/index_{page}.html"
        print(f"\n[Page {page+1}] {url}")
        items, _ = parse_list_page(url)
        if not items:
            print("  (empty)")
            continue
        all_items.extend(items)

    print(f"\n=== 共 {len(all_items)} 条列表项 ===")

    # 去重 (用标题去重)
    seen_titles = set()
    unique_items = []
    for item in all_items:
        if item['title'] not in seen_titles:
            seen_titles.add(item['title'])
            unique_items.append(item)
    print(f"去重后: {len(unique_items)} 条")

    # 爬详情
    count = 0
    for item in unique_items:
        print(f"  [{count+1}/{len(unique_items)}] {item['title'][:40]}...")
        content, attachments = parse_detail(item['url'])
        item['content'] = content
        item['attachment_info'] = json.dumps(attachments, ensure_ascii=False) if attachments else ''
        if content:
            item['summary'] = re.sub(r'\s+', '', content)[:200]
        else:
            item['summary'] = ''
        count += 1

    # 入库
    for item in unique_items:
        item['site_name'] = SITE_NAME
        item['source_url'] = item['url']
        item['pub_date'] = item.get('date', '')
        if item.get('attachment_info'):
            item['attachments'] = item['attachment_info']

    try:
        from crawler_lib import push_to_searchdb
        push_to_searchdb(unique_items, SITE_NAME)
        print(f"\n✅ 入库完成: {len(unique_items)} 条")
    except Exception as e:
        print(f"\n❌ 入库异常: {e}")
        import traceback
        traceback.print_exc()
        # fallback: 输出JSON
        with open(f'/root/pingluo_hjbh.json', 'w', encoding='utf-8') as f:
            json.dump(unique_items, f, ensure_ascii=False, indent=2)
        print(f"  数据已保存到 /root/pingluo_hjbh.json")


if __name__ == '__main__':
    main()
