#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""华宁县-建设项目环境影响评价信息爬虫（玉溪市政府信息公开平台）"""
import requests
from bs4 import BeautifulSoup
import re
import sys
import os
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
import urllib.parse

def make_absolute(href, base):
    if href.startswith('http://') or href.startswith('https://'):
        return href
    if href.startswith('/'):
        return base.rstrip('/') + href
    return base.rstrip('/') + '/' + href.lstrip('/')

SITE_NAME = "华宁县-环评公示"
BASE_URL = "https://www.huaning.gov.cn"
LIST_URL = "https://www.huaning.gov.cn/yxgovfront/newDepartmentContentds.jspx?path=hnxzfxxgk&channelId=7495&pageNo={}"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def get_list_items(page):
    url = LIST_URL.format(page)
    resp = requests.get(url, headers=HEADERS, timeout=30)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'html.parser')
    items = []
    list_div = soup.select_one('#listChangeDiv .conRight') or soup.select_one('.conRight')
    if not list_div:
        return items
    for li in list_div.find_all('li', recursive=False):
        a_tag = li.find('a')
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        title = a_tag.get('title', '') or a_tag.get_text(strip=True)
        spans = li.find_all('span')
        date_str = spans[-1].get_text(strip=True) if len(spans) >= 2 else (spans[0].get_text(strip=True) if spans else '')
        if not href or not title:
            continue
        items.append({
            'title': title.strip(),
            'url': make_absolute(href, BASE_URL),
            'date': date_str.strip()
        })
    return items

def fetch_detail(item):
    url = item['url']
    resp = requests.get(url, headers=HEADERS, timeout=30)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    title_tag = soup.select_one('div.article-head')
    title = title_tag.get_text(strip=True) if title_tag else item['title']
    
    date_str = item['date']
    time_div = soup.select_one('div.article-time')
    if time_div:
        m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', time_div.get_text())
        if m:
            date_str = m.group(1).replace('/', '-')
    
    # 正文：article-main-text (display:none 隐藏的版本反而有完整内容)
    content_div = soup.select_one('div.article-main-text') or soup.select_one('div.article-main')
    paragraphs = []
    attachments = []
    
    if content_div:
        for el in content_div.children:
            if el.name == 'p':
                for a in el.find_all('a'):
                    href = a.get('href', '')
                    if re.search(r'\.(docx?|pdf|xlsx?|rar|zip)$', href, re.I):
                        attach_name = a.get_text(strip=True) or os.path.basename(href)
                        attachments.append({'name': attach_name, 'url': make_absolute(href, BASE_URL)})
                # FIX: use strip=True not '\n' to avoid span-induced line breaks
                p_text = el.get_text(strip=True)
                if p_text:
                    paragraphs.append(p_text)
            elif el.name == 'table':
                md = html_table_to_html(el)
                if md:
                    paragraphs.append(md)
            elif el.name == 'div' and el.find('table'):
                for table in el.find_all('table'):
                    md = html_table_to_html(table)
                    if md:
                        paragraphs.append(md)
    
    if not attachments and content_div:
        for a in content_div.find_all('a'):
            href = a.get('href', '')
            if re.search(r'\.(docx?|pdf|xlsx?|rar|zip)$', href, re.I):
                attach_name = a.get_text(strip=True) or os.path.basename(href)
                attachments.append({'name': attach_name, 'url': make_absolute(href, BASE_URL)})
    
    content = '\n\n'.join(paragraphs)
    if attachments:
        if content:
            content += '\n\n'
        content += '\n'.join(['<p><a href="{}">{}</a></p>'.format(a['url'], a['name']) for a in attachments])
    
    return {
        'title': title,
        'content': content,
        'date': date_str,
        'url': url,
        'attachments': ','.join(a['url'] for a in attachments)
    }

def main():
    import argparse
    parser = argparse.ArgumentParser(description='华宁县环评公示爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数')
    args = parser.parse_args()
    
    all_items = []
    for page in range(1, args.pages + 1):
        items = get_list_items(page)
        if not items:
            print(f"第{page}页无数据，停止")
            break
        print(f"第{page}页 共{len(items)}条")
        all_items.extend(items)
    
    print(f"\n共获取 {len(all_items)} 条列表项")
    results = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] ", end="")
        try:
            detail = fetch_detail(item)
        except Exception as e:
            print(f"  ✗ 异常: {item['title'][:30]} - {e}")
            continue
        
        if not detail['content'] or len(detail['content'].strip()) < 5:
            print(f"  跳过(空内容): {detail['title'][:30]}")
            continue
        
        print(f"{detail['title'][:50]}...")
        results.append({
            "title": detail['title'],
            "url": detail['url'],
            "source_url": detail['url'],
            "content": detail['content'],
            "pub_date": detail['date'] or item['date'],
            "site_name": SITE_NAME,
            "attachments": detail['attachments']
        })
        print(f"    正文: {len(detail['content'])}字 | 日期: {detail['date'] or item['date']}")
    
    if not results:
        print("未获取到数据")
        return
    
    print(f"\n入库 {len(results)} 条...")
    push_to_searchdb(results, SITE_NAME)
    print(f"完成！共入库 {len(results)} 条")

if __name__ == '__main__':
    main()
