#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫：寿光市人民政府 - 环评公示
站点：www.shouguang.gov.cn
CMS：寿光政府门户（静态分页 + TRS_Editor 正文）
"""

import sys, os, re, requests, time
from bs4 import BeautifulSoup

_HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, _HERE)
from crawler_lib import push_to_searchdb

SITE_NAME = "寿光市-环评公示"

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36'
}

BASE_URL = 'https://www.shouguang.gov.cn'
LIST_URL = 'https://www.shouguang.gov.cn/news/hpgs'

session = requests.Session()
session.headers.update(HEADERS)


def safe_get(url, max_retries=3):
    for attempt in range(max_retries):
        try:
            r = session.get(url, timeout=30)
            r.encoding = 'utf-8'
            return r
        except (requests.ConnectionError, requests.Timeout) as e:
            if attempt < max_retries - 1:
                time.sleep(2)
                continue
            print(f"    ⚠️ 请求失败({url}): {e}")
            return None
    return None


def fetch_list_page(page):
    """获取列表页"""
    if page == 1:
        url = f"{LIST_URL}/index.html"
    else:
        # index_1.html, index_2.html, ... (page 2 = index_1.html)
        url = f"{LIST_URL}/index_{page-1}.html"

    r = safe_get(url)
    if r is None:
        return [], 0

    soup = BeautifulSoup(r.text, 'html.parser')

    # 解析总页数
    total_pages = 0
    script_tags = soup.find_all('script', text=re.compile(r'countPage\s*=\s*(\d+)'))
    for st in script_tags:
        m = re.search(r'countPage\s*=\s*(\d+)', st.string or '')
        if m:
            total_pages = int(m.group(1))
            break

    items = []
    # 找列表ul
    ul = soup.find('ul')
    if not ul:
        return [], total_pages

    for li in ul.find_all('li', recursive=False):
        a_tag = li.find('a', href=True)
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        title = a_tag.get('title', '') or a_tag.get_text(strip=True)
        if not title or not href:
            continue

        # 相对路径转绝对
        if href.startswith('./'):
            href = LIST_URL + '/' + href[2:]
        elif not href.startswith('http'):
            if href.startswith('/'):
                href = BASE_URL + href
            else:
                href = LIST_URL + '/' + href

        date_span = li.find('span')
        date_str = date_span.get_text(strip=True) if date_span else ''

        items.append({'title': title.strip(), 'url': href, 'date': date_str})

    return items, total_pages


def fetch_detail(url):
    """获取详情页正文"""
    r = safe_get(url)
    if r is None:
        return ''

    soup = BeautifulSoup(r.text, 'html.parser')

    # 正文容器
    content_div = soup.find('div', class_='content', id='mainText')
    if not content_div:
        content_div = soup.find('div', class_=re.compile(r'content|article|nrzw|zoom', re.I))
    if not content_div:
        content_div = soup.find('div', id=re.compile(r'content|article|zoom|main', re.I))
    if not content_div:
        return ''

    # TRS_Editor
    editor = content_div.find('div', class_=re.compile(r'TRS_Editor|TRS_UEDITOR', re.I))
    if editor:
        target = editor
    else:
        target = content_div

    parts = []
    for child in target.children:
        if child.name == 'p':
            text = child.get_text(' ', strip=True)
            if text:
                parts.append(text)
        elif child.name == 'table':
            md = html_table_to_markdown(str(child))
            if md:
                parts.append(md)
        elif child.name in ('div', 'section', 'article'):
            _extract_nodes(child, parts)

    if not parts:
        text = target.get_text('\n', strip=True)
        if text:
            parts = [text]

    body = '\n\n'.join(parts)
    body = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])', '', body)

    # 附件
    attachment_links = []
    # 查找附件区域
    att_div = content_div.find('div', class_=re.compile(r'attach|file|down', re.I))
    if not att_div:
        att_div = content_div.find('ul', class_=re.compile(r'attach|file', re.I))
    if att_div:
        for a in att_div.find_all('a', href=True):
            href = a.get('href', '')
            text = a.get_text(strip=True) or os.path.basename(href)
            if href.startswith('/'):
                href = BASE_URL + href
            elif href.startswith('./'):
                href = url.rsplit('/', 1)[0] + '/' + href[2:]
            elif not href.startswith('http'):
                href = url.rsplit('/', 1)[0] + '/' + href
            attachment_links.append(f"[{text}]({href})")

    if attachment_links:
        body += '\n\n**附件：**\n' + '\n'.join(attachment_links)

    return body


def _extract_nodes(node, parts):
    for child in node.children:
        if child.name == 'p':
            text = child.get_text(' ', strip=True)
            if text:
                parts.append(text)
        elif child.name == 'table':
            md = html_table_to_markdown(str(child))
            if md:
                parts.append(md)
        elif child.name in ('div', 'section', 'article'):
            _extract_nodes(child, parts)


def html_table_to_markdown(html):
    soup = BeautifulSoup(html, 'html.parser')
    tables = soup.find_all('table')
    if not tables:
        return ''
    result = []
    for table in tables:
        rows = table.find_all('tr')
        if not rows:
            continue
        md_rows = []
        for row in rows:
            cells = row.find_all(['td', 'th'])
            if not cells:
                continue
            colspans = [int(c.get('colspan', 1)) for c in cells]
            if all(cs > 1 for cs in colspans):
                for cell in cells:
                    text = cell.get_text(' ', strip=True).replace('\n', ' ')
                    if text:
                        result.append(text)
                result.append('')
            else:
                row_data = []
                for cell in cells:
                    text = cell.get_text(' ', strip=True).replace('\n', ' ')
                    row_data.append(text)
                md_rows.append('| ' + ' | '.join(row_data) + ' |')
        if md_rows:
            result.append(md_rows[0])
            n_cols = len(md_rows[0].split('|')) - 2
            result.append('|' + '---|' * n_cols)
            for row in md_rows[1:]:
                result.append(row)
            result.append('')
    return '\n'.join(result).strip()


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--pages', type=int, default=0, help='仅爬前N页（0=全部）')
    args = parser.parse_args()
    max_pages = args.pages if args.pages else 0

    print(f"[{SITE_NAME}] 开始")

    all_items = []
    total_pages_found = 0

    for page in range(1, 100):  # 最多100页
        items, pagecount = fetch_list_page(page)
        if not items:
            if page == 1:
                print("  ⚠️ 第1页为空，退出")
                return
            print(f"  第{page}页: 空，结束")
            break
        if page == 1:
            total_pages_found = pagecount
            if total_pages_found:
                print(f"  总计 {total_pages_found} 页")
        print(f"  第{page}页: {len(items)} 条")
        all_items.extend(items)
        if total_pages_found and page >= total_pages_found:
            break
        if max_pages and page >= max_pages:
            print(f"  --pages={max_pages} 限制，停止")
            break

    print(f"\n列表合计: {len(all_items)} 条")

    if not all_items:
        print("  列表为空，退出")
        return

    db_items = []
    success = 0
    failed = 0
    for i, item in enumerate(all_items):
        url = item['url']
        title = item['title']
        date_str = item['date']
        print(f"  [{i+1}/{len(all_items)}] {title[:40]}...", end=' ')
        body = fetch_detail(url)
        if not body:
            print("⚠️ 空正文")
            failed += 1
        else:
            print(f"✅ {len(body)}字")
            success += 1

        db_items.append({
            'site_name': SITE_NAME,
            'source_url': url,
            'url': url,
            'title': title,
            'pub_date': date_str,
            'summary': body[:500] if body else '',
            'content': body,
        })

    push_to_searchdb(db_items, batch_label=SITE_NAME)
    print(f"\n[{SITE_NAME}] 完成: 共{len(all_items)}条, 正文成功{success}, 空{failed}")


if __name__ == '__main__':
    main()
