#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_linli_hp.py - 临澧县-建设项目环评
CMS: PowerEasy (动易)
列表: AJAX endpoint /zwgk/site/label/8888 (pageIndex=N, pageSize=15)
详情: /zwgk/public/6616604/ID.html
内容: div.wzcon.clearfix > p, table
标题: meta[ArticleTitle] (完整)
日期: meta[PubDate]
"""
import requests
import re
import sys
import os

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
import urllib.parse

SITE_NAME = "临澧县-建设项目环评"
BASE_URL = "https://www.linli.gov.cn"
LIST_API = "https://www.linli.gov.cn/zwgk/site/label/8888"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "X-Requested-With": "XMLHttpRequest"
}


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def get_list_items(page):
    """获取列表项（新 div.xxgk_navli 结构）"""
    params = {
        "labelName": "publicInfoList",
        "siteId": "6796514",
        "organId": "6616604",
        "pageSize": "15",
        "pageIndex": str(page),
        "catId": "1143471951",
        "type": "4",
        "length": "50",
        "dateFormat": "yyyy年MM月dd日",
        "file": "/c1/xxgk/publicInfoList_newest_zc"
    }
    try:
        r = requests.get(LIST_API, params=params, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        print(f"  FAIL: {e}")
        return []

    from bs4 import BeautifulSoup as BS
    soup = BS(r.text, 'html.parser')
    items = []

    for navli in soup.select('div.xxgk_navli'):
        a_tag = navli.find('a')
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        # 取 a 标签文本作为标题（完整，无省略号）
        title = a_tag.get_text(strip=True)
        
        # 日期（兼容新旧结构）
        date_el = navli.select_one('li.rq')
        date_str = date_el.get_text(strip=True) if date_el else ''
        m = re.search(r'(\d{4})年(\d{1,2})月(\d{1,2})日', date_str)
        if m:
            date_str = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"

        if not href or not title:
            continue
        if href.startswith('/'):
            full_url = BASE_URL + href
        elif href.startswith('http'):
            full_url = href
        else:
            full_url = f"{BASE_URL}/{href.lstrip('/')}"

        items.append({
            'title': title.strip(),
            'url': full_url,
            'date': date_str
        })

    # 兼容旧结构（tr.xxgk_nav_con）
    if not items:
        for tr in soup.select('tr.xxgk_nav_con'):
            a_tag = tr.find('a')
            if not a_tag:
                continue
            href = a_tag.get('href', '')
            title = a_tag.get('title', '') or a_tag.get_text(strip=True)
            date_td = tr.select_one('td.fbrq')
            date_str = date_td.get_text(strip=True) if date_td else ''
            m = re.search(r'(\d{4})年(\d{1,2})月(\d{1,2})日', date_str)
            if m:
                date_str = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"
            if not href or not title:
                continue
            if href.startswith('/'):
                full_url = BASE_URL + href
            elif href.startswith('http'):
                full_url = href
            else:
                full_url = f"{BASE_URL}/{href.lstrip('/')}"
            items.append({
                'title': title.strip(),
                'url': full_url,
                'date': date_str
            })

    return items


def parse_detail_page(html):
    """解析详情页：标题、日期、正文、附件"""
    from bs4 import BeautifulSoup as BS
    soup = BS(html, 'html.parser')

    # 标题
    meta_title = soup.select_one('meta[ArticleTitle]')
    title = meta_title.get('content', '').strip() if meta_title else ''
    if not title:
        h1 = soup.select_one('h1')
        if h1:
            title = h1.get_text(strip=True)

    # 日期
    meta_date = soup.select_one('meta[PubDate]')
    date_str = ''
    if meta_date:
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', meta_date.get('content', ''))
        if m:
            date_str = m.group(1)

    # 正文容器
    content_div = (soup.select_one('div.wzcon.clearfix') or
                   soup.select_one('div.xxgk-wzcon') or
                   soup.select_one('div.j-fontContent'))

    body_parts = []
    attachments = []

    if content_div:
        # 遍历 p, table, img
        for elem in content_div.find_all(['p', 'table', 'img'], recursive=True):
            # 跳过 display:none
            style = elem.get('style', '')
            if 'display:none' in style.replace(' ', ''):
                continue
            # 跳过表格内的 p（表格已整体输出）
            if elem.name == 'p' and elem.find_parent('table'):
                continue

            if elem.name == 'p':
                for a in elem.find_all('a'):
                    href = a.get('href', '')
                    if re.search(r'\.(docx?|pdf|xlsx?|rar|zip)$', href, re.I):
                        attach_name = a.get_text(strip=True) or os.path.basename(href)
                        if href.startswith('/'):
                            full_url = BASE_URL + href
                        elif href.startswith('http'):
                            full_url = href
                        else:
                            full_url = f"{BASE_URL}/{href.lstrip('/')}"
                        attachments.append({'name': attach_name, 'url': full_url})
                # strip=True 避免 span 边界产生换行
                p_text = elem.get_text(strip=True)
                if p_text:
                    body_parts.append(p_text)

            elif elem.name == 'table':
                md = html_table_to_html(elem)
                if md:
                    body_parts.append(md)

            elif elem.name == 'img':
                src = elem.get('src', '')
                alt = elem.get('alt', '')
                if src and not src.endswith('.gif'):
                    if src.startswith('/'):
                        full_src = BASE_URL + src
                    elif src.startswith('http'):
                        full_src = src
                    else:
                        full_src = f"{BASE_URL}/{src.lstrip('/')}"
                    body_parts.append(f"![{alt}]({full_src})")

    # 附件兜底（从全文找）
    if not attachments and content_div:
        for a in content_div.find_all('a', href=True):
            href = a.get('href', '')
            if re.search(r'\.(docx?|pdf|xlsx?|rar|zip)$', href, re.I):
                attach_name = a.get_text(strip=True) or os.path.basename(href)
                if href.startswith('/'):
                    full_url = BASE_URL + href
                elif href.startswith('http'):
                    full_url = href
                else:
                    full_url = f"{BASE_URL}/{href.lstrip('/')}"
                attachments.append({'name': attach_name, 'url': full_url})

    body = '\n\n'.join(body_parts)

    # 附件嵌入正文
    if attachments:
        attach_lines = []
        for a in attachments:
            attach_lines.append('[附件：{}]({})'.format(a['name'], a['url']))
        if attach_lines:
            if body:
                body += '\n\n'
            body += '\n'.join(attach_lines)

    attach_str = ','.join(a['url'] for a in attachments)

    return title, date_str, body, attach_str


def main():
    import argparse
    parser = argparse.ArgumentParser(description='临澧县-建设项目环评爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数')
    args = parser.parse_args()

    all_list_items = []
    for page in range(1, args.pages + 1):
        print(f"  列表页 {page}/{args.pages}")
        items = get_list_items(page)
        if not items:
            print(f"    → 0条，停止")
            break
        print(f"    → {len(items)}条")
        all_list_items.extend(items)

    print(f"\n共获取 {len(all_list_items)} 条列表项")

    results = []
    for i, item in enumerate(all_list_items):
        print(f"  [{i+1}/{len(all_list_items)}] ", end="")
        try:
            r = requests.get(item['url'], headers={
                "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
            }, timeout=30)
            # 强制 UTF-8 编码（页面可能缺 charset 头）
            r.encoding = r.apparent_encoding or 'utf-8'
        except Exception as e:
            print(f"  FAIL (fetch): {e}")
            continue

        title, date, body, attach = parse_detail_page(r.text)

        if not title:
            title = item['title']

        if not body or len(body.strip()) < 5:
            print(f"  跳过(空内容): {title[:30]}")
            continue

        print(f"{title[:50]}...")

        results.append({
            "title": title,
            "url": item['url'],
            "source_url": item['url'],
            "content": body,
            "pub_date": date or item['date'],
            "site_name": SITE_NAME,
            "attachments": attach
        })
        print(f"    正文: {len(body)}字 | 日期: {date or item['date']}")

    if not results:
        print("未获取到数据")
        return

    print(f"\n入库 {len(results)} 条...")
    push_to_searchdb(results, SITE_NAME)
    print(f"完成！共入库 {len(results)} 条")


if __name__ == '__main__':
    main()
