#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
建峰集团 - 环境公示 爬虫 (cnjf.com)
栏目: 信息公开 > 环境公示 (classid=26)
CMS: ASP.NET 自建
列表: ul.attractbid-nav > li > a
分页: ?classid=26&page=N
详情: div.video-detail > h3(标题), span(日期), p(正文)
"""
import re
import sys
import time
import os
import subprocess
from urllib.parse import urljoin

import requests
from bs4 import BeautifulSoup

# ── 配置 ──────────────────────────────────────────
BASE_URL = "https://www.cnjf.com"
LIST_PATH = "/aspx/ch/dutylist.aspx"
LIST_PARAMS = {"classid": 26}
SITE_NAME = "建峰集团-环境公示"
GROUP = "企业"
DB_PATH = os.getenv("SEARCH_DB", "/mnt/data/search.db")
MAX_PAGES = 5  # 默认5页

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

session = requests.Session()
session.headers.update(HEADERS)


def clean_text(text):
    """清理标题：strip &middot;&nbsp;实体前缀和多余空白"""
    if not text:
        return ""
    text = re.sub(r'(&middot;|&nbsp;|\s)+', ' ', text)
    return text.strip()


def clean_content_html(html_content, detail_url):
    """提取正文：段落\n\n分隔、表格HTML保留、附件嵌入"""
    if not html_content:
        return ""
    soup = BeautifulSoup(html_content, 'html.parser')
    for tag in soup(['script', 'style']):
        tag.decompose()

    parts = []
    for child in soup.children:
        if not child.name:
            continue
        if child.name == 'p':
            # 附件链接嵌入
            for a_tag in child.find_all('a', href=True):
                href = a_tag['href']
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                    a_text = a_tag.get_text(strip=True) or href
                    full_url = urljoin(detail_url, href) if not href.startswith('http') else href
                    a_tag.replace_with(soup.new_string(f' [{a_text}]({full_url}) '))
            t = child.get_text(strip=True)
            if t:
                parts.append(t)
        elif child.name == 'table':
            parts.append(str(child))
        elif child.name in ('div', 'section'):
            for p in child.find_all('p', recursive=True):
                if p.find_parent('table'):
                    continue
                t = p.get_text(strip=True)
                if t:
                    parts.append(t)
            for tbl in child.find_all('table', recursive=True):
                parts.append(str(tbl))
        elif child.name == 'img':
            src = child.get('src', '')
            alt = child.get('alt', '')
            if src:
                full_src = urljoin(detail_url, src) if not src.startswith('http') else src
                parts.append(f'![{alt}]({full_src})')

    return '\n\n'.join(parts)


def fetch(url, retries=3):
    for i in range(retries):
        try:
            r = session.get(url, timeout=30)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
            print(f"  ⚠️ HTTP {r.status_code} for {url[:80]}")
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
                continue
            print(f"  ❌ 请求失败: {url[:80]}: {e}")
    return None


def parse_list(html):
    """解析列表页，返回 [(title, url, date_str), ...]"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.select_one('ul.attractbid-nav')
    if not ul:
        return items

    for li in ul.find_all('li', recursive=False):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href']
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)

        # 日期
        time_div = a.select_one('div.attractbid-time')
        if time_div:
            year_tag = time_div.find('p')
            month_tag = time_div.find('span')
            year = year_tag.get_text(strip=True) if year_tag else ''
            month = month_tag.get_text(strip=True) if month_tag else ''
            date_str = f"{year}-{month}" if year and month else year or month
        else:
            date_str = ''

        # 标题
        menu_div = a.select_one('div.attractbid-menu')
        title = ''
        if menu_div:
            p = menu_div.find('p')
            if p:
                title = clean_text(p.get_text(strip=True))

        if title and href:
            items.append((title, href, date_str))

    return items


def parse_total_pages(html):
    """解析总页数"""
    soup = BeautifulSoup(html, 'html.parser')
    page_div = soup.select_one('div.page')
    if not page_div:
        return 1
    links = page_div.find_all('a')
    max_page = 1
    for a in links:
        href = a.get('href', '')
        m = re.search(r'page=(\d+)', href)
        if m:
            p = int(m.group(1))
            if p > max_page:
                max_page = p
    return max_page


def fetch_detail(url):
    """获取详情页，返回 (title, date_str, content_html)"""
    html = fetch(url)
    if not html:
        return None, None, None

    soup = BeautifulSoup(html, 'html.parser')
    detail_div = soup.select_one('div.video-detail')
    if not detail_div:
        return None, None, None

    # 标题
    h3 = detail_div.find('h3')
    title = clean_text(h3.get_text(strip=True)) if h3 else ''

    # 日期
    date_span = detail_div.find('span')
    date_str = ''
    if date_span:
        text = date_span.get_text(strip=True)
        m = re.search(r'(\d{4}-\d{2}-\d{2})', text)
        if m:
            date_str = m.group(1)

    # 正文 (所有p标签span内容)
    content_parts = []
    for p in detail_div.find_all('p', recursive=True):
        # 跳过上下页导航
        if p.find_parent('div', class_='product-page') or p.find_parent('div', class_='masses-page'):
            continue
        t = p.get_text(strip=True)
        if t and len(t) > 5:
            content_parts.append(str(p))

    content_html = '\n'.join(content_parts)
    return title, date_str, content_html


def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT,
        content TEXT,
        publish_date TEXT,
        source_url TEXT UNIQUE,
        page_url TEXT,
        site_name TEXT,
        summary TEXT
    )''')
    conn.commit()
    conn.close()


def sync_fts():
    """跨进程同步FTS"""
    subprocess.run([
        sys.executable, '-c', f'''
import sqlite3, os
db = "{DB_PATH}"
conn = sqlite3.connect(db)
conn.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(title, site_name, summary, tokenize=trigram)")
conn.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) SELECT rowid, title, site_name, summary FROM gov_raw WHERE rowid NOT IN (SELECT rowid FROM gov_search)")
conn.commit()
conn.close()
print("FTS synced")
'''], capture_output=True)


def main():
    import argparse
    parser = argparse.ArgumentParser(description=f"爬虫: {SITE_NAME}")
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help=f"页数（默认{MAX_PAGES}）")
    args = parser.parse_args()
    max_pages = args.pages

    print(f"=== {SITE_NAME} 爬虫 ===")
    print(f"URL: {BASE_URL}{LIST_PATH}")
    print(f"页数: {max_pages}")

    init_db()
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    # 获取总页数
    html = fetch(f"{BASE_URL}{LIST_PATH}?classid=26")
    if not html:
        print("❌ 无法获取列表页")
        return

    total_pages = parse_total_pages(html)
    total_pages = min(total_pages, max_pages)
    print(f"总页数: {total_pages}")

    all_items = []
    for page in range(1, total_pages + 1):
        url = f"{BASE_URL}{LIST_PATH}?classid=26"
        if page > 1:
            url += f"&page={page}"

        print(f"\n第{page}页...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("❌ 失败")
            continue

        items = parse_list(html)
        print(f"{len(items)}条")
        all_items.extend(items)
        time.sleep(0.5)

    print(f"\n列表总计: {len(all_items)} 条")

    # 获取已存在URL
    existing = set()
    for row in c.execute("SELECT source_url FROM gov_raw"):
        existing.add(row[0])

    print(f"已知URL: {len(existing)}")

    new_count = 0
    skip_count = 0
    error_count = 0

    for i, (title, url, date_str) in enumerate(all_items, 1):
        if url in existing:
            skip_count += 1
            continue

        # 获取详情
        detail_title, detail_date, content_html = fetch_detail(url)
        if not detail_title:
            error_count += 1
            continue

        # 标题优先使用详情页的
        final_title = detail_title or title
        final_date = detail_date or date_str

        # 内容
        content = clean_content_html(content_html, url)
        summary = re.sub(r'<[^>]+>', '', content)[:200].strip() if content else ''

        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, content, publish_date, source_url, page_url, site_name, summary) VALUES (?, ?, ?, ?, ?, ?, ?)",
                (final_title, content, final_date, url, url, SITE_NAME, summary)
            )
            if c.lastrowid:
                new_count += 1
        except Exception as e:
            error_count += 1
            continue

        if new_count % 10 == 0 and new_count > 0:
            conn.commit()
            print(f"  进度: {new_count}条新, {skip_count}条跳过, {error_count}条错误")

    conn.commit()
    conn.close()

    print(f"\n=== 完成 ===")
    print(f"  新增: {new_count} 条")
    print(f"  跳过: {skip_count} 条")
    print(f"  错误: {error_count} 条")

    if new_count > 0:
        sync_fts()
        print(f"  FTS同步完成")

    print(f"  新增 {new_count} 条")


if __name__ == "__main__":
    main()
