#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
福建环保网-一次公示 爬虫
https://www.fjhb.org/huanping/yici/
CMS: 自定义
分页: list_N.html (20条/页)
仅全量前5页，日跑仅第1页
"""

import requests
import re
import json
import time
import sys
import os
import sqlite3
from bs4 import BeautifulSoup

BASE_URL = "https://www.fjhb.org"
LIST_BASE = BASE_URL + "/huanping/yici"
SITE_NAME = "福建环保网-一次公示"
GROUP = "企业"
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')
MAX_PAGES = 5  # 前5页 = ~100条

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(HEADERS)


def fetch(url, timeout=30):
    resp = session.get(url, timeout=timeout)
    resp.encoding = 'utf-8'
    return resp


def crawl_list(page_num):
    """爬取列表页"""
    if page_num == 1:
        url = f"{LIST_BASE}/list_1.html"
    else:
        url = f"{LIST_BASE}/list_{page_num}.html"
    
    print(f"  Fetching page {page_num}: {url}")
    
    try:
        resp = fetch(url)
    except Exception as e:
        print(f"  Error: {e}")
        return []
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    items = []
    
    rows = soup.select('table tr')
    for row in rows:
        cells = row.select('td')
        if len(cells) < 3:
            continue
        
        link = cells[0].find('a')
        if not link:
            continue
        
        title = link.get_text(strip=True)
        if not title:
            continue
        
        href = link.get('href', '')
        if not href.startswith('http'):
            if href.startswith('/'):
                href = f"{BASE_URL}{href}"
            else:
                href = f"{LIST_BASE}/{href}"
        
        date_text = cells[2].get_text(strip=True)
        
        items.append({
            'title': title,
            'url': href,
            'date': date_text,
        })
    
    return items


def crawl_detail(item):
    """爬取详情页"""
    url = item['url']
    print(f"    Detail: {item['title'][:40]}...")
    
    try:
        resp = fetch(url)
    except Exception as e:
        print(f"    Error: {e}")
        return None
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    # 标题 - 从列表页取（已验证完整），或从 <title> 标签提取
    title = item['title']
    if not title:
        # 从 <title> 提取（格式: "文章标题_福建环保网"）
        t_tag = soup.find('title')
        if t_tag:
            t_text = t_tag.get_text(strip=True)
            if '_' in t_text:
                title = t_text.rsplit('_', 1)[0].strip()
    
    # 日期
    date = item['date']
    date_match = re.search(r'日期[：:]\s*(\d{4}-\d{2}-\d{2})', resp.text)
    if date_match:
        date = date_match.group(1)
    
    # 正文区域
    content_div = soup.find('div', class_=re.compile(r'content|article|detail|text|body'))
    if not content_div:
        content_div = soup.find('article')
    if not content_div:
        content_div = soup.find(id=re.compile(r'content|article|detail|text|body'))
    if not content_div:
        main_div = soup.find('div', class_=re.compile(r'main|container|wrap'))
        if main_div:
            for tag in main_div.find_all(['header', 'footer', 'nav', 'aside']):
                tag.decompose()
            content_div = main_div
        else:
            body = soup.find('body')
            if body:
                for tag in body.find_all(['header', 'footer', 'nav', 'aside', 'script', 'style']):
                    tag.decompose()
                content_div = body
    
    content_parts = []
    attachments = []
    
    if content_div:
        # 附件链接
        for a_tag in content_div.find_all('a', href=True):
            href = a_tag['href']
            if not href.startswith('http'):
                if href.startswith('/'):
                    href = f"{BASE_URL}{href}"
                else:
                    href = f"{url.rstrip('/')}/{href}"
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar)$', href, re.I):
                attach_title = a_tag.get_text(strip=True)
                attachments.append({
                    'title': attach_title or os.path.basename(href),
                    'url': href
                })
        
        # 正文段落 - 保留HTML表格
        for p in content_div.find_all(['p', 'div']):
            if not p.get_text(strip=True):
                continue
            if p.find_parent(['form', 'textarea']):
                continue
            
            text = p.get_text(strip=True)
            if text in ['附件下载', '文章评论'] or re.match(r'^共 \d+ 条评论', text):
                continue
            
            tables = p.find_all('table')
            if tables:
                for table in tables:
                    content_parts.append(str(table))
                continue
            
            # 过滤模板元数据行
            if re.match(r'^(责任编辑|初审|复审|终审|编辑[：:]|\[纠错\]).*', text):
                continue
            
            # 保留 <strong> 子标题
            if p.find('strong'):
                content_parts.append(text)
            else:
                content_parts.append(text)
    
    content = '\n\n'.join(part for part in content_parts if part)
    
    # 空内容回退
    if len(content.strip()) < 20:
        attach_links = '\n'.join(f"[{a['title']}]({a['url']})" for a in attachments)
        if attach_links:
            content = f'<p><a href="{url}">{title}</a></p>\n\n附件：\n{attach_links}'
        else:
            content = f'<p><a href="{url}">{title}</a></p>'
    
    summary = content[:200] if content else title
    
    return {
        'site_name': SITE_NAME,
        'source_url': '',
        'page_url': url,
        'title': title,
        'publish_date': date,
        'content': content,
        'summary': summary,
        'category': GROUP,
        'status': 'published',
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
    }


def save_to_db(records, db_path=DB_PATH, incremental=False):
    """保存到数据库"""
    if not records:
        return 0
    
    conn = sqlite3.connect(db_path, timeout=60)
    c = conn.cursor()
    
    new_count = 0
    for i, rec in enumerate(records):
        try:
            c.execute('''INSERT OR IGNORE INTO gov_raw 
                (site_name, source_url, page_url, title, publish_date,
                 content, summary, category, status, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
                (rec['site_name'], rec['source_url'], rec['page_url'],
                 rec['title'], rec['publish_date'], rec['content'],
                 rec['summary'], rec['category'], rec['status'],
                 rec['attachments']))
            if c.rowcount > 0:
                new_count += 1
                if new_count <= 5 or (i+1) % 20 == 0:
                    print(f"  [{i+1}/{len(records)}] {rec['title'][:40]} +{new_count}")
        except Exception as e:
            print(f"  ERROR [{rec['page_url']}]: {e}")
    
    conn.commit()
    conn.close()
    return new_count


def main():
    print(f"=== {SITE_NAME} - 全量爬取 ===")
    
    all_items = []
    seen_urls = set()
    
    for page in range(1, MAX_PAGES + 1):
        items = crawl_list(page)
        if not items:
            print(f"  -> 0 items on page {page}")
            break
        print(f"  -> {len(items)} items on page {page}")
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
    
    print(f"\nTotal unique items: {len(all_items)}")
    
    records = []
    for idx, item in enumerate(all_items):
        print(f"  [{idx+1}/{len(all_items)}] {item['title'][:30]}...")
        detail = crawl_detail(item)
        if detail:
            records.append(detail)
        time.sleep(0.5)
    
    if records:
        count = save_to_db(records)
        print(f"\nInserted {count} new records (trigger auto-syncs FTS)")
    else:
        print("\nNo records to insert")


if __name__ == '__main__':
    # 如果增量模式，仅爬第1页
    if '--incremental' in sys.argv:
        MAX_PAGES = 1
    main()
