#!/usr/bin/env python3
"""Crawler for 贵州汇景森环保工程有限公司 - 公告公示"""
import requests
import re
import sqlite3
import os
import sys
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'http://www.huijingsen.cn'
LIST_URL = '/index.php/lists/48.html'
DB_PATH = '/root/search.db'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}
GROUP = '贵州'

def extract_date_from_item(item):
    """Extract date from list item - <span>2026-06-23</span> in the footer area"""
    spans = item.find_all('span')
    for span in spans:
        txt = span.get_text(strip=True)
        if re.match(r'\d{4}-\d{2}-\d{2}', txt):
            return txt
    return None

def get_detail(url):
    """Get title, content, and attachments from detail page"""
    full_url = urljoin(BASE_URL, url)
    try:
        r = requests.get(full_url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        print(f"  FAIL fetch {url}: {e}", file=sys.stderr)
        return None, None, []
    
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # Title from <title> tag
    title_tag = soup.find('title')
    title = title_tag.get_text(strip=True) if title_tag else ''
    # Strip suffix
    title = re.sub(r'\s*-\s*公告公示\s*-\s*贵州汇景森环保工程有限公司\s*$', '', title).strip()
    
    # Content from <div class="showkuang mt15" style="min-height: 300px;">
    divs = soup.find_all('div', class_='showkuang')
    content_div = None
    for d in divs:
        classes = d.get('class', [])
        if 'mt15' in classes:
            content_div = d
            break
    if not content_div and divs:
        content_div = divs[0]
    if not content_div:
        print(f"  No content div found", file=sys.stderr)
        return title, None, []
    
    # Extract attachments (PDF/doc links)
    attachments = []
    for a_tag in content_div.find_all('a', href=True):
        href = a_tag['href']
        if re.search(r'\.(pdf|doc|docx|xls|xlsx)$', href, re.I):
            full_url_attach = urljoin(BASE_URL, href) if not href.startswith('http') else href
            link_text = a_tag.get_text(strip=True)
            attachments.append((full_url_attach, link_text or href))
    
    # Strip inline formatting tags before text extraction to avoid span-splitting
    for tag in content_div.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.replace_with(tag.get_text())
    
    # Get content HTML for table preservation
    content_html = str(content_div)
    
    # Handle <table> - keep as HTML
    tables = []
    def save_table(m):
        idx = len(tables)
        tables.append(m.group(0))
        return f'__TABLE_{idx}__'
    content_html = re.sub(r'<table[^>]*>.*?</table>', save_table, content_html, flags=re.DOTALL)
    
    # Convert remaining HTML to text with paragraph separation
    content_soup = BeautifulSoup(content_html, 'html.parser')
    content_text = content_soup.get_text(separator='\n', strip=True)
    
    # Restore tables
    for i, t in enumerate(tables):
        content_text = content_text.replace(f'__TABLE_{i}__', t)
    
    # Strip leading/trailing whitespace per line and remove empty lines
    lines = content_text.split('\n')
    lines = [l.strip() for l in lines if l.strip()]
    content_text = '\n\n'.join(lines)
    
    return title, content_text, attachments

def crawl():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    inserted = 0
    updated = 0
    errors = 0
    
    for page in range(1, 14):
        if page == 1:
            url = urljoin(BASE_URL, LIST_URL)
        else:
            url = urljoin(BASE_URL, LIST_URL + f'?page={page}')
        
        print(f"Fetching page {page}...", file=sys.stderr)
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
        except Exception as e:
            print(f"  FAIL page {page}: {e}", file=sys.stderr)
            errors += 1
            continue
        
        soup = BeautifulSoup(r.text, 'html.parser')
        items = soup.select('li.acea-row')
        
        if not items:
            print(f"  No items found on page {page}", file=sys.stderr)
            break
        
        for item in items:
            # Get title link
            title_a = item.find('a', href=True)
            if not title_a:
                continue
            
            href = title_a['href']
            detail_url = href if href.startswith('/') else '/' + href.lstrip('/')
            list_title = title_a.get_text(strip=True)
            
            # Extract date
            date_str = extract_date_from_item(item)
            
            # Check if exists
            exists = c.execute(
                'SELECT id FROM gov_raw WHERE page_url = ?',
                (detail_url,)
            ).fetchone()
            
            if exists:
                # Skip if already exists (日跑增量模式)
                print(f"  SKIP (exists): {detail_url}", file=sys.stderr)
                continue
            
            # Fetch detail
            print(f"  Fetch: {detail_url}", file=sys.stderr)
            title, content, attachments = get_detail(detail_url)
            
            if not title:
                title = list_title
            
            # Build attachment markdown
            attach_md = ''
            if attachments:
                parts = []
                for attach_url, label in attachments:
                    attach_url_full = urljoin(BASE_URL, attach_url) if not attach_url.startswith('http') else attach_url
                    parts.append(f'[{label}]({attach_url_full})')
                attach_md = '\n\n' + '\n'.join(parts)
            
            full_content = content + attach_md if content else attach_md
            
            # Fallback: if no content, use title + detail URL
            if not full_content:
                full_content = f'<p><a href="{urljoin(BASE_URL, detail_url)}">{title}</a></p>'
            
            try:
                c.execute(
                    'INSERT OR IGNORE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url) VALUES (?, ?, ?, ?, ?, ?)',
                    (detail_url, title, full_content, date_str, GROUP, urljoin(BASE_URL, detail_url))
                )
                conn.commit()
                inserted += 1
                print(f"  OK: {title[:60]}", file=sys.stderr)
            except Exception as e:
                print(f"  FAIL insert {title[:40]}: {e}", file=sys.stderr)
                conn.rollback()
                errors += 1
    
    conn.close()
    print(f"\nDone. Inserted: {inserted}, Errors: {errors}", file=sys.stderr)
    return inserted, errors

if __name__ == '__main__':
    result = crawl()
    print(f"RESULT:{result[0]}:{result[1]}")
