#!/usr/bin/env python3
"""Crawler for sxx.gov.cn - 濉溪县人民政府信息公开-招标公示及招标采购预算
CMS: Lonsun 龙讯政府信息公开平台
List: /zwgk/public/column/1981?type=4&catId=4741861&action=list&pageIndex=N
Detail: /zwgk/public/1981/{id}.html
"""

import requests
from bs4 import BeautifulSoup
import json
import re
import sys
import os
from datetime import datetime

requests.packages.urllib3.disable_warnings()

BASE = 'https://www.sxx.gov.cn'
OUTPUT_DIR = '/root/gov_crawler/output'
os.makedirs(OUTPUT_DIR, exist_ok=True)

session = requests.Session()
session.headers.update({
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
})

LIST_BASE = '/zwgk/public/column/1981?type=4&catId=4741861&action=list'

def get_articles_from_page(page_num):
    """Extract article links and dates from a list page."""
    url = f'{BASE}{LIST_BASE}'
    if page_num > 1:
        url += f'&pageIndex={page_num}'
    
    r = session.get(url, timeout=30, verify=False)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.content, 'html.parser')
    
    articles = []
    for li in soup.select('li.clearfix'):
        a_tag = li.select_one('a.title')
        date_span = li.select_one('span.date')
        if not a_tag:
            continue
        href = a_tag.get('href', '').strip()
        # Use link text for title (it's the full title)
        title = a_tag.get_text(strip=True)
        pub_date = date_span.get_text(strip=True) if date_span else ''
        
        if href and title:
            if not href.startswith('http'):
                href = BASE + href if href.startswith('/') else BASE + '/' + href
            articles.append({'href': href, 'title': title, 'date': pub_date})
    
    return articles

def extract_article(detail_url):
    """Extract metadata and content from an article detail page."""
    r = session.get(detail_url, timeout=30, verify=False)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.content, 'html.parser')
    
    # Title from meta
    title = ''
    meta = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta:
        title = meta.get('content', '')
    
    # Date
    pub_date = ''
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta:
        date_str = meta.get('content', '').strip()
        if date_str:
            try:
                dt = datetime.strptime(date_str[:10], '%Y-%m-%d')
                pub_date = dt.strftime('%Y-%m-%d')
            except:
                pub_date = date_str[:10]
    
    # Source
    source = ''
    meta = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta:
        source = meta.get('content', '')
    
    # Content
    content = ''
    wzcon_div = soup.select_one('div.wzcon.j-fontContent')
    
    if wzcon_div:
        html_content = str(wzcon_div)
        
        # Preserve tables as HTML
        tables = []
        table_counter = [0]
        
        def save_table(match):
            tbl_html = match.group(0)
            tables.append(tbl_html)
            idx = table_counter[0]
            table_counter[0] += 1
            return f'\n__TABLE_{idx}__\n'
        
        html_content = re.sub(r'<table[^>]*>.*?</table>', save_table, html_content, flags=re.DOTALL | re.IGNORECASE)
        
        # Replace <br> and </p> with newlines for paragraph breaks
        html_content = re.sub(r'<br\s*/?>', '\n', html_content, flags=re.IGNORECASE)
        html_content = re.sub(r'</p>', '\n', html_content, flags=re.IGNORECASE)
        html_content = re.sub(r'</div>', '\n', html_content, flags=re.IGNORECASE)
        
        # Get text without separator (prevent span splitting)
        text_soup = BeautifulSoup(html_content, 'html.parser')
        content_text = text_soup.get_text(separator='', strip=False)
        
        # Clean up
        lines = []
        for line in content_text.split('\n'):
            line = line.strip()
            if not line:
                continue
            placeholder_match = re.match(r'__TABLE_(\d+)__', line)
            if placeholder_match:
                idx = int(placeholder_match.group(1))
                if idx < len(tables):
                    lines.append(tables[idx])
                continue
            lines.append(line)
        
        content = '\n\n'.join(lines)
    
    # Attachments - check content for download links
    attachments = []
    if wzcon_div:
        for a_tag in wzcon_div.select('a[href]'):
            a_href = a_tag.get('href', '')
            a_text = a_tag.get_text(strip=True) or '附件'
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar|ppt|pptx)$', a_href, re.I):
                if not a_href.startswith('http'):
                    a_href = BASE + a_href if a_href.startswith('/') else detail_url.rsplit('/', 1)[0] + '/' + a_href
                attachments.append({'name': a_text, 'url': a_href})
    
    result = {
        'title': title,
        'pub_date': pub_date,
        'source': source,
        'url': detail_url,
        'content': content,
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else ''
    }
    return result

def main():
    max_pages = int(sys.argv[1]) if len(sys.argv) > 1 else 2
    
    # Collect all articles
    all_articles = []
    for pg in range(1, max_pages + 1):
        print(f'Fetching page {pg}...', file=sys.stderr)
        articles = get_articles_from_page(pg)
        print(f'  Found {len(articles)} articles', file=sys.stderr)
        all_articles.extend(articles)
        if len(articles) < 5:  # Last page or no items
            break
    
    print(f'Total articles to process: {len(all_articles)}', file=sys.stderr)
    
    # Process each article
    results = []
    for i, art in enumerate(all_articles):
        display_title = art['title'][:40] if len(art['title']) > 40 else art['title']
        print(f'  [{i+1}/{len(all_articles)}] {display_title}', file=sys.stderr)
        try:
            data = extract_article(art['href'])
            if not data['title']:
                data['title'] = art['title']
            if not data['pub_date']:
                data['pub_date'] = art['date']
            results.append(data)
        except Exception as e:
            print(f'  ERROR: {e}', file=sys.stderr)
            results.append({
                'title': art['title'],
                'pub_date': art['date'],
                'source': '',
                'url': art['href'],
                'content': '',
                'attachments': ''
            })
    
    # Write output
    timestamp = datetime.now().strftime('%Y%m%d_%H%M%S')
    output_file = os.path.join(OUTPUT_DIR, f'sxx_tzgg_{timestamp}.jsonl')
    with open(output_file, 'w', encoding='utf-8') as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + '\n')
    
    print(f'Output: {output_file}', file=sys.stderr)
    print(f'Total items: {len(results)}')

if __name__ == '__main__':
    main()
