#!/usr/bin/env python3
"""Crawler for tl.gov.cn - 铜陵市生态环境局信息公开-最新公开
CMS: 政府信息公开通用平台 (tl.gov.cn/openness)
Lists: showList/770/1948/page_{N}.html, 15条/页, <a class="biaoti-title"> (用title属性获取完整标题)
Detail: show/{id}.html, <div id="zoom"> 正文含表格HTML
"""

import requests
from bs4 import BeautifulSoup
import json
import re
import sys
import os
from datetime import datetime

requests.packages.urllib3.disable_warnings()

BASE = 'https://www.tl.gov.cn'
LIST_PATH = '/openness/OpennessContent/showList/770/1948'
OUTPUT_DIR = '/root/gov_crawler/output'
os.makedirs(OUTPUT_DIR, exist_ok=True)

session = requests.Session()
session.headers.update({
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
})

def get_articles_from_page(page_num):
    """Extract article links and dates from a list page."""
    url = f'{BASE}{LIST_PATH}/page_{page_num}.html'
    
    r = session.get(url, timeout=30, verify=False)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.content, 'html.parser')
    
    articles = []
    for li in soup.select('li[style*="height"]'):
        a_tag = li.select_one('a.biaoti-title')
        if not a_tag:
            continue
        href = a_tag.get('href', '').strip()
        # Use title attribute for full title (no ellipsis)
        title = a_tag.get('title', '').strip()
        if not title:
            title = a_tag.get_text(strip=True)
        date_span = li.select_one('span')
        pub_date = date_span.get_text(strip=True) if date_span else ''
        
        if href and title:
            if not href.startswith('http'):
                href = BASE + href if href.startswith('/') else BASE + '/' + href
            articles.append({'href': href, 'title': title, 'date': pub_date})
    
    return articles

def extract_article(detail_url):
    """Extract metadata and content from an article detail page."""
    r = session.get(detail_url, timeout=30, verify=False)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.content, 'html.parser')
    
    # Title from meta
    title = ''
    meta = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta:
        title = meta.get('content', '')
    if not title:
        title_div = soup.select_one('#title')
        if title_div:
            title = title_div.get_text(strip=True)
    
    # Date
    pub_date = ''
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta:
        date_str = meta.get('content', '').strip()
        if date_str:
            try:
                dt = datetime.strptime(date_str[:10], '%Y-%m-%d')
                pub_date = dt.strftime('%Y-%m-%d')
            except:
                pub_date = date_str[:10]
    
    # Source
    source = ''
    meta = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta:
        source = meta.get('content', '')
    
    # Author
    author = ''
    meta = soup.find('meta', attrs={'name': 'Author'})
    if meta:
        author = meta.get('content', '')
    
    # Content - preserve tables as HTML
    content = ''
    zoom_div = soup.select_one('div#zoom')
    if not zoom_div:
        zoom_div = soup.select_one('div.m-scrollbar')
    if not zoom_div:
        zoom_div = soup.select_one('div.g-detailbox')
    
    if zoom_div:
        # Process tables: keep as HTML
        html_content = str(zoom_div)
        
        # Replace tables with placeholder
        tables = []
        table_counter = [0]
        
        def save_table(match):
            tbl_html = match.group(0)
            tables.append(tbl_html)
            idx = table_counter[0]
            table_counter[0] += 1
            return f'\n__TABLE_{idx}__\n'
        
        html_with_placeholders = re.sub(r'<table[^>]*>.*?</table>', save_table, html_content, flags=re.DOTALL | re.IGNORECASE)
        
        # Replace <br> with newline
        html_with_placeholders = re.sub(r'<br\s*/?>', '\n', html_with_placeholders, flags=re.IGNORECASE)
        # Replace </p> with newline
        html_with_placeholders = re.sub(r'</p>', '\n', html_with_placeholders, flags=re.IGNORECASE)
        # Replace </div> with newline  
        html_with_placeholders = re.sub(r'</div>', '\n', html_with_placeholders, flags=re.IGNORECASE)
        
        # Strip all other tags, using separator='' to prevent span splitting
        text_soup = BeautifulSoup(html_with_placeholders, 'html.parser')
        content_text = text_soup.get_text(separator='', strip=False)
        
        # Clean up whitespace
        lines = []
        for line in content_text.split('\n'):
            line = line.strip()
            if not line:
                continue
            # Check for table placeholders
            placeholder_match = re.match(r'__TABLE_(\d+)__', line)
            if placeholder_match:
                idx = int(placeholder_match.group(1))
                if idx < len(tables):
                    lines.append(tables[idx])
                continue
            lines.append(line)
        
        content = '\n\n'.join(lines)
    
    # Attachments
    attachments = []
    # Check readFileList
    read_file_div = soup.select_one('div.readFileList')
    if read_file_div:
        for a_tag in read_file_div.select('a[href]'):
            a_href = a_tag.get('href', '')
            a_text = a_tag.get_text(strip=True) or '附件'
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', a_href, re.I):
                if not a_href.startswith('http'):
                    a_href = BASE + a_href if a_href.startswith('/') else detail_url.rsplit('/', 1)[0] + '/' + a_href
                attachments.append({'name': a_text, 'url': a_href})
    
    # Check download links in m-detail-downlist
    downlist = soup.select_one('div.m-detail-downlist')
    if downlist:
        for a_tag in downlist.select('a[href]'):
            a_href = a_tag.get('href', '')
            a_text = a_tag.get_text(strip=True) or '附件'
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', a_href, re.I):
                if not a_href.startswith('http'):
                    a_href = BASE + a_href if a_href.startswith('/') else detail_url.rsplit('/', 1)[0] + '/' + a_href
                if not any(a['url'] == a_href for a in attachments):
                    attachments.append({'name': a_text, 'url': a_href})
    
    # Check content for download links
    if zoom_div:
        for a_tag in zoom_div.select('a[href]'):
            a_href = a_tag.get('href', '')
            a_text = a_tag.get_text(strip=True) or '附件'
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', a_href, re.I):
                if not a_href.startswith('http'):
                    a_href = BASE + a_href if a_href.startswith('/') else detail_url.rsplit('/', 1)[0] + '/' + a_href
                if not any(a['url'] == a_href for a in attachments):
                    attachments.append({'name': a_text, 'url': a_href})
    
    result = {
        'title': title,
        'pub_date': pub_date,
        'source': source,
        'author': author,
        'url': detail_url,
        'content': content,
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else ''
    }
    return result

def main():
    max_pages = int(sys.argv[1]) if len(sys.argv) > 1 else 5
    
    # Collect all articles
    all_articles = []
    for pg in range(1, max_pages + 1):
        print(f'Fetching page {pg}...', file=sys.stderr)
        articles = get_articles_from_page(pg)
        print(f'  Found {len(articles)} articles', file=sys.stderr)
        all_articles.extend(articles)
        if len(articles) < 15:
            break  # Last page
    
    print(f'Total articles to process: {len(all_articles)}', file=sys.stderr)
    
    # Process each article
    results = []
    for i, art in enumerate(all_articles):
        print(f'  [{i+1}/{len(all_articles)}] {art["title"][:40]}...', file=sys.stderr)
        try:
            data = extract_article(art['href'])
            # If no content found from detail, use what we have
            if not data['title']:
                data['title'] = art['title']
            if not data['pub_date']:
                data['pub_date'] = art['date']
            results.append(data)
        except Exception as e:
            print(f'  ERROR: {e}', file=sys.stderr)
            # Fallback: use list data
            results.append({
                'title': art['title'],
                'pub_date': art['date'],
                'source': '',
                'author': '',
                'url': art['href'],
                'content': '',
                'attachments': ''
            })
    
    # Write output
    timestamp = datetime.now().strftime('%Y%m%d_%H%M%S')
    output_file = os.path.join(OUTPUT_DIR, f'tl_tzgg_{timestamp}.jsonl')
    with open(output_file, 'w', encoding='utf-8') as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + '\n')
    
    print(f'Output: {output_file}', file=sys.stderr)
    print(f'Total items: {len(results)}')

if __name__ == '__main__':
    main()
