#!/usr/bin/env python3
"""延长县人民政府 - 公示公告 爬虫
URL: https://www.yanchangxian.gov.cn/xwzx/gsgg/{page}.html
系统: 延长自有CMS (自定义)
详情容器: div.m-txt-article → <p>段落
"""

import requests
import json
import sys
import os
import re
from bs4 import BeautifulSoup
from datetime import datetime

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
}
BASE_URL = 'https://www.yanchangxian.gov.cn'
DOMAIN = 'www.yanchangxian.gov.cn'
SITE_NAME = '延长县人民政府公示公告'
GROUP = '陕西'

def fetch_list(page):
    """Fetch list page, return list of (title, url, date)"""
    url = f'{BASE_URL}/xwzx/gsgg/{page}.html'
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')
    
    items = []
    ul = soup.select_one('div.m-lst36 ul')
    if not ul:
        print(f'  [WARN] No m-lst36 ul found on page {page}')
        return items
    
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        span = li.find('span')
        if a:
            href = a.get('href', '')
            title = a.get('title') or a.get_text(strip=True)
            date = span.get_text(strip=True) if span else ''
            # Build full URL
            if href.startswith('/'):
                href = BASE_URL + href
            items.append((title, href, date))
    
    return items

def fetch_detail(url):
    """Fetch detail page, return (title, date, content_html, attachments)"""
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')
    
    # Title - from h1 or meta
    title = ''
    h1 = soup.find('h1')
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        mt = soup.find('meta', attrs={'name': re.compile(r'ArticleTitle|ContentTitle', re.I)})
        if mt and mt.get('content'):
            title = mt['content'].strip()
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            title = title_tag.get_text(strip=True).replace('--延长县人民政府', '').strip()
    
    # Date - from meta pubdate
    pubdate = ''
    meta_pub = soup.find('meta', attrs={'name': re.compile(r'pubdate|publicdate|publishdate', re.I)})
    if meta_pub and meta_pub.get('content'):
        pubdate = meta_pub['content'].strip()[:10]
    if not pubdate:
        # Try from text
        date_match = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', soup.get_text())
        if date_match:
            pubdate = date_match.group(1).replace('/', '-')
    
    # Content - div.m-txt-article → <p> paragraphs
    content_div = soup.select_one('div.m-txt-article')
    if not content_div:
        content_div = soup.select_one('div.m-lst36-article')
    
    content_parts = []
    attachments = []
    
    if content_div:
        # Extract text from direct <p> children
        for child in content_div.find_all('p', recursive=False):
            # Check for attachments inside this p
            for a in child.find_all('a'):
                href = a.get('href', '')
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href.lower()):
                    attach_text = a.get_text(strip=True) or os.path.basename(href)
                    attachments.append({'url': href if href.startswith('http') else BASE_URL + href if href.startswith('/') else url + '/' + href, 'name': attach_text})
            
            # Get clean text
            text = child.get_text('\n', strip=True)
            if text:
                content_parts.append(text)
        
        # Also handle tables
        for table in content_div.find_all('table'):
            table_html = str(table)
            content_parts.append(f'[表格]\n{table_html}')
        
        # Handle non-<p> text
        for child in content_div.children:
            if child.name is None and child.strip():
                content_parts.append(child.strip())
    
    content_text = '\n\n'.join(content_parts)
    
    # Also find attachments outside content div (in the page)
    all_links = soup.find_all('a')
    for a in all_links:
        href = a.get('href', '')
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href.lower()):
            if not any(a['url'] == href or a['name'] == a.get_text(strip=True) for a in attachments):
                attach_text = a.get_text(strip=True) or os.path.basename(href)
                attach_url = href if href.startswith('http') else BASE_URL + href if href.startswith('/') else url + '/' + href
                # Deduplicate
                if not any(att['url'] == attach_url for att in attachments):
                    attachments.append({'url': attach_url, 'name': attach_text})
    
    return title, pubdate, content_text, attachments

def crawl(max_pages=5):
    results = []
    total_new = 0
    
    for page in range(1, max_pages + 1):
        print(f'Fetching page {page}...')
        items = fetch_list(page)
        if not items:
            print(f'  No more items at page {page}, stopping')
            break
        
        print(f'  Found {len(items)} items')
        
        for i, (title, url, date) in enumerate(items):
            print(f'  [{i+1}/{len(items)}] {title[:40]}...')
            
            try:
                detail_title, pubdate, content, attachments = fetch_detail(url)
                final_title = detail_title or title
                final_date = pubdate or date
                
                # Build attachment info
                attach_info = ''
                if attachments:
                    attach_list = [f"{att['name']}: {att['url']}" for att in attachments]
                    attach_info = '\n附件:\n' + '\n'.join(attach_list)
                
                full_content = content + ('\n\n' + attach_info if attach_info else '')
                
                result = {
                    'title': final_title,
                    'url': url,
                    'date': final_date,
                    'content': full_content,
                    'site_name': SITE_NAME,
                    'source': DOMAIN,
                    'attachments': attachments
                }
                results.append(result)
                total_new += 1
                
            except Exception as e:
                print(f'  [ERROR] Failed to fetch detail: {url} - {e}')
        
        # Small delay between pages
        if page < max_pages:
            import time
            time.sleep(1.5)
    
    return results

def output_json(results):
    """Output as JSON for the pipeline"""
    output = {
        'total': len(results),
        'items': results
    }
    print('=' * 60)
    print(f'Total items crawled: {len(results)}')
    print(json.dumps(output, ensure_ascii=False, indent=2))

if __name__ == '__main__':
    pages = 5
    if len(sys.argv) > 1:
        try:
            pages = int(sys.argv[1])
        except ValueError:
            pass
    
    print(f'Crawling yanchangxian.gov.cn gsgg - {pages} pages')
    data = crawl(max_pages=pages)
    output_json(data)
