#!/usr/bin/env python3
"""
内丘县人民政府网 (hbnq.gov.cn) - 公告公示爬虫
站点：https://www.hbnq.gov.cn/channelList/11049.html
CMS: 自定义政府网站
列表：/channelList/11049.html 及 /channelList/11049_{pn}.html 分页(10页~200条)
"""
import requests
import re
import json
import time
import os
import sys
from bs4 import BeautifulSoup

BASE_URL = 'https://www.hbnq.gov.cn'
LIST_URL = BASE_URL + '/channelList/11049.html'
SITE_NAME = '内丘县人民政府-公告公示'
GROUP = '内丘县'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

def fetch(url, encoding='utf-8'):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = encoding
    return r.text

def extract_list_items(html):
    """解析列表页"""
    items = []
    # Find all article links - some in <li>, some in other containers
    soup = BeautifulSoup(html, 'html.parser')
    
    for a_tag in soup.find_all('a', href=re.compile(r'/content/11049/\d+\.html')):
        href = a_tag.get('href', '')
        if not href.startswith('http'):
            href = BASE_URL + href
        
        # Get title
        title = a_tag.get('title', '')
        if not title:
            title = a_tag.get_text(strip=True)
        if not title:
            continue
        
        # Find date (look for closest span.time)
        date_str = ''
        date_span = a_tag.find_next('span', class_=lambda c: c and 'time' in c)
        if date_span:
            date_str = date_span.get_text(strip=True)
        else:
            # Sometimes date is in a sibling span
            parent = a_tag.parent
            if parent:
                date_span = parent.find('span', class_=lambda c: c and 'time' in c)
                if date_span:
                    date_str = date_span.get_text(strip=True)
        
        items.append({'title': title, 'link': href, 'date': date_str})
    
    return items

def extract_detail(html, url):
    """解析详情页"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from meta
    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)
    
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            t = title_tag.get_text(strip=True)
            if '_' in t:
                title = t.split('_')[0]
            else:
                title = t
    
    # Date from meta
    date_str = ''
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date and meta_date.get('content'):
        date_match = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', meta_date['content'])
        if date_match:
            date_str = date_match.group(1)
    
    # Content - try various containers
    content = ''
    for selector in [{'class': 'detail'}, {'class': 'content'}, {'class': 'TRS_Editor'}, 
                     {'class': 'ewb-article'}, {'id': 'UCAP-CONTENT'}, {'class': 'article'}]:
        if 'class' in selector:
            div = soup.find('div', class_=lambda c, cls=selector['class']: c and cls in c)
        else:
            div = soup.find('div', id=selector['id'])
        if div:
            content = str(div)
            break
    
    if not content:
        # Try to find content after <h1>
        h1 = soup.find('h1')
        if h1:
            # Get everything after h1 until footer-like elements
            content_parts = []
            for sibling in h1.find_next_siblings():
                if sibling.name in ['div'] and sibling.get('class'):
                    cls_str = ' '.join(sibling.get('class', []))
                    if any(x in cls_str for x in ['foot', 'print', 'share', 'btn']):
                        continue
                    content_parts.append(str(sibling))
                elif sibling.name == 'div':
                    content_parts.append(str(sibling))
            content = '\n'.join(content_parts)
    
    # Attachments
    attachments = []
    for a_tag in soup.find_all('a', href=re.compile(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', re.I)):
        href = a_tag.get('href', '')
        if href.startswith('/'):
            href = 'https:' + href if href.startswith('//') else BASE_URL + href
        
        name = a_tag.get_text(strip=True) or href.split('/')[-1]
        attachments.append({'url': href, 'name': name})
    
    return {
        'title': title,
        'date': date_str,
        'content': content.strip(),
        'attachments': attachments
    }

def crawl_all(incremental=False):
    """爬取全部文章"""
    total_pages = 10
    
    if incremental:
        page_range = [1]
    else:
        page_range = list(range(1, total_pages + 1))
    
    results = []
    
    for page_num in page_range:
        if page_num == 1:
            url = LIST_URL
        else:
            url = BASE_URL + '/channelList/11049_%d.html' % page_num
        
        print('  Fetching list page %d/%d...' % (page_num, total_pages))
        html = fetch(url)
        items = extract_list_items(html)
        
        if not items:
            print('    No more items, stopping')
            break
        
        print('    Found %d items' % len(items))
        
        for item in items:
            print('  Fetching: %s' % item['title'][:50])
            try:
                detail_html = fetch(item['link'])
                detail = extract_detail(detail_html, item['link'])
                
                # Use detail title (full) instead of potentially truncated list title
                result = {
                    'title': detail['title'] or item['title'],
                    'link': item['link'],
                    'date': detail['date'] or item['date'],
                    'content': detail['content'],
                    'attachments': detail['attachments'],
                    'site_name': SITE_NAME,
                    'group': GROUP
                }
                results.append(result)
                time.sleep(0.3)
            except Exception as e:
                print('    ERROR fetching detail: %s' % e)
                results.append({
                    'title': item['title'],
                    'link': item['link'],
                    'date': item['date'],
                    'content': '',
                    'attachments': [],
                    'site_name': SITE_NAME,
                    'group': GROUP
                })
    
    return results

def export_jsonl(results, output_path):
    with open(output_path, 'w', encoding='utf-8') as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + '\n')
    print('Exported %d records to %s' % (len(results), output_path))

if __name__ == '__main__':
    output_dir = '/root/gov_crawler/output'
    os.makedirs(output_dir, exist_ok=True)
    
    import sys
    incremental = '--incremental' in sys.argv or '1' in sys.argv
    
    print('Crawling %s (incremental=%s)' % (SITE_NAME, incremental))
    results = crawl_all(incremental=incremental)
    
    if results:
        jsonl_path = os.path.join(output_dir, 'hbnq_gsgg.jsonl')
        export_jsonl(results, jsonl_path)
        print('\nDone! %d articles total' % len(results))
    else:
        print('No results found!')
