#!/usr/bin/env python3
"""Crawl cqcs.gov.cn - 长寿区人民政府 部门街镇"""
import requests, re, json, time, os, sys
from bs4 import BeautifulSoup

BASE_URL = 'https://www.cqcs.gov.cn'
LIST_URL = BASE_URL + '/zwxx_164/bmjz/'
SITE_NAME = '长寿区人民政府-部门街镇'
GROUP = '长寿区'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9',
}

def fetch(url, timeout=30):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = 'utf-8'
    return r.text

def extract_list_items(html):
    """Extract items from list page"""
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for a in soup.select('ul.new-list li a.e'):
        title = a.get('title', '').strip()
        if not title:
            title = a.get_text(strip=True)
        href = a.get('href', '')
        if href.startswith('./'):
            href = BASE_URL + '/zwxx_164/bmjz/' + href[2:]
        elif href.startswith('/'):
            href = BASE_URL + href
        elif not href.startswith('http'):
            href = BASE_URL + '/zwxx_164/bmjz/' + href
        
        # Date from sibling <span>
        date_span = a.find_next_sibling('span')
        date_str = date_span.get_text(strip=True) if date_span else ''
        
        items.append({'title': title, 'link': href, 'date': date_str})
    return items

def extract_detail(html, url):
    """Extract detail page content"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title - from h1.zwxl-title or meta
    title = ''
    h1 = soup.find('h1', class_=lambda c: c and 'zwxl-title' in c)
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            t = title_tag.get_text(strip=True)
            if '_' in t:
                title = t.split('_')[0]
            else:
                title = t
    
    # Date
    date_str = ''
    date_span = soup.find('span', id='NewsArticlePubDay')
    if date_span:
        date_str = date_span.get_text(strip=True)
    if not date_str:
        # Fallback: look for date pattern in page
        date_match = re.search(r'(\d{4}-\d{2}-\d{2})', html)
        if date_match:
            date_str = date_match.group(1)
    
    # Content - TRS UEDITOR
    content_html = ''
    trs_div = soup.find('div', class_=lambda c: c and 'TRS_UEDITOR' in c)
    if trs_div:
        # Keep HTML structure: tables, paragraphs
        content_html = str(trs_div)
        # Clean
        content_html = re.sub(r'\s*style="[^"]*"', '', content_html)
        content_html = re.sub(r'class="[^"]*"', '', content_html)
    if not content_html:
        # Fallback: any content div
        content_div = soup.find('div', class_='zwxl-content')
        if content_div:
            content_html = str(content_div)
    
    # Clean content: keep tables, paragraphs
    content_soup = BeautifulSoup(content_html, 'html.parser')
    # Remove unwanted elements
    for tag in content_soup.find_all(['script', 'style', 'iframe']):
        tag.decompose()
    
    content = str(content_soup)
    content = re.sub(r'\s{2,}', ' ', content).strip()
    
    # Attachments
    attachments = []
    for a_tag in soup.find_all('a', href=re.compile(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', re.I)):
        href = a_tag.get('href', '')
        if href.startswith('./'):
            href = BASE_URL + '/zwxx_164/bmjz/' + href[2:]
        elif href.startswith('/'):
            href = BASE_URL + href
        elif not href.startswith('http'):
            href = BASE_URL + '/zwxx_164/bmjz/' + href
        name = a_tag.get_text(strip=True) or href.split('/')[-1]
        attachments.append({'url': href, 'name': name})
    
    return {'title': title, 'date': date_str, 'content': content, 'attachments': attachments}

def crawl_all(incremental=False):
    results = []
    
    if incremental:
        total_pages = 1
    else:
        total_pages = 33  # 33 pages total
    
    for page_num in range(1, total_pages + 1):
        if page_num == 1:
            url = LIST_URL + 'index.html'
        else:
            url = LIST_URL + f'index_{page_num}.html'
        
        print(f'  Fetching page {page_num}/{total_pages}...')
        try:
            html = fetch(url)
            items = extract_list_items(html)
            if not items:
                print(f'    No items, stopping')
                break
            print(f'    Found {len(items)} items')
            
            for item in items:
                print(f'  Fetching detail: {item["title"][:50]}')
                try:
                    detail_html = fetch(item['link'])
                    detail = extract_detail(detail_html, item['link'])
                    
                    result = {
                        'title': detail['title'] or item['title'],
                        'link': item['link'],
                        'date': detail['date'] or item['date'],
                        'content': detail['content'],
                        'attachments': detail['attachments'],
                        'site_name': SITE_NAME,
                        'group': GROUP
                    }
                    results.append(result)
                    time.sleep(0.5)
                except Exception as e:
                    print(f'    ERROR detail: {e}')
                    results.append({
                        'title': item['title'], 'link': item['link'], 'date': item['date'],
                        'content': '', 'attachments': [], 'site_name': SITE_NAME, 'group': GROUP
                    })
        except Exception as e:
            print(f'    ERROR page {page_num}: {e}')
    
    return results

def export_jsonl(results, path):
    with open(path, 'w', encoding='utf-8') as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + '\n')
    print(f'Exported {len(results)} to {path}')

def export_sql(results, path):
    with open(path, 'w', encoding='utf-8') as f:
        f.write(f'-- Records for {SITE_NAME}\n')
        for r in results:
            content = r['content'].replace("'", "''")
            title = r['title'].replace("'", "''")
            att = json.dumps(r['attachments'], ensure_ascii=False)
            sql = f"INSERT INTO gov_raw (title, content, publish_date, page_url, site_name, source, attachments) VALUES ('{title}', '{content}', '{r['date']}', '{r['link']}', '{SITE_NAME}', '{GROUP}', '{att}');\n"
            f.write(sql)
    print(f'Exported SQL to {path}')

if __name__ == '__main__':
    output_dir = '/root/gov_crawler/output'
    os.makedirs(output_dir, exist_ok=True)
    
    incremental = '--incremental' in sys.argv
    
    print(f'Crawling {SITE_NAME} (incremental={incremental})')
    results = crawl_all(incremental=incremental)
    
    if results:
        jsonl_path = os.path.join(output_dir, 'cqcs_bmjz.jsonl')
        export_jsonl(results, jsonl_path)
        
        sql_path = os.path.join(output_dir, 'cqcs_bmjz.sql')
        export_sql(results, sql_path)
        
        print(f'\nDone! {len(results)} articles')
    else:
        print('No results!')
