#!/usr/bin/env python3
"""
祥云股份 (harvin.cn) - 招标公告爬虫
站点：https://www.harvin.cn/news/10/
CMS: 中企动力 (yun300.cn / 300.cn)
列表：AJAX /comp/portalResNews/list.do?currentPage=N
"""
import requests
import re
import json
import time
import os
import sys
from bs4 import BeautifulSoup

BASE_URL = 'https://www.harvin.cn'
LIST_URL = BASE_URL + '/news/10/'
SITE_NAME = '祥云股份-招标公告'
GROUP = '企业'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

def fetch(url, encoding='utf-8'):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = encoding
    return r.text

def extract_list_items(html):
    """解析AJAX列表页面"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    
    for h2 in soup.find_all('h2', class_=lambda c: c and 'newTitle' in c):
        title = h2.get_text(strip=True)
        if not title:
            continue
        # Find parent <a> tag
        parent_a = h2.find_parent('a')
        if parent_a and parent_a.get('href'):
            link = parent_a['href']
            if link.startswith('/'):
                link = BASE_URL + link
        else:
            continue
        
        # Extract date
        date_str = ''
        # Date is in the format: 发布时间 : YYYY-MM--DD
        time_div = h2.find_next('div', class_='newToolBox')
        if time_div:
            time_text = time_div.get_text(strip=True)
            date_match = re.search(r'(\d{4})[-/](\d{1,2})[-/](\d{1,2})', time_text)
            if date_match:
                y, m, d = date_match.groups()
                date_str = '%s-%s-%s' % (y, m.zfill(2), d.zfill(2))
        
        items.append({'title': title, 'link': link, 'date': date_str})
    
    return items

def extract_detail(html, url):
    """解析详情页"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title - from h1.p_headA > div.font
    title = ''
    h1 = soup.find('h1', class_=lambda c: c and 'p_headA' in c)
    if h1:
        font_div = h1.find('div', class_='font')
        if font_div:
            icon = font_div.find('i')
            if icon:
                icon.decompose()
            title = font_div.get_text(strip=True)
    
    # Fallback: title from <title> tag
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            t = title_tag.get_text(strip=True)
            if '_' in t:
                title = t.split('_')[0]
            else:
                title = t
    
    # Date
    date_str = ''
    pub_span = soup.find('span', class_='i_pubDate')
    if pub_span:
        # Get the parent li and find the date
        parent_li = pub_span.find_parent('li')
        if parent_li:
            date_match = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', parent_li.get_text(strip=True))
            if date_match:
                date_str = date_match.group(1)
    
    # Content - extract from div.e_box.p_articles (the actual article body)
    # This div follows p_infoBox and p_articlesTitle in the p_NewsDetail structure
    articles_div = soup.find('div', class_=lambda c: c and 'p_articles' in c and 'e_box' in c)
    
    content_lines = []
    
    if articles_div:
        # Remove "详情" title if present
        title_div = articles_div.find('div', class_='p_articlesTitle')
        if title_div:
            title_div.decompose()
        
        # Process tables into readable text
        for table in articles_div.find_all('table'):
            rows = []
            for tr in table.find_all('tr'):
                cells = []
                for td in tr.find_all(['td', 'th']):
                    cell_text = td.get_text(strip=True, separator=' ')
                    cell_text = re.sub(r'\s+', ' ', cell_text).strip()
                    if cell_text:
                        cells.append(cell_text)
                if cells:
                    rows.append(' | '.join(cells))
            if rows:
                content_lines.append('\n'.join(rows))
        
        # Remove table elements from soup before text extraction
        for table in articles_div.find_all('table'):
            table.decompose()
        
        # Get remaining text with paragraph separators
        paragraphs = []
        for p in articles_div.find_all(['p', 'div']):
            text = p.get_text(strip=True)
            if text:
                text = text.replace('\xa0', ' ').replace('\u3000', ' ').strip()
                if text:
                    paragraphs.append(text)
        
        if paragraphs:
            content_lines.append('\n\n'.join(paragraphs))
        
        content = '\n\n'.join(content_lines) if content_lines else ''
    else:
        # Fallback: try p_NewsDetail
        news_detail = soup.find('div', class_=lambda c: c and 'p_NewsDetail' in c)
        if news_detail:
            for cls in ['p_topBox', 'p_infoBox', 'p_PrevAndNext', 'p_PrevAndNextMo', 'p_articlesTitle']:
                for tag in news_detail.find_all('div', class_=lambda c, cls=cls: c and cls in c):
                    tag.decompose()
            
            text_soup = BeautifulSoup(str(news_detail), 'html.parser')
            paragraphs = []
            for tag in text_soup.find_all(['p', 'div']):
                text = tag.get_text(strip=True)
                if text and len(text) > 5:
                    text = text.replace('\xa0', ' ').replace('\u3000', ' ').strip()
                    paragraphs.append(text)
            content = '\n\n'.join(paragraphs)
    
    # Attachments
    attachments = []
    for a_tag in soup.find_all('a', href=re.compile(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', re.I)):
        href = a_tag.get('href', '')
        if href.startswith('/'):
            href = BASE_URL + href
        elif href.startswith('//'):
            href = 'https:' + href
        name = a_tag.get_text(strip=True) or href.split('/')[-1]
        attachments.append({'url': href, 'name': name})
    
    # Clean content
    content = re.sub(r'\n{3,}', '\n\n', content).strip()
    
    return {
        'title': title,
        'date': date_str,
        'content': content,
        'attachments': attachments
    }

def crawl_all(incremental=False):
    """爬取全部文章"""
    ajax_url = BASE_URL + '/comp/portalResNews/list.do'
    comp_id = 'portalResNews_list-15724885164848993'
    cid = '10'
    
    results = []
    
    if incremental:
        total_pages = 1
    else:
        total_pages = 5
    
    for page_num in range(1, total_pages + 1):
        params = {
            'compId': comp_id,
            'cid': cid,
            'currentPage': str(page_num)
        }
        print('  Fetching list page %d/%d...' % (page_num, total_pages))
        
        try:
            html = requests.get(ajax_url, params=params, headers=HEADERS, timeout=30).text
            items = extract_list_items(html)
            
            if not items:
                print('    No more items, stopping')
                break
            
            print('    Found %d items' % len(items))
            
            for item in items:
                print('  Fetching: %s' % item['title'][:50])
                try:
                    detail_html = fetch(item['link'])
                    detail = extract_detail(detail_html, item['link'])
                    
                    result = {
                        'title': detail['title'] or item['title'],
                        'link': item['link'],
                        'date': detail['date'] or item['date'],
                        'content': detail['content'],
                        'attachments': detail['attachments'],
                        'site_name': SITE_NAME,
                        'group': GROUP
                    }
                    
                    results.append(result)
                    time.sleep(0.5)
                except Exception as e:
                    print('    ERROR fetching detail: %s' % e)
                    # Still record with list data
                    results.append({
                        'title': item['title'],
                        'link': item['link'],
                        'date': item['date'],
                        'content': '',
                        'attachments': [],
                        'site_name': SITE_NAME,
                        'group': GROUP
                    })
        except Exception as e:
            print('    ERROR fetching list page %d: %s' % (page_num, e))
    
    return results

def export_jsonl(results, output_path):
    """导出为JSONL格式"""
    with open(output_path, 'w', encoding='utf-8') as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + '\n')
    print('Exported %d records to %s' % (len(results), output_path))

def export_sql(results, output_path):
    """导出为SQL插入语句"""
    with open(output_path, 'w', encoding='utf-8') as f:
        f.write('-- Records for %s\n' % SITE_NAME)
        f.write('DELETE FROM gov_raw WHERE link LIKE \'%s/news/%%\';\n' % BASE_URL)
        for r in results:
            content_escaped = r['content'].replace("'", "''")
            title_escaped = r['title'].replace("'", "''")
            attachments_json = json.dumps(r['attachments'], ensure_ascii=False)
            
            sql = "INSERT INTO gov_raw (title, link, date, content, attachments, site_name, \"group\") VALUES ('%s', '%s', '%s', '%s', '%s', '%s', '%s');\n" % (
                title_escaped, r['link'], r['date'], content_escaped, attachments_json, r['site_name'], r['group']
            )
            f.write(sql)
    print('Exported %d records to %s' % (len(results), output_path))

if __name__ == '__main__':
    output_dir = '/root/gov_crawler/output'
    os.makedirs(output_dir, exist_ok=True)
    
    import sys
    incremental = '--incremental' in sys.argv or '1' in sys.argv
    
    print('Crawling %s (incremental=%s)' % (SITE_NAME, incremental))
    results = crawl_all(incremental=incremental)
    
    if results:
        jsonl_path = os.path.join(output_dir, 'harvin.jsonl')
        export_jsonl(results, jsonl_path)
        
        sql_path = os.path.join(output_dir, 'harvin.sql')
        export_sql(results, sql_path)
        
        print('\nDone! %d articles total' % len(results))
        
        # Auto-import to DB
        print('Importing to DB...')
        os.system('cd /root/gov_crawler && python3 import_harvin2.py')
        
        # Restart search app
        os.system('systemctl restart search_app')
        print('search_app restarted')
    else:
        print('No results found!')
