#!/usr/bin/env python3
"""
恒申集团 (hscc.com) - 信息公示爬虫
站点：https://www.hscc.com/news/20/
CMS: 233 Technology (yun300.cn)
"""
import requests
import re
import json
import time
import sys
import os
from bs4 import BeautifulSoup

BASE_URL = 'https://www.hscc.com'
LIST_URL = BASE_URL + '/news/20/'
SITE_NAME = '恒申集团'
GROUP = '企业'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

def fetch(url, encoding='utf-8'):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = encoding
    return r.text

def extract_list_items(html):
    items = []
    soup = BeautifulSoup(html, 'html.parser')

    for h3 in soup.find_all('h3', class_=lambda c: c and ('newTitle' in c or 'titile' in c or 'p_title' in c)):
        title = h3.get_text(strip=True)
        if not title:
            continue
        parent_a = h3.find_parent('a')
        if not parent_a:
            parent_a = h3.parent
            while parent_a and parent_a.name != 'a':
                parent_a = parent_a.parent
        if not parent_a or not parent_a.get('href'):
            continue
        link = parent_a['href']
        if link.startswith('/'):
            link = BASE_URL + link
        date_str = ''
        time_box = h3.find_next('p', class_='TimeBox')
        if time_box:
            date_str = time_box.get_text(strip=True)
        else:
            p_time = h3.find_next('div', class_=lambda c: c and 'p_time' in c)
            if p_time:
                h6 = p_time.find('h6')
                if h6:
                    date_str = h6.get_text(strip=True)
        if '/' in date_str:
            parts = date_str.split('/')
            if len(parts) == 3:
                date_str = f'{parts[0]}-{parts[1].zfill(2)}-{parts[2].zfill(2)}'
        items.append({'title': title, 'link': link, 'date': date_str})

    seen = set()
    unique = []
    for item in items:
        if item['link'] not in seen:
            seen.add(item['link'])
            unique.append(item)
    return unique

def extract_detail(html, url):
    soup = BeautifulSoup(html, 'html.parser')
    title = ''
    h1 = soup.find('h1', class_=lambda c: c and 'p_headA' in c)
    if h1:
        font_div = h1.find('div', class_='font')
        if font_div:
            icon = font_div.find('i')
            if icon:
                icon.decompose()
            title = font_div.get_text(strip=True)
    if not title:
        for h1 in soup.find_all('h1'):
            font_div = h1.find('div', class_='font')
            if font_div:
                icon = font_div.find('i')
                if icon:
                    icon.decompose()
                t = font_div.get_text(strip=True)
                if t:
                    title = t
                    break
    date_str = ''
    # Main article date: <li><span class="i_pubDate">发布时间：</span>DATE</li>
    pub_span = soup.find('span', class_='i_pubDate')
    if pub_span and pub_span.parent:
        dm = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', pub_span.parent.get_text(strip=True))
        if dm:
            date_str = dm.group(1)
    if not date_str:
        time_div = soup.find('div', class_='time')
        if time_div:
            ts = time_div.find('span')
            if ts:
                date_str = ts.get_text(strip=True)
    content = ''
    articles_div = soup.find('div', class_=lambda c: c and 'p_articles' in c)
    if articles_div:
        for s in articles_div.find_all(['script', 'style']):
            s.decompose()
        for p in articles_div.find_all(['p', 'div']):
            p_text = p.get_text(strip=True)
            if p_text:
                content += p_text + '\n\n'
        for img in articles_div.find_all('img'):
            src = img.get('src', '')
            if src and not src.startswith('http'):
                src = BASE_URL + src
            alt = img.get('alt', '')
            if src:
                content += f'<p><a href="{src}">查看图片</a></p>\n\n'
    content = content.strip()
    return {'title': title, 'date': date_str, 'content': content, 'source_url': url}

def main(max_pages=5):
    print(f'[恒申集团] 爬取列表页: {LIST_URL}')
    html = fetch(LIST_URL)
    items = extract_list_items(html)
    print(f'[恒申集团] 列表页获取到 {len(items)} 条文章')

    api_url = BASE_URL + '/comp/portalResNews/list.do?compId=portalResNews_list-17646463194693745&cid=20&page={p}&size=14'
    for p in range(2, max_pages + 1):
        try:
            api_html = fetch(api_url.format(p=p))
            new_items = extract_list_items(api_html)
            if not new_items:
                break
            seen_links = set(i['link'] for i in items)
            added = 0
            for item in new_items:
                if item['link'] not in seen_links:
                    items.append(item)
                    seen_links.add(item['link'])
                    added += 1
            if added == 0:
                break
            time.sleep(0.5)
        except Exception as e:
            break

    print(f'\n[恒申集团] 共获取 {len(items)} 条，开始抓取详情...')
    results = []
    for i, item in enumerate(items):
        try:
            html = fetch(item['link'])
            detail = extract_detail(html, item['link'])
            detail['publish_date'] = detail['date'] or item['date']
            detail['site_name'] = SITE_NAME
            detail['category'] = GROUP
            results.append(detail)
            print(f'  [{i+1}/{len(items)}] {detail["title"][:40]}... {detail.get("publish_date", "")}')
            time.sleep(0.5)
        except Exception as e:
            print(f'  [{i+1}/{len(items)}] ERROR: {item["link"]}: {e}')

    output_file = '/tmp/hscc_output.jsonl'
    with open(output_file, 'w', encoding='utf-8') as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + '\n')

    print(f'\n[恒申集团] 完成！共 {len(results)} 条，已保存到 {output_file}')
    return results

if __name__ == '__main__':
    main(int(sys.argv[1]) if len(sys.argv) > 1 else 5)
