#!/usr/bin/env python3
"""
山东潍焦控股集团 - 新闻资讯爬虫
https://www.sdcoke.com/news/
CMS: 自定义PHP - static HTML pagination
"""
import sys, re, json, time, requests
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

BASE = "https://www.sdcoke.com"
LIST_PATH = "/news/index.html"
PAGINATION_TPL = "/news/{}.html"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def crawl_list(url):
    """Parse list page, return articles and whether there's a next page."""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except:
        return [], False

    if resp.status_code != 200:
        return [], False

    soup = BeautifulSoup(resp.text, 'html.parser')
    articles = []

    # Parse cat-news-list items (the main list)
    for item in soup.find_all('div', class_='cat-news-list'):
        title_div = item.find('div', class_='cat-xw-title')
        if not title_div:
            continue
        a_tag = title_div.find('a', href=True)
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        if href.startswith('/'):
            href = BASE + href
        elif not href.startswith('http'):
            href = BASE + '/' + href.lstrip('/')

        title = a_tag.get_text(strip=True)

        # Date from cat-xw-time (day + year/month)
        day_span = item.find('div', class_='cat-xw-day')
        year_span = item.find('div', class_='cat-xw-year')
        date_text = ''
        if day_span and year_span:
            date_text = f"{year_span.get_text(strip=True)}-{day_span.get_text(strip=True)}"
            # Normalize
            date_text = date_text.replace('年', '-').replace('月', '')

        if not title or len(title) < 5:
            continue

        articles.append({'title': title, 'url': href, 'date': date_text})

    # Also parse cat-news-item (featured top items)
    for item in soup.find_all('div', class_='cat-news-item'):
        title_div = item.find('div', class_='home-notice-title')
        if not title_div:
            continue
        a_tag = title_div.find('a', href=True)
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        if href.startswith('/'):
            href = BASE + href
        elif not href.startswith('http'):
            href = BASE + '/' + href.lstrip('/')

        title = a_tag.get_text(strip=True)

        time_div = item.find('div', class_='cat-news-time')
        date_text = time_div.get_text(strip=True) if time_div else ''

        if not title or len(title) < 5:
            continue

        # Deduplicate by URL
        if not any(a['url'] == href for a in articles):
            articles.append({'title': title, 'url': href, 'date': date_text})

    # Check pagination
    pages_div = soup.find('div', id='pages')
    has_next = bool(pages_div and '下一页' in pages_div.get_text())

    return articles, has_next


def crawl_list_all(max_pages):
    """Crawl multiple pages."""
    all_articles = []
    for page in range(max_pages):
        if page == 0:
            url = BASE + LIST_PATH
        else:
            url = BASE + PAGINATION_TPL.format(page + 1)

        print(f"[LIST] Page {page+1}: {url}", file=sys.stderr)
        articles, has_next = crawl_list(url)
        print(f"  Found {len(articles)} articles", file=sys.stderr)
        if not articles:
            break
        all_articles.extend(articles)
        if not has_next:
            break
        time.sleep(1)

    return all_articles


def crawl_detail(article):
    url = article['url']
    print(f"[DETAIL] {article['title'][:40]}...", file=sys.stderr)

    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    if resp.status_code != 200:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    soup = BeautifulSoup(resp.text, 'html.parser')

    # Title
    title_div = soup.find('div', class_='show-news-title')
    if title_div:
        t = title_div.get_text(strip=True)
        if t:
            article['title'] = t

    # Date from show-news-info
    date_text = article.get('date', '')
    info_div = soup.find('div', class_='show-news-info')
    if info_div:
        m = re.search(r'(\d{4}[-/]\d{2}[-/]\d{2})', info_div.get_text())
        if m:
            date_text = m.group(1)
    article['date_pub'] = date_text

    # Content from second show-news-info div (first is metadata, second is content)
    info_divs = soup.find_all('div', class_='show-news-info')
    content_div = None
    if len(info_divs) >= 2:
        content_div = info_divs[1]
    if not content_div:
        # Fallback: find all show-news-info, skip first
        content_all = soup.find_all('div', class_=re.compile(r'show-news-info'))
        if len(content_all) >= 2:
            content_div = content_all[1]

    if not content_div:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        return article

    # Find attachments: check for downloadable files
    attachments = []
    for a_tag in content_div.find_all('a', href=re.compile(r'\.(doc|pdf|docx|xls|xlsx|rar|zip)(\?|$)', re.I)):
        ahref = a_tag.get('href', '')
        if ahref.startswith('/'):
            ahref = BASE + ahref
        if not ahref.startswith('http'):
            ahref = BASE + '/' + ahref.lstrip('/')
        atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
        if not any(a['url'] == ahref for a in attachments):
            attachments.append({'name': atext, 'url': ahref})

    # Replace attachment links with markdown
    for a_tag in content_div.find_all('a'):
        ahref = a_tag.get('href', '')
        if re.search(r'\.(doc|pdf|docx|xls|xlsx|rar|zip)(\?|$)', ahref, re.I):
            if ahref.startswith('/'):
                ahref = BASE + ahref
            if not ahref.startswith('http'):
                ahref = BASE + '/' + ahref.lstrip('/')
            atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
            a_tag.replace_with(f"[{atext}]({ahref})")

    # Build content
    parts = []
    for child in content_div.children:
        if child.name == 'table':
            parts.append(str(child))
        elif child.name in ('p', 'div'):
            text = child.get_text(separator='', strip=True)
            if text:
                # Skip empty/whitespace-only divs
                parts.append(text)
        elif isinstance(child, str):
            text = child.strip()
            if text:
                parts.append(text)

    content_text = '\n\n'.join(parts) if parts else ''

    if len(content_text.strip()) < 20:
        content_text = f"[{article['title']}]({url})"

    article['content'] = content_text
    article['attachments'] = attachments
    return article


def main():
    max_pages = MAX_PAGES
    if len(sys.argv) > 1 and sys.argv[1] == '--incremental':
        max_pages = 1

    articles = crawl_list_all(max_pages)
    print(f"[LIST] Total: {len(articles)}", file=sys.stderr)
    if not articles:
        print("[]")
        return

    for i, a in enumerate(articles):
        print(f"[{i+1}/{len(articles)}] Crawling detail...", file=sys.stderr)
        a = crawl_detail(a)
        time.sleep(0.5)

    output = [{
        'title': a.get('title',''),
        'url': a.get('url',''),
        'date': a.get('date_pub', a.get('date','')),
        'content': a.get('content',''),
        'attachments': a.get('attachments',[]),
    } for a in articles]

    print(json.dumps(output, ensure_ascii=False, indent=2))


if __name__ == '__main__':
    main()
