#!/usr/bin/env python3
"""
朔州新闻网 - 公告爬虫
https://www.shuozhounews.cn/folder2074/folder2076/folder2218/
CMS: m2o (媒体通) - offset-based pagination
"""
import sys, re, json, time, requests
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

BASE = "https://www.shuozhounews.cn"
LIST_PATH = "/folder2074/folder2076/folder2218/"
PAGINATION_TPL = "/folder2074/folder2076/folder2218/?pp={}"
PAGE_SIZE = 30
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def crawl_list(url):
    """Parse list page HTML, return articles."""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        return [], False

    if resp.status_code != 200:
        return [], False

    soup = BeautifulSoup(resp.text, 'html.parser')
    articles = []

    # Find txtList2 ul
    ul = soup.find('ul', class_=re.compile(r'txtList2'))
    if not ul:
        return [], False

    for li in ul.find_all('li', recursive=False):
        a_tag = li.find('a', class_='truncate', href=True)
        if not a_tag:
            continue

        href = a_tag.get('href', '')
        if href.startswith('//'):
            href = 'https:' + href
        elif href.startswith('/'):
            href = BASE + href
        elif not href.startswith('http'):
            href = BASE + '/' + href.lstrip('/')

        # Get title from link text (may be truncated)
        title = a_tag.get_text(strip=True)
        # Also check title attribute
        title_attr = a_tag.get('title', '')
        if title_attr and len(title_attr) > len(title):
            title = title_attr

        # Date from sibling span
        date_span = li.find('span', class_='listDate')
        date_text = date_span.get_text(strip=True) if date_span else ''

        if not title or len(title) < 5:
            continue

        articles.append({'title': title, 'url': href, 'date': date_text})

    # Check if there's a next page / pagination
    has_next = bool(soup.find('div', class_='meneame'))

    return articles, has_next


def crawl_list_all(max_pages):
    """Crawl multiple list pages."""
    all_articles = []
    for page in range(max_pages):
        if page == 0:
            url = BASE + LIST_PATH
        else:
            offset = page * PAGE_SIZE
            url = BASE + PAGINATION_TPL.format(offset)

        print(f"[LIST] Page {page+1}: {url}", file=sys.stderr)
        articles, has_next = crawl_list(url)
        print(f"  Found {len(articles)} articles", file=sys.stderr)
        if not articles:
            break
        all_articles.extend(articles)
        if not has_next:
            break
        time.sleep(1)

    return all_articles


def crawl_detail(article):
    url = article['url']
    print(f"[DETAIL] {article['title'][:40]}...", file=sys.stderr)

    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    if resp.status_code != 200:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    soup = BeautifulSoup(resp.text, 'html.parser')

    # Title from h1 inside div.article-title, or from m2o_content JS
    title_h1 = soup.find('h1')
    if title_h1:
        t = title_h1.get_text(strip=True)
        if t and len(t) > 5:
            article['title'] = t
    if not title_h1 or len(article['title']) < 5:
        m = re.search(r'm2o_content\s*=\s*\{[^}]*"title"\s*:\s*"([^"]+)"', resp.text)
        if m:
            article['title'] = m.group(1)

    # Date: from URL path /YYYY-MM-DD/ or article page
    date_text = article.get('date', '')
    m = re.search(r'/(\d{4}-\d{2}-\d{2})/', url)
    if m:
        date_text = m.group(1)
    # Also check for meta or span
    if not date_text:
        m2 = re.search(r'(\d{4}-\d{2}-\d{2})', resp.text)
        if m2:
            date_text = m2.group(1)
    article['date_pub'] = date_text

    # Content from div.article-main
    content_div = soup.find('div', class_='article-main')
    if not content_div:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        return article

    # Find attachments: file download links
    attachments = []
    for a_tag in content_div.find_all('a', href=re.compile(r'/download/material', re.I)):
        ahref = a_tag.get('href', '')
        if ahref.startswith('//'):
            ahref = 'https:' + ahref
        elif ahref.startswith('/'):
            ahref = BASE + ahref
        if not ahref.startswith('http'):
            ahref = BASE + '/' + ahref.lstrip('/')
        # Get name from title attribute or img title or URL
        name = a_tag.get('title', '') or ''
        img = a_tag.find('img')
        if img and not name:
            name = img.get('title', '') or img.get('alt', '') or ''
        if not name:
            name = ahref.split('name=')[-1].split('&')[0] if 'name=' in ahref else ahref.split('/')[-1]
        if not any(a['url'] == ahref for a in attachments):
            attachments.append({'name': name, 'url': ahref})

    # Also check for image attachments (.doc/.pdf images)
    for img in content_div.find_all('img', class_='image-refer'):
        parent_a = img.find_parent('a')
        if parent_a and parent_a.get('href'):
            ahref = parent_a['href']
            if ahref.startswith('//'):
                ahref = 'https:' + ahref
            elif ahref.startswith('/'):
                ahref = BASE + ahref
            name = img.get('title', '') or parent_a.get('title', '') or ahref.split('/')[-1]
            if not any(a['url'] == ahref for a in attachments):
                attachments.append({'name': name, 'url': ahref})

    # Replace attachment links with markdown in content
    for a_tag in content_div.find_all('a'):
        ahref = a_tag.get('href', '')
        if '/download/material' in ahref:
            if ahref.startswith('//'):
                ahref = 'https:' + ahref
            elif ahref.startswith('/'):
                ahref = BASE + ahref
            atext = a_tag.get_text(strip=True) or ''
            img = a_tag.find('img')
            if img:
                atext = img.get('title', '') or img.get('alt', '') or atext
            if not atext:
                atext = ahref.split('name=')[-1].split('&')[0] if 'name=' in ahref else '附件'
            md_link = f"[{atext}]({ahref})"
            a_tag.replace_with(md_link)

    # For img.image-refer inside content, replace with the parent link
    for img in content_div.find_all('img', class_='image-refer'):
        parent_a = img.find_parent('a')
        if parent_a:
            # Already handled above - the a tag was replaced
            pass
        else:
            # Standalone doc preview image - keep the alt text
            alt_text = img.get('alt', '') or img.get('title', '') or ''
            if alt_text:
                img.replace_with(f'[{alt_text}]({url})')

    # Build content: paragraphs with \n\n
    parts = []
    for child in content_div.children:
        if child.name == 'table':
            parts.append(str(child))
        elif child.name == 'p':
            text = child.get_text(separator='', strip=True)
            if text and not re.match(r'^(责任编辑|初审|复审|终审|编辑)', text):
                parts.append(text)
        elif child.name == 'div':
            text = child.get_text(separator='', strip=True)
            if text and not re.match(r'^(责任编辑|编辑|扫码)', text):
                classes = child.get('class', [])
                if not any(c in str(classes) for c in ['editor', 'clear']):
                    # Check for inner table
                    inner_table = child.find('table')
                    if inner_table:
                        parts.append(str(inner_table))
                    else:
                        parts.append(text)
        elif isinstance(child, str):
            text = child.strip()
            if text:
                parts.append(text)

    content_text = '\n\n'.join(parts) if parts else ''

    # Fallback: remove editor line at the end
    content_text = re.sub(r'\n+编辑[：:][^\n]*$', '', content_text).strip()

    if len(content_text.strip()) < 20:
        content_text = f"[{article['title']}]({url})"

    article['content'] = content_text
    article['attachments'] = attachments
    return article


def main():
    max_pages = MAX_PAGES
    if len(sys.argv) > 1 and sys.argv[1] == '--incremental':
        max_pages = 1

    articles = crawl_list_all(max_pages)
    print(f"[LIST] Total: {len(articles)}", file=sys.stderr)
    if not articles:
        print("[]")
        return

    for i, a in enumerate(articles):
        print(f"[{i+1}/{len(articles)}] Crawling detail...", file=sys.stderr)
        a = crawl_detail(a)
        time.sleep(0.5)

    output = [{
        'title': a.get('title',''),
        'url': a.get('url',''),
        'date': a.get('date_pub', a.get('date','')),
        'content': a.get('content',''),
        'attachments': a.get('attachments',[]),
    } for a in articles]

    print(json.dumps(output, ensure_ascii=False, indent=2))


if __name__ == '__main__':
    main()
