#!/usr/bin/env python3
"""
安康高新技术产业开发区管理委员会 - 环境保护公告爬虫
https://www.sxakhidz.gov.cn/Node-35041.html
CMS: 国微CMS (华社) - static HTML pagination
"""
import sys
import re
import json
import time
import requests
from bs4 import BeautifulSoup
from datetime import datetime

BASE = "https://www.sxakhidz.gov.cn"
LIST_URL = "/Node-35041.html"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}


def extract_title_from_attr(title_attr):
    """title attribute format: '标题：xxx&#xD;点击数：N&#xD;发表时间：YY年MM月DD日'"""
    if not title_attr:
        return ""
    m = re.search(r'标题[：:]\s*(.*?)(?:&#[xX]?0*[dD];|\\r?\\n)', title_attr)
    if m:
        return m.group(1).strip()
    t = re.sub(r'&#[xX]?0*[dD];.*', '', title_attr).strip()
    t = re.sub(r'点击数.*', '', t).strip()
    t = re.sub(r'发表时间.*', '', t).strip()
    t = re.sub(r'^标题[：:]\s*', '', t).strip()
    return t


def crawl_list(page_url, max_pages):
    articles = []
    for page_num in range(1, max_pages + 1):
        if page_num == 1:
            url = BASE + page_url
        else:
            base_name = page_url.replace('.html', '')
            url = BASE + f"{base_name}_{page_num}.html"

        print(f"[LIST] Page {page_num}: {url}", file=sys.stderr)
        try:
            resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
            resp.encoding = 'utf-8'
        except Exception as e:
            print(f"[ERROR] Failed to fetch page {page_num}: {e}", file=sys.stderr)
            break

        if resp.status_code != 200:
            print(f"[WARN] Page {page_num} returned {resp.status_code}", file=sys.stderr)
            break

        soup = BeautifulSoup(resp.text, 'html.parser')
        items = soup.select('li a[title]')
        if not items:
            items = soup.select('ul li a[title]')

        found = 0
        for a_tag in items:
            href = a_tag.get('href', '')
            title_attr = a_tag.get('title', '')

            if '/govsub/' in href or 'category-' in href:
                continue
            if '/Content-' not in href and not href.startswith('/Content-') and 'Content-' not in href:
                continue

            if href.startswith('http'):
                full_url = href
            elif href.startswith('/'):
                full_url = BASE + href
            else:
                full_url = BASE + '/' + href

            title = extract_title_from_attr(title_attr)
            date_text = ""
            span = a_tag.find_next('span', class_='dateRight')
            if span:
                date_text = span.get_text(strip=True)

            if not title:
                title = a_tag.get_text(strip=True)
            if not title or len(title) < 5:
                continue

            articles.append({
                'title': title,
                'url': full_url,
                'date': date_text,
            })
            found += 1

        print(f"  Found {found} articles on page {page_num}", file=sys.stderr)
        if found == 0:
            break
        time.sleep(1)

    return articles


def crawl_detail(article):
    url = article['url']
    print(f"[DETAIL] {article['title'][:40]}...", file=sys.stderr)

    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"[ERROR] Failed: {e}", file=sys.stderr)
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    if resp.status_code != 200:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    soup = BeautifulSoup(resp.text, 'html.parser')

    # Title from <h4> inside .detaTit
    title_tag = soup.select_one('.detaTit h4')
    if title_tag:
        article['title'] = title_tag.get_text(strip=True)

    # Date from meta tag
    pub_date = ""
    meta = soup.find('meta', attrs={'name': re.compile(r'PubDate|publishDate', re.I)})
    if meta and meta.get('content'):
        pub_date = meta['content'].strip()
    if not pub_date:
        time_span = soup.select_one('.shij span')
        if time_span:
            t = time_span.get_text(strip=True)
            m = re.search(r'(\d{4}[-/]\d{2}[-/]\d{2})', t)
            if m:
                pub_date = m.group(1)
    if not pub_date:
        pub_date = article.get('date', '')
    article['date_pub'] = pub_date

    # Content from .wenz
    content_div = soup.select_one('div.wenz')
    if not content_div:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        return article

    # Attachments
    attachments = []
    for a_tag in content_div.find_all('a', href=re.compile(r'\.(doc|pdf|docx|xls|xlsx|rar|zip)(\?|$)', re.I)):
        ahref = a_tag.get('href', '')
        if ahref.startswith('/'):
            ahref = BASE + ahref
        if not ahref.startswith('http'):
            ahref = BASE + '/' + ahref.lstrip('/')
        atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
        attachments.append({'name': atext, 'url': ahref})

    for a_tag in content_div.find_all('a', class_='ke-insertfile'):
        ahref = a_tag.get('href', '')
        if ahref.startswith('/'):
            ahref = BASE + ahref
        if not ahref.startswith('http'):
            ahref = BASE + '/' + ahref.lstrip('/')
        atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
        if not any(a['url'] == ahref for a in attachments):
            attachments.append({'name': atext, 'url': ahref})

    # Convert attachment links to markdown in content HTML
    for a_tag in content_div.find_all('a'):
        ahref = a_tag.get('href', '')
        if re.search(r'\.(doc|pdf|docx|xls|xlsx|rar|zip)(\?|$)', ahref, re.I) or 'UploadFiles' in ahref:
            if ahref.startswith('/'):
                ahref = BASE + ahref
            if not ahref.startswith('http'):
                ahref = BASE + '/' + ahref.lstrip('/')
            atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
            md_link = f"[{atext}]({ahref})"
            a_tag.replace_with(md_link)

    # Build text: paragraphs, keep table HTML
    parts = []
    for child in content_div.children:
        if child.name == 'table':
            parts.append(str(child))
        elif child.name == 'p':
            text = child.get_text(separator='', strip=True)
            if text:
                parts.append(text)
        elif child.name == 'div':
            text = child.get_text(separator='', strip=True)
            if text:
                parts.append(text)
        elif isinstance(child, str):
            text = child.strip()
            if text:
                parts.append(text)

    # Filter metadata lines
    filtered = []
    for p in parts:
        p_clean = p.strip()
        if re.match(r'^(索引号|公开目录|公开形式|公开日期|文号|发文时间|文件名称)[：:]\s*', p_clean):
            continue
        filtered.append(p_clean)

    content_text = '\n\n'.join(filtered) if filtered else ''

    if len(content_text.strip()) < 20:
        content_text = f"[{article['title']}]({url})"

    article['content'] = content_text
    article['attachments'] = attachments
    return article


def main():
    import urllib3
    urllib3.disable_warnings()

    max_pages = MAX_PAGES
    if len(sys.argv) > 1 and sys.argv[1] == '--incremental':
        max_pages = 1

    articles = crawl_list(LIST_URL, max_pages)
    print(f"[LIST] Total articles found: {len(articles)}", file=sys.stderr)
    if not articles:
        print("[]")
        return

    for i, article in enumerate(articles):
        print(f"[{i+1}/{len(articles)}] Crawling detail...", file=sys.stderr)
        article = crawl_detail(article)
        time.sleep(0.5)

    output = [{
        'title': a.get('title', ''),
        'url': a.get('url', ''),
        'date': a.get('date_pub', a.get('date', '')),
        'content': a.get('content', ''),
        'attachments': a.get('attachments', []),
    } for a in articles]

    print(json.dumps(output, ensure_ascii=False, indent=2))


if __name__ == '__main__':
    main()
