#!/usr/bin/env python3
"""
舒城县人民政府 - 公示公告爬虫
https://www.shucheng.gov.cn/zwzx/gsgg/index.html
CMS: 龙讯科技 (Lonsun) - HTML pagination
"""
import sys, re, json, time, requests
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

BASE = "https://www.shucheng.gov.cn"
PAGE1_URL = "/zwzx/gsgg/index.html"
PAGINATION_TPL = "https://www.shucheng.gov.cn/content/column/6788901?pageIndex={}"
MAX_PAGES = 5  # 初始5页; 增量用--incremental只跑第1页

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Connection": "keep-alive",
}

def crawl_list_html(url):
    """Parse list page HTML, return articles."""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"[ERROR] {e}", file=sys.stderr)
        return [], False

    if resp.status_code != 200:
        return [], False

    soup = BeautifulSoup(resp.text, 'html.parser')
    articles = []
    ul = soup.find('ul', class_=re.compile(r'doc_list'))
    if not ul:
        return [], False

    for li in ul.find_all('li', recursive=False):
        a_tag = li.find('a', href=True)
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        title_attr = a_tag.get('title', '')
        link_text_span = a_tag.find('span')
        title = title_attr or (link_text_span.get_text(strip=True) if link_text_span else a_tag.get_text(strip=True))

        # Date from sibling span
        date_span = li.find('span', class_='date')
        date_text = date_span.get_text(strip=True) if date_span else ''

        if not title or len(title) < 5:
            continue

        if href.startswith('http'):
            full_url = href
        elif href.startswith('/'):
            full_url = BASE + href
        else:
            full_url = BASE + '/' + href

        articles.append({'title': title, 'url': full_url, 'date': date_text})

    return articles, True


def crawl_list_all(max_pages):
    """Crawl multiple pages."""
    all_articles = []
    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE + PAGE1_URL
        else:
            url = PAGINATION_TPL.format(page)

        print(f"[LIST] Page {page}: {url}", file=sys.stderr)
        articles, ok = crawl_list_html(url)
        print(f"  Found {len(articles)} articles", file=sys.stderr)
        if not ok or not articles:
            break
        all_articles.extend(articles)
        time.sleep(1)

    return all_articles


def crawl_detail(article):
    url = article['url']
    print(f"[DETAIL] {article['title'][:40]}...", file=sys.stderr)

    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    if resp.status_code != 200:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    soup = BeautifulSoup(resp.text, 'html.parser')

    # Title from <h1 class="newstitle">
    h1 = soup.find('h1', class_='newstitle')
    if h1:
        article['title'] = h1.get_text(strip=True)

    # Date from meta
    pub_date = ""
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        pub_date = meta['content'].strip()[:10]
    if not pub_date:
        span = soup.find('span', string=re.compile(r'发布日期'))
        if span:
            m = re.search(r'(\d{4}[-/]\d{2}[-/]\d{2})', span.get_text())
            if m:
                pub_date = m.group(1)
    if not pub_date:
        pub_date = article.get('date', '')
    article['date_pub'] = pub_date

    # Content from div#zoom.j-fontContent
    content_div = soup.find('div', id='zoom')
    if not content_div:
        content_div = soup.find('div', class_=re.compile(r'j-fontContent|newscontnet'))
    if not content_div:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        return article

    # Attachments: find links to files (.xlsx, .doc, .pdf, etc.)
    attachments = []
    for a_tag in content_div.find_all('a', href=re.compile(r'\.(doc|pdf|docx|xls|xlsx|rar|zip|xlsm|ppt|pptx)(\?|$)', re.I)):
        ahref = a_tag.get('href', '')
        if ahref.startswith('/'):
            ahref = BASE + ahref
        if not ahref.startswith('http'):
            ahref = BASE + '/' + ahref.lstrip('/')
        atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
        if not any(a['url'] == ahref for a in attachments):
            attachments.append({'name': atext, 'url': ahref})

    # Also check /group3/... paths (Lonsun file storage)
    for a_tag in content_div.find_all('a', href=re.compile(r'/group\d/')):
        ahref = a_tag.get('href', '')
        if ahref.startswith('/'):
            ahref = BASE + ahref
        if not ahref.startswith('http'):
            ahref = BASE + '/' + ahref.lstrip('/')
        # Skip non-file links
        ext = ahref.split('.')[-1].lower() if '.' in ahref else ''
        if ext in ('doc','pdf','docx','xls','xlsx','rar','zip','xlsm','ppt','pptx','jpg','png','gif','jpeg'):
            atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
            if not any(a['url'] == ahref for a in attachments):
                attachments.append({'name': atext, 'url': ahref})

    # Convert attachment links to markdown in content
    for a_tag in content_div.find_all('a'):
        ahref = a_tag.get('href', '')
        is_attach = bool(re.search(r'\.(doc|pdf|docx|xls|xlsx|rar|zip|xlsm|ppt|pptx)(\?|$)', ahref, re.I)) or '/group' in ahref
        if is_attach:
            if ahref.startswith('/'):
                ahref = BASE + ahref
            if not ahref.startswith('http'):
                ahref = BASE + '/' + ahref.lstrip('/')
            atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
            md_link = f"[{atext}]({ahref})"
            a_tag.replace_with(md_link)

    # Build content: paragraphs with \n\n, tables kept as HTML
    parts = []
    for child in content_div.children:
        if child.name == 'table':
            # Keep table HTML
            parts.append(str(child))
        elif child.name == 'p':
            text = child.get_text(separator='', strip=True)
            if text and not re.match(r'^(责任编辑|初审|复审|终审|\[纠错\])', text):
                parts.append(text)
        elif child.name == 'div':
            # Check if this div contains a table
            inner_table = child.find('table')
            if inner_table:
                # Keep table HTML, and get text from sibling content before/after
                parts.append(str(inner_table))
                # Also get any text siblings
                for sibling in child.children:
                    if sibling.name != 'table' and isinstance(sibling, str):
                        t = sibling.strip()
                        if t:
                            parts.append(t)
            else:
                text = child.get_text(separator='', strip=True)
                if text and not re.match(r'^(责任编辑|初审|复审|终审|\[纠错\]|标签)', text):
                    classes = child.get('class', [])
                    if not any('clear' in (c or '') for c in classes):
                        parts.append(text)
        elif isinstance(child, str):
            text = child.strip()
            if text:
                parts.append(text)

    content_text = '\n\n'.join(parts) if parts else ''

    # Fallback: if too short
    if len(content_text.strip()) < 20:
        content_text = f"[{article['title']}]({url})"

    article['content'] = content_text
    article['attachments'] = attachments
    return article


def main():
    max_pages = MAX_PAGES
    if len(sys.argv) > 1 and sys.argv[1] == '--incremental':
        max_pages = 1

    articles = crawl_list_all(max_pages)
    print(f"[LIST] Total: {len(articles)}", file=sys.stderr)
    if not articles:
        print("[]")
        return

    for i, a in enumerate(articles):
        print(f"[{i+1}/{len(articles)}] Crawling detail...", file=sys.stderr)
        a = crawl_detail(a)
        time.sleep(0.5)

    output = [{
        'title': a.get('title',''),
        'url': a.get('url',''),
        'date': a.get('date_pub', a.get('date','')),
        'content': a.get('content',''),
        'attachments': a.get('attachments',[]),
    } for a in articles]

    print(json.dumps(output, ensure_ascii=False, indent=2))


if __name__ == '__main__':
    main()
