#!/usr/bin/env python3
"""
重庆市黔江区人民政府 - 周矶管理区其他法定信息爬虫
https://www.qianjiang.gov.cn/bmjd/xzfgzbm/zygyyqgwh/zwgk_48887/fdzdgknr_48889/qtfdzdgk/
CMS: WCM (重庆政府网站群) - index_N.html pagination
"""
import sys, re, json, time, requests
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

BASE = "https://www.qianjiang.gov.cn"
LIST_PATH = "/bmjd/xzfgzbm/zygyyqgwh/zwgk_48887/fdzdgknr_48889/qtfdzdgk/index.html"
PAGINATION_TPL = "/bmjd/xzfgzbm/zygyyqgwh/zwgk_48887/fdzdgknr_48889/qtfdzdgk/index_{}.html"
TOTAL_PAGES = 4  # createPage(4, 0, "index", "html")
MAX_PAGES = 4  # 全量4页; 增量跑第1页

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def crawl_list(url):
    """Parse list page HTML, return articles."""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except:
        return []

    if resp.status_code != 200:
        return []

    soup = BeautifulSoup(resp.text, 'html.parser')
    articles = []

    ul = soup.find('ul', class_='news-list')
    if not ul:
        return []

    for li in ul.find_all('li', recursive=False):
        a_tag = li.find('a', href=True)
        if not a_tag:
            continue

        href = a_tag.get('href', '')
        # Relative path like "./202605/t20260518_15684488.html"
        if href.startswith('./'):
            href = BASE + LIST_PATH.rsplit('/', 1)[0] + '/' + href[2:]
        elif href.startswith('/'):
            href = BASE + href
        elif not href.startswith('http'):
            href = BASE + '/' + href.lstrip('/')

        # Title from title attribute (full version) or link text
        title = a_tag.get('title', '') or a_tag.get_text(strip=True)

        # Date from <span> sibling
        date_span = li.find('span')
        date_text = date_span.get_text(strip=True) if date_span else ''

        if not title or len(title) < 5:
            continue

        articles.append({'title': title, 'url': href, 'date': date_text})

    return articles


def crawl_list_all(max_pages):
    all_articles = []
    for page in range(max_pages):
        if page == 0:
            url = BASE + LIST_PATH
        else:
            url = BASE + PAGINATION_TPL.format(page)

        print(f"[LIST] Page {page+1}: {url}", file=sys.stderr)
        articles = crawl_list(url)
        print(f"  Found {len(articles)} articles", file=sys.stderr)
        if not articles:
            break
        all_articles.extend(articles)
        time.sleep(1)

    return all_articles


def crawl_detail(article):
    url = article['url']
    print(f"[DETAIL] {article['title'][:40]}...", file=sys.stderr)

    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    if resp.status_code != 200:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        article['date_pub'] = article.get('date', '')
        return article

    soup = BeautifulSoup(resp.text, 'html.parser')

    # Date from article or list
    article['date_pub'] = article.get('date', '')

    # Content from div.content.qjcontent
    content_div = soup.find('div', class_='content')
    if not content_div or 'qjcontent' not in (content_div.get('class', []) or []):
        content_div = soup.find('div', class_=lambda c: c and 'content' in c and 'qjcontent' in c)
    if not content_div:
        content_div = soup.select_one('.content.qjcontent')

    if not content_div:
        article['content'] = f"[{article['title']}]({url})"
        article['attachments'] = []
        return article

    # Content from the TRS editor div inside content.qjcontent
    # Skip video, startprint/endprint markers
    trs_div = content_div.find('div', class_=re.compile(r'trs_editor|TRS_UEDITOR'))
    if trs_div:
        content_div = trs_div

    # Find attachments in file-box (separate from content_div)
    attachments = []
    file_box = soup.find('div', class_='file-box')
    if file_box:
        for a_tag in file_box.find_all('a', href=True):
            ahref = a_tag.get('href', '')
            if ahref.startswith('/'):
                ahref = BASE + ahref
            elif ahref.startswith('./'):
                ahref = BASE + LIST_PATH.rsplit('/', 1)[0] + '/' + ahref[2:]
            elif not ahref.startswith('http'):
                ahref = BASE + '/' + ahref.lstrip('/')
            atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
            attachments.append({'name': atext, 'url': ahref})

    # Also check for file links inside content_div
    for a_tag in content_div.find_all('a', href=re.compile(r'\.(doc|pdf|docx|xls|xlsx|rar|zip)(\?|$)', re.I)):
        ahref = a_tag.get('href', '')
        if ahref.startswith('/'):
            ahref = BASE + ahref
        elif ahref.startswith('./'):
            ahref = BASE + LIST_PATH.rsplit('/', 1)[0] + '/' + ahref[2:]
        elif not ahref.startswith('http'):
            ahref = BASE + '/' + ahref.lstrip('/')
        atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
        if not any(a['url'] == ahref for a in attachments):
            attachments.append({'name': atext, 'url': ahref})

    # Replace attachment links with markdown
    for a_tag in content_div.find_all('a'):
        ahref = a_tag.get('href', '')
        if re.search(r'\.(doc|pdf|docx|xls|xlsx|rar|zip)(\?|$)', ahref, re.I):
            if ahref.startswith('/'):
                ahref = BASE + ahref
            elif ahref.startswith('./'):
                ahref = BASE + LIST_PATH.rsplit('/', 1)[0] + '/' + ahref[2:]
            elif not ahref.startswith('http'):
                ahref = BASE + '/' + ahref.lstrip('/')
            atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
            a_tag.replace_with(f"[{atext}]({ahref})")

    # Build content: keep table HTML, paragraphs with \n\n
    parts = []
    for child in content_div.children:
        if child.name == 'table':
            parts.append(str(child))
        elif child.name in ('p', 'div'):
            # Skip print markers
            child_str = str(child)
            if 'startprint' in child_str or 'endprint' in child_str:
                continue
            text = child.get_text(separator='', strip=True)
            if text and not re.match(r'^(责任编辑|初审|复审|终审|\[纠错\])', text):
                # Check for nested table
                inner_table = child.find('table')
                if inner_table and text:
                    parts.append(str(inner_table))
                elif text:
                    parts.append(text)
        elif isinstance(child, str):
            text = child.strip()
            if text:
                parts.append(text)

    content_text = '\n\n'.join(parts) if parts else ''

    if len(content_text.strip()) < 20:
        content_text = f"[{article['title']}]({url})"

    article['content'] = content_text
    article['attachments'] = attachments
    return article


def main():
    max_pages = MAX_PAGES
    if len(sys.argv) > 1 and sys.argv[1] == '--incremental':
        max_pages = 1

    articles = crawl_list_all(max_pages)
    print(f"[LIST] Total: {len(articles)}", file=sys.stderr)
    if not articles:
        print("[]")
        return

    for i, a in enumerate(articles):
        print(f"[{i+1}/{len(articles)}] Crawling detail...", file=sys.stderr)
        a = crawl_detail(a)
        time.sleep(0.5)

    output = [{
        'title': a.get('title',''),
        'url': a.get('url',''),
        'date': a.get('date_pub', a.get('date','')),
        'content': a.get('content',''),
        'attachments': a.get('attachments',[]),
    } for a in articles]

    print(json.dumps(output, ensure_ascii=False, indent=2))


if __name__ == '__main__':
    main()
