#!/usr/bin/env python3
"""crawl_nc_hjbha.py - 南昌市生态环境信息公开 (UCAP CMS)

List: /ncszf/hjbha/2021zfxxgk_list.shtml
Pagination: _N.shtml (page 1 = no suffix)
Detail: /ncszf/hjbha/YYYYMM/uuid.shtml
"""

import requests, sys, re, time, json
from bs4 import BeautifulSoup, Tag, NavigableString
from urllib.parse import urljoin, urlparse

BASE_URL = "https://www.nc.gov.cn"
LIST_URL = BASE_URL + "/ncszf/hjbha/2021zfxxgk_list.shtml"
PER_PAGE = 20

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}

session = requests.Session()
session.headers.update(HEADERS)


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def is_external_url(url):
    """Check if URL points to an external domain (not nc.gov.cn)."""
    parsed = urlparse(url)
    if not parsed.netloc:
        return False  # relative URL
    return "nc.gov.cn" not in parsed.netloc


def fetch_list_page(page_num):
    """Fetch list page. Page 1 = no suffix, pages 2+ = _N.shtml."""
    if page_num == 1:
        url = LIST_URL
    else:
        url = LIST_URL.replace(".shtml", f"_{page_num}.shtml")
    try:
        r = session.get(url, timeout=20)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  [ERROR] Fetch list page {page_num}: {e}", file=sys.stderr)
        return None


def extract_list_items(html):
    """Extract items from list page."""
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="pageList")
    if not ul:
        return []
    items = []
    for li in ul.find_all("li"):
        a = li.find("a")
        span = li.find("span", class_="time")
        if a and span:
            href = a.get("href", "")
            title = a.get("title", "") or a.get_text(strip=True)
            date = span.get_text(strip=True)
            if href and title:
                full_url = urljoin(BASE_URL, href)
                # Skip external URLs
                if is_external_url(full_url):
                    continue
                items.append({
                    "title": title,
                    "url": full_url,
                    "date": date,
                })
    return items


def extract_detail(detail_url):
    """Extract title, date, content from detail page."""
    try:
        r = session.get(detail_url, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Fetch detail: {e}", file=sys.stderr)
        return None

    soup = BeautifulSoup(r.text, "html.parser")

    # Title from h1 or meta
    h1 = soup.find("h1")
    title = h1.get_text(strip=True) if h1 else ""
    if not title:
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()

    # Date from meta PubDate
    date = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
        if m:
            date = m.group(1)

    # Content area
    content_area = soup.find("div", class_="article-content-body")
    if not content_area:
        detail_div = soup.find("div", class_="xxgk-detail")
        if detail_div:
            content_area = detail_div.find("div", class_="article-content-body")
        if not content_area:
            content_area = detail_div

    if not content_area:
        print(f"  [WARN] No content area, using title", file=sys.stderr)
        return {"title": title, "date": date, "content": title}

    # Check for ucapcontent wrapper
    ucap = content_area.find("ucapcontent")
    if ucap:
        content_area = ucap

    # Extract paragraphs and tables
    paragraphs = []
    for child in content_area.children:
        if isinstance(child, NavigableString):
            t = child.strip()
            if t:
                paragraphs.append(t)
            continue
        if not isinstance(child, Tag):
            continue

        if child.name == "p":
            txt = child.get_text(strip=True)
            if txt:
                paragraphs.append(txt)
        elif child.name == 'table':
            tbl_html = html_table_to_html(child, detail_url)
            if tbl_html:
                content_parts.append(tbl_html)
        elif child.name == "div":
            txt = child.get_text(strip=True)
            if txt:
                paragraphs.append(txt)
        elif child.name == "img":
            src = child.get("src", "")
            alt = child.get("alt", "")
            if src:
                paragraphs.append(f"![{alt}]({urljoin(detail_url, src)})")

    content = "\n\n".join(paragraphs)

    # If content is just a bare URL (external system redirect), use title
    if re.match(r'^https?://[^\s]+$', content.strip()):
        content = title

    # Check for attachments
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if re.search(r'\.(pdf|docx?|xlsx?)$', href, re.I):
            text = a.get_text(strip=True) or href.split("/")[-1]
            full_url = urljoin(detail_url, href)
            # Only add if not already in content
            if full_url not in content:
                content += f"\n\n[{text}]({full_url})"

    return {
        "title": title,
        "date": date,
        "content": content,
    }


def main():
    pages = 5
    if len(sys.argv) > 1:
        pages = int(sys.argv[1])

    # Fetch list pages
    all_items = []
    for p in range(1, pages + 1):
        print(f"Fetching list page {p}...", file=sys.stderr)
        html = fetch_list_page(p)
        if not html:
            break
        items = extract_list_items(html)
        if not items:
            print(f"  No items found on page {p}, stopping", file=sys.stderr)
            break
        print(f"  Found {len(items)} items", file=sys.stderr)
        all_items.extend(items)
        if p < pages:
            time.sleep(1)

    total_needed = pages * PER_PAGE
    all_items = all_items[:total_needed]
    print(f"Total items: {len(all_items)}", file=sys.stderr)

    # Fetch details
    results = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}", file=sys.stderr)
        detail = extract_detail(item["url"])
        if detail:
            results.append({
                "title": detail["title"],
                "page_url": item["url"],
                "date": detail["date"] or item["date"],
                "site_name": "南昌市生态环境",
                "content": detail["content"],
                "summary": detail["content"][:200] if detail["content"] else "",
            })
        if (i + 1) % 5 == 0:
            time.sleep(1)

    print(json.dumps(results, ensure_ascii=False))


if __name__ == "__main__":
    main()
