#!/usr/bin/env python3
"""crawl_napo_hjbh.py - 那坡县生态环境-建设项目环评审批 (TRS CMS)

List: /zdlyxx/shgysyjslygk/hjbhly/jsxmhpyxpj/
Pagination: index.shtml (p1), index_N.shtml (pN)
Items: ul.npx-common-ul > li > a + span
Detail: ./tXXXXX.shtml
"""

import requests, sys, re, time, json
from bs4 import BeautifulSoup, Tag, NavigableString
from urllib.parse import urljoin

BASE_URL = "http://www.napo.gov.cn"
LIST_PATH = "/zdlyxx/shgysyjslygk/hjbhly/jsxmhpyxpj/"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}

session = requests.Session()
session.headers.update(HEADERS)


def extract_table(table_tag, paragraphs):
    """Convert HTML table to Markdown pipe table and append to paragraphs."""
    rows = table_tag.find_all("tr")
    col_count = 0
    table_data = []
    for tr in rows:
        cells = tr.find_all(["td", "th"])
        if not cells:
            continue
        row_data = [c.get_text(" ", strip=True) for c in cells]
        col_count = max(col_count, len(row_data))
        table_data.append(row_data)
    if table_data:
        header = table_data[0]
        while len(header) < col_count:
            header.append("")
        lines = ["| " + " | ".join(header) + " |"]
        lines.append("|" + "|".join([" --- "] * col_count) + "|")
        for row in table_data[1:]:
            while len(row) < col_count:
                row.append("")
            lines.append("| " + " | ".join(row) + " |")
        paragraphs.append("\n".join(lines))


def main():
    pages = 5
    if len(sys.argv) > 1:
        pages = int(sys.argv[1])

    # Step 1: Fetch list pages (stop at first 404)
    all_items = []
    for p in range(1, pages + 1):
        if p == 1:
            url = BASE_URL + LIST_PATH + "index.shtml"
        else:
            url = BASE_URL + LIST_PATH + f"index_{p}.shtml"

        print(f"Fetching list page {p}: {url}", file=sys.stderr)
        try:
            r = session.get(url, timeout=20)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERROR] {e}", file=sys.stderr)
            break

        if r.status_code != 200:
            print(f"  HTTP {r.status_code}, stopping", file=sys.stderr)
            break

        soup = BeautifulSoup(r.text, "html.parser")
        ul = soup.find("ul", class_="npx-common-ul")
        if not ul:
            print(f"  No list found, stopping", file=sys.stderr)
            break

        items = []
        for li in ul.find_all("li"):
            a = li.find("a")
            span = li.find("span")
            if a and span:
                href = a.get("href", "")
                title = a.get("title", "") or a.get_text(strip=True)
                date = span.get_text(strip=True)
                if href and title:
                    items.append({
                        "title": title,
                        "url": urljoin(BASE_URL + LIST_PATH, href),
                        "date": date,
                    })

        print(f"  Found {len(items)} items", file=sys.stderr)
        if not items:
            break
        all_items.extend(items)
        time.sleep(0.5)

    print(f"Total items: {len(all_items)}", file=sys.stderr)

    # Step 2: Fetch details
    results = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}", file=sys.stderr)

        try:
            r = session.get(item["url"], timeout=20)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [SKIP] Fetch failed: {e}", file=sys.stderr)
            continue

        soup = BeautifulSoup(r.text, "html.parser")

        # Title
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        title = meta.get("content", "").strip() if meta else item["title"]

        # Date
        date = item["date"]
        meta_pub = soup.find("meta", attrs={"name": "PubDate"})
        if meta_pub and meta_pub.get("content"):
            m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_pub["content"])
            if m:
                date = m.group(1)

        # Content
        con = soup.find("div", class_="article-con")
        if not con:
            print(f"  [SKIP] No content div", file=sys.stderr)
            continue

        paragraphs = []
        for child in con.children:
            # Skip comment nodes (NavigableStrings that are HTML comments)
            if isinstance(child, NavigableString):
                t = child.strip()
                if t and not t.startswith("视频") and not t.startswith("图集") and not t.startswith("正文") and not t.startswith("附件") and not t.startswith("其他"):
                    paragraphs.append(t)
                continue
            if not isinstance(child, Tag):
                continue

            if child.name == "p":
                txt = child.get_text(strip=True)
                if txt:
                    paragraphs.append(txt)
            elif child.name == "table":
                extract_table(child, paragraphs)
            elif child.name == "div":
                # Check if this is a TRS editor or other content wrapper
                cls_str = " ".join(child.get("class", [])) if child.get("class") else ""
                if "downloadfile" in cls_str:
                    # Skip download/attachment file containers (handled later)
                    continue
                # Recurse into content wrapper divs to get individual p/table
                has_children = False
                for sub in child.find_all(["p", "table"], recursive=False):
                    if sub.name == "p":
                        txt = sub.get_text(strip=True)
                        if txt:
                            paragraphs.append(txt)
                            has_children = True
                    elif sub.name == "table":
                        extract_table(sub, paragraphs)
                        has_children = True
                if not has_children:
                    txt = child.get_text(strip=True)
                    if txt and len(txt) > 10:
                        paragraphs.append(txt)

        content = "\n\n".join(paragraphs)

        # Skip if content is just title (no real body)
        c = content.strip().rstrip(". ")
        t = title.strip().rstrip(". ")
        if c == t:
            print(f"  [SKIP] No real content", file=sys.stderr)
            continue

        # Check for attachments
        for a in soup.find_all("a", href=True):
            href = a["href"]
            if re.search(r'\.(pdf|docx?|xlsx?)$', href, re.I):
                text = a.get_text(strip=True) or href.split("/")[-1]
                full_url = urljoin(item["url"], href)
                if full_url not in content:
                    content += f"\n\n[{text}]({full_url})"

        results.append({
            "title": title,
            "page_url": item["url"],
            "date": date,
            "site_name": "那坡县生态环境",
            "content": content,
            "summary": content[:200] if content else "",
        })

        if (i + 1) % 5 == 0:
            time.sleep(0.5)

    print(json.dumps(results, ensure_ascii=False))


if __name__ == "__main__":
    main()
