#!/usr/bin/env python3
"""crawl_netda_tzgg.py - 南通经济技术开发区通知公告

TrueCMS Variant B: Inline (first 30) + API pagination
List: /ntjjkfqrmzf/tzgg/tzgg.html
Detail: /ntjjkfqrmzf/tzgg/content/{uuid}.html
"""

import requests, sys, re, time, json
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.netda.gov.cn"
LIST_URL = BASE_URL + "/ntjjkfqrmzf/tzgg/tzgg.html"
API_URL = BASE_URL + "/truecms/messageController/getMessage.do"
COLUMN_ID = "24981568-7339-431d-80a7-86e939bf6728"
PER_PAGE = 10  # API perPage
INLINE_COUNT = 30  # First 30 items are inline in HTML

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": LIST_URL,
}
DETAIL_HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}

session = requests.Session()


def extract_inline_items(html):
    """Extract items from inline initData div (first 30 items)."""
    soup = BeautifulSoup(html, "html.parser")
    init = soup.find("div", id="initData")
    if not init:
        return []
    items = []
    for ul in init.find_all("ul", class_="con"):
        li = ul.find("li")
        if not li:
            continue
        a = li.find("a")
        label = li.find("label")
        if a and label:
            href = a.get("href", "")
            title = a.get("title", "") or a.text.strip()
            date = label.text.strip()
            items.append({
                "title": title,
                "url": urljoin(BASE_URL, href),
                "date": date,
            })
    return items


def fetch_api_page(page_num):
    """Fetch one page via API (starts from page 4, items 31+)."""
    startrecord = (page_num - 1) * PER_PAGE + 1
    endrecord = startrecord + PER_PAGE - 1
    params = {
        "startrecord": str(startrecord),
        "endrecord": str(endrecord),
        "perpage": str(PER_PAGE),
        "columnId": COLUMN_ID,
    }
    r = session.get(API_URL, params=params, headers=HEADERS, timeout=20)
    data = r.json()
    result_html = data.get("result", "")
    # Strip CDATA markers
    result_html = result_html.replace("<![CDATA[", "").replace("]]>", "")
    # Unescape backslash-escaped slashes
    result_html = result_html.replace("\\/", "/")
    soup = BeautifulSoup(result_html, "html.parser")
    items = []
    for li in soup.find_all("li"):
        a = li.find("a")
        label = li.find("label")
        if a and label:
            href = a.get("href", "")
            title = a.get("title", "") or a.text.strip()
            date = label.text.strip()
            items.append({
                "title": title,
                "url": urljoin(BASE_URL, href),
                "date": date,
            })
    return items


def extract_detail(detail_url):
    """Extract title, date, content from detail page."""
    try:
        r = session.get(detail_url, headers=DETAIL_HEADERS, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Fetch detail failed: {e}", file=sys.stderr)
        return None

    soup = BeautifulSoup(r.text, "html.parser")
    zoom = soup.find("div", class_="content", id="zoom")
    if not zoom:
        zoom = soup.find("div", id="zoom")
    if not zoom:
        print(f"  [ERROR] No zoom div found", file=sys.stderr)
        return None

    # Title from div.title
    title_div = zoom.find("div", class_="title")
    title = title_div.text.strip() if title_div else ""

    # Date from div.stitle > span.fl
    stitle = zoom.find("div", class_="stitle")
    date = ""
    if stitle:
        span = stitle.find("span", class_="fl")
        if span:
            date_text = span.text.strip()
            # "发布时间：2026-07-10" or "2026-07-10"
            m = re.search(r"(\d{4}-\d{2}-\d{2})", date_text)
            if m:
                date = m.group(1)

    # Fallback: meta PubDate
    if not date:
        meta = soup.find("meta", attrs={"name": "PubDate"})
        if meta and meta.get("content"):
            m = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
            if m:
                date = m.group(1)

    # Content from div.con
    con = zoom.find("div", class_="con")
    if not con:
        print(f"  [ERROR] No div.con found", file=sys.stderr)
        content = title  # fallback
    else:
        paragraphs = []
        for child in con.children:
            if child.name == "p":
                txt = child.get_text(strip=True)
                if txt:
                    paragraphs.append(txt)
            elif child.name == "table":
                # Render table as Markdown pipe table
                md_table = []
                rows = child.find_all("tr")
                col_count = 0
                for tr in rows:
                    cells = tr.find_all(["td", "th"])
                    if not cells:
                        continue
                    row_data = [c.get_text(strip=True) for c in cells]
                    col_count = max(col_count, len(row_data))
                    md_table.append(row_data)
                if md_table:
                    # Build pipe table
                    lines = []
                    # Header row
                    header = md_table[0]
                    # Pad header to col_count
                    while len(header) < col_count:
                        header.append("")
                    lines.append("| " + " | ".join(header) + " |")
                    # Separator
                    lines.append("|" + "|".join([" --- "] * col_count) + "|")
                    # Data rows
                    for row in md_table[1:]:
                        while len(row) < col_count:
                            row.append("")
                        lines.append("| " + " | ".join(row) + " |")
                    paragraphs.append("\n".join(lines))

        content = "\n\n".join(paragraphs)

    return {
        "title": title,
        "date": date,
        "content": content,
    }


def main():
    pages = 5  # default
    if len(sys.argv) > 1:
        pages = int(sys.argv[1])

    # Step 1: Fetch list page
    print(f"Fetching list page: {LIST_URL}", file=sys.stderr)
    r = session.get(LIST_URL, headers=HEADERS, timeout=20)
    r.encoding = "utf-8"
    html = r.text

    # Step 2: Extract inline items
    all_items = extract_inline_items(html)
    print(f"Inline items: {len(all_items)}", file=sys.stderr)

    # Step 3: Fetch API pages if needed
    needed = pages * PER_PAGE
    if len(all_items) < needed:
        # First 3 pages (30 items) are inline, pages 4+ via API
        api_start_page = (INLINE_COUNT // PER_PAGE) + 1  # page 4
        while len(all_items) < needed:
            page = api_start_page + (len(all_items) - INLINE_COUNT) // PER_PAGE
            if page < api_start_page:
                page = api_start_page
            print(f"  Fetching API page {page}...", file=sys.stderr)
            api_items = fetch_api_page(page)
            if not api_items:
                print(f"  [WARN] Empty API response for page {page}", file=sys.stderr)
                break
            all_items.extend(api_items)
            time.sleep(0.5)

    # Trim to exact count
    all_items = all_items[:needed]
    print(f"Total items to process: {len(all_items)}", file=sys.stderr)

    # Step 4: Fetch details
    results = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}", file=sys.stderr)
        detail = extract_detail(item["url"])
        if detail:
            results.append({
                "title": detail["title"],
                "page_url": item["url"],
                "date": detail["date"] or item["date"],
                "site_name": "南通经济技术开发区",
                "content": detail["content"],
                "summary": detail["content"][:200] if detail["content"] else "",
            })
        time.sleep(0.5)

    # Step 5: Output as JSON for piping to DB insert
    print(json.dumps(results, ensure_ascii=False))


if __name__ == "__main__":
    main()
