#!/usr/bin/env python3
"""crawl_muchuan_gggs.py - 沐川县公告公示 (TRS CMS)

List: /mcx/gggs/list.shtml
Pagination: list_N.shtml (p1=list.shtml, p2=list_2.shtml)
Items: div.reclist > dl > a (title after ||) + p (date in 「」)
Detail: /mcx/gggs/YYYYMM/uuid.shtml
"""

import requests, sys, re, time, json
from bs4 import BeautifulSoup, Tag, NavigableString
from urllib.parse import urljoin

BASE_URL = "http://www.muchuan.gov.cn"
LIST_URL = BASE_URL + "/mcx/gggs/list.shtml"
PER_PAGE = 25

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}

session = requests.Session()
session.headers.update(HEADERS)


def extract_table(table_tag, paragraphs):
    """Convert HTML table to Markdown pipe table."""
    rows = table_tag.find_all("tr")
    col_count = 0
    table_data = []
    for tr in rows:
        cells = tr.find_all(["td", "th"])
        if not cells:
            continue
        row_data = [c.get_text(" ", strip=True) for c in cells]
        col_count = max(col_count, len(row_data))
        table_data.append(row_data)
    if table_data:
        header = table_data[0]
        while len(header) < col_count:
            header.append("")
        lines = ["| " + " | ".join(header) + " |"]
        lines.append("|" + "|".join([" --- "] * col_count) + "|")
        for row in table_data[1:]:
            while len(row) < col_count:
                row.append("")
            lines.append("| " + " | ".join(row) + " |")
        paragraphs.append("\n".join(lines))


def extract_detail(detail_url):
    """Extract title, date, content from detail page."""
    try:
        r = session.get(detail_url, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Fetch detail: {e}", file=sys.stderr)
        return None

    soup = BeautifulSoup(r.text, "html.parser")

    # Title from meta
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    title = meta.get("content", "").strip() if meta else ""

    # Date from meta PubDate
    date = ""
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta["content"])
        if m:
            date = m.group(1)

    # Content from div.wrap
    wrap = soup.find("div", class_="wrap")
    if not wrap:
        print(f"  [WARN] No wrap div", file=sys.stderr)
        return {"title": title, "date": date, "content": title}

    paragraphs = []
    for child in wrap.children:
        if isinstance(child, NavigableString):
            t = child.strip()
            # Skip interference like "二维码开始", "二维码结束"
            if t and not re.search(r"二维码开始|二维码结束|打印本页|关闭窗口", t):
                paragraphs.append(t)
            continue
        if not isinstance(child, Tag) or child.name not in ["div", "p", "table"]:
            continue

        txt = child.get_text(strip=True)
        if not txt:
            continue

        cls_str = " ".join(child.get("class", [])) if child.get("class") else ""

        # Skip footer/header/UI elements
        if any(x in cls_str for x in ["share", "footer", "header", "nav"]):
            continue
        if re.search(r"扫一扫在手机打开|打印本页|关闭窗口|字体|分享", txt[:20]):
            continue
        # Skip title/info divs (first 2 divs in wrap - handled by meta)
        if re.match(r"发布时间：", txt[:10]):
            continue

        # Check if this is the article body div
        if child.find_all(["p", "table"]):
            # Has paragraph/table children - extract individually
            for sub in child.find_all(["p", "table"], recursive=False):
                if sub.name == "p":
                    st = sub.get_text(strip=True)
                    if st:
                        paragraphs.append(st)
                elif sub.name == "table":
                    extract_table(sub, paragraphs)
        else:
            # Plain text div
            if txt and len(txt) > 5:
                paragraphs.append(txt)

    content = "\n\n".join(paragraphs)

    # Skip if no real content
    if not content.strip() or content.strip() == title:
        content = title

    # Check for attachments
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if re.search(r'\.(pdf|docx?|xlsx?)$', href, re.I):
            text = a.get_text(strip=True) or href.split("/")[-1]
            full_url = urljoin(detail_url, href)
            if full_url not in content:
                content += f'\n\n<p><a href="{full_url}">{text}</a></p>'

    return {"title": title, "date": date, "content": content}


def main():
    pages = 5
    if len(sys.argv) > 1:
        pages = int(sys.argv[1])

    all_items = []
    for p in range(1, pages + 1):
        url = LIST_URL if p == 1 else LIST_URL.replace(".shtml", f"_{p}.shtml")
        print(f"Fetching list page {p}: {url}", file=sys.stderr)
        try:
            r = session.get(url, timeout=20)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERROR] {e}", file=sys.stderr)
            break

        if r.status_code != 200:
            print(f"  HTTP {r.status_code}, stopping", file=sys.stderr)
            break

        soup = BeautifulSoup(r.text, "html.parser")
        reclist = soup.find("div", class_="reclist")
        if not reclist:
            break

        items = []
        for dl in reclist.find_all("dl"):
            a = dl.find("a")
            p = dl.find("p")
            if a and p:
                href = a.get("href", "")
                # Title is after || separator
                a_text = a.get_text(" ", strip=True)
                if "||" in a_text:
                    title = a_text.split("||", 1)[1].strip()
                else:
                    title = a_text
                date = p.get_text(strip=True).strip("「」").strip()
                if href and title:
                    items.append({
                        "title": title,
                        "url": urljoin(BASE_URL, href),
                        "date": date,
                    })

        print(f"  Found {len(items)} items", file=sys.stderr)
        if not items:
            break
        all_items.extend(items)
        time.sleep(0.5)

    print(f"Total items: {len(all_items)}", file=sys.stderr)

    results = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}", file=sys.stderr)
        detail = extract_detail(item["url"])
        if detail:
            c = detail["content"].strip().rstrip(". ")
            t = detail["title"].strip().rstrip(". ")
            if c == t:
                print(f"  [SKIP] No real content", file=sys.stderr)
                continue
            results.append({
                "title": detail["title"],
                "page_url": item["url"],
                "date": detail["date"] or item["date"],
                "site_name": "沐川县公告公示",
                "content": detail["content"],
                "summary": detail["content"][:200] if detail["content"] else "",
            })
        if (i + 1) % 10 == 0:
            time.sleep(0.5)

    print(json.dumps(results, ensure_ascii=False))


if __name__ == "__main__":
    main()
