#!/usr/bin/env python3
"""
Crawler for 原平市人民政府 - 通知公告
CMS: TRS (拓尔思), UTF-8
Site: www.yuanping.gov.cn/zwyw/tzgg/
"""
import requests, re, json, sys, os
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.yuanping.gov.cn"
LIST_URL = BASE_URL + "/zwyw/tzgg/"
SITE_NAME = "原平市-通知公告"
INCREMENTAL_DAYS = 7
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 20
session = requests.Session()
session.headers.update(HEADERS)


def fetch(url):
    resp = session.get(url, timeout=TIMEOUT)
    resp.encoding = "utf-8"
    return resp.text


def parse_list(html):
    """Parse list — div.jyfw > ul.jyfw2 > li > a + span(date)"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for ul in soup.find_all("ul", class_="jyfw2"):
        for li in ul.find_all("li", recursive=False):
            a = li.find("a", href=True)
            if not a:
                continue
            href = a["href"].strip()
            title = a.get_text(strip=True)
            if not title or not href:
                continue
            date_span = li.find("span")
            date_str = date_span.get_text(strip=True) if date_span else ""
            if href.startswith("http"):
                pass
            elif href.startswith("./"):
                href = urljoin(LIST_URL, href[2:])
            elif href.startswith("/"):
                href = BASE_URL + href
            else:
                href = urljoin(LIST_URL, href)
            items.append((title.strip(), href, date_str[:10].replace(".", "-")))
    return items


def get_page_urls():
    urls = [LIST_URL]
    for i in range(2, MAX_PAGES + 1):
        if i - 1 <= 87:
            urls.append(LIST_URL + f"index_{i-1}.html")
    return urls


ATTACH_RE = re.compile(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|txt)$", re.I)


def _render_node_inline(node, url):
    """Render a node to inline HTML, preserving <a href> links and converting
    <img> to <a href=full_src>查看图片</a>. Returns (html_str, attachments).
    """
    if isinstance(node, str):
        txt = node.strip()
        if txt and txt != "&nbsp;":
            return txt, []
        return "", []
    if not node.name:
        return "", []
    tag = node.name.lower()
    if tag == "a":
        href = node.get("href", "").strip()
        text = node.get_text(" ", strip=True)
        if not text:
            return "", []
        if href:
            full = urljoin(url, href)
            if ATTACH_RE.search(full):
                return f'<a href="{full}">{text}</a>', [{"title": text, "url": full}]
            return f'<a href="{full}">{text}</a>', []
        return text, []
    if tag == "img":
        src = node.get("src", "")
        if not src:
            return "", []
        alt = node.get("alt", "") or "查看图片"
        full_src = urljoin(url, src)
        return f'<a href="{full_src}">{alt}</a>', []
    if tag in ("br", "style", "script", "iframe", "object", "embed", "param"):
        return "", []
    # Recurse children for other inline tags
    out = []
    atts = []
    for ch in node.children:
        h, a = _render_node_inline(ch, url)
        if h:
            out.append(h)
        atts.extend(a)
    return "".join(out), atts


def _render_trs_children(node, url, processed=None):
    """Recursively render content container children into paragraph parts.
    Produces <p>...</p> HTML segments (text + inline <a href> links),
    images as <p><a href="full_src">查看图片</a></p>.
    """
    if processed is None:
        processed = set()
    parts = []
    for child in node.children:
        if isinstance(child, str):
            txt = child.strip()
            if txt and txt != "&nbsp;":
                parts.append(f"<p>{txt}</p>")
            continue
        if not child.name:
            continue
        if id(child) in processed:
            continue
        processed.add(id(child))
        tag = child.name.lower()
        if tag == "p":
            inner = []
            for ch in child.children:
                if isinstance(ch, str):
                    t = ch.strip()
                    if t and t != "&nbsp;":
                        inner.append(t)
                elif ch.name:
                    h, _ = _render_node_inline(ch, url)
                    if h:
                        inner.append(h)
            p_text = "".join(inner).strip()
            if p_text:
                parts.append(f"<p>{p_text}</p>")
        elif tag == "img":
            src = child.get("src", "")
            if src:
                alt = child.get("alt", "") or "查看图片"
                full_src = urljoin(url, src)
                parts.append(f'<p><a href="{full_src}">{alt}</a></p>')
        elif tag in ("div", "section"):
            parts.extend(_render_trs_children(child, url, processed))
        elif tag in ("br", "style", "script", "iframe", "object", "embed", "param"):
            pass
        elif tag == "table":
            parts.append(str(child))
        else:
            h, _ = _render_node_inline(child, url)
            if h:
                parts.append(f"<p>{h}</p>")
    return parts


def _clean_trs_spaces(text):
    """Clean excessive spaces from TRS inline font/span fragments.
    Use [^\\S\\n] (whitespace excluding newline) to avoid crossing paragraph boundaries.
    """
    # Between Chinese chars
    text = re.sub(r"([\u4e00-\u9fff])[^\S\n]+([\u4e00-\u9fff])", r"\1\2", text)
    # Chinese char before Chinese opening punctuation
    text = re.sub(r"([\u4e00-\u9fff])[^\S\n]+([《（【「『\[({])", r"\1\2", text)
    # Before Chinese closing punctuation
    text = re.sub(r"[^\S\n]+([，。、；：？！）】」』)\"\]])", r"\1", text)
    # After Chinese opening punctuation
    text = re.sub(r"([（【「『\[({《])[^\S\n]+", r"\1", text)
    # Between number and Chinese
    text = re.sub(r"(\d)[^\S\n]+([\u4e00-\u9fff])", r"\1\2", text)
    text = re.sub(r"([\u4e00-\u9fff])[^\S\n]+(\d)", r"\1\2", text)
    # Number-space-period: "2 ." -> "2."
    text = re.sub(r"(\d)[^\S\n]+\.", r"\1.", text)
    # Space within numbers
    text = re.sub(r"(\d)[^\S\n]+(\d)", r"\1\2", text)
    # Space in time "8: 00" -> "8:00"
    text = re.sub(r"(:\s*)(\d)", r":\2", text)
    # Multiple spaces -> single
    text = re.sub(r" {2,}", " ", text)
    # Leading/trailing spaces on lines
    text = re.sub(r" +\n", "\n", text)
    text = re.sub(r"\n +", "\n", text)
    return text.strip()


def _find_content_div(soup):
    """Find the main content container with fallback chain."""
    for cls in ("TRS_Editor", "text", "content", "article", "zwxl", "detail_con", "wzcon"):
        div = soup.find("div", class_=cls)
        if div:
            return div
    return None


def parse_detail(html, url):
    """Parse detail — h1 title + date span + content container (fallback chain)."""
    soup = BeautifulSoup(html, "html.parser")

    # Title from h1 in div.jz (after breadcrumb)
    title = ""
    jz_divs = soup.find_all("div", class_="jz")
    for jz in jz_divs:
        h1 = jz.find("h1")
        if h1:
            title = h1.get_text(strip=True)
            break
    if not title:
        t = soup.find("title")
        if t:
            title = t.get_text(strip=True)

    # Date: only trust dates found in the head area (BEFORE the content
    # container). Scanning the whole page wrongly picks dates inside the
    # article body (e.g. 2026年8月31日起 or a deadline 2026年10月19日).
    # When there is no content container at all (JS-only pages like weixin),
    # return empty so the caller falls back to the list-page date.
    pub_date = ""
    content_div = _find_content_div(soup)
    if content_div is not None:
        # Only trust dates in the head area (BEFORE the content container).
        # Scanning the whole page wrongly picks dates inside the article body
        # (e.g. 2026年8月31日起 or a deadline 2026年10月19日).
        # Prefer a 发布时间： label; support both dash and CJK formats.
        prev_spans = content_div.find_all_previous("span")
        for span in prev_spans:
            txt = span.get_text(strip=True)
            m = re.search(r"发布时间[:：]\s*(\d{4})[-年/](\d{1,2})[-月/](\d{1,2})日?", txt)
            if m:
                pub_date = f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
                break
        if not pub_date:
            for span in prev_spans:
                txt = span.get_text(strip=True)
                m = re.search(r"(\d{4})[-年/](\d{1,2})[-月/](\d{1,2})日?", txt)
                if m:
                    pub_date = f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
                    break

    # Content from fallback container chain — recursive
    content_html = ""
    content_div = _find_content_div(soup)
    if content_div:
        parts = _render_trs_children(content_div, url)
        content_html = "\n".join(parts)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html)
        content_html = _clean_trs_spaces(content_html)

    # Attachments — search content container first, fall back to whole page
    # (some TRS pages put attachment <p> outside TRS_Editor), filtering out
    # nav/footer and PDF-player tool links.
    def _collect_attachments(scope):
        out = []
        seen = set()
        for a in scope.find_all("a", href=True):
            ahref = a["href"].strip()
            if not ATTACH_RE.search(ahref):
                continue
            text = a.get_text(strip=True)
            if not text:
                continue
            if "pdf软件" in text or "下载软件" in text:
                continue
            if a.find_parent(["nav", "footer"]):
                continue
            if any(a.find_parent(class_=c) for c in ("footer", "nav", "pagefooter", "foot", "top", "header")):
                continue
            full_url = urljoin(url, ahref)
            if full_url not in seen:
                seen.add(full_url)
                out.append({"title": text, "url": full_url})
        return out

    attachments = _collect_attachments(content_div) if content_div else []
    if not attachments:
        attachments = _collect_attachments(soup)

    # PDF-embedded pages (TRS pdfurl script): extract the pdf link as attachment
    if not attachments and not content_html.strip():
        m = re.search(r"pdfurl\s*=\s*['\"]([^'\"]+\.pdf)['\"]", html, re.I)
        if m:
            full_url = urljoin(url, m.group(1))
            attachments.append({"title": full_url.split("/")[-1], "url": full_url})

    # Build final content: real body + attachment <p><a href> segments.
    # If an attachment URL is already embedded in the body (e.g. "附件一：<a>"),
    # do not duplicate it at the end — only append attachments missing from body.
    att_segments = ""
    for att in attachments:
        if att["url"] not in content_html:
            att_segments += f'<p><a href="{att["url"]}">{att["title"]}</a></p>\n'
    if content_html.strip():
        body = content_html.strip()
    else:
        body = f'<p><a href="{url}">{title}</a></p>\n' if title else ""
    if att_segments:
        content_html = body + "\n\n" + att_segments if body else att_segments
    else:
        content_html = body

    return title, pub_date, content_html, attachments


def incremental_filter(items):
    cutoff = datetime.now() - timedelta(days=INCREMENTAL_DAYS)
    filtered = []
    for title, url, date_str in items:
        if date_str:
            try:
                item_date = datetime.strptime(date_str, "%Y-%m-%d")
                if item_date >= cutoff:
                    filtered.append((title, url, date_str))
            except ValueError:
                filtered.append((title, url, date_str))
        else:
            filtered.append((title, url, date_str))
    return filtered


def main():
    is_incremental = any(arg in sys.argv for arg in ["--incremental", "incremental", "inc"])
    page_urls = get_page_urls()

    all_items = []
    for idx, url in enumerate(page_urls):
        try:
            html = fetch(url)
            items = parse_list(html)
            if not items:
                break
            print(f"Page {idx+1}: {len(items)} items", file=sys.stderr)
            all_items.extend(items)
        except Exception as e:
            print(f"Page {idx+1} error ({url}): {e}", file=sys.stderr)

    print(f"Total: {len(all_items)}", file=sys.stderr)
    if is_incremental:
        all_items = incremental_filter(all_items)
        print(f"Incremental: {len(all_items)}", file=sys.stderr)

    seen = set()
    unique_items = []
    for item in all_items:
        if item[1] not in seen:
            seen.add(item[1])
            unique_items.append(item)

    print(f"Unique: {len(unique_items)}", file=sys.stderr)

    results = []
    for title, url, list_date in unique_items:
        try:
            html = fetch(url)
            det_title, det_date, content, attachments = parse_detail(html, url)
            final_title = det_title or title
            final_date = det_date or list_date
            results.append({
                "title": final_title,
                "page_url": url,
                "publish_date": final_date,
                "content": content,
                "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
                "site_name": SITE_NAME,
            })
            print(f"  OK: {final_title[:50]}", file=sys.stderr)
        except Exception as e:
            print(f"  ERR {url}: {e}", file=sys.stderr)

    for r in results:
        print(json.dumps(r, ensure_ascii=False))


if __name__ == "__main__":
    main()
