#!/usr/bin/env python3
"""crawl_ruichang_gggs.py — 瑞昌市人民政府 公告公示 (ruichang.gov.cn)"""
import re, sys, time, json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.ruichang.gov.cn"
SITE_NAME = "瑞昌市人民政府-公告公示"
LIST_PATH = "/ywzx/gggs"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
session = requests.Session()
session.headers.update(HEADERS)


def fetch_list_page(page_num):
    if page_num == 1:
        url = f"{BASE_URL}{LIST_PATH}/"
    else:
        url = f"{BASE_URL}{LIST_PATH}/index_{page_num - 1}.html"
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"  \u274c \u5217\u8868\u9875{page_num}\u5931\u8d25: {e}")
        return None


def parse_list(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.select("ul.newsli li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        title = ""
        # Title is text before <span>
        for child in a.children:
            if child.name == "span":
                break
            if isinstance(child, str):
                title += child
        title = title.strip()
        if not title:
            title = a.get_text(strip=True)
        span = a.find("span")
        pub_date = span.get_text(strip=True) if span else ""
        full_url = urljoin(f"{BASE_URL}{LIST_PATH}/", href)
        items.append({"title": title, "url": full_url, "date": pub_date})
    return items


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")

    # Title from td.title
    title = ""
    td = soup.find("td", class_="title")
    if td:
        title = td.get_text(strip=True)

    # Date
    date = ""
    m = re.search(r"发布日期[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if m:
        date = m.group(1)
    if not date:
        m = re.search(r"生成日期[：:]\s*(\d{4}-\d{2}-\d{2})", html)
        if m:
            date = m.group(1)

    # Content: div#zoom > div.trs_editor_view
    content = ""
    zoom = soup.find("div", id="zoom")
    if zoom:
        editor = zoom.find("div", class_="trs_editor_view")
        if not editor:
            editor = zoom
        content = extract_content(editor, url)
    else:
        # Fallback: try trs_editor_view directly
        editor = soup.find("div", class_="trs_editor_view")
        if editor:
            content = extract_content(editor, url)

    # Attachments
    attachments = []
    for a_tag in soup.find_all("a", href=True):
        href = a_tag["href"]
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", href, re.I) or "downfile" in href:
            fname = a_tag.get_text(strip=True) or href.split("/")[-1]
            full_url = urljoin(url, href)
            if not any(a["url"] == full_url for a in attachments):
                attachments.append({"name": fname, "url": full_url})

    return title, content, date, attachments


def extract_content(container, url):
    parts = []
    processed = set()
    for el in container.find_all(["p", "table", "img", "h1", "h2", "h3", "h4"], recursive=True):
        if id(el) in processed:
            continue
        processed.add(id(el))
        if el.name == "table":
            md = table_to_markdown(el)
            if md:
                parts.append(md)
        elif el.name == "img":
            src = el.get("src", "")
            if src:
                alt = el.get("alt", "")
                full_src = src if src.startswith("http") else urljoin(url, src)
                parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
        elif el.name == "p":
            txt = el.get_text(" ", strip=True)
            if txt:
                txt = re.sub(r"[^\S\n]+", " ", txt)
                parts.append(txt)
        else:
            txt = el.get_text(" ", strip=True)
            if txt:
                parts.append(txt)
    return "\n\n".join(parts)


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def main():
    import argparse
    parser = argparse.ArgumentParser(description="瑞昌市人民政府公告公示爬虫")
    parser.add_argument("--pages", type=int, default=5)
    args = parser.parse_args()

    sys.path.insert(0, "/root/gov_crawler")
    from crawler_lib import push_to_searchdb

    total = 0
    for page in range(1, args.pages + 1):
        print(f"[\u5217\u8868 {page}/{args.pages}]", end=" ", flush=True)
        html = fetch_list_page(page)
        if not html:
            continue
        items = parse_list(html)
        if not items:
            print("\u26a0\ufe0f \u65e0\u6570\u636e")
            break
        print(f"{len(items)} \u6761")
        for item in items:
            detail_url = item["url"]
            print(f"  \u2192 {item['title'][:40]}...", end=" ", flush=True)
            try:
                resp = session.get(detail_url, timeout=30)
                resp.encoding = "utf-8"
            except Exception as e:
                print(f"\u274c {e}")
                continue
            title, content, det_date, attachments = extract_detail(resp.text, detail_url)
            use_title = title or item["title"]
            use_date = det_date or item["date"]
            summary = content[:200].replace("\n", " ") if content else use_title

            if attachments:
                links = "\n\n**\u9644\u4ef6\uff1a**\n"
                for att in attachments:
                    links += f"- [{att['name']}]({att['url']})\n"
                content += links

            push_to_searchdb(
                items=[{
                    "title": use_title, "content": content, "summary": summary,
                    "url": detail_url, "source_url": detail_url,
                    "site_name": SITE_NAME,
                    "pub_date": use_date,
                    "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
                }]
            )
            total += 1
            print(f"\u2705 {len(content)}\u5b57" if content else "\u26a0\ufe0f \u7a7a")
            time.sleep(0.3)
        time.sleep(0.5)

    print(f"\n=== \u5b8c\u6210\uff01\u5171\u5165\u5e93 {total} \u6761 ===")


if __name__ == "__main__":
    sys.exit(main())
