#!/usr/bin/env python3
"""
yncn.gov.cn (昌宁县人民政府) — 通知公告 爬虫
CMS: VSB (Visual Site Builder, 西安博达)
列表: /cxyw/tzgg.htm (第1页), /cxyw/tzgg/{page}.htm (第2页起)
详情: /info/{columnID}/{articleID}.htm
"""

import requests, re, sys, os, time
from bs4 import BeautifulSoup

BASE = "https://www.yncn.gov.cn"
LIST_URL = BASE + "/cxyw/tzgg.htm"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}

CRAWLER_DIR = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, CRAWLER_DIR)
from crawler_lib import push_to_searchdb

MAX_PAGES = int(sys.argv[sys.argv.index("--pages") + 1]) if "--pages" in sys.argv else 5

SESS = requests.Session()
SESS.headers.update(HEADERS)


def fetch(url):
    try:
        r = SESS.get(url, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("[ERROR] fetch %s: %s" % (url, e))
        return ""


def parse_list(html, base_url=LIST_URL):
    """Parse list items from a page"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    tbl = soup.find("div", class_="tablebody")
    if not tbl:
        return items

    for tr in tbl.find_all("div", class_="tr"):
        tds = tr.find_all("div", class_="td")
        if len(tds) < 2:
            continue
        a = tds[0].find("a")
        if not a:
            continue

        href = a.get("href", "")
        if "info/" not in href:
            continue

        # Build absolute URL
        if href.startswith("../"):
            href = BASE + href[2:]
        elif not href.startswith("http"):
            href = BASE + "/" + href.lstrip("/")

        # Title from title attribute (full) or text
        title = a.get("title", "") or a.get_text(strip=True)

        # Date from second td
        date_text = tds[1].get_text(strip=True) if len(tds) > 1 else ""
        # Convert "2026年07月21日" to "2026-07-21"
        pub_date = ""
        m = re.search(r'(\d{4})年(\d{1,2})月(\d{1,2})日', date_text)
        if m:
            pub_date = "%s-%s-%s" % (m.group(1), m.group(2).zfill(2), m.group(3).zfill(2))

        items.append({"title": title, "url": href, "pub_date": pub_date})

    return items


def fetch_detail(url):
    """Fetch article detail and extract content"""
    html = fetch(url)
    if not html or len(html) < 2000:
        return None

    soup = BeautifulSoup(html, "html.parser")

    # Title from h1
    title = ""
    h1 = soup.select_one("div.article div.title h1")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"]

    # Date from meta
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"][:10]  # "2026-07-21 16:17" -> "2026-07-21"

    # Content from vsb_content div
    content = ""
    for cid in ["vsb_content_4", "vsb_content"]:
        content_div = soup.find("div", id=cid)
        if content_div:
            break

    if content_div:
        parts = []
        # Use v_news_content if present
        vnc = content_div.find("div", class_="v_news_content")
        target = vnc if vnc else content_div

        for elem in target.find_all(["p", "table"], recursive=True):
            if elem.name == "p":
                if elem.find_parent("table"):
                    continue
                text = elem.get_text(strip=True)
                if text:
                    parts.append(text)
            elif elem.name == "table":
                parts.append(str(elem))

        # Also extract attachments
        for a in content_div.find_all("a", href=re.compile(r"virtual_attach_file|attach")):
            href = a.get("href", "")
            fname = a.get_text(strip=True)
            if fname:
                if href.startswith("../"):
                    href = BASE + href[2:]
                elif not href.startswith("http"):
                    href = BASE + "/" + href.lstrip("/")
                parts.append("[附件: %s](%s)" % (fname, href))

        content = "\n\n".join(parts) if parts else ""

    if not content:
        content = "[无正文内容]"

    return {"title": title, "date": pub_date, "content": content}


def main():
    # Page 1
    html_p1 = fetch(LIST_URL)
    if not html_p1 or len(html_p1) < 5000:
        print("[ERROR] Cannot fetch page 1")
        sys.exit(1)

    all_items = parse_list(html_p1)
    print("[INFO] Page 1: %d items" % len(all_items))

    # Find total pages from pagebar
    soup = BeautifulSoup(html_p1, "html.parser")
    total_pages = 1
    pbar = soup.find("div", class_="pagebar")
    if pbar:
        # Links are like tzgg/1.htm
        page_links = pbar.find_all("a", href=re.compile(r"tzgg/\d+"))
        for link in page_links:
            m = re.search(r'tzgg/(\d+)', link.get("href", ""))
            if m:
                pg = int(m.group(1)) + 1  # 0-indexed: tzgg/1.htm = page 2
                if pg > total_pages:
                    total_pages = pg

    pages_to_fetch = min(total_pages, MAX_PAGES)
    print("[INFO] Total pages: %d" % total_pages)

    for page in range(2, pages_to_fetch + 1):
        url = BASE + "/cxyw/tzgg/%d.htm" % (page - 1)
        html = fetch(url)
        if html and len(html) > 5000:
            items = parse_list(html)
            if items:
                all_items.extend(items)
                print("[INFO] Page %d: %d items" % (page, len(items)))
        time.sleep(0.5)

    print("[INFO] Total items: %d" % len(all_items))

    batch = []
    for idx, item in enumerate(all_items):
        title = item["title"]
        url = item["url"]
        pub_date = item["pub_date"]

        detail = fetch_detail(url)
        if detail is None:
            continue

        # Use detail's full title if available
        if detail.get("title"):
            title = detail["title"]
        # Use detail date as fallback
        if not pub_date and detail.get("date"):
            pub_date = detail["date"]

        batch.append({
            "url": url,
            "title": title,
            "content": detail["content"],
            "pub_date": pub_date,
            "site_name": "昌宁县人民政府-通知公告",
            "source_url": url,
        })

        print("  [%d] %s... %s" % (idx + 1, title[:45], pub_date))
        time.sleep(0.3)

    if batch:
        result = push_to_searchdb(batch)
        print("\n[DONE] 完成: %d 条" % len(batch))
    else:
        print("[DONE] 无数据")


if __name__ == "__main__":
    main()
