#!/usr/bin/env python3
"""
六枝特区人民政府 - 公示公告 爬虫
CMS: 贵州统一平台 (static pagination)
URL: https://www.liuzhi.gov.cn/newsite/zwdt_5753277/gsgg_5753281/
"""
import sys, os, re, time, json
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import requests
from bs4 import BeautifulSoup
from crawler_lib import push_to_searchdb

SITE_NAME = "六枝特区-公示公告"
BASE_URL = "https://www.liuzhi.gov.cn/newsite/zwdt_5753277/gsgg_5753281/"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")


def crawl_list_page(page_num):
    """Crawl a single list page"""
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"{BASE_URL}index_{page_num}.html"

    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")
    except Exception as e:
        print(f"[List] Error page {page_num}: {e}")
        return []

    ul = soup.find("ul", class_="wtp-xwlb")
    if not ul:
        return []

    items = []
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a or not a.get("href"):
            continue

        href = a["href"].strip()
        txt = li.get_text(strip=True)

        # Date is last 10 chars "YYYY-MM-DD"
        date_str = txt[-10:] if re.match(r'\d{4}-\d{2}-\d{2}', txt[-10:]) else ""
        title = txt[:-10].strip() if date_str else txt

        items.append({
            "title": title,
            "url": href,
            "date_str": date_str,
        })

    return items


def crawl_detail(item):
    """Fetch detail page"""
    url = item["url"]
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")
    except Exception as e:
        print(f"[Detail] Error: {e}")
        return None

    # Title
    title = item["title"]
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()

    # Date
    date_str = item["date_str"]
    pd = soup.find("meta", attrs={"name": "PubDate"})
    if pd and pd.get("content"):
        date_str = pd["content"].strip()[:10]

    # Content - use div.container div.detail or trs_editor_view
    body_html = ""
    for sel in ["div.container div.detail", "div.detail", "div.container",
                "div.trs_editor_view", "div.lzzf-mainbox"]:
        div = soup.select_one(sel)
        if div:
            body_html = str(div)
            break

    # Summary
    summary_text = ""
    if body_html:
        ts = BeautifulSoup(body_html, "html.parser")
        summary_text = ts.get_text(separator=" ", strip=True)[:300]

    return {
        "site_name": SITE_NAME,
        "title": title,
        "url": url,
        "pub_date": date_str,
        "source_url": url,
        "content": body_html,
        "summary": summary_text,
    }


def main():
    pages = 30
    for a in sys.argv[1:]:
        if a.startswith('--args='):
            try:
                pages = int(a.split('=', 1)[1])
            except:
                pass
    print(f"=== {SITE_NAME} ===")
    print(f"Pages: {pages}, 3年截止: {THREE_YEARS_AGO}")

    # Step 1: List pages
    print("\n--- Step 1: List pages ---")
    all_items = []
    for p in range(1, pages + 1):
        items = crawl_list_page(p)
        if not items:
            print(f"[List] Page {p}: empty — end")
            break
        items = [it for it in items if it["date_str"] >= THREE_YEARS_AGO]
        all_items.extend(items)
        dates = [it["date_str"] for it in items if it["date_str"]]
        dr = f"{dates[0]}..{dates[-1]}" if dates else "?"
        print(f"[List] Page {p}: {len(items)} items ({dr}) total: {len(all_items)}")
        time.sleep(0.3)

    print(f"\nTotal: {len(all_items)} items")
    if not all_items:
        print("No items")
        return

    # Step 2: Details
    print(f"\n--- Step 2: {len(all_items)} details ---")
    records = []
    errors = 0

    with ThreadPoolExecutor(max_workers=3) as ex:
        fut_map = {ex.submit(crawl_detail, it): it for it in all_items}
        for i, fut in enumerate(as_completed(fut_map)):
            rec = fut.result()
            if rec:
                records.append(rec)
            else:
                errors += 1
            if (i + 1) % 50 == 0:
                print(f"[Detail] {i+1}/{len(all_items)} done")

    print(f"\nDetail: {len(records)} success, {errors} errors")

    # Step 3: Save via push_to_searchdb (handles FTS sync)
    print("\n--- Step 3: Save ---")
    push_to_searchdb(records, batch_label="liuzhi_gsgg")

    print("\n=== Done ===")


if __name__ == "__main__":
    main()
