#!/usr/bin/env python3
"""
panzhou.gov.cn - 盘州市人民政府 / 通知公告
站点: www.panzhou.gov.cn/jjpz/tzgg/
CMS: TRS（清华同方思创）
分组: 贵州
"""

import requests
import re
import json
import sys
import os
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.panzhou.gov.cn/jjpz/tzgg/"
SITE_NAME = "盘州市人民政府"
GROUP = "贵州"
MAX_PAGES = 5

DB_PATH = "/root/search.db"

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(headers)


def fetch_page(url, retries=3):
    for attempt in range(retries):
        try:
            r = session.get(url, timeout=15)
            r.encoding = "utf-8"
            return r
        except Exception as e:
            if attempt < retries - 1:
                import time
                time.sleep(2)
            else:
                raise e
    return None


def parse_list_items(html, list_url):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    pmb = soup.find("div", class_="PageMainBox")
    if not pmb:
        return items

    ul = pmb.find("ul", class_="NewsList")
    if not ul:
        return items

    for li in ul.find_all("li", recursive=False):
        a = li.find("a")
        span = li.find("span")
        if not a or not a.get("href"):
            continue

        href = a["href"]
        title = a.get("title", "") or a.get_text(strip=True)
        date_str = span.get_text(strip=True) if span else ""

        detail_url = urljoin(list_url, href)
        items.append((title, detail_url, date_str))
    return items


def parse_detail(detail_url):
    r = fetch_page(detail_url)
    if not r:
        return None, "[]", "", ""

    soup = BeautifulSoup(r.text, "html.parser")

    # --- Title ---
    title = ""
    meta_t = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_t:
        title = meta_t.get("content", "")

    # --- Date ---
    date_str = ""
    meta_d = soup.find("meta", attrs={"name": "PubDate"})
    if meta_d:
        raw = meta_d.get("content", "")
        m = re.search(r"(\d{4}[-/]\d{1,2}[-/]\d{1,2})", raw)
        if m:
            date_str = m.group(1).replace("/", "-")

    # --- Body ---
    body_parts = []
    attachments = []

    # TRS CMS 正文容器
    te = soup.find("div", class_="trs_editor_view")
    if not te:
        te = soup.find("div", class_=lambda c: c and "TRS" in (c or "").upper())

    if te:
        for child in te.children:
            if child.name == "p":
                txt = child.get_text(strip=True)
                if txt:
                    body_parts.append(txt)

        # Attachments inside TRS body
        for atag in te.find_all("a", href=True):
            href = atag["href"]
            if any(ext in href.lower() for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar"]):
                fname = atag.get_text(strip=True) or href.split("/")[-1]
                full_url = urljoin(detail_url, href)
                attachments.append({"name": fname, "url": full_url})

    # Also check ContentPageBox for title/date
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)

    body = "\n\n".join(body_parts)
    return body, json.dumps(attachments, ensure_ascii=False), date_str, title


def crawl(max_pages=MAX_PAGES):
    all_items = []

    for page_idx in range(max_pages):
        if page_idx == 0:
            url = BASE_URL + "index.html"
        else:
            url = f"{BASE_URL}index_{page_idx}.html"

        print(f"  [{page_idx + 1}/{max_pages}] {url}", flush=True)
        r = fetch_page(url)
        if not r:
            print(f"    FAILED to fetch", flush=True)
            continue

        items = parse_list_items(r.text, url)
        if not items:
            print(f"    No items found", flush=True)
            continue
        print(f"    Found {len(items)} items", flush=True)

        for title, detail_url, list_date in items:
            body, attachments_json, detail_date, detail_title = parse_detail(detail_url)
            use_title = detail_title or title
            use_date = detail_date or list_date

            body_for_check = body or ""
            has_content = len(body_for_check) > 0
            seg_count = len(body_for_check.split("\n\n")) if body_for_check else 0

            all_items.append({
                "title": use_title,
                "page_url": detail_url,
                "date": use_date,
                "body": body,
                "attachments": attachments_json,
                "site_name": SITE_NAME,
                "group": GROUP,
                "_has_content": has_content,
                "_seg_count": seg_count,
            })

            print(f"    {use_title[:40]:40s} | {use_date:12s} | seg={seg_count} | attach={'YES' if attachments_json != '[]' else 'no'}", flush=True)

    return all_items


def import_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()

# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)     c.execute("DELETE FROM gov_raw WHERE site_name=?", (SITE_NAME,))
# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)     c.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))

    count = 0
    for item in items:
        c.execute("""
            INSERT OR REPLACE INTO gov_raw (title, page_url, site_name, publish_date, content, attachments, date_rank, group_name, script_name) VALUES (?, ?, ?, ?, ?, ?, 0, ?, 'crawl_pz_tzgg.py')
        """, (
            item["title"],
            item["page_url"],
            item["site_name"],
            item["date"],
            item["body"],
            item["attachments"],
            item["group"],
        ))
        count += 1

    conn.commit()
    conn.close()
    return count


def main():
    import time
    start = time.time()

    max_p = MAX_PAGES
    if len(sys.argv) > 1 and sys.argv[1] == "--pages":
        max_p = int(sys.argv[2])

    print(f"=== panzhou.gov.cn 盘州市 通知公告 ===", flush=True)
    print(f"Pages: {max_p}", flush=True)

    items = crawl(max_pages=max_p)

    if not items:
        print("No items collected!", flush=True)
        return

    total = len(items)
    with_body = sum(1 for i in items if i["_has_content"])
    avg_seg = sum(i["_seg_count"] for i in items) / total if total else 0
    with_attach = sum(1 for i in items if i["attachments"] != "[]")

    print(f"\n=== Results ===", flush=True)
    print(f"Total: {total}", flush=True)
    print(f"With body: {with_body} ({with_body / total * 100:.1f}%)", flush=True)
    print(f"Avg segments: {avg_seg:.1f}", flush=True)
    print(f"With attachments: {with_attach}", flush=True)

    imported = import_to_db(items)
    print(f"Imported: {imported}", flush=True)

    elapsed = time.time() - start
    print(f"Elapsed: {elapsed:.1f}s", flush=True)


if __name__ == "__main__":
    main()
