#!/usr/bin/env python3
"""
panzhihua.gov.cn - 攀枝花市人民政府 / 公告公示
站点: www.panzhihua.gov.cn/zwgk/gzdt/gggs/
CMS: 自建，div#Zoom 正文，div.boxPzh15_con 列表
分组: 四川
"""

import requests
import re
import json
import sys
import os
from bs4 import BeautifulSoup
from urllib.parse import urljoin
import urllib.parse

BASE_URL = "http://www.panzhihua.gov.cn/zwgk/gzdt/gggs/"
SITE_NAME = "攀枝花市人民政府"
GROUP = "四川"
MAX_PAGES = 3  # 共3页

DB_PATH = "/root/search.db"

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(headers)


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def fetch_page(url, retries=3):
    for attempt in range(retries):
        try:
            r = session.get(url, timeout=15)
            r.encoding = "utf-8"
            return r
        except Exception as e:
            if attempt < retries - 1:
                import time
                time.sleep(2)
            else:
                raise e
    return None


def parse_list_items(html, list_url):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    bpc = soup.find("div", class_="boxPzh15_con")
    if not bpc:
        return items

    for ul in bpc.find_all("ul", class_="List_list"):
        for li in ul.find_all("li", recursive=False):
            a = li.find("a")
            span = li.find("span", class_="grey999")
            if not a or not a.get("href"):
                continue

            href = a["href"]
            title = a.get("title", "") or a.get_text(strip=True)
            date_str = span.get_text(strip=True) if span else ""

            detail_url = urljoin(list_url, href)
            items.append((title, detail_url, date_str))
    return items


def parse_detail(detail_url):
    r = fetch_page(detail_url)
    if not r:
        return None, "[]", "", ""

    soup = BeautifulSoup(r.text, "html.parser")

    # --- Title ---
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        meta_t = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta_t:
            title = meta_t.get("content", "")

    # --- Date ---
    date_str = ""
    meta_d = soup.find("meta", attrs={"name": "PubDate"})
    if meta_d:
        raw = meta_d.get("content", "")
        m = re.search(r"(\d{4}[-/]\d{1,2}[-/]\d{1,2})", raw)
        if m:
            date_str = m.group(1).replace("/", "-")
    if not date_str:
        explain = soup.find("p", class_="explain")
        if explain:
            txt = explain.get_text()
            m = re.search(r"发布时间[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})", txt)
            if m:
                date_str = m.group(1).replace("/", "-")

    # --- Body ---
    zoom = soup.find("div", id="Zoom")
    body_parts = []
    attachments = []

    if zoom:
        for child in zoom.children:
            if child.name == "p":
                txt = child.get_text(strip=True)
                if txt:
                    body_parts.append(txt)

        # Tables: keep as HTML
        for table in zoom.find_all("table"):
            body_parts.append(html_table_to_html(table, detail_url))

        # Attachments inside Zoom
        for atag in zoom.find_all("a", href=True):
            href = atag["href"]
            if any(ext in href.lower() for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar"]):
                fname = atag.get_text(strip=True) or href.split("/")[-1]
                full_url = urljoin(detail_url, href)
                attachments.append({"name": fname, "url": full_url})

    # Attachments outside Zoom
    for atag in soup.find_all("a", href=True):
        href = atag["href"]
        if any(ext in href.lower() for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar"]):
            fname = atag.get_text(strip=True) or href.split("/")[-1]
            full_url = urljoin(detail_url, href)
            if not any(a["url"] == full_url for a in attachments):
                attachments.append({"name": fname, "url": full_url})

    body = "\n\n".join(body_parts)
    return body, json.dumps(attachments, ensure_ascii=False), date_str, title


def crawl(max_pages=MAX_PAGES):
    all_items = []

    for page_idx in range(max_pages):
        if page_idx == 0:
            url = BASE_URL + "index.shtml"
        else:
            url = f"{BASE_URL}index_{page_idx + 1}.shtml"

        print(f"  [{page_idx + 1}/{max_pages}] {url}", flush=True)
        r = fetch_page(url)
        if not r:
            print(f"    FAILED to fetch", flush=True)
            continue

        items = parse_list_items(r.text, url)
        if not items:
            print(f"    No items found", flush=True)
            continue
        print(f"    Found {len(items)} items", flush=True)

        for title, detail_url, list_date in items:
            body, attachments_json, detail_date, detail_title = parse_detail(detail_url)
            use_title = detail_title or title
            use_date = detail_date or list_date

            body_for_check = body or ""
            has_content = len(body_for_check) > 0
            seg_count = len(body_for_check.split("\n\n")) if body_for_check else 0

            all_items.append({
                "title": use_title,
                "page_url": detail_url,
                "date": use_date,
                "body": body,
                "attachments": attachments_json,
                "site_name": SITE_NAME,
                "group": GROUP,
                "_has_content": has_content,
                "_seg_count": seg_count,
            })

            print(f"    {use_title[:40]:40s} | {use_date:12s} | seg={seg_count} | attach={'YES' if attachments_json != '[]' else 'no'}", flush=True)

    return all_items


def import_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    c.execute("DELETE FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    c.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))

    count = 0
    for item in items:
        c.execute("""
            INSERT OR REPLACE INTO gov_raw (title, page_url, site_name, publish_date, content, attachments, date_rank, group_name, script_name) VALUES (?, ?, ?, ?, ?, ?, 0, ?, 'crawl_pzh_gggs.py')
        """, (
            item["title"],
            item["page_url"],
            item["site_name"],
            item["date"],
            item["body"],
            item["attachments"],
            item["group"],
        ))
        count += 1

    conn.commit()
    conn.close()
    return count


def main():
    import time
    start = time.time()

    max_p = MAX_PAGES
    if len(sys.argv) > 1 and sys.argv[1] == "--pages":
        max_p = int(sys.argv[2])

    print(f"=== panzhihua.gov.cn 攀枝花市 公告公示 ===", flush=True)
    print(f"Pages: {max_p}", flush=True)

    items = crawl(max_pages=max_p)

    if not items:
        print("No items collected!", flush=True)
        return

    total = len(items)
    with_body = sum(1 for i in items if i["_has_content"])
    avg_seg = sum(i["_seg_count"] for i in items) / total if total else 0
    with_attach = sum(1 for i in items if i["attachments"] != "[]")

    print(f"\n=== Results ===", flush=True)
    print(f"Total: {total}", flush=True)
    print(f"With body: {with_body} ({with_body / total * 100:.1f}%)", flush=True)
    print(f"Avg segments: {avg_seg:.1f}", flush=True)
    print(f"With attachments: {with_attach}", flush=True)

    imported = import_to_db(items)
    print(f"Imported: {imported}", flush=True)

    elapsed = time.time() - start
    print(f"Elapsed: {elapsed:.1f}s", flush=True)


if __name__ == "__main__":
    main()
