#!/usr/bin/env python3
"""
日照市行政审批服务局 — 通知公告栏目爬虫
URL: http://xzspj.rizhao.gov.cn/col/col116809/index.html
CMS: jPage 分页系统 (政务网站)
List API: POST /module/web/jpage/dataproxy.jsp?startrecord=N&endrecord=M
需设置 r.encoding = 'utf-8' 解决无charset导致乱码问题
"""

import requests
from bs4 import BeautifulSoup
import re
import sys
import os
import argparse
from datetime import datetime
from urllib.parse import urljoin

# ─── 配置 ─────────────────────────────────────────────
BASE_URL = "http://xzspj.rizhao.gov.cn"
SITE_NAME = "日照市行政审批服务局"
COLUMN_NAME = "通知公告"
GROUP = "山东省"
INDUSTRY = "政府公告"
PER_PAGE = 15
BATCH_SIZE = 46  # jPage groupSize=3, 每批约3页

# dataproxy配置
PROXY_URL = "http://xzspj.rizhao.gov.cn/module/web/jpage/dataproxy.jsp"
PARAMS_TEMPLATE = {
    "webid": "190",
    "path": "http://xzspj.rizhao.gov.cn/",
    "columnid": "116809",
    "unitid": "467097",
    "webname": "日照市行政审批服务局 日照市政务服务管理办公室",
    "col": "1",
    "sourceContentType": "1",
    "permissiontype": "0",
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": "http://xzspj.rizhao.gov.cn/col/col116809/index.html",
}

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb


def fetch_list_page(start_record: int) -> list:
    """
    获取列表页，返回 [(title, url, date_str), ...]
    start_record: 起始记录号 (1-based)
    """
    end_record = start_record + PER_PAGE - 1
    # URL解码传body，编码传query string
    try:
        data_params = {**PARAMS_TEMPLATE, "startrecord": str(start_record), "endrecord": str(end_record), "perpage": str(PER_PAGE)}
        resp = requests.post(PROXY_URL, params=data_params, data=data_params,
                             headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        xml_text = resp.text
        items = []
        # 提取所有 <record><![CDATA[...]]></record>
        records = re.findall(r'<record><!\[CDATA\[(.*?)\]\]></record>', xml_text, re.DOTALL)
        for record in records:
            a_soup = BeautifulSoup(record, "html.parser")
            a_tag = a_soup.find("a")
            if not a_tag or not a_tag.get("href"):
                continue
            url = a_tag["href"].strip()
            title = a_tag.get("title", "").strip() or a_tag.get_text(strip=True)
            date_str = ""
            t_span = a_tag.find("span", class_="t")
            if t_span:
                date_str = t_span.get_text(strip=True).strip("[]")
            if not title:
                continue
            items.append((title, url, date_str))

        page_no = (start_record - 1) // PER_PAGE + 1
        print(f"  📄 第{page_no}页 (记录{start_record}-{end_record}): {len(items)}条", flush=True)
        return items
    except Exception as e:
        print(f"  ❌ 列表页请求失败: {e}", flush=True)
        return []


def get_total_count() -> int:
    """获取总条数"""
    try:
        data_params = {**PARAMS_TEMPLATE, "startrecord": "1", "endrecord": "15", "perpage": "15"}
        resp = requests.post(PROXY_URL, params=data_params, data=data_params, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        m = re.search(r'<totalrecord>(\d+)</totalrecord>', resp.text)
        return int(m.group(1)) if m else 0
    except:
        return 0


def extract_content_with_tables(div):
    """递归提取正文，保留表格HTML"""
    content_parts = []
    def walk_element(el):
        for child in el.children:
            if child.name == "table":
                content_parts.append(str(child))
            elif child.name and child.name not in ("script", "style"):
                if child.find_all("table"):
                    walk_element(child)
                else:
                    sep = "" if child.name == "p" else " "
                    txt = child.get_text(sep, strip=True)
                    if txt:
                        content_parts.append(txt)
    walk_element(div)
    return "\n".join(content_parts) if content_parts else div.get_text(" ", strip=True)


def fetch_detail(url: str) -> tuple:
    """获取详情页，返回 (title, content, publish_date)"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        if r.status_code != 200:
            return None, "", ""
        r.encoding = "utf-8"
    except Exception as e:
        print(f"    ❌ 请求失败: {e}", flush=True)
        return None, "", ""

    soup = BeautifulSoup(r.text, "html.parser")

    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        t = soup.find("title")
        if t:
            title = t.get_text(strip=True)

    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "pubdate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    content_div = soup.find("div", id="zoom")
    if not content_div:
        content_div = soup.find("div", id=lambda x: x and "zoom" in str(x).lower())
    content = extract_content_with_tables(content_div) if content_div else ""

    return title, content, pub_date


def crawl(pages: int = 5, push: bool = True):
    """主爬虫"""
    total = get_total_count()
    total_pages = max(1, (total + PER_PAGE - 1) // PER_PAGE) if total else 127

    print(f"🔍 {SITE_NAME} - {COLUMN_NAME}", flush=True)
    print(f"📊 总计 {total} 条, {total_pages} 页", flush=True)

    all_items = []
    for start_record in range(1, total + 1, BATCH_SIZE):
        if pages > 0 and len(all_items) >= pages * PER_PAGE:
            break
        items = fetch_list_page(start_record)
        if not items:
            print(f"  ⚠️ start_record={start_record}为空，停止翻页", flush=True)
            break
        all_items.extend(items)

    print(f"\n📋 列表共 {len(all_items)} 条", flush=True)
    print(f"{'='*60}", flush=True)

    push_items = []
    fail_count = 0

    for idx, (title, url, date_str) in enumerate(all_items, 1):
        print(f"\n[{idx}/{len(all_items)}] {title[:60]}", flush=True)
        print(f"    📎 {url}", flush=True)
        print(f"    📅 {date_str}", flush=True)

        detail_title, content, pub_date = fetch_detail(url)
        if detail_title is None:
            print(f"    ❌ 详情获取失败", flush=True)
            fail_count += 1
            continue

        final_title = detail_title or title
        final_date = pub_date or date_str

        content_preview = content[:80].replace('\n', ' ')
        print(f"    ✅ 正文 {len(content)} 字: {content_preview}...", flush=True)

        push_items.append({
            "title": final_title,
            "url": url,
            "source_url": url,
            "pub_date": final_date,
            "content": content,
            "summary": content[:500] if content else final_title,
            "site_name": SITE_NAME,
            "group_name": GROUP,
            "industry": INDUSTRY,
            "category": COLUMN_NAME,
        })

    if push and push_items:
        print(f"\n{'='*60}", flush=True)
        print(f"📤 正在推送 {len(push_items)} 条到 search.db...", flush=True)
        push_to_searchdb(push_items, batch_label=f"{SITE_NAME}-{COLUMN_NAME}")
    elif push:
        print(f"\n⚠️ 无数据可推送", flush=True)

    print(f"\n{'='*60}", flush=True)
    print(f"✅ 完成: 共{len(push_items)}条, 失败{fail_count}", flush=True)


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=f"{SITE_NAME} - {COLUMN_NAME} 爬虫")
    parser.add_argument("--pages", type=int, default=5, help="爬取页数 (0=全部)")
    parser.add_argument("--no-push", action="store_true", help="不推送到search.db")
    args = parser.parse_args()

    crawl(pages=args.pages, push=not args.no_push)
