#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
烟台市人民政府 - 行政许可公示爬虫
CMS: HanWeb/JPAAS (烟台市政府门户网站)
列表: API /api-gateway/jpaas-publish-server/front/page/build/unit (JSON返回HTML)
分页: paramJson={pageNo:N,pageSize:15}, 前5页
详情: div.text#zoom 正文内容 + table附件
标题: a[title] 取完整标题（避免省略号）
"""

import re, sys, json, time, requests, subprocess
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.yantai.gov.cn"
LIST_API = "https://www.yantai.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
DB_PATH = "/mnt/data/search.db"
SITE_NAME = "烟台市-行政许可公示"
CATEGORY = "hjxx"
GROUP = "山东烟台"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "application/json, text/plain, */*",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://www.yantai.gov.cn/col/col43294/index.html?vc_xxgkarea=113706000042603877-1&number=C130107&jh=263",
}
MAX_PAGES = 5
QUERY_DATA = {
    "webId": "1",
    "pageId": "43294",
    "parseType": "bulidstatic",
    "pageType": "column",
    "tagId": "当前栏目列表",
    "tplSetId": "uo2n7cvvGR8iPKdzxdBuu",
}


def parse_date(date_str):
    """从列表页的日期文本中提取 YYYY-MM-DD"""
    cleaned = date_str.replace("\n", "").replace(" ", "").strip()
    match = re.search(r"(\d{4})\s*-\s*(\d{1,2})\s*-\s*(\d{1,2})", cleaned)
    if match:
        return f"{match.group(1)}-{match.group(2).zfill(2)}-{match.group(3).zfill(2)}"
    return ""


def get_list_page(pageno=1):
    """调用列表API获取一页数据"""
    params = dict(QUERY_DATA)
    params["paramJson"] = json.dumps({"pageNo": pageno, "pageSize": 15})
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, timeout=30)
        if resp.status_code != 200:
            print(f"列表页{pageno} HTTP {resp.status_code}")
            return []
        data = resp.json()
        if not data.get("success"):
            print(f"列表页{pageno} API失败: {data.get('message','')}")
            return []
        html = data["data"]["html"]
        soup = BeautifulSoup(html, "html.parser")
        items = []
        for li in soup.select("li.bt-main-r-ul-li"):
            a_tag = li.find("a")
            if not a_tag:
                continue
            title = (a_tag.get("title") or "").strip()
            if not title:
                title = a_tag.get_text(strip=True)
            title = title.replace("&middot;", "·").replace("&nbsp;", " ").strip()
            href = a_tag.get("href", "")
            url = urljoin(BASE_URL, href)
            span = li.find("span")
            date_str = span.get_text() if span else ""
            pub_date = parse_date(date_str)
            items.append({"title": title, "url": url, "date": pub_date})
        print(f"列表页{pageno}: 解析到 {len(items)} 条")
        return items
    except Exception as e:
        print(f"列表页{pageno} 异常: {e}")
        return []


def extract_content(url):
    """抓取详情页，提取正文+附件 (保留 HTML 表格/链接/图片)"""
    try:
        resp = requests.get(url, headers={
            "User-Agent": HEADERS["User-Agent"],
            "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
            "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
        }, timeout=30)
    except Exception as e:
        print(f"  请求失败 {url}: {e}")
        return "", "", ""
    if resp.status_code != 200:
        return "", "", ""
    soup = BeautifulSoup(resp.text, "html.parser")

    title = ""
    h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        mt = soup.find("meta", attrs={"name": "ArticleTitle"})
        if mt and mt.get("content"):
            title = mt["content"].strip()

    zoom = soup.find("div", class_="text", id="zoom")
    if not zoom:
        zoom = soup.find("div", id="zoom")
    if not zoom:
        zoom = soup.find("div", class_="text")

    attachments = []
    content = ""

    if zoom:
        # 1) 附件提取 + 链接绝对化
        for a in zoom.find_all("a", href=True):
            href = a["href"]
            full_url = urljoin(BASE_URL, href) if not href.startswith("http") else href
            a["href"] = full_url
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|ofd|wps)$", href, re.I) or "document/download" in href:
                a_text = a.get_text(strip=True) or href.split("/")[-1].split("?")[0]
                attachments.append({"name": a_text, "url": full_url})

        # 2) img src 绝对化
        for img in zoom.find_all("img", src=True):
            src = img["src"]
            img["src"] = urljoin(BASE_URL, src) if not src.startswith(("http", "data:")) else src

        # 3) 移除尾部无关区块
        for bad in zoom.find_all(attrs={"class": re.compile(r"(scan|share|qrcode|print|toolbar|bshare)", re.I)}):
            bad.decompose()
        for bad in zoom.find_all(attrs={"id": re.compile(r"(scan|share|qrcode|print|toolbar|bshare)", re.I)}):
            bad.decompose()
        for node in zoom.find_all(string=re.compile(r"打印本页|关闭窗口|扫码在手机|分享到")):
            p = node.find_parent()
            if p:
                p.decompose()

        # 4) 保留 HTML: 去掉内联 style, 保留表格/链接/图片
        content = str(zoom)
        content = re.sub(r"<script[\s\S]*?</script>", "", content)
        content = re.sub(r"<style[\s\S]*?</style>", "", content)
        content = re.sub(r'\sstyle="[^"]*"', "", content)
        content = re.sub(r"\n{3,}", "\n\n", content)
        content = content.strip()

    if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 20:
        content = f'<p><a href="{url}">{title or "原文"}</a></p>'
        if attachments:
            for a in attachments:
                content += f'<p><a href="{a["url"]}">{a["name"]}</a></p>'

    date_str = ""
    date_match = re.search(r"日期[：:]\s*(\d{4}-\d{1,2}-\d{1,2})", resp.text)
    if date_match:
        date_str = date_match.group(1)
    if not date_str:
        mt2 = re.search(r'<meta name="PubDate" content="([^"]*)"', resp.text)
        if mt2:
            date_str = mt2.group(1)[:10]

    return title, content, date_str, json.dumps(attachments, ensure_ascii=False)



def push_to_searchdb(items):
    """通过sqlite3 CLI批量入库 (避免Python sqlite3 WAL损坏)"""
    new_count = 0
    skip_count = 0

    result = subprocess.run(
        ["sqlite3", "-cmd", ".timeout 60000", DB_PATH, "SELECT page_url FROM gov_raw WHERE site_name='%s'" % SITE_NAME],
        capture_output=True, text=True, timeout=60
    )
    existing = set()
    if result.returncode == 0 and result.stdout.strip():
        existing = set(result.stdout.strip().split('\n'))

    batch = []
    for item in items:
        url = item["url"]
        if url in existing:
            skip_count += 1
            continue

        t = item["title"].replace("'", "''")
        c = (item.get("content", "") or "").replace("'", "''")
        d = item.get("date", "")
        dr = int(d.replace("-", "")) if d else 0
        a = (item.get("attachments", "") or "").replace("'", "''")
        s = (c[:200] if c else "").replace("'", "''")

        batch.append(
            "INSERT INTO gov_raw (site_name, title, page_url, publish_date, content, date_rank, category, group_name, attachments) "
            "VALUES ('%s', '%s', '%s', '%s', '%s', %d, '%s', '%s', '%s');"
            % (SITE_NAME, t, url, d, c, dr, CATEGORY, GROUP, a)
        )
        batch.append(
            "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
            "VALUES (last_insert_rowid(), '%s', '%s', '%s');"
            % (t, SITE_NAME, s)
        )
        new_count += 1
        existing.add(url)

    if batch:
        sql = "\n".join(batch)
        r = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH], input=sql, capture_output=True, text=True, timeout=120)
        if r.returncode != 0:
            print(f"  写入失败: {r.stderr[:200]}")
            # 回退仅写gov_raw
            raw_sql = "\n".join(batch[::2])
            r2 = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH], input=raw_sql, capture_output=True, text=True, timeout=120)
            if r2.returncode == 0:
                print(f"  FTS跳过，仅写入gov_raw {new_count}条")
            else:
                print(f"  gov_raw也写入失败: {r2.stderr[:200]}")
                new_count = 0

    return new_count, skip_count


def main():
    import argparse
    parser = argparse.ArgumentParser(description="烟台市-行政许可公示爬虫")
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="抓取页数")
    parser.add_argument("--dry-run", action="store_true", help="仅测试不写入DB")
    args = parser.parse_args()

    pages = min(args.pages, MAX_PAGES)
    print(f"开始爬取烟台市行政许可公示，共 {pages} 页...")

    all_items = []
    for page in range(1, pages + 1):
        items = get_list_page(page)
        if not items:
            print(f"第{page}页无数据，停止")
            break
        for item in items:
            print(f"  详情: {item['title'][:50]}...")
            title, content, date_str, attachments = extract_content(item["url"])
            if title:
                item["title"] = title
            item["content"] = content
            if date_str:
                item["date"] = date_str
            item["attachments"] = attachments
            all_items.append(item)
            time.sleep(1)

    print(f"\n共采集 {len(all_items)} 条")

    if not args.dry_run and all_items:
        new_count, skip_count = push_to_searchdb(all_items)
        print(f"入库完成: 新增 {new_count}, 跳过 {skip_count}")
    elif args.dry_run:
        print("dry-run模式，未写入数据库")
        for item in all_items[:3]:
            print(f"  [{item['date']}] {item['title']}")


if __name__ == "__main__":
    main()
