#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
开化县人民政府 - 政府信息公开→公告公示 爬虫
CMS: JPAAS发布平台 (Zhongke)
列表: API /api-gateway/jpaas-publish-server/front/page/build/unit 返回HTML
      解析 table > tr > a.bt_link(标题) + td.bt_time(日期)
详情: div.title(标题), div.ly(日期), div.article#zoom(正文)
"""

import re
import sys
import time
import subprocess
from urllib.parse import urljoin

import requests
from bs4 import BeautifulSoup, Tag

# ── 配置 ──────────────────────────────────────────
BASE_URL = "https://www.kaihua.gov.cn"
LIST_PATH = "/col/col1229090858"
SITE_NAME = "开化县人民政府"
GROUP = "浙江"
DB_PATH = "/mnt/data/search.db"

# JPAAS API参数
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
API_PARAMS = {
    'parseType': 'bulidstatic',
    'webId': '2681',
    'tplSetId': '3SlHl21KjjsAamu0rkzk4',
    'pageType': 'column',
    'tagId': '当前栏目list',
    'editType': 'null',
    'pageId': '1229090858',
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": BASE_URL + LIST_PATH + "/index.html",
}

session = requests.Session()
session.headers.update(HEADERS)


def clean_text(text):
    if not text:
        return ""
    text = re.sub(r'(&middot;|&nbsp;|\s)+', ' ', text)
    return text.strip()


def clean_content_html(html_content, detail_url):
    """提取正文，表格保留HTML，附件嵌入"""
    if not html_content:
        return ""
    soup = BeautifulSoup(html_content, 'html.parser')
    for tag in soup(['script', 'style']):
        tag.decompose()

    parts = []
    for child in soup.children:
        if not child.name:
            continue

        if child.name == 'p':
            t = child.get_text(strip=True)
            if t:
                # 附件链接嵌入
                for a_tag in child.find_all('a', href=True):
                    href = a_tag['href']
                    if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                        a_text = a_tag.get_text(strip=True) or href
                        full_url = urljoin(detail_url, href) if not href.startswith('http') else href
                        a_tag.replace_with(soup.new_string(f' [{a_text}]({full_url}) '))
                parts.append(t)

        elif child.name == 'table':
            parts.append(str(child))

        elif child.name in ('div', 'section', 'article'):
            for p in child.find_all('p', recursive=True):
                if p.find_parent('table'):
                    continue
                t = p.get_text(strip=True)
                if t:
                    for a_tag in p.find_all('a', href=True):
                        href = a_tag['href']
                        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                            a_text = a_tag.get_text(strip=True) or href
                            full_url = urljoin(detail_url, href) if not href.startswith('http') else href
                            a_tag.replace_with(soup.new_string(f' [{a_text}]({full_url}) '))
                    parts.append(t)
            for tbl in child.find_all('table', recursive=True):
                parts.append(str(tbl))
            for img in child.find_all('img'):
                src = img.get('src', '')
                alt = img.get('alt', '')
                if src:
                    full_src = urljoin(detail_url, src) if not src.startswith('http') else src
                    parts.append(f'![{alt}]({full_src})')

        elif child.name == 'img':
            src = child.get('src', '')
            alt = child.get('alt', '')
            if src:
                full_src = urljoin(detail_url, src) if not src.startswith('http') else src
                parts.append(f'![{alt}]({full_src})')

    return '\n\n'.join(parts)


def fetch_list(max_pages=5):
    """通过JPAAS API获取文章列表（单页返回全部）"""
    all_items = []
    print(f"  第1页(API)...", end=' ')
    try:
        r = session.get(API_URL, params=API_PARAMS, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            print(f"HTTP {r.status_code}")
            return all_items
        data = r.json()
        if not data.get('success') or not data.get('data', {}).get('html'):
            print("API响应异常")
            return all_items
        html = data['data']['html']
    except Exception as e:
        print(f"ERR {e}")
        return all_items

    soup = BeautifulSoup(html, 'html.parser')
    rows = soup.select('table tr')
    count = 0
    for row in rows:
        a = row.find('a', class_='bt_link')
        if not a:
            continue
        href = a.get('href', '')
        if not href:
            continue
        title = clean_text(a.get_text())
        # 优先用title属性（完整标题）
        title_attr = a.get('title', '')
        if title_attr:
            title = clean_text(title_attr)
        # 日期 td.bt_time
        td_date = row.find('td', class_='bt_time')
        date = ''
        if td_date:
            date = clean_text(td_date.get_text())
            # 日期被换行分割: "2026-\n\t\t\t07-\n\t\t\t21"
            date = date.replace(' ', '').replace('\n', '').replace('\t', '').replace('\r', '')
        if not href or not title:
            continue
        full_url = urljoin(BASE_URL, href)
        all_items.append({'url': full_url, 'title': title, 'date': date})
        count += 1

    print(f"{count}条")
    # 单页全部返回，不需要更多页
    return all_items


def parse_detail(html, url):
    """解析详情页"""
    soup = BeautifulSoup(html, 'html.parser')
    title = ''
    publish_date = ''
    source = ''
    content = ''

    # 1. 标题 - div.title
    title_div = soup.find('div', class_='title')
    if title_div:
        title = clean_text(title_div.get_text())

    # 2. 日期 - 优先从 div.xxgklist_new 的 公开日期: 提取
    xxtd = soup.find('div', class_='xxgklist_new')
    if xxtd:
        txt = xxtd.get_text()
        dm = re.search(r'公开日期[：:]\s*(\d{4}[-年]\d{1,2}[-月]\d{1,2})', txt)
        if dm:
            raw = dm.group(1)
            raw = raw.replace('年', '-').replace('月', '-').replace('日', '')
            publish_date = raw

    # 备选: div.ly 时间字段
    if not publish_date:
        ly_div = soup.find('div', class_='ly')
        if ly_div:
            txt = ly_div.get_text()
            dm = re.search(r'(\d{4}[-年]\d{1,2}[-月]\d{1,2})', txt)
            if dm:
                raw = dm.group(1)
                raw = raw.replace('年', '-').replace('月', '-').replace('日', '')
                publish_date = raw

    # 3. 正文 - div.article#zoom
    article = soup.find('div', class_='article')
    if not article:
        article = soup.find('div', id='zoom')
    if article:
        content = clean_content_html(str(article), url)

    return title, publish_date, source, content


def insert_article(item, detail_title, pub_date, source, content):
    page_url = item['url'].replace("'", "''")
    title = (detail_title or item['title']).replace("'", "''")
    publish_date = pub_date or item.get('date', '')
    source_name = source or SITE_NAME
    source_name = source_name.replace("'", "''")
    content_safe = content.replace("'", "''")
    summary = title[:200]
    site_name = SITE_NAME
    has_table = 1 if "<table" in content else 0

    sql_raw = f"""INSERT OR IGNORE INTO gov_raw 
        (page_url, title, content, publish_date, source_url, site_name, summary, group_name, industry, has_table)
    VALUES 
        ('{page_url}', '{title}', '{content_safe}', '{publish_date}', '{page_url}', '{site_name}', '{summary}', '{GROUP}', '政府公告', {has_table});
"""
    combined_sql = sql_raw + """SELECT CASE WHEN changes() > 0 THEN last_insert_rowid() ELSE 0 END;
"""
    try:
        result = subprocess.run(
            ['sqlite3', "-cmd", ".timeout 60000", DB_PATH],
            input=combined_sql,
            capture_output=True, text=True, timeout=30
        )
        if result.returncode == 0:
            out = result.stdout.strip()
            try:
                rowid = int(out.strip())
            except ValueError:
                rowid = 0
            if rowid > 0:
                fts_sql = f"""INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary)
SELECT rowid, title, site_name, summary FROM gov_raw WHERE rowid = {rowid};
"""
                subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH], input=fts_sql, capture_output=True, text=True, timeout=30)
                return True
        else:
            if 'UNIQUE constraint' not in result.stderr:
                print(f"    [DB] {result.stderr[:200]}")
    except Exception as e:
        print(f"    [DB] 错误: {e}")
    return False


def main(max_pages=5):
    print(f"=== 开化县 - 公告公示 爬虫 ===\n")
    articles = fetch_list(max_pages)
    if not articles:
        print("无文章列表")
        return
    print(f"\n列表总计: {len(articles)} 条")

    known_urls = set()
    try:
        result = subprocess.run(
            ['sqlite3', "-cmd", ".timeout 60000", DB_PATH, "SELECT page_url FROM gov_raw WHERE page_url LIKE '%kaihua.gov.cn%'"],
            capture_output=True, text=True, timeout=30
        )
        if result.returncode == 0 and result.stdout.strip():
            known_urls = set(result.stdout.strip().split('\n'))
    except Exception:
        pass
    print(f"已知URL: {len(known_urls)}")

    added = 0
    skipped = 0
    for i, a in enumerate(articles):
        url = a['url']
        if url in known_urls:
            skipped += 1
            continue

        print(f"  [{i+1}/{len(articles)}] {a['title'][:50]}...", end=' ')
        try:
            r = session.get(url, timeout=30)
            r.encoding = 'utf-8'
        except Exception as e:
            print(f"ERR {e}")
            time.sleep(1)
            continue

        d_title, d_date, d_source, content = parse_detail(r.text, url)
        if not content.strip():
            print("SKIP 无正文")
            skipped += 1
            continue

        if insert_article(a, d_title, d_date, d_source, content):
            added += 1
            print(f"OK ({len(content.strip())}字)")
        else:
            skipped += 1
            print("SKIP (重复或入库失败)")
        time.sleep(0.3)

    print(f"\n结果: 新增 {added} 条, 跳过 {skipped} 条")
    print("=== 完成 ===")


if __name__ == '__main__':
    pages = 5
    if len(sys.argv) > 1:
        for arg in sys.argv[1:]:
            if arg.startswith('--pages='):
                try:
                    pages = int(arg.split('=')[1])
                except ValueError:
                    pass
    main(max_pages=pages)
