#!/usr/bin/env python3
"""
济南圣泉集团股份有限公司 - 信息公开（标题内嵌URL）
CMS: Portal SaaS (DCloud/FastMake)
列表: /info_public.html (1页, 9条PDF/DOCX附件)
正文: 标题内嵌文件下载URL [title](file_url)
"""

import re
import sys
import json
import time
import subprocess
import sqlite3
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin, unquote

BASE_URL = 'https://www.shengquan.com'
LIST_URL = BASE_URL + '/info_public.html'
DB_PATH = '/root/search.db'
SITE_NAME = 'shengquan.com-信息公开'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
    'Referer': BASE_URL + '/info_public.html',
}


def get_publish_date(file_url):
    """
    列表页无发布日期字段, 发布日期在文件 URL 中: .../pg2025072113072960956/cms/file/xxx.pdf
    提取 pg 后 8 位日期 (20250721 → 2025-07-21)。
    失败回退当天日期。
    """
    m = re.search(r'pg(\d{8})', file_url)
    if m:
        raw = m.group(1)  # YYYYMMDD
        return f'{raw[:4]}-{raw[4:6]}-{raw[6:8]}'
    return time.strftime('%Y-%m-%d')


def extract_items_from_html(html):
    """
    用正则提取列表项（避免BeautifulSoup截断URL中的中文字符）。
    返回 [{title, url, file_type}]
    """
    items = []
    for m in re.finditer(r'<li[^>]*data-href="([^"]+)"[^>]*>', html):
        data_href = m.group(1)
        li_end = html.find('</li>', m.end())
        li_content = html[m.end():li_end] if li_end > 0 else ''

        # 标题
        title = ''
        t_m = re.search(r'<p class="dd2">(.*?)</p>', li_content)
        if t_m:
            title = t_m.group(1).strip()
            title = title.replace('&amp;', '&').replace('&lt;', '<').replace('&gt;', '>')

        # 文件类型
        file_type = ''
        f_m = re.search(r'<p class="dd1">(.*?)</p>', li_content)
        if f_m:
            file_type = f_m.group(1).strip()

        if not title or not data_href:
            continue

        # 提取实际文件URL（查看器链接需解析file参数）
        if '/viewer.html' in data_href and 'file=' in data_href:
            fm = re.search(r'file=([^&"]+)', data_href)
            if fm:
                file_url = unquote(fm.group(1))
            else:
                file_url = data_href
        else:
            file_url = data_href

        if not file_url.startswith('http'):
            file_url = urljoin(BASE_URL, file_url)

        items.append({
            'title': title,
            'url': file_url,
            'file_type': file_type,
        })

    return items


def main():
    import argparse
    parser = argparse.ArgumentParser(description='济南圣泉集团-信息公开爬虫')
    parser.add_argument('--max-pages', type=int, default=1, help='最大爬取页数')
    args = parser.parse_args()

    session = requests.Session()
    session.headers.update(HEADERS)

    print(f'爬取 {SITE_NAME}...')

    r = session.get(LIST_URL, timeout=30)
    r.encoding = 'utf-8'
    items = extract_items_from_html(r.text)

    if not items:
        print('没有获取到任何条目')
        return

    print(f'共获取{len(items)}个文件')

    conn = sqlite3.connect(DB_PATH, timeout=60)
    inserted = 0
    updated = 0
    errors = 0
    fts_sqls = []

    for idx, item in enumerate(items):
        try:
            title = item['title']
            file_url = item['url']
            file_type = item['file_type']
            print(f'  [{idx+1}/{len(items)}] [{file_type}] {title[:50]}...')

            # 正文 = 纯PDF外链站: 标题+URL内嵌段（禁markdown链接）
            content = f'<p><a href="{file_url}">{title}</a></p>'
            summary = content[:300] if len(content) > 300 else content
            publish_date = get_publish_date(file_url)
            if publish_date:
                print(f'    发布日期: {publish_date} (Last-Modified)')
            source_url = file_url.split('/')[-1] if '/' in file_url else file_url

            # DB操作
            cur = conn.execute(
                'SELECT id FROM gov_raw WHERE page_url=? AND site_name=?',
                (file_url, SITE_NAME)
            )
            row = cur.fetchone()

            if row:
                conn.execute('''
                    UPDATE gov_raw SET title=?, content=?, summary=?, publish_date=?,
                    source_url=? WHERE id=?
                ''', (title, content, summary, publish_date, source_url, row[0]))
                updated += 1
                fts_sqls.append(
                    "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) "
                    "VALUES ({},'{}','{}','{}');".format(
                        row[0],
                        title.replace("'", "''"),
                        SITE_NAME.replace("'", "''"),
                        summary.replace("'", "''")
                    )
                )
            else:
                cur2 = conn.execute('''
                    INSERT INTO gov_raw (title, content, summary, site_name,
                    page_url, publish_date, source_url)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                ''', (title, content, summary, SITE_NAME,
                      file_url, publish_date, source_url))
                inserted += 1
                rid = cur2.lastrowid
                fts_sqls.append(
                    "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) "
                    "VALUES ({},'{}','{}','{}');".format(
                        rid,
                        title.replace("'", "''"),
                        SITE_NAME.replace("'", "''"),
                        summary.replace("'", "''")
                    )
                )

        except Exception as e:
            print(f'  ERROR: {e}')
            errors += 1

    conn.commit()
    conn.close()

    # 批量写FTS
    if fts_sqls:
        print(f'  同步FTS索引 {len(fts_sqls)} 条...')
        batch_sql = 'BEGIN;\n' + '\n'.join(fts_sqls) + '\nCOMMIT;'
        try:
            r = subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH, batch_sql],
                             capture_output=True, text=True, timeout=60)
            if r.stderr:
                print(f'  FTS batch error: {r.stderr.strip()}')
            else:
                print(f'  FTS同步完成')
        except Exception as e:
            print(f'  FTS batch failed: {e}')

    print(f'\n完成！新增{inserted}条，更新{updated}条，错误{errors}条')


if __name__ == '__main__':
    main()
