#!/usr/bin/env python3
"""
苏尼特右旗人民政府 - 通知公告爬虫
CMS: 内容管理系统(eportal/dynamic JS)
列表: /sntyq/gk/tzgg/index.html (第1页) / 83ef7296-N.html (第N页, 10条/页, 共158页)
详情: /sntyq/gk/tzgg/{timestamp}/index.html
标题: a[title] 属性(列表) / meta ArticleTitle(详情)
正文: div#docContent (p/table/img)
"""

import re
import sys
import json
import time
import subprocess
import sqlite3
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta

BASE_URL = 'https://www.sntyq.gov.cn'
LIST_URL_P1 = BASE_URL + '/sntyq/gk/tzgg/index.html'
LIST_URL_PN = BASE_URL + '/sntyq/gk/tzgg/83ef7296-{}.html'
DB_PATH = '/root/search.db'
SITE_NAME = 'sntyq.gov.cn-通知公告'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
    'Referer': BASE_URL + '/sntyq/gk/tzgg/',
}

MAX_PAGES_DEFAULT = 5  # 默认前5页


def get_soup(url, session=None):
    s = session or requests.Session()
    try:
        resp = s.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            print(f'  WARNING: HTTP {resp.status_code} for {url}')
            return None
        return BeautifulSoup(resp.text, 'html.parser')
    except Exception as e:
        print(f'  ERROR: fetch failed: {url} - {e}')
        return None


def extract_list_page(soup):
    """从列表页提取文章链接、标题和日期"""
    items = []
    # 匹配所有带 istitle="true" 的链接（文章链接）
    for a in soup.find_all('a', attrs={'istitle': 'true'}):
        href = a.get('href', '')
        title = (a.get('title') or '').strip()
        if not title:
            title = a.get_text(strip=True)
        if not title or not href:
            continue

        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)

        # 从URL路径中提取日期（时间戳前8位 YYYYMMDD）
        date_str = ''
        m = re.search(r'/(\d{8})\d+', href)
        if m:
            ts = m.group(1)
            date_str = f'{ts[:4]}-{ts[4:6]}-{ts[6:8]}'

        items.append({'title': title, 'url': href, 'date': date_str})

    return items


def extract_detail(soup, url):
    """提取详情页内容"""
    # 标题：从meta
    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()

    # 回退：从面包屑或页面标题文本
    if not title:
        for el in soup.find_all(['span', 'div', 'h1', 'h2']):
            if el.get('class') and any(c in ' '.join(el.get('class', [])).lower() for c in ['tit', 'title', 'bt', 'con-tt']):
                t = el.get_text(strip=True)
                if t and len(t) > 5:
                    title = t
                    break

    if not title:
        print(f'  WARNING: no title for {url}')
        return None

    # 发布日期
    publish_date = ''
    meta_pub = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_pub and meta_pub.get('content'):
        publish_date = meta_pub['content'].strip()[:10]

    # 回退：从页面文本
    if not publish_date:
        page_text = soup.get_text()
        m = re.search(r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})', page_text)
        if m:
            publish_date = m.group(1)

    # 来源
    source_url = ''
    meta_src = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_src and meta_src.get('content'):
        source_url = meta_src['content'].strip()

    if not source_url:
        page_text = soup.get_text()
        m = re.search(r'来源[：:]\s*([^\n\r]+)', page_text)
        if m:
            source_url = m.group(1).strip()

    # 正文
    content_div = soup.find('div', id='docContent')
    if not content_div:
        content_div = soup.find('div', class_='docContent')
    if not content_div:
        content_div = soup.find('div', class_='con-nr')

    content_parts = []
    attachments = []
    images = []

    if content_div:
        # 遍历所有直接子元素，兼容<p>和<div>两种渲染方式
        for child in content_div.find_all(recursive=False):
            if child.name in ('p', 'div', 'section'):
                txt = child.get_text(strip=True)
                if txt and len(txt) > 2:
                    content_parts.append(txt)
            elif child.name == 'table':
                html_str = str(child)
                content_parts.append(html_str)
            elif child.name == 'img':
                src = child.get('src', '')
                alt = child.get('alt', '')
                if src:
                    if not src.startswith('http'):
                        src = urljoin(BASE_URL, src)
                    images.append({'src': src, 'alt': alt})
                    content_parts.append(f'![{alt}]({src})')
            elif child.name == 'a':
                href = child.get('href', '')
                if href and re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                    if not href.startswith('http'):
                        href = urljoin(BASE_URL, href)
                    a_title = child.get_text(strip=True) or child.get('title', '') or '附件'
                    attachments.append({'name': a_title, 'url': href})
                    content_parts.append(f'[{a_title}]({href})')

        # 如果recursive=False没取到（某些情况子元素嵌套更深），
        # 回退到递归查找p/div
        if not content_parts:
            for child in content_div.find_all(['p', 'div', 'table', 'img', 'a']):
                if child.name in ('p', 'div'):
                    txt = child.get_text(strip=True)
                    if txt and len(txt) > 2:
                        # 跳过表格内div的文本（避免重复）
                        parent_table = child.find_parent('table')
                        if not parent_table:
                            content_parts.append(txt)
                elif child.name == 'table':
                    content_parts.append(str(child))
                elif child.name == 'img':
                    src = child.get('src', '')
                    alt = child.get('alt', '')
                    if src:
                        if not src.startswith('http'):
                            src = urljoin(BASE_URL, src)
                        content_parts.append(f'![{alt}]({src})')
                elif child.name == 'a':
                    href = child.get('href', '')
                    if href and re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                        if not href.startswith('http'):
                            href = urljoin(BASE_URL, href)
                        a_title = child.get_text(strip=True) or child.get('title', '') or '附件'
                        attachments.append({'name': a_title, 'url': href})
                        content_parts.append(f'[{a_title}]({href})')

    content = '\n\n'.join(content_parts)

    # PDF-only 降级
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>'
        for img in images:
            content += f'\n![{img["alt"]}]({img["src"]})'
        for att in attachments:
            content += f'\n[{att["name"]}]({att["url"]})'

    summary = content[:300] if len(content) > 300 else content

    return {
        'title': title,
        'content': content,
        'summary': summary,
        'publish_date': publish_date,
        'source_url': source_url,
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
    }


def main():
    import argparse
    parser = argparse.ArgumentParser(description='苏尼特右旗人民政府-通知公告爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES_DEFAULT, help='最大爬取页数')
    args = parser.parse_args()

    max_pages = args.max_pages
    session = requests.Session()
    all_items = []
    seen_urls = set()
    empty_pages = 0

    print(f'爬取 {SITE_NAME}，前{max_pages}页...')

    for page in range(1, max_pages + 1):
        if page == 1:
            url = LIST_URL_P1
        else:
            url = LIST_URL_PN.format(page)

        print(f'\n列表页 {page}/{max_pages}: {url}')
        soup = get_soup(url, session)
        if not soup:
            print(f'  FAILED: 无法获取页面')
            empty_pages += 1
            if empty_pages >= 3:
                print('  连续3页失败，停止')
                break
            continue

        items = extract_list_page(soup)
        if not items:
            print(f'  空列表，停止分页')
            break

        # 去重
        new_count = 0
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
                new_count += 1

        empty_pages = 0
        print(f'  本页{len(items)}条，新增{new_count}条，累计{len(all_items)}条')

        if new_count == 0:
            print('  无新内容，停止分页')
            break

        time.sleep(0.5)

    print(f'\n共获取{len(all_items)}篇文章，开始抓取详情...')

    conn = sqlite3.connect(DB_PATH, timeout=60)
    inserted = 0
    updated = 0
    errors = 0
    fts_sqls = []  # 收集FTS语句，commit后批量执行

    for idx, item in enumerate(all_items):
        try:
            print(f'  [{idx+1}/{len(all_items)}] {item["title"][:40]}...')
            soup = get_soup(item['url'], session)
            if not soup:
                errors += 1
                continue

            detail = extract_detail(soup, item['url'])
            if not detail:
                errors += 1
                continue

            if not detail['title']:
                detail['title'] = item['title']

            page_url = item['url']
            publish_date = detail['publish_date'] or item['date']
            content = detail['content']
            summary = detail['summary']
            title = detail['title']
            source_url = detail['source_url']
            attachments = detail['attachments']

            # 插入/更新 + FTS同步
            cur = conn.execute(
                'SELECT id FROM gov_raw WHERE page_url=? AND site_name=?',
                (page_url, SITE_NAME)
            )
            row = cur.fetchone()

            if row:
                conn.execute('''
                    UPDATE gov_raw SET title=?, content=?, summary=?, publish_date=?,
                    source_url=?, attachments=? WHERE id=?
                ''', (title, content, summary, publish_date, source_url, attachments, row[0]))
                updated += 1
                fts_sqls.append("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES ({},'{}','{}','{}');".format(
                    row[0],
                    title.replace("'", "''"),
                    SITE_NAME.replace("'", "''"),
                    summary.replace("'", "''")))
            else:
                cur2 = conn.execute('''
                    INSERT INTO gov_raw (title, content, summary, site_name,
                    page_url, publish_date, source_url, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                ''', (title, content, summary, SITE_NAME,
                      page_url, publish_date, source_url, attachments))
                inserted += 1
                rid = cur2.lastrowid
                fts_sqls.append("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES ({},'{}','{}','{}');".format(
                    rid,
                    title.replace("'", "''"),
                    SITE_NAME.replace("'", "''"),
                    summary.replace("'", "''")))

            time.sleep(0.3)

        except Exception as e:
            print(f'  ERROR: {e}')
            errors += 1

    conn.commit()
    conn.close()

    # commit后批量写FTS（避免WAL锁定问题）
    if fts_sqls:
        print(f'  同步FTS索引 {len(fts_sqls)} 条...')
        batch_sql = 'BEGIN;\n' + '\n'.join(fts_sqls) + '\nCOMMIT;'
        try:
            r = subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH, batch_sql],
                             capture_output=True, text=True, timeout=60)
            if r.stderr:
                print(f'  FTS batch error: {r.stderr.strip()}')
            else:
                print(f'  FTS同步完成')
        except Exception as e:
            print(f'  FTS batch failed: {e}')

    print(f'\n完成！新增{inserted}条，更新{updated}条，错误{errors}条')


if __name__ == '__main__':
    main()
