#!/usr/bin/env python3
"""
第二师铁门关市 - 意见征集爬虫
CMS: 自定义JSP政府CMS (tmg.gov.cn)
列表: /info/iList.jsp?cat_id=10039&cur_page=N (6条/页, 共153条26页)
详情: /gzhd/yjzj/{id}.htm
标题: div.con-tt (完整标题无省略号)
正文: div.con-nr (p/table/img)
"""

import re
import sys
import json
import time
import subprocess
import sqlite3
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime

BASE_URL = 'https://www.tmg.gov.cn'
LIST_URL = BASE_URL + '/info/iList.jsp?cat_id=10039&cur_page={}'
DB_PATH = '/root/search.db'
SITE_NAME = 'tmg.gov.cn-意见征集'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
    'Referer': BASE_URL + '/info/iList.jsp?cat_id=10039',
}

MAX_PAGES_DEFAULT = 5  # 默认前5页


def get_soup(url, session=None):
    s = session or requests.Session()
    try:
        resp = s.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            print(f'  WARNING: HTTP {resp.status_code} for {url}')
            return None
        return BeautifulSoup(resp.text, 'html.parser')
    except Exception as e:
        print(f'  ERROR: fetch failed: {url} - {e}')
        return None


def extract_list_page(soup):
    """从列表页提取文章链接、标题和日期"""
    items = []
    ul = soup.find('ul', class_='news-ul')
    if not ul:
        print('  WARNING: ul.news-ul not found')
        return items

    for li in ul.find_all('li', recursive=False):
        h2 = li.find('h2', class_='news-cont-tt')
        if not h2:
            continue
        a = h2.find('a')
        if not a or not a.get('href'):
            continue

        # 标题：优先用 title 属性（完整无省略号），回退到文本
        title = (a.get('title') or '').strip()
        if not title:
            title = a.get_text(strip=True)
        if not title:
            continue

        href = a['href'].strip()
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)

        # 日期：span.day + span.year
        time_div = li.find('div', class_='news-ul-time')
        day = ''
        year = ''
        if time_div:
            day_el = time_div.find('span', class_='day')
            year_el = time_div.find('span', class_='year')
            if day_el:
                day = day_el.get_text(strip=True)
            if year_el:
                year = year_el.get_text(strip=True)

        # 组合日期 YYYY-MM-DD
        date_str = ''
        if year and day:
            # year is like "2026-03", day is like "25"
            date_str = f'{year}-{day.zfill(2)}'

        items.append({'title': title, 'url': href, 'date': date_str})

    return items


def extract_detail(soup, url):
    """提取详情页内容"""
    # 标题
    title = ''
    title_el = soup.find('div', class_='con-tt')
    if title_el:
        title = title_el.get_text(strip=True)

    if not title:
        print(f'  WARNING: no title for {url}')
        return None

    # 发布日期
    publish_date = ''
    time_el = soup.find('div', class_='con-time')
    if time_el:
        time_text = time_el.get_text()
        m = re.search(r'发布时间：(\d{4}-\d{2}-\d{2})', time_text)
        if m:
            publish_date = m.group(1)

    # 来源
    source_url = ''
    if time_el:
        time_text = time_el.get_text()
        m = re.search(r'来源：(.+?)(?:浏览量|编辑|\n|$)', time_text)
        if m:
            source_url = m.group(1).strip()

    # 正文
    content_div = soup.find('div', class_='con-nr')
    content_parts = []
    attachments = []
    images = []

    if content_div:
        for child in content_div.find_all(['p', 'table', 'img', 'a']):
            if child.name == 'p':
                txt = child.get_text(strip=True)
                if txt and len(txt) > 2:
                    content_parts.append(txt)
            elif child.name == 'table':
                html_str = str(child)
                content_parts.append(html_str)
            elif child.name == 'img':
                src = child.get('src', '')
                alt = child.get('alt', '')
                if src:
                    if not src.startswith('http'):
                        src = urljoin(BASE_URL, src)
                    images.append({'src': src, 'alt': alt})
                    content_parts.append(f'![{alt}]({src})')
            elif child.name == 'a':
                # 附件链接检测
                href = child.get('href', '')
                if href and re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                    if not href.startswith('http'):
                        href = urljoin(BASE_URL, href)
                    a_title = child.get_text(strip=True) or child.get('title', '') or '附件'
                    attachments.append({'name': a_title, 'url': href})
                    content_parts.append(f'[{a_title}]({href})')

    content = '\n\n'.join(content_parts)

    # PDF-only 降级
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>'
        for img in images:
            content += f'\n![{img["alt"]}]({img["src"]})'
        for att in attachments:
            content += f'\n[{att["name"]}]({att["url"]})'

    summary = content[:300] if len(content) > 300 else content

    return {
        'title': title,
        'content': content,
        'summary': summary,
        'publish_date': publish_date,
        'source_url': source_url,
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
    }


def main():
    import argparse
    parser = argparse.ArgumentParser(description='第二师铁门关市-意见征集爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES_DEFAULT, help='最大爬取页数')
    args = parser.parse_args()

    max_pages = args.max_pages
    session = requests.Session()
    all_items = []
    seen_urls = set()
    empty_pages = 0

    print(f'爬取 {SITE_NAME}，前{max_pages}页...')

    for page in range(1, max_pages + 1):
        url = LIST_URL.format(page)
        print(f'\n列表页 {page}/{max_pages}: {url}')
        soup = get_soup(url, session)
        if not soup:
            print(f'  FAILED: 无法获取页面')
            empty_pages += 1
            if empty_pages >= 3:
                print('  连续3页失败，停止')
                break
            continue

        items = extract_list_page(soup)
        if not items:
            print(f'  空列表，停止分页')
            break

        # 去重
        new_count = 0
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
                new_count += 1

        empty_pages = 0
        print(f'  本页{len(items)}条，新增{new_count}条，累计{len(all_items)}条')

        if new_count == 0:
            print('  无新内容，停止分页')
            break

        time.sleep(0.5)

    print(f'\n共获取{len(all_items)}篇文章，开始抓取详情...')

    conn = sqlite3.connect(DB_PATH, timeout=60)
    inserted = 0
    updated = 0
    errors = 0
    fts_sqls = []  # 收集FTS语句，commit后批量执行

    for idx, item in enumerate(all_items):
        try:
            print(f'  [{idx+1}/{len(all_items)}] {item["title"][:40]}...')
            soup = get_soup(item['url'], session)
            if not soup:
                errors += 1
                continue

            detail = extract_detail(soup, item['url'])
            if not detail:
                errors += 1
                continue

            if not detail['title']:
                detail['title'] = item['title']

            page_url = item['url']
            publish_date = detail['publish_date'] or item['date']
            content = detail['content']
            summary = detail['summary']
            title = detail['title']
            source_url = detail['source_url']
            attachments = detail['attachments']

            # 插入/更新
            cur = conn.execute(
                'SELECT id FROM gov_raw WHERE page_url=? AND site_name=?',
                (page_url, SITE_NAME)
            )
            row = cur.fetchone()

            if row:
                conn.execute('''
                    UPDATE gov_raw SET title=?, content=?, summary=?, publish_date=?,
                    source_url=?, attachments=? WHERE id=?
                ''', (title, content, summary, publish_date, source_url, attachments, row[0]))
                updated += 1
                fts_sqls.append("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES ({},'{}','{}','{}');".format(
                    row[0],
                    title.replace("'", "''"),
                    SITE_NAME.replace("'", "''"),
                    summary.replace("'", "''")))
            else:
                cur2 = conn.execute('''
                    INSERT INTO gov_raw (title, content, summary, site_name,
                    page_url, publish_date, source_url, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                ''', (title, content, summary, SITE_NAME,
                      page_url, publish_date, source_url, attachments))
                inserted += 1
                rid = cur2.lastrowid
                fts_sqls.append("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES ({},'{}','{}','{}');".format(
                    rid,
                    title.replace("'", "''"),
                    SITE_NAME.replace("'", "''"),
                    summary.replace("'", "''")))

            time.sleep(0.3)

        except Exception as e:
            print(f'  ERROR: {e}')
            errors += 1

    conn.commit()
    conn.close()

    # commit后批量写FTS（避免WAL锁定问题）
    if fts_sqls:
        print(f'  同步FTS索引 {len(fts_sqls)} 条...')
        batch_sql = 'BEGIN;\n' + '\n'.join(fts_sqls) + '\nCOMMIT;'
        try:
            r = subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH, batch_sql],
                             capture_output=True, text=True, timeout=60)
            if r.stderr:
                print(f'  FTS batch error: {r.stderr.strip()}')
            else:
                print(f'  FTS同步完成')
        except Exception as e:
            print(f'  FTS batch failed: {e}')

    print(f'\n完成！新增{inserted}条，更新{updated}条，错误{errors}条')


if __name__ == '__main__':
    main()
