#!/usr/bin/env python3
import os
# -*- coding: utf-8 -*-
"""
青铜峡市人民政府 - 社会保障 爬虫 (fix: 表格渲染 + 去重)
http://www.qtx.gov.cn/xxgk/zfxxgkml/shbz/
CMS: TRS (Hanweb) - createPageHTML分页
"""
import requests, re, time, json, sys, os
from bs4 import BeautifulSoup, NavigableString, Tag
from datetime import datetime

SITE_NAME = "青铜峡市人民政府-社会保障"
BASE_URL = "http://www.qtx.gov.cn/xxgk/zfxxgkml/shbz/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def fetch(url, retries=3):
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=20)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if i < retries-1:
                time.sleep(2)
    return None

def get_page_urls():
    """获取列表页所有URL"""
    html = fetch(BASE_URL)
    if not html:
        print("ERROR: 无法获取首页")
        return []
    
    m = re.search(r'createPageHTML\((\d+),\s*\d+,\s*"index",\s*"html"\)', html)
    total_pages = int(m.group(1)) if m else 1
    print(f"总页数: {total_pages}")
    
    page_urls = [BASE_URL]
    for i in range(1, total_pages):
        page_urls.append(BASE_URL + f"index_{i}.html")
    
    return page_urls

def parse_list(html):
    """解析列表页"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='pageList')
    if not ul:
        return items
    
    for li in ul.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get_text(strip=True)
        span = li.find('span', class_='time')
        date = span.get_text(strip=True) if span else ''
        if href and title:
            items.append((href, title, date))
    
    print(f"  列表页解析: {len(items)} 条")
    return items

def render_table_to_markdown(table):
    """将HTML表格渲染为原生HTML <table>（避免markdown库管道表兼容性问题）"""
    rows = []
    for tr in table.find_all('tr'):
        cells = []
        for td in tr.find_all(['td', 'th']):
            colspan = int(td.get('colspan', 1))
            cell_text = td.get_text(strip=True)
            tag = 'th' if td.name == 'th' else 'td'
            cells.append(f'<{tag}>{cell_text}</{tag}>')
            if colspan > 1:
                cells.extend([f'<{tag}></{tag}>'] * (colspan - 1))
        if any(c for c in cells):
            rows.append('<tr>' + ''.join(cells) + '</tr>')
    
    if rows:
        return '<table>' + ''.join(rows) + '</table>'
    return ''

def extract_content_text(element):
    """从内容容器中提取正文，跳过表格内部的p（避免重复）"""
    blocks = []
    
    for child in element.children:
        if isinstance(child, NavigableString):
            text = child.strip()
            if text:
                blocks.append(text)
            continue
        
        if not isinstance(child, Tag):
            continue
        
        # 表格 → 渲染为Markdown管道表
        if child.name == 'table':
            md = render_table_to_markdown(child)
            if md:
                blocks.append(md)
            continue
        
        # 图片
        if child.name == 'img':
            src = child.get('src', '')
            if src:
                alt = child.get('alt', '')
                full_src = src if src.startswith('http') else ('http://www.qtx.gov.cn' + src if src.startswith('/') else None)
                if full_src:
                    blocks.append(f'![{alt}]({full_src})')
            continue
        
        # p → 提取文本（跳过p内部的表格）
        if child.name == 'p':
            # 跳过嵌套p（TRS常见）
            if child.find('p'):
                continue
            
            # 检查p内是否有table
            inner_table = child.find('table')
            if inner_table:
                md = render_table_to_markdown(inner_table)
                if md:
                    blocks.append(md)
                continue
            
            # 检查是否有附件链接
            text_parts = []
            for c in child.children:
                if isinstance(c, NavigableString):
                    t = str(c).strip()
                    if t:
                        text_parts.append(t)
                elif isinstance(c, Tag):
                    if c.name == 'a':
                        href = c.get('href', '')
                        txt = c.get_text(strip=True)
                        if href and txt:
                            text_parts.append(f'[{txt}]({href})')
                        elif txt:
                            text_parts.append(txt)
                    elif c.name == 'br':
                        text_parts.append('\n')
                    elif c.name in ['span', 'strong', 'em', 'b', 'u', 'i']:
                        t = c.get_text(strip=True)
                        if t:
                            text_parts.append(t)
                    else:
                        t = c.get_text(strip=True)
                        if t:
                            text_parts.append(t)
            
            text = ''.join(text_parts).strip()
            if text:
                blocks.append(text)
            continue
        
        # div → 递归处理
        if child.name == 'div':
            sub = extract_content_text(child)
            if sub:
                blocks.extend(sub)
            continue
    
    return blocks

def parse_detail(href, list_title, list_date):
    """解析详情页"""
    if href.startswith('http'):
        url = href
    elif href.startswith('../'):
        url = BASE_URL.rsplit('/', 2)[0] + '/' + href.lstrip('../')
    elif href.startswith('/'):
        url = 'http://www.qtx.gov.cn' + href
    else:
        url = BASE_URL.rstrip('/') + '/' + href
    
    html = fetch(url)
    if not html:
        print(f"  ERROR: 无法获取详情页 {url}")
        return {
            'title': list_title,
            'date': list_date,
            'content': '',
            'attachments': []
        }
    
    soup = BeautifulSoup(html, 'html.parser')
    
    # 标题
    title = list_title
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    
    # 日期
    date = list_date
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date and meta_date.get('content'):
        date = meta_date['content'].strip()
    
    # 正文
    content = ''
    attachments = []
    
    content_div = soup.find('div', class_='xl_con_con')
    if content_div:
        blocks = extract_content_text(content_div)
        content = '\n\n'.join(blocks)
    else:
        # 备用
        for p in soup.find_all('p'):
            text = p.get_text(strip=True)
            if text:
                content += text + '\n\n'
        content = content.strip()
    
    # 全文搜索附件
    for a_tag in soup.find_all('a', href=re.compile(r'\.(doc|docx|pdf|xls|xlsx|rar|zip)$', re.I)):
        a_href = a_tag.get('href', '')
        a_text = a_tag.get_text(strip=True) or '附件'
        full_href = a_href if a_href.startswith('http') else ('http://www.qtx.gov.cn' + a_href if a_href.startswith('/') else url.rsplit('/', 1)[0] + '/' + a_href)
        # 去重
        if not any(a['url'] == full_href for a in attachments):
            attachments.append({'name': a_text, 'url': full_href})
    
    # summary
    clean_content = re.sub(r'\s+', ' ', content).strip()
    summary = clean_content[:200]
    
    # date_rank
    date_parsed = date[:10] if date else ''
    date_rank = 0
    if date_parsed:
        try:
            dt = datetime.strptime(date_parsed, '%Y-%m-%d')
            date_rank = int(dt.timestamp())
        except:
            pass
    
    print(f"  OK: {title[:30]}... | {date_parsed} | content={len(content)}c")
    return {
        'title': title,
        'date': date_parsed,
        'content': content,
        'summary': summary,
        'attachments': json.dumps(attachments, ensure_ascii=False),
        'date_rank': date_rank,
        'source_url': url
    }

def main():
    print(f"=== {SITE_NAME} 爬虫 (fix版) ===")
    
    page_urls = get_page_urls()
    print(f"共 {len(page_urls)} 页待爬")
    
    all_items = []
    for page_url in page_urls:
        html = fetch(page_url)
        if not html:
            continue
        items = parse_list(html)
        time.sleep(1)
        all_items.extend(items)
    
    print(f"列表共 {len(all_items)} 条")
    
    # DB
    if os.path.exists('/root/search.db'):
        db_path = '/root/search.db'
    else:
        db_path = '/mnt/data/search.db'
    
    import sqlite3
    conn = sqlite3.connect(db_path, timeout=60)
    c = conn.cursor()
    
    # 清理旧数据（注意：先清gov_search再清gov_raw）
    c.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
    c.execute("DELETE FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    conn.commit()
    print(f"已清理 {SITE_NAME} 旧数据")
    
    count = 0
    for href, list_title, list_date in all_items:
        detail = parse_detail(href, list_title, list_date)
        
        c.execute("""
            INSERT OR REPLACE INTO gov_raw (page_url, title, publish_date, content, summary, site_name, attachments, date_rank, source_url, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'crawl_qtxgov_shbz.py')
        """, (
            detail.get('source_url', ''),
            detail['title'],
            detail.get('date', ''),
            detail.get('content', ''),
            detail.get('summary', ''),
            SITE_NAME,
            detail.get('attachments', '[]'),
            detail.get('date_rank', 0),
            detail.get('source_url', '')
        ))
        count += 1
        time.sleep(1.5)
    
    conn.commit()
    print(f"入库 {count} 条")
    
    # 验证
    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    raw_count = c.fetchone()[0]
    c.execute("SELECT COUNT(*) FROM gov_search WHERE site_name=?", (SITE_NAME,))
    search_count = c.fetchone()[0]
    
    conn.close()
    print(f"验证: gov_raw={raw_count}, gov_search={search_count}")
    
    print(f"\n配置条目:")
    print(json.dumps({
        "name": SITE_NAME,
        "script": f"crawl_qtxgov_shbz.py",
        "url": BASE_URL,
        "group": "宁夏"
    }, ensure_ascii=False, indent=2))

if __name__ == '__main__':
    main()
