#!/usr/bin/env python3
"""
浙石化 (www.zpc-cn.com) 爬虫
SPA架构，通过API获取数据
公示公告栏目（columnId=58）
"""
import re
import requests
import sqlite3
import os
import sys

SITE_NAME = '浙石化'
SITE_DOMAIN = 'www.zpc-cn.com'
API_BASE = 'https://skr.xinlantech.cn/api'
COLUMN_IDS = {'第四/公示公告': 58}

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

def fetch_list(column_id, max_pages=10):
    """获取栏目列表"""
    items = []
    seen = set()
    
    for page in range(1, max_pages + 1):
        url = f'{API_BASE}/news/getNews?columnId={column_id}&linkerId=7&page={page}'
        resp = requests.get(url, timeout=30)
        data = resp.json()
        
        if data.get('state') != 'success':
            break
        
        news = data.get('news', [])
        if not news:
            break
        
        for n in news:
            nid = n.get('id')
            if not nid or nid in seen:
                continue
            seen.add(nid)
            
            title = n.get('title', '').strip()
            if not title:
                continue
            
            # 从列表API获取content（完整HTML）
            content = (n.get('content') or '').strip()
            
            # 如果列表API的content不完整，从详情API获取
            if not content:
                detail = fetch_detail(nid)
                content = detail.get('content', '')
            
            # 发布日期
            pub_date = n.get('addtime') or n.get('create_time', '')
            if pub_date:
                pub_date = pub_date[:10]  # YYYY-MM-DD
            
            items.append({
                'id': nid,
                'title': title,
                'content': content,
                'publish_date': pub_date,
            })
        
        # 如果返回数量小于分页数，说明没更多了
        if len(news) < 10:
            break
    
    return items

def fetch_detail(nid):
    """获取文章详情"""
    try:
        resp = requests.get(f'{API_BASE}/news/getNewsContent?id={nid}', timeout=15)
        data = resp.json()
        if data.get('state') == 'success':
            return data.get('news', {})
    except:
        pass
    return {}

def init_db():
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        site_name TEXT,
        source_url TEXT,
        page_url TEXT,
        title TEXT,
        publish_date TEXT,
        date_rank INTEGER DEFAULT 0,
        summary TEXT,
        status TEXT,
        category TEXT DEFAULT '',
        visits INTEGER DEFAULT 0,
        content TEXT DEFAULT '',
        tags TEXT DEFAULT ''
    )''')
    try:
        db.execute('''CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(
            title, content, site_name,
            content='gov_raw', content_rowid='id',
            tokenize='unicode61'
        )''')
    except sqlite3.OperationalError:
        pass
    db.commit()
    db.close()

def save_items(items):
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = db.cursor()
    new_count = 0
    total = len(items)
    
    for item in items:
        page_url = f'https://www.zpc-cn.com/#/ArticleDetail?id={item["id"]}'
        try:
            cur.execute('''INSERT OR IGNORE INTO gov_raw 
                (title, content, publish_date, page_url, source_url, site_name, status)
                VALUES (?, ?, ?, ?, ?, ?, ?)''', (
                item['title'],
                item['content'],
                item['publish_date'],
                page_url,
                SITE_DOMAIN,
                SITE_NAME,
                'published',
            ))
            if cur.rowcount > 0:
                new_count += 1
                row_id = cur.lastrowid
                try:
                    db.execute('INSERT OR REPLACE INTO gov_search(rowid, title, site_name) VALUES (?,?,?,?)',
                               (row_id, item['title'], item['content'] or '', SITE_NAME))
                except sqlite3.IntegrityError:
                    pass
        except Exception as e:
            print(f'  入库失败: {item["title"][:30]}... {e}')
    
    db.commit()
    total_rows = cur.execute('SELECT COUNT(*) FROM gov_raw WHERE site_name=?', (SITE_NAME,)).fetchone()[0]
    db.close()
    print(f'共获取 {total} 条有效记录')
    print(f'入库: 新增{new_count}, 累计{total_rows}')
    return new_count


if __name__ == '__main__':
    print(f'=== {SITE_NAME} 爬虫 ({SITE_DOMAIN}) ===')
    init_db()
    
    all_items = []
    for col_name, col_id in COLUMN_IDS.items():
        print(f'获取栏目: {col_name} (columnId={col_id})')
        items = fetch_list(col_id)
        print(f'  找到 {len(items)} 条')
        all_items.extend(items)
    
    save_items(all_items)
