#!/usr/bin/env python3
"""
石家庄新闻网 (news.sjzdaily.com.cn) 爬虫 - 公示公告栏目
JSON API + 静态HTML详情页
"""
import re
import json
import requests
import sqlite3
import os
import sys
import time

SITE_NAME = '石家庄新闻网'
SITE_DOMAIN = 'news.sjzdaily.com.cn'
BASE_URL = 'http://news.sjzdaily.com.cn'

DB_PATH = os.environ.get('DB_PATH', os.path.join(os.path.dirname(os.path.abspath(__file__)), 'search.db'))

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

TOTAL_PAGES = 20
ITEMS_PER_PAGE = 50


def fetch_list_json(page):
    """获取JSON分页数据 (0-indexed)"""
    url = f'{BASE_URL}/428/NewsList_{page}.json'
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        data = resp.json()
    except Exception as e:
        print(f"  [ERROR] 第{page+1}页JSON请求失败: {e}")
        return []

    items = []
    for item in data.get('info', []):
        title = item.get('title', '').strip()
        url = item.get('url', '').strip()
        issue_time = item.get('issueTime', '').strip()
        if not title or not url:
            continue
        # 补全URL
        if not url.startswith('http'):
            url = f'{BASE_URL}{url}' if url.startswith('/') else f'{BASE_URL}/{url}'
        items.append({
            'title': title.strip('\u200b').strip(),  # 移除零宽空格
            'url': url,
            'publish_date': issue_time[:10] if issue_time else '',
        })

    print(f"  第{page+1}页: 找到 {len(items)} 条")
    return items


def fetch_detail(url):
    """获取详情页正文"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f"    [ERROR] 详情页请求失败 {url}: {e}")
        return None

    result = {}

    # 提取完整标题: <title>标题_石家庄新闻网</title>
    title_match = re.search(r'<title>([^<]+?)(?:_石家庄新闻网)?</title>', html)
    if title_match:
        result['title'] = title_match.group(1).strip()

    # 提取正文: class="news_txt"
    for class_name in ['news_txt', 'newscontent', 'news-content', 'article-content', 'article', 'content']:
        content_match = re.search(r'class="' + class_name + '"[^>]*>(.*?)</div>', html, re.DOTALL)
        if content_match:
            content_html = content_match.group(1).strip()
            # 移除脚本和样式
            content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL)
            content_html = re.sub(r'<style[^>]*>.*?</style>', '', content_html, flags=re.DOTALL)
            # 补全链接
            content_html = re.sub(r'src="(?!https?://)', f'src="{BASE_URL}', content_html)
            content_html = re.sub(r'href="(?!https?://)', f'href="{BASE_URL}', content_html)
            # 只保留正文中图片和文字部分，清理标题重复
            result['content'] = content_html
            break

    if 'content' not in result:
        result['content'] = ''

    return result


def get_db():
    """获取数据库连接"""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT,
        content TEXT,
        publish_date TEXT,
        url TEXT UNIQUE,
        site_name TEXT,
        source TEXT,
        attachments TEXT,
        created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    conn.commit()
    return conn


def save_item(conn, item):
    """保存单条数据"""
    c = conn.cursor()
    try:
        c.execute('''
            INSERT OR IGNORE INTO gov_raw (title, content, publish_date, url, site_name, source)
            VALUES (?, ?, ?, ?, ?, ?)
        ''', (
            item['title'],
            item.get('content', ''),
            item.get('publish_date', ''),
            item['url'],
            SITE_NAME,
            f'{SITE_NAME} - 公示公告',
        ))
        if c.rowcount > 0:
            print(f"    ✅ 新增: {item['title'][:50]}...")
        else:
            print(f"    ⏭️  已存在: {item['title'][:50]}...")
        conn.commit()
    except Exception as e:
        print(f"    ❌ 写入失败: {e}")


def run(max_pages=None):
    """主运行函数"""
    print(f"={'='*50}")
    print(f"  🌐 {SITE_NAME} ({SITE_DOMAIN}) 爬虫")
    print(f"={'='*50}")

    conn = get_db()

    pages = max_pages if max_pages else TOTAL_PAGES
    print(f"\n📊 共 {TOTAL_PAGES} 页, 每页 {ITEMS_PER_PAGE} 条")

    for p in range(pages):
        items = fetch_list_json(p)
        if not items:
            continue

        print(f"\n📄 第{p+1}页 ({len(items)} 条)")
        for idx, item in enumerate(items):
            title_display = item['title'][:50] if item['title'] else '(无标题)'
            print(f"  ▶️  [{idx+1}/{len(items)}] {title_display}...")
            detail = fetch_detail(item['url'])
            if detail:
                if detail.get('title'):
                    item['title'] = detail['title']
                item['content'] = detail.get('content', '')
            save_item(conn, item)
            time.sleep(0.5)

        # 如果没到20页但要限制，就停
        if max_pages and p + 1 >= max_pages:
            break

    conn.close()

    # 统计
    conn2 = sqlite3.connect(DB_PATH, timeout=60)
    count = conn2.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
    conn2.close()
    print(f"\n={'='*50}")
    print(f"  ✅ 完成! {SITE_NAME} 共导入 {count} 条数据到 {DB_PATH}")
    print(f"={'='*50}")


if __name__ == '__main__':
    max_pages = None
    if len(sys.argv) > 1:
        max_pages = int(sys.argv[1])
    run(max_pages)
