#!/usr/bin/env python3
"""仪征市人民政府-公示公告 爬虫 (翰Web, Playwright渲染列表)"""
import re, requests, sqlite3, os, sys, time, json
from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

SITE_NAME = '仪征市人民政府-公示公告'
BASE = 'https://www.yizheng.gov.cn'
LIST = '/xxfb/gsgg/index.html'
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

def fetch_all_list():
    """用 Playwright 逐页渲染并提取列表数据"""
    all_items = []
    seen_urls = set()
    PAGE_SIZE = 15
    
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=['--no-sandbox'])
        ctx = browser.new_context(
            user_agent='Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
        )
        page = ctx.new_page()
        
        try:
            # Get total pages from page 1
            page.goto(BASE + LIST, wait_until='networkidle', timeout=30000)
            
            total_pages = page.evaluate('''() => {
                let max = 1;
                document.querySelectorAll('.page-content a, .pagination a, .bt-pagination a').forEach(a => {
                    if (a.getAttribute('href') && a.getAttribute('href').includes('index_')) {
                        const n = parseInt(a.textContent);
                        if (!isNaN(n) && n > max) max = n;
                    }
                });
                return max;
            }''')
            
            if total_pages <= 1:
                total_pages = 177  # fallback
            
            print(f'📋 共 {total_pages} 页')
            
            for pg in range(1, total_pages + 1):
                if pg > 1:
                    pg_url = BASE + f'/xxfb/gsgg/index_{pg}.html'
                    try:
                        page.goto(pg_url, wait_until='networkidle', timeout=15000)
                    except:
                        page.goto(pg_url, wait_until='load', timeout=15000)
                
                # Extract items
                items = page.evaluate('''() => {
                    const result = [];
                    document.querySelectorAll('.bt-list-new').forEach(li => {
                        const a = li.querySelector('a');
                        const span = li.querySelector('.bt-list-time');
                        if (a) {
                            result.push({
                                title: (a.getAttribute('title') || a.textContent).trim(),
                                url: a.getAttribute('href'),
                                date: span ? span.textContent.trim() : ''
                            });
                        }
                    });
                    return result;
                }''')
                
                if not items:
                    print(f'  ⚠️ 第{pg}页无数据')
                    break
                
                new_count = 0
                for item in items:
                    url = item['url']
                    if not url.startswith('http'):
                        url = BASE + url
                    if url not in seen_urls:
                        seen_urls.add(url)
                        all_items.append({
                            'title': item['title'],
                            'url': url,
                            'date': item['date'],
                        })
                        new_count += 1
                
                print(f'  📄 第{pg}页: {new_count}条新')
                
                if pg % 50 == 0:
                    print(f'    已收集 {len(all_items)} 条...')
        
        except Exception as e:
            print(f'  ⚠️ Playwright错误: {e}')
        finally:
            browser.close()
    
    return all_items

def fetch_first_page():
    """只获取第一页（日增量用）"""
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=['--no-sandbox'])
        ctx = browser.new_context(user_agent=HEADERS['User-Agent'])
        page = ctx.new_page()
        page.goto(BASE + LIST, wait_until='networkidle', timeout=30000)
        items = page.evaluate('''() => {
            const result = [];
            document.querySelectorAll('.bt-list-new').forEach(li => {
                const a = li.querySelector('a');
                const span = li.querySelector('.bt-list-time');
                if (a) {
                    result.push({
                        title: (a.getAttribute('title') || a.textContent).trim(),
                        url: a.getAttribute('href'),
                        date: span ? span.textContent.trim() : ''
                    });
                }
            });
            return result;
        }''')
        browser.close()
        result = []
        for item in items:
            url = item['url']
            if not url.startswith('http'):
                url = BASE + url
            result.append({'title': item['title'], 'url': url, 'date': item['date']})
        return result

def parse_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        for cls in ['bt-content', 'bt-article', 'article-content', 'content', 'page-content', 'theme-showcase']:
            div = soup.find('div', class_=cls)
            if div:
                for tag in div(['script', 'style']):
                    tag.decompose()
                return str(div)
        for id_name in ['zoom', 'UCAP-CONTENT']:
            div = soup.find('div', id=id_name)
            if div:
                for tag in div(['script', 'style']):
                    tag.decompose()
                return str(div)
    except:
        pass
    return ''

def run(mode='full'):
    if mode == 'incremental':
        items = fetch_first_page()
        print(f'📋 增量模式: 第1页 {len(items)}条')
    else:
        items = fetch_all_list()
        print(f'\n📋 共采集 {len(items)} 条')
    
    if not items:
        return
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    new, skip = 0, 0
    
    for idx, item in enumerate(items):
        body = parse_detail(item['url'])
        try:
            conn.execute("""
                INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (SITE_NAME, item['url'], item['url'], item['title'],
                  item['date'], body, item['title']))
            if conn.total_changes > 0:
                new += 1
            else:
                skip += 1
        except Exception as e:
            print(f'  ❌ - {e}')
        
        if (idx + 1) % 20 == 0:
            print(f'  💾 [{idx+1}/{len(items)}] 新增{new}, 跳过{skip}')
            conn.commit()
        time.sleep(0.3)
    
    conn.commit()
    conn.close()
    print(f'\n✅ {SITE_NAME}: 新增{new}, 跳过{skip}, 共{len(items)}条')

if __name__ == '__main__':
    mode = sys.argv[1] if len(sys.argv) > 1 else 'full'
    if mode == 'test':
        with sync_playwright() as p:
            browser = p.chromium.launch(headless=True, args=['--no-sandbox'])
            ctx = browser.new_context(user_agent=HEADERS['User-Agent'])
            page = ctx.new_page()
            page.goto('https://www.yizheng.gov.cn/xxfb/gsgg/index.html', wait_until='networkidle', timeout=30000)
            total = page.evaluate('''() => Math.max(...Array.from(document.querySelectorAll('.page-content a, .pagination a')).filter(a => a.getAttribute('href') && a.getAttribute('href').includes('index_')).map(a => parseInt(a.textContent)).filter(n => !isNaN(n)), 1)''')
            print(f'Total pages: {total}')
            browser.close()
    else:
        run(mode)
