#!/usr/bin/env python3
"""
靖远县人民政府 - 公告公示 爬虫 (Chrome CDP 翻页)
"""
import subprocess, json, time, sqlite3, requests, urllib.request
from datetime import datetime, date
from bs4 import BeautifulSoup
import websocket
import os

CHROME = '/root/.cache/ms-playwright/chromium-1217/chrome-linux64/chrome'
SITE_NAME = '靖远县人民政府-公告公示'
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = date(2023, 6, 1)
BASE = 'https://www.jingyuan.gov.cn'

def collect_all_articles():
    port = 9224
    proc = subprocess.Popen(
        [CHROME, '--headless', '--no-sandbox', '--disable-gpu',
         '--disable-dev-shm-usage', f'--remote-debugging-port={port}',
         '--remote-allow-origins=*', f'{BASE}/xwzx/gggs/index.html'],
        stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, env={}
    )
    time.sleep(4)
    
    try:
        resp = urllib.request.urlopen(f'http://127.0.0.1:{port}/json')
        targets = json.loads(resp.read())
        ws_url = targets[0]['webSocketDebuggerUrl']
        
        ws = websocket.create_connection(ws_url, timeout=15)
        next_id = 1
        
        def send(method, params=None):
            nonlocal next_id
            cmd = {"id": next_id, "method": method}
            if params: cmd["params"] = params
            ws.send(json.dumps(cmd))
            next_id += 1
            return next_id - 1
        
        def wait_result(cmd_id, timeout_ms=10000):
            """Wait for specific command result, skipping console messages"""
            import select
            deadline = time.time() + timeout_ms / 1000
            while time.time() < deadline:
                if select.select([ws.sock], [], [], 0.5)[0]:
                    msg = json.loads(ws.recv())
                    if msg.get('id') == cmd_id:
                        return msg
            return None
        
        send("Runtime.enable")
        time.sleep(3)
        
        # Collect all articles by clicking through pages
        script = """
(async () => {
    const sleep = ms => new Promise(r => setTimeout(r, ms));
    let all = [];
    for (let page = 1; page <= 35; page++) {
        await sleep(1500);
        const items = Array.from(document.querySelectorAll('.u-lst-bor li')).map(li => {
            const a = li.querySelector('a');
            const span = li.querySelector('span');
            return {
                href: a ? a.getAttribute('href') : '',
                title: a ? a.textContent.trim().replace(/^-/, '') : '',
                date: span ? span.textContent.trim() : ''
            };
        });
        if (items.length === 0) break;
        all = all.concat(items);
        const lastD = items[items.length-1].date;
        if (lastD < '2023-06-01') break;
        const nxt = document.querySelector('.layui-laypage-next');
        if (!nxt || nxt.classList.contains('layui-disabled')) break;
        nxt.click();
    }
    // Dedup
    const seen = new Set();
    return JSON.stringify(all.filter(x => { if (seen.has(x.href)) return false; seen.add(x.href); return true; }));
})()
"""
        cid = send("Runtime.evaluate", {
            "expression": script,
            "awaitPromise": True,
            "returnByValue": True,
            "timeout": 60000
        })
        
        result = None
        import select
        deadline = time.time() + 60
        while time.time() < deadline:
            if select.select([ws.sock], [], [], 1)[0]:
                msg = json.loads(ws.recv())
                if msg.get('id') == cid:
                    result = msg
                    break
        
        ws.close()
        
        if result:
            if 'exceptionDetails' in str(result.get('result',{})):
                print(f"JS error: {result['result'].get('exceptionDetails',{}).get('text','')}")
                return []
            val = result.get('result',{}).get('result',{}).get('value','[]')
            if isinstance(val, str):
                return json.loads(val)
            return val or []
        return []
    except Exception as e:
        print(f"CDP error: {e}")
        return []
    finally:
        proc.kill()
        try: proc.wait(timeout=3)
        except: pass

def fetch_detail(url):
    try:
        r = requests.get(url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=20, verify=False)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        pc = soup.find(id='printContent')
        if pc:
            d = soup.new_tag('div')
            for p in pc.find_all('p'):
                d.append(p)
            html = str(d)
            return html if len(d.get_text(strip=True)) > 50 else str(pc)
        return ''
    except:
        return ''

def main():
    print('Step 1: Collecting all articles via Chrome CDP...')
    items = collect_all_articles()
    print(f'\nTotal collected: {len(items)}')
    
    if not items:
        print("Fallback: API only (1 page, 20 items)")
        items = api_fallback()
    
    if items:
        dates = [i.get('date','') for i in items if i.get('date')]
        if dates: print(f'Range: {min(dates)} ~ {max(dates)}')
        
        filtered = [it for it in items if it.get('date') and 
                    datetime.strptime(it['date'], '%Y-%m-%d').date() >= CUTOFF_DATE]
        print(f'After filter: {len(filtered)}')
        
        conn = sqlite3.connect(DB_PATH)
        c = conn.cursor()
        added = no_content = 0
        for i, it in enumerate(filtered):
            url = it['href'] if it['href'].startswith('http') else BASE + it['href']
            c.execute('SELECT 1 FROM gov_raw WHERE page_url = ?', (url,))
            if c.fetchone(): continue
            content = fetch_detail(url)
            if not content: no_content += 1
            c.execute('''INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)''',
                (SITE_NAME, url, url, it['title'], it['date'], content, it['title']))
            added += 1
            if (i+1) % 10 == 0:
                conn.commit()
                time.sleep(0.2)
        conn.commit(); conn.close()
        print(f'\nDone! Added: {added}, NoContent: {no_content}')

def api_fallback():
    params = {
        'parseType': 'bulidstatic', 'webId': 'a56204ae810e47f3980d37aaecd6be5e',
        'tplSetId': '5004e1c65f564ef99a4d2646c2f2b1e5', 'pageType': 'column',
        'tagId': 'Ajax分页', 'pageId': '58fb923cd15b4bd3bea25ab66c8dd656',
        'pageNo': '1', 'pageSize': '20'
    }
    r = requests.get(f'{BASE}/api-gateway/jpaas-publish-server/front/page/build/unit',
        params=params, headers={'User-Agent': 'Mozilla/5.0'}, timeout=15, verify=False)
    d = r.json()
    h = d.get('data', {}).get('html', '')
    soup = BeautifulSoup(h, 'lxml')
    items = []
    for li in soup.select('.u-lst-bor li'):
        a = li.find('a'); span = li.find('span')
        if a and span:
            href = a.get('href','')
            if href and not href.startswith('http'): href = BASE + href
            items.append({'href': href, 'title': a.get_text(strip=True).lstrip('-').strip(), 'date': span.get_text(strip=True)})
    return items

if __name__ == '__main__':
    main()
