#!/usr/bin/env python3
"""
荆州市人民政府 - 环境影响评价 爬虫 (Chrome渲染)
"""

import subprocess, time, sqlite3, requests
from datetime import datetime, date
from bs4 import BeautifulSoup
import os

CHROME = '/root/.cache/ms-playwright/chromium-1217/chrome-linux64/chrome'
SITE_NAME = '荆州市人民政府-环境影响评价'
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = date(2023, 6, 1)
BASE = 'https://zwgk.jingzhou.gov.cn'

def render_page(url):
    proc = subprocess.run(
        [CHROME, '--headless', '--no-sandbox', '--disable-gpu',
         '--disable-dev-shm-usage', '--ignore-certificate-errors', '--dump-dom', url],
        capture_output=True, timeout=25, env={}
    )
    return proc.stdout.decode('utf-8', errors='replace')

def get_articles():
    html = render_page(f'{BASE}/documentList.shtml?column_id=54164')
    soup = BeautifulSoup(html, 'lxml')
    ul = soup.find('ul', class_='zw-list')
    if not ul: return []
    items = []
    for li in ul.find_all('li'):
        a = li.find('a')
        span = li.find('span')
        if not a or not span: continue
        href = a.get('href', '')
        title = a.get_text(strip=True)
        pub_date = span.get_text(strip=True)
        if href and title and pub_date:
            items.append({'url': href, 'title': title, 'pub_date': pub_date})
    return items

def fetch_detail(url):
    try:
        r = requests.get(url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=20)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        view = soup.find('div', class_='view')
        if view:
            return str(view)
        body = soup.find('div', class_='jzgov-article-body')
        if body:
            return str(body)
        return ''
    except:
        return ''

def main():
    print('Rendering page...')
    items = get_articles()
    print(f'Total: {len(items)}')
    
    dates = [i['pub_date'] for i in items if i.get('pub_date')]
    if dates: print(f'Range: {min(dates)} ~ {max(dates)}')
    
    filtered = [it for it in items if it.get('pub_date') and 
                datetime.strptime(it['pub_date'], '%Y-%m-%d').date() >= CUTOFF_DATE]
    print(f'After filter: {len(filtered)}')
    
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    added = no_content = 0
    
    for it in filtered:
        c.execute('SELECT 1 FROM gov_raw WHERE page_url = ?', (it['url'],))
        if c.fetchone(): continue
        content = fetch_detail(it['url'])
        if not content: no_content += 1
        c.execute('''INSERT OR IGNORE INTO gov_raw
            (site_name, source_url, page_url, title, publish_date, content, summary)
            VALUES (?, ?, ?, ?, ?, ?, ?)''',
            (SITE_NAME, it['url'], it['url'], it['title'],
             it['pub_date'], content, it['title']))
        added += 1
        time.sleep(0.3)
    
    conn.commit(); conn.close()
    print(f'\nDone! Added: {added}, NoContent: {no_content}')

if __name__ == '__main__':
    main()
