#!/usr/bin/env python3
"""
惠州市惠东县黄埠镇人民政府 - 信息公开 爬虫
使用 headless Chrome 渲染获取数据
"""

import subprocess, re, time, sqlite3, requests
from datetime import datetime, date
from bs4 import BeautifulSoup
import os

CHROME = '/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome'
SITE_NAME = '惠东县黄埠镇人民政府-镇街信息'
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = date(2023, 6, 1)
BASE = 'http://www.huidong.gov.cn'

def render_page(url):
    proc = subprocess.run(
        [CHROME, '--headless', '--no-sandbox', '--disable-gpu',
         '--disable-dev-shm-usage', '--dump-dom', url],
        capture_output=True, timeout=25, env={}
    )
    return proc.stdout.decode('utf-8', errors='replace')

def get_articles():
    html = render_page(f'{BASE}/hzhdhbz/gkmlpt/index#10846')
    soup = BeautifulSoup(html, 'lxml')
    items = []
    for tr in soup.select('table.table-content tbody tr'):
        tds = tr.find_all('td')
        if len(tds) >= 2:
            a = tds[0].find('a')
            if a:
                href = a.get('href', '')
                if href.startswith('//'):
                    href = 'http:' + href
                title = a.get_text(strip=True)
                pub_date = tds[1].get_text(strip=True)
                if href and title and pub_date:
                    items.append({'url': href, 'title': title, 'pub_date': pub_date})
    return items

def get_content(url):
    try:
        r = requests.get(url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=20)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        # Extract article-content
        ac = soup.find('div', class_='article-content')
        if ac:
            return str(ac)
        # Fallback: content div
        c = soup.find('div', class_='content')
        if c:
            return str(c)
        return ''
    except:
        return ''

def main():
    print('Step 1: Rendering page...')
    items = get_articles()
    print(f'Total: {len(items)}')
    
    dates = [i['pub_date'] for i in items if i.get('pub_date')]
    if dates:
        print(f'Range: {min(dates)} ~ {max(dates)}')
    
    filtered = [it for it in items if it.get('pub_date') and 
                datetime.strptime(it['pub_date'], '%Y-%m-%d').date() >= CUTOFF_DATE]
    print(f'After filter: {len(filtered)}')
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    added = no_content = 0
    
    for it in filtered:
        c.execute('SELECT 1 FROM gov_raw WHERE page_url = ?', (it['url'],))
        if c.fetchone():
            continue
        
        content = get_content(it['url'])
        if not content:
            no_content += 1
        
        c.execute('''INSERT OR IGNORE INTO gov_raw
            (site_name, source_url, page_url, title, publish_date, content, summary)
            VALUES (?, ?, ?, ?, ?, ?, ?)''',
            (SITE_NAME, it['url'], it['url'], it['title'],
             it['pub_date'], content, it['title']))
        added += 1
        time.sleep(0.3)
    
    conn.commit()
    conn.close()
    print(f'\nDone! Added: {added}, NoContent: {no_content}')

if __name__ == '__main__':
    main()
