#!/usr/bin/env python3
"""
庐江县人民政府-生态环境分局 - 建设项目环境影响评价审批 爬虫
CMS: Lonsun (龙讯)
WAF: Cloudflare JSL (列表页) + Knownsec 创宇盾 (详情页)
需 Playwright 浏览器绕过

列表页: https://www.hefei.gov.cn/public/column/19081?catId=6999451&action=list&type=4&pageIndex=N
详情页: https://www.lj.gov.cn/public/19081/{id}.html
"""

import os, sys, re, json, asyncio, sqlite3
from datetime import datetime
from playwright.async_api import async_playwright

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "庐江县人民政府-建设项目环评审批"
BASE_LIST = "https://www.hefei.gov.cn/public/column/19081"
BASE_DETAIL = "https://www.lj.gov.cn/public/19081"
CUTOFF = "2023-06-18"

def get_list_url(page_idx):
    if page_idx == 0:
        return f"{BASE_LIST}?type=4&catId=6999451&action=list"
    return f"{BASE_LIST}?catId=6999451&action=list&type=4&pageIndex={page_idx}"
async def goto_page(page, url, max_wait=90):
    """Navigate to URL and wait for ALL Cloudflare challenge layers to resolve.
    Returns True when real page content detected."""
    print(f"    Loading page...", end=' ', flush=True)
    try:
        await page.goto(url, wait_until='networkidle', timeout=45000)
    except:
        pass
    
    # Poll for real content (not challenge page)
    for i in range(max_wait):
        content = await page.content()
        # Real page: >50K and has Lonsun content markers
        if len(content) > 50000 and ('xxgk_nav_list' in content or '庐江县' in content or 'ColumnName' in content):
            print(f"OK ({i}s)", end=' ', flush=True)
            return True
        await asyncio.sleep(2)
    
    print(f"timeout")
    return False

async def scrape_list_page(page, page_idx):
    """Scrape one list page, return [(href, title, date), ...]"""
    url = get_list_url(page_idx)
    ok = await goto_page(page, url)
    if not ok:
        return []
    
    await page.wait_for_timeout(2000)
    
    items = await page.evaluate('''
        () => {
            const list = document.querySelector('ul.xxgk_nav_list');
            if (!list) return [];
            return Array.from(list.querySelectorAll('li')).map(li => {
                const a = li.querySelector('a');
                const spans = li.querySelectorAll('span');
                return {
                    href: a ? a.href : '',
                    text: a ? a.textContent.trim() : '',
                    date: spans.length > 0 ? spans[spans.length-1].textContent.trim() : '',
                };
            });
        }
    ''')
    return items

async def scrape_detail(page, url):
    """Scrape one detail page, return (title, date, content)"""
    ok = await goto_page(page, url)
    if not ok:
        return '', '', ''
    
    result = await page.evaluate('''
        () => {
            const getMeta = (name) => {
                const el = document.querySelector('meta[name="' + name + '"]');
                return el ? el.getAttribute('content') : '';
            };
            const title = getMeta('ArticleTitle') || document.title || '';
            const pubDate = getMeta('PubDate') || '';
            
            let content = '';
            const selectors = [
                'div#zoom.newscontnet', 'div.j-fontContent.newscontnet',
                'div.con_main', 'div.wenzhang', 'div.contentbox',
                '.bt-content', '.zoom', '.TRS_UEDITOR',
                '.article-content', '.xxgk-content',
                '#content', '.main-content'
            ];
            for (const sel of selectors) {
                const el = document.querySelector(sel);
                if (el && el.textContent.trim().length > 50) {
                    content = el.innerHTML;
                    break;
                }
            }
            
            if (!content) {
                const divs = document.querySelectorAll('div');
                let best = null, bestLen = 0;
                divs.forEach(d => {
                    const cls = d.className || '';
                    const text = d.textContent.trim();
                    if (text.length > 200 && text.length > bestLen &&
                        !cls.includes('header') && !cls.includes('footer') && 
                        !cls.includes('nav') && !cls.includes('sidebar')) {
                        best = d;
                        bestLen = text.length;
                    }
                });
                if (best) content = best.innerHTML;
            }
            
            return { title, pubDate, content };
        }
    ''')
    
    title = result.get('title', '') or url.split('/')[-1].replace('.html', '')
    pub_date = result.get('pubDate', '')
    content = result.get('content', '')
    
    if content:
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
        content = content.strip()
    
    return title, pub_date[:10], content

def save_to_db(items):
    conn = sqlite3.connect(SEARCH_DB)
    c = conn.cursor()
    new_count = 0
    for url, title, date, content in items:
        try:
            c.execute("""
                INSERT OR IGNORE INTO gov_raw (page_url, title, publish_date, content, site_name, summary)
                VALUES (?, ?, ?, ?, ?, ?)
            """, (url, title, date, content, SITE_NAME, title))
            if c.rowcount > 0:
                new_count += 1
        except Exception as e:
            print(f"  DB error: {e}", file=sys.stderr)
    conn.commit()
    conn.close()
    return new_count

async def main():
    print(f"Starting: {SITE_NAME}")
    print(f"DB: {SEARCH_DB}")
    
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        context = await browser.new_context(
            user_agent='Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
            viewport={'width': 1920, 'height': 1080},
            locale='zh-CN',
        )
        await context.add_init_script('''
            Object.defineProperty(navigator, 'webdriver', {get: () => undefined});
            Object.defineProperty(navigator, 'plugins', {get: () => [1,2,3,4,5]});
        ''')
        
        page = await context.new_page()
        
        # --- Phase 1: Scrape list pages ---
        print("\n=== Phase 1: Scraping list pages ===")
        all_items = []
        max_pages = 174
        
        for page_idx in range(max_pages):
            print(f"  Page {page_idx+1}/{max_pages}: ", end='', flush=True)
            items = await scrape_list_page(page, page_idx)
            
            if not items:
                print("  -> empty results, may be at end")
                break
            
            recent = [it for it in items if it['date'] >= CUTOFF]
            older = [it for it in items if it['date'] < CUTOFF if it['date']]
            
            dates_str = ''
            if items[0].get('date'):
                dates_str = f", dates {items[-1]['date']}~{items[0]['date']}"
            
            print(f"    {len(items)} items, {len(recent)} recent{dates_str}")
            
            for it in recent:
                all_items.append(it)
            
            # If all items on this page are older, we're done
            if len(older) == len(items) and len(items) > 0 and older[0]['date'] < CUTOFF:
                print(f"  -> All older than cutoff, stopping")
                break
            
            await asyncio.sleep(2)
        
        print(f"\nTotal items within 3 years: {len(all_items)}")
        
        if not all_items:
            print("No items to process.")
            await browser.close()
            return
        
        # --- Phase 2: Scrape detail pages ---
        print("\n=== Phase 2: Scraping detail pages ===")
        detail_results = []
        
        for i, item in enumerate(all_items):
            print(f"  [{i+1}/{len(all_items)}] {item['text'][:40]}...", end=' ', flush=True)
            
            title, date, content = await scrape_detail(page, item['href'])
            
            if not date or date < CUTOFF:
                date = item['date']
            
            if not content or len(content) < 30:
                print("NO CONTENT")
                detail_results.append((item['href'], title or item['text'], date, ''))
            else:
                print(f"OK ({len(content)} chars)")
                detail_results.append((item['href'], title or item['text'], date, content))
            
            await asyncio.sleep(1)
        
        # --- Phase 3: Save to DB ---
        print(f"\n=== Phase 3: Saving to DB ===")
        new_count = save_to_db(detail_results)
        print(f"Inserted {new_count} new records")
        
        if new_count > 0:
            try:
                conn = sqlite3.connect(SEARCH_DB)
                conn.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
                conn.commit()
                conn.close()
                print("FTS rebuilt")
            except Exception as e:
                print(f"FTS error: {e}")
        
        await browser.close()
    
    print("\nDone!")

if __name__ == "__main__":
    asyncio.run(main())
