#!/usr/bin/env python3
import os
"""crawl_susong.py - 宿松县人民政府-县生态环境分局-行政许可"""
import sys, os, re, asyncio, time
from datetime import datetime, timedelta

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime('%Y-%m-%d')
SITE_NAME = '宿松县人民政府-行政许可'
MAX_PAGES = 10
API_TPL = 'https://www.susong.gov.cn/site/label/8888?_={rand}&labelName=publicInfoList&siteId=2000005581&organId=2000003861&pageSize=20&pageIndex={page}&isDate=true&dateFormat=yyyy-MM-dd&length=50&type=4&action=list&isDriving=true&result=&isJson=true&keyWords=&isSetValue=true&catIds=&catId=41336664'

def clean_title(t):
    t = re.sub(r'\s+', ' ', t).strip()
    for p in ['【我要纠错】', '分享到：', '打印 下载 收藏']:
        t = t.replace(p, '')
    return t.strip()

async def run_async(incremental=False):
    from playwright.async_api import async_playwright
    
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        context = await browser.new_context(
            user_agent='Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36',
            locale='zh-CN',
        )
        await context.add_init_script("""
            Object.defineProperty(navigator, 'webdriver', {get: () => undefined});
        """)
        page = await context.new_page()
        
        # Load list page to bypass JS challenge
        list_url = 'https://www.susong.gov.cn/public/column/2000003861?type=4&catId=41336664&action=list&nav=3'
        await page.goto(list_url, wait_until='domcontentloaded', timeout=30000)
        await page.wait_for_timeout(3000)
        
        records = []
        for pn in range(1, MAX_PAGES + 1):
            rand = str(abs(hash(f"susong_{pn}")))[-17:]
            api_url = API_TPL.format(rand=rand, page=pn)
            
            data = await page.evaluate(f'''
                async () => {{
                    try {{
                        const r = await fetch("{api_url}");
                        return await r.json();
                    }} catch(e) {{ return {{ error: e.message }}; }}
                }}
            ''')
            
            if 'error' in data or 'data' not in data:
                break
            
            items = data.get('data', [])
            if not items:
                break
            
            for item in items:
                pub_date = item.get('publishDate', '').split(' ')[0]
                if pub_date < CUTOFF:
                    continue
                
                title = clean_title(item.get('title', ''))
                link = item.get('link', '')
                
                records.append({
                    'title': title,
                    'url': link,
                    'pub_date': pub_date,
                    'site_name': SITE_NAME,
                    'content': '',
                    'summary': '',
                })
            
            if incremental:
                break
            await page.wait_for_timeout(500)
        
        print(f"  List: {len(records)} items")
        
        # Fetch detail content
        for i, rec in enumerate(records):
            try:
                await page.goto(rec['url'], wait_until='domcontentloaded', timeout=20000)
                await page.wait_for_timeout(1500)
                
                content_el = await page.query_selector('.clearfix.xxgkcontent')
                if content_el:
                    html = await content_el.inner_html()
                    html = re.sub(r'分享到：.*?收藏', '', html, flags=re.DOTALL)
                    html = re.sub(r'【我要纠错】', '', html)
                    rec['content'] = html.strip()
                else:
                    rec['content'] = ''
            except Exception as e:
                print(f"  Detail error [{i+1}/{len(records)}]: {rec['url']} - {e}")
                rec['content'] = ''
            
            if (i+1) % 5 == 0:
                print(f"  Detail: {i+1}/{len(records)}")
        
        await browser.close()
    
    valid = [r for r in records if r['content'].strip()]
    print(f"  With content: {len(valid)}/{len(records)}")
    
    if valid:
        push_to_searchdb(valid, "susong_xzxk")
    
    return len(valid)

def run(incremental=False):
    return asyncio.run(run_async(incremental=incremental))

if __name__ == '__main__':
    inc = '--incremental' in sys.argv
    cnt = run(incremental=inc)
    print(f"Done: {cnt} records")
