#!/usr/bin/env python3
"""crawl_susong_towns.py - 宿松县乡镇公示公告（高岭乡/河塌乡/北浴乡）"""
import sys, os, re, asyncio, time
from datetime import datetime, timedelta

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime('%Y-%m-%d')
MAX_PAGES = 30

ORGANS = [
    ('2000004191', '宿松县-高岭乡-公示公告'),
    ('2000004081', '宿松县-河塌乡-公示公告'),
    ('2000004231', '宿松县-北浴乡-公示公告'),
]

def clean_title(t):
    t = re.sub(r'\s+', ' ', t).strip()
    for p in ['【我要纠错】', '分享到：', '打印 下载 收藏']:
        t = t.replace(p, '')
    return t.strip()

async def run_async(incremental=False):
    from playwright.async_api import async_playwright
    
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        context = await browser.new_context(
            user_agent='Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36',
            locale='zh-CN',
        )
        await context.add_init_script("""
            Object.defineProperty(navigator, 'webdriver', {get: () => undefined});
        """)
        page = await context.new_page()
        
        # Bypass JS challenge once
        await page.goto('https://www.susong.gov.cn/public/column/2000004191?type=4&catId=41335247&action=list', wait_until='domcontentloaded', timeout=30000)
        await page.wait_for_timeout(3000)
        
        all_records = []
        
        for oid, site_name in ORGANS:
            print(f"\n=== {site_name} ===")
            records = []
            
            for pn in range(1, MAX_PAGES + 1):
                rand = str(abs(hash(f"{oid}_{pn}")))[-17:]
                api_url = f'https://www.susong.gov.cn/site/label/8888?_={rand}&labelName=publicInfoList&siteId=2000005581&organId={oid}&pageSize=20&pageIndex={pn}&isDate=true&dateFormat=yyyy-MM-dd&length=50&type=4&action=list&isDriving=true&result=&isJson=true&keyWords=&isSetValue=true&catIds=&catId=41335247'
                
                data = await page.evaluate(f'''
                    async () => {{
                        try {{
                            const r = await fetch("{api_url}");
                            return await r.json();
                        }} catch(e) {{ return {{ error: e.message }}; }}
                    }}
                ''')
                
                if 'error' in data or 'data' not in data:
                    break
                
                items = data.get('data', [])
                if not items:
                    break
                
                for item in items:
                    pub_date = item.get('publishDate', '').split(' ')[0]
                    if pub_date < CUTOFF:
                        continue
                    
                    records.append({
                        'title': clean_title(item.get('title', '')),
                        'url': item.get('link', ''),
                        'pub_date': pub_date,
                        'site_name': site_name,
                        'content': '',
                        'summary': '',
                    })
                
                if incremental:
                    break
                await page.wait_for_timeout(300)
            
            print(f"  List: {len(records)} items")
            
            # Fetch detail content
            for i, rec in enumerate(records):
                try:
                    await page.goto(rec['url'], wait_until='domcontentloaded', timeout=20000)
                    await page.wait_for_timeout(1500)
                    
                    content_el = await page.query_selector('.clearfix.xxgkcontent')
                    if content_el:
                        html = await content_el.inner_html()
                        html = re.sub(r'分享到：.*?收藏', '', html, flags=re.DOTALL)
                        html = re.sub(r'【我要纠错】', '', html)
                        rec['content'] = html.strip()
                    else:
                        rec['content'] = ''
                except Exception as e:
                    print(f"    Detail error [{i+1}/{len(records)}]: {rec['url']} - {e}")
                    rec['content'] = ''
                
                if (i+1) % 5 == 0:
                    print(f"    Detail: {i+1}/{len(records)}")
            
            valid = [r for r in records if r['content'].strip()]
            print(f"  With content: {len(valid)}/{len(records)}")
            all_records.extend(valid)
        
        await browser.close()
    
    if all_records:
        push_to_searchdb(all_records, "susong_towns")
    
    return len(all_records)

def run(incremental=False):
    return asyncio.run(run_async(incremental=incremental))

if __name__ == '__main__':
    inc = '--incremental' in sys.argv
    cnt = run(incremental=inc)
    print(f"Done: {cnt} records total")
