#!/usr/bin/env python3
"""
crawl_guzhen.py - 固镇县人民政府-建设项目环评审批
CloudWAF绕过: Playwright获取列表+详情页cookies
"""
import os, re, sys, json, time, asyncio, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from playwright.async_api import async_playwright
import requests

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE = "https://www.guzhen.gov.cn"
LIST_URL = BASE + "/zfxxgk/public/column/29641?type=4&catId=18193621&action=list&nav=3"
SITE_NAME = "固镇县政府-建设项目环评审批"
CATEGORY = "环保公示"
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
COOKIES_FILE = "/tmp/guzhen_cookies.json"

print(f"[guzhen] 3-year cutoff: {CUTOFF_DATE}")

async def get_all_items():
    """Use Playwright to get all list items"""
    all_items = []
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True)
        context = await browser.new_context(user_agent=HEADERS['User-Agent'])
        page = await context.new_page()
        
        await page.goto(LIST_URL, wait_until="networkidle", timeout=30000)
        await page.wait_for_timeout(5000)
        
        # Save cookies
        cookies = await context.cookies()
        with open(COOKIES_FILE, "w") as f:
            json.dump(cookies, f)
        
        total_pages = await page.evaluate('''() => {
            var lastA = document.querySelector('a[aria-label="跳转至尾页"]');
            return lastA ? parseInt(lastA.getAttribute('paged')) || 1 : 1;
        }''')
        print(f"[guzhen] Total pages: {total_pages}")
        
        for pg in range(1, total_pages + 1):
            if pg > 1:
                await page.evaluate(f'''() => {{
                    var input = document.querySelector('.inputBar input');
                    var btn = document.querySelector('.go-page');
                    if (input && btn) {{ input.value = "{pg}"; btn.click(); return true; }}
                    return false;
                }}''')
                await page.wait_for_timeout(3000)
            
            items = await page.evaluate('''() => {
                var lis = document.querySelectorAll('li.clearfix');
                return Array.from(lis).map(function(li) {
                    var a = li.querySelector('a.title');
                    var span = li.querySelector('span.date');
                    return {
                        title: a ? (a.getAttribute('title') || a.textContent.trim()) : '',
                        url: a ? a.getAttribute('href') : '',
                        date: span ? span.textContent.trim() : ''
                    };
                });
            }''')
            all_items.extend(items)
            print(f"  Page {pg}: {len(items)} items")
        
        await browser.close()
    return all_items

def fetch_detail(url, cookies):
    resp = requests.get(url, headers=HEADERS, cookies=cookies, timeout=20, verify=False)
    resp.encoding = "utf-8"
    return resp.text if resp.status_code == 200 and 'CloudWAF' not in resp.text else ""

def parse_detail(html):
    soup = BeautifulSoup(html, 'html.parser')
    title = ''
    mt = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if mt and mt.get('content'): title = mt['content'].strip()
    if not title:
        h1 = soup.find('h1')
        if h1: title = h1.get_text(strip=True)
    
    date_str = ''
    md = soup.find('meta', attrs={'name': 'PubDate'})
    if md and md.get('content'):
        m = re.search(r'(\d{4}-\d{2}-\d{2})', md['content'])
        if m: date_str = m.group(1)
    
    body = ''
    zoom = soup.find('div', id='zoom')
    if zoom: body = str(zoom)
    
    return title, date_str, body

# Step 1: Get items via Playwright
print("[guzhen] Fetching list via Playwright...")
all_items = asyncio.run(get_all_items())
print(f"[guzhen] Total items: {len(all_items)}")

# Filter by cutoff
active = [(item['title'], item['url'], item['date']) for item in all_items if item['date'] >= CUTOFF_DATE]
skipped = len(all_items) - len(active)
print(f"[guzhen] Within 3yr: {len(active)}, older: {skipped}")

# Step 2: Fetch details
with open(COOKIES_FILE, "r") as f:
    cookies_data = json.load(f)
cookie_dict = {c['name']: c['value'] for c in cookies_data}

conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()
new_count = dup_count = error_count = 0

for idx, (title, href, date_str) in enumerate(active, 1):
    detail_url = href if href.startswith("http") else BASE + href
    
    try:
        cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
        if cur.fetchone():
            dup_count += 1
            continue
        
        html = fetch_detail(detail_url, cookie_dict)
        if not html:
            error_count += 1
            if idx <= 3:
                print(f"  [WARN] Empty/WAF for: {detail_url}")
            continue
        
        d_title, d_date, body = parse_detail(html)
        if not d_title: d_title = title
        if not d_date and date_str: d_date = date_str
        
        date_rank = int(d_date.replace("-", "")) if d_date and "-" in d_date else 0
        summary = ""
        if body:
            text_soup = BeautifulSoup(body, 'html.parser')
            plain = text_soup.get_text(strip=True)
            summary = plain[:200]
        
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, d_title, detail_url, d_date, body, date_rank, CATEGORY, summary)
        )
        if cur.rowcount > 0:
            new_count += 1
    except Exception as e:
        error_count += 1
        if idx <= 3:
            print(f"  ERROR [{idx}] {title[:40]}: {e}")
    
    conn.commit()
    time.sleep(0.3)
    
    if idx % 10 == 0 or idx == len(active):
        print(f"  Progress {idx}/{len(active)}: +{new_count} new, {dup_count} dup, {error_count} err")

conn.close()
print(f"\n[guzhen] Summary: +{new_count} new, {dup_count} dup, {error_count} err")
print(json.dumps({"site": SITE_NAME, "new": new_count, "dup": dup_count, "err": error_count}, ensure_ascii=False))
