#!/usr/bin/env python3
"""
crawl_lnjp.py - 建平县人民政府-建设项目环境影响评价文件审批
需要Playwright获取列表页（CMS API需浏览器环境）
"""
import os, re, sys, json, time, asyncio, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from playwright.async_api import async_playwright
import requests

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "建平县人民政府-建设项目环境影响评价文件审批"
CATEGORY = "环评公示"
BASE = "https://www.lnjp.gov.cn"
LIST_URL = BASE + "/jpxzf/26gly/sthj/xzxk/jsxmhjyxpjwjsp/glist.html"
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}

print(f"[lnjp] 3-year cutoff: {CUTOFF_DATE}")

async def get_all_items():
    """Use Playwright to get list items from all pages"""
    all_items = []
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True)
        page = await browser.new_page(user_agent=HEADERS['User-Agent'])
        
        await page.goto(LIST_URL, wait_until="networkidle", timeout=30000)
        await page.wait_for_timeout(3000)
        
        # Extract total pages
        total_pages_text = await page.text_content('.nowpage')
        total_pages = 4  # default from initial analysis
        
        # Get current page items
        items = await page.evaluate('''() => {
            var lis = document.querySelectorAll('ul.list-wrap1 li');
            return Array.from(lis).map(function(li) {
                var a = li.querySelector('a');
                var span = li.querySelector('span');
                return {
                    title: a ? a.textContent.trim() : '',
                    url: a ? a.getAttribute('href') : '',
                    date: span ? span.textContent.trim() : ''
                };
            });
        }''')
        all_items.extend(items)
        print(f"  Page 1: {len(items)} items")
        
        # Navigate to remaining pages
        for pg in range(2, total_pages + 1):
            clicked = await page.evaluate('''(pg) => {
                var links = document.querySelectorAll('a.nextpage, a.lastpage');
                if (links.length > 0) {
                    // Click the next page link
                    var nextLink = document.querySelector('a.nextpage');
                    if (nextLink) { nextLink.click(); return true; }
                    var lastLink = document.querySelector('a.lastpage');
                    if (lastLink) { lastLink.click(); return true; }
                }
                return false;
            }''', pg)
            
            if not clicked:
                # try pageAutoLink function directly
                await page.evaluate(f'pageAutoLink({pg - 1})')
            
            await page.wait_for_timeout(3000)
            
            items = await page.evaluate('''() => {
                var lis = document.querySelectorAll('ul.list-wrap1 li');
                return Array.from(lis).map(function(li) {
                    var a = li.querySelector('a');
                    var span = li.querySelector('span');
                    return {
                        title: a ? a.textContent.trim() : '',
                        url: a ? a.getAttribute('href') : '',
                        date: span ? span.textContent.trim() : ''
                    };
                });
            }''')
            all_items.extend(items)
            print(f"  Page {pg}: {len(items)} items")
        
        await browser.close()
    return all_items

print("[lnjp] Fetching list via Playwright...")
all_items = asyncio.run(get_all_items())
print(f"[lnjp] Total items: {len(all_items)}")

# Filter by cutoff
active = [(item['title'], item['url'], item['date']) for item in all_items if item['date'] >= CUTOFF_DATE]
skipped = len(all_items) - len(active)
print(f"[lnjp] Within 3yr: {len(active)}, older: {skipped}")

conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()
new_count = dup_count = error_count = 0

for idx, (title, href, date_str) in enumerate(active, 1):
    detail_url = href if href.startswith("http") else BASE + href
    
    cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
    if cur.fetchone():
        dup_count += 1
        continue
    
    try:
        resp = requests.get(detail_url, headers=HEADERS, timeout=20)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        error_count += 1
        if idx <= 3:
            print(f"  [WARN] Fetch error: {detail_url[:60]} -> {e}")
        continue
    
    d_title = title
    d_date = date_str
    body = ''
    
    # Parse meta tags
    mt = re.search(r'ArticleTitle"\s*content="([^"]+)"', html)
    if mt and mt.group(1).strip():
        d_title = mt.group(1).strip()
    
    md = re.search(r'PubDate"\s*content="([^"]+)"', html)
    if md:
        m = re.search(r'(\d{4}-\d{2}-\d{2})', md.group(1))
        if m:
            d_date = m.group(1)
    
    # Body in div.center-info
    ci = re.search(r'<div class="center-info"[^>]*>(.*?)</div>\s*<div class="other"', html, re.DOTALL)
    if ci:
        body = ci.group(1).strip()
    else:
        ci2 = re.search(r'<div class="center-info"[^>]*>(.*?)</div>', html, re.DOTALL)
        if ci2:
            body = ci2.group(1).strip()
    
    if not body:
        error_count += 1
        continue
    
    date_rank = int(d_date.replace("-", "")) if d_date else 0
    text_soup = BeautifulSoup(body, 'html.parser')
    summary = text_soup.get_text(strip=True)[:200]
    
    cur.execute(
        "INSERT OR IGNORE INTO gov_raw "
        "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
        "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
        (SITE_NAME, d_title, detail_url, d_date, body, date_rank, CATEGORY, summary)
    )
    if cur.rowcount > 0:
        new_count += 1
    
    conn.commit()
    if idx % 10 == 0 or idx == len(active):
        print(f"  Progress {idx}/{len(active)}: +{new_count} new, {dup_count} dup, {error_count} err")
    
    time.sleep(0.3)

conn.close()
print(f"\n[lnjp] Summary: +{new_count} new, {dup_count} dup, {error_count} err")
print(json.dumps({"site": SITE_NAME, "new": new_count, "dup": dup_count, "err": error_count}, ensure_ascii=False))
