#!/usr/bin/env python3
"""
彭州市人民政府 - 环境影响评价审批 爬虫
CMS: EJS (AJAX pagination via ejspagination_go)
URL: https://www.pengzhou.gov.cn/pzs/c143008/list.shtml
WAF: JS Challenge (requires Playwright with stealth)
"""

import sys
import os
import re
import json
import time
from datetime import datetime, timedelta
from urllib.parse import urljoin

# Add project root
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))

from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

# DB
DB_PATH = "/root/search.db"
import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES


def get_connection():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

SITE_NAME = "彭州市生态环境局-环评审批"
BASE_URL = "https://www.pengzhou.gov.cn/pzs/c143008/list.shtml"
THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")


def create_browser():
    """Create a Playwright browser with WAF bypass"""
    p = sync_playwright().start()
    browser = p.chromium.launch(
        headless=True,
        args=["--disable-blink-features=AutomationControlled", "--no-sandbox"]
    )
    ctx = browser.new_context(
        user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
        locale="zh-CN",
        viewport={"width": 1920, "height": 1080}
    )
    ctx.add_init_script("delete Object.getPrototypeOf(navigator).__proto__.webdriver;")
    return p, browser, ctx


def crawl_list_pages(max_pages=9):
    """Crawl list pages via AJAX pagination, return list of item dicts"""
    all_items = []
    
    p, browser, ctx = create_browser()
    page = ctx.new_page()
    
    try:
        page.goto(BASE_URL, wait_until="networkidle", timeout=30000)
        time.sleep(3)
        
        for page_num in range(1, min(max_pages, _MAX_PG or max_pages) + 1):
            print(f"[List] Fetching page {page_num}...")
            
            time.sleep(1.5)
            content = page.content()
            soup = BeautifulSoup(content, "html.parser")
            ul = soup.find("ul", class_=lambda c: c and "list" in c and "ejs-list" in c)
            
            if not ul:
                print(f"[List] Page {page_num}: no list found, stopping")
                break
            
            items_on_page = 0
            for li in ul.find_all("li"):
                a = li.find("a")
                if not a or not a.get("href"):
                    continue
                
                href = a["href"].strip()
                text = li.get_text(strip=True)
                
                # Parse date from text (format: "YYYY-MM-DDTitle")
                date_match = re.match(r'(\d{4}-\d{2}-\d{2})', text)
                date_str = date_match.group(1) if date_match else ""
                
                title = text[len(date_str):] if date_str else text
                
                all_items.append({
                    "title": title,
                    "url": href,
                    "date_str": date_str,
                })
                items_on_page += 1
            
            print(f"[List] Page {page_num}: {items_on_page} items (total: {len(all_items)})")
            
            # Filter out older than 3 years
            items_within_3y = [it for it in all_items if it["date_str"] >= THREE_YEARS_AGO]
            items_outside = len(all_items) - len(items_within_3y)
            if items_outside > 5:
                print(f"[List] {items_outside} items outside 3-year window, stopping")
                all_items = items_within_3y
                break
            
            # Click next page
            if page_num < max_pages:
                try:
                    next_btn = page.locator(f"a.pagination-num:has-text('{page_num + 1}')")
                    if next_btn.count() > 0:
                        next_btn.first.click()
                        time.sleep(2)
                    else:
                        print(f"[List] No more pages, stopping")
                        break
                except Exception as e:
                    print(f"[List] Click error on page {page_num}: {e}")
                    break
    
    except Exception as e:
        print(f"[List] Error: {e}")
    finally:
        browser.close()
        p.stop()
    
    # Final 3-year filter
    all_items = [it for it in all_items if it["date_str"] >= THREE_YEARS_AGO]
    print(f"[List] Total items within 3 years: {len(all_items)}")
    return all_items


def crawl_detail(item, page):
    """Fetch a detail page and extract content."""
    url = item["url"]
    
    try:
        page.goto(url, wait_until="networkidle", timeout=30000)
        time.sleep(1)
        
        content = page.content()
        soup = BeautifulSoup(content, "html.parser")
        
        # Title from meta
        title = item["title"]
        mt = soup.find("meta", attrs={"name": "ArticleTitle"})
        if mt and mt.get("content"):
            title = mt["content"].strip()
        
        # Date from meta
        date_str = item["date_str"]
        pd = soup.find("meta", attrs={"name": "PubDate"})
        if pd and pd.get("content"):
            date_str = pd["content"].strip()[:10]
        
        # Content from article__content div
        content_div = soup.find("div", id="article__content") or soup.find("div", class_="article__content")
        body_html = ""
        if content_div:
            body_html = str(content_div)
        else:
            wrapper = soup.find("div", class_=lambda c: c and "wrapper-detail" in str(c))
            if wrapper:
                body_html = str(wrapper)
        
        # Source URL
        source_url = page.url
        
        # Generate clean summary
        summary_text = ""
        if body_html:
            txt_soup = BeautifulSoup(body_html, "html.parser")
            summary_text = txt_soup.get_text(separator=" ", strip=True)[:300]
        
        return {
            "site_name": SITE_NAME,
            "title": title,
            "page_url": source_url,
            "publish_date": date_str,
            "source_url": source_url,
            "content": body_html,
            "summary": summary_text,
        }
    
    except Exception as e:
        print(f"[Detail] Error {url}: {e}")
        return None


def save_to_db(records):
    """Save records to gov_raw table"""
    conn = get_connection()
    cursor = conn.cursor()
    
    inserted = 0
    for rec in records:
        try:
            cursor.execute("""
                INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (
                rec["site_name"], rec["title"],
                rec["page_url"], rec["publish_date"],
                rec["source_url"], rec["content"],
                rec.get("summary", "")
            ))
            if cursor.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"[DB] Error inserting {rec.get('title', '?')}: {e}")
    
    conn.commit()
    conn.close()
    return inserted


def main():
    print(f"=== 彭州市生态环境局-环评审批 爬虫 ===")
    print(f"3年截止日期: {THREE_YEARS_AGO}")
    
    # Step 1: Crawl list pages
    print("\n--- Step 1: Crawl list pages ---")
    items = crawl_list_pages(max_pages=9)
    print(f"Total items from list: {len(items)}")
    
    if not items:
        print("No items found, exiting")
        return
    
    # Step 2: Crawl detail pages
    print(f"\n--- Step 2: Crawl {len(items)} detail pages ---")
    
    p, browser, ctx = create_browser()
    page = ctx.new_page()
    
    records = []
    errors = 0
    for i, item in enumerate(items):
        rec = crawl_detail(item, page)
        if rec:
            records.append(rec)
            if (i + 1) % 10 == 0:
                print(f"[Detail] {i+1}/{len(items)} done")
        else:
            errors += 1
    
    browser.close()
    p.stop()
    
    print(f"\nDetail crawl complete: {len(records)} success, {errors} errors")
    
    # Step 3: Save to DB
    print("\n--- Step 3: Save to DB ---")
    inserted = save_to_db(records)
    print(f"Inserted {inserted} new records (total fetched: {len(records)})")
    
    print("\n=== Done ===")


if __name__ == "__main__":
    main()
