#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_fengjie.py — 奉节网-要闻
http://www.xfjw.net/newspage/news-list/?name=要闻&columnId=30013
CMS: Nuxt.js SSR + Playwright detail rendering
"""
import re
import sys
import os
import time
import json
import subprocess
from bs4 import BeautifulSoup
import requests

SITE_NAME = "奉节网-要闻"
BASE = "http://www.xfjw.net"
LIST_PATH = "/newspage/news-list/"
COLUMN_ID = "30013"
SEARCH_DB = "/mnt/data/search.db"
GROUP = "重庆"

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
})

def fetch_list(page):
    params = {"name": "要闻", "columnId": COLUMN_ID, "currentNav": "1"}
    if page > 1:
        params["page"] = str(page)
    r = session.get(f"{BASE}{LIST_PATH}", params=params, timeout=30)
    r.encoding = "utf-8"
    return r.text

def parse_list(html):
    """Extract articles from SSR HTML"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    
    # Find the newsitem wrapper div
    newsitem = soup.find("div", class_="newsitem")
    if not newsitem:
        print("  [ERROR] No newsitem div found!", file=sys.stderr)
        return items
    
    # Find all <a> tags with detail links inside newsitem
    all_links = newsitem.find_all("a", href=re.compile(r"infoClassifyId"))
    print(f"  [Debug] Links found: {len(all_links)}", file=sys.stderr)
    
    seen_urls = set()
    for a in all_links:
        href = a.get("href", "").strip()
        if not href or href in seen_urls:
            continue
        if "infoClassifyId" not in href:
            continue
        seen_urls.add(href)
        
        text = a.get_text(strip=True)
        # Extract date from end of text: "标题2026-07-24"
        m = re.search(r'(\d{4}-\d{2}-\d{2})$', text)
        pub_date = m.group(1) if m else ""
        title = text[:m.start()].strip() if m else text
        
        items.append({
            "url": href,
            "title": title.strip(),
            "pub_date": pub_date,
        })
    
    return items

def get_total_pages(html):
    soup = BeautifulSoup(html, "html.parser")
    page_items = soup.find_all(class_="ivu-page-item")
    nums = []
    for el in page_items:
        t = el.get_text(strip=True)
        if t.isdigit():
            nums.append(int(t))
    return max(nums) if nums else 1

def parse_detail_playwright(url):
    """Use Playwright to render detail page and extract content"""
    from playwright.sync_api import sync_playwright
    
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--no-sandbox", "--disable-gpu"])
        page = browser.new_page()
        page.set_default_timeout(15000)
        
        try:
            page.goto(url, wait_until="networkidle", timeout=20000)
            
            # Get title
            title = page.title()
            
            # Get date from meta or page
            m = re.search(r'(\d{4}-\d{2}-\d{2}\s\d{2}:\d{2}:\d{2})', page.inner_text("body"))
            pub_date = m.group(1) if m else ""
            
            # Get content - try common content divs
            content = ""
            for sel in ["[class*=content] > div", ".detail-content", "article", 
                        "[class*=detail]", ".news-content", "#content"]:
                try:
                    el = page.query_selector(sel)
                    if el:
                        txt = el.inner_html()
                        if len(txt) > 200:
                            content = txt
                            break
                except:
                    pass
            
            if not content:
                # Use full body minus header/footer
                body = page.eval_on_selector("body", "el => el.innerHTML")
                # Strip script/style
                body = re.sub(r'<script[^>]*>.*?</script>', '', body, flags=re.DOTALL)
                body = re.sub(r'<style[^>]*>.*?</style>', '', body, flags=re.DOTALL)
                content = body
            
            summary = re.sub(r'<[^>]+>', '', content)[:200] if content else ""
            
            return {
                "title": title,
                "pub_date": pub_date.split(" ")[0] if pub_date else "",
                "content": content,
                "summary": summary,
            }
        except Exception as e:
            print(f"  PW Error: {e}")
            return {"title": "", "pub_date": "", "content": "", "summary": ""}
        finally:
            browser.close()

def push_to_searchdb(items, batch_label):
    sql_parts = []
    for item in items:
        t = item["title"].replace("'", "''")
        site = SITE_NAME.replace("'", "''")
        pu = (item.get("url") or "").replace("'", "''")
        pd = (item.get("pub_date") or "").replace("'", "''")
        s = (item.get("summary") or "").replace("'", "''")
        c = (item.get("content") or "").replace("'", "''")
        sn = "fengjie"
        
        sql = (
            f"INSERT OR IGNORE INTO gov_raw"
            f"(site_name, source_url, page_url, title, publish_date, summary, content, status, script_name) "
            f"VALUES('{site}','{pu}','{pu}','{t}','{pd}','{s}','{c}','active','{sn}');\n"
        )
        sql_parts.append(sql)
    
    full_sql = "BEGIN;\n" + "".join(sql_parts) + "COMMIT;\n"
    r = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", SEARCH_DB], input=full_sql, capture_output=True, text=True, timeout=60)
    return len(sql_parts), r.stderr

def sync_fts(urls):
    for u in urls:
        ue = u.replace("'", "''")
        sql = (
            f"INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) "
            f"SELECT rowid, title, site_name, summary FROM gov_raw "
            f"WHERE page_url='{ue}' AND rowid NOT IN (SELECT rowid FROM gov_search);\n"
        )
        subprocess.run(["sqlite3", "-cmd", ".timeout 60000", SEARCH_DB], input=sql, capture_output=True, text=True, timeout=30)

def main():
    max_pages = 3
    if "--pages" in sys.argv:
        idx = sys.argv.index("--pages")
        if idx + 1 < len(sys.argv):
            max_pages = int(sys.argv[idx + 1])
    
    incremental = "--incremental" in sys.argv
    skip_details = "--no-detail" in sys.argv  # for testing list only
    
    print(f"爬取: {SITE_NAME}")
    
    # Fetch page 1
    html = fetch_list(1)
    total_pages = get_total_pages(html)
    print(f"总页数: {total_pages}")
    
    if max_pages and max_pages < total_pages:
        total_pages = max_pages
    
    all_items = []
    for page in range(1, total_pages + 1):
        print(f"  列表页 {page}/{total_pages}...", end=" ", flush=True)
        html = fetch_list(page)
        items = parse_list(html)
        all_items.extend(items)
        print(f"{len(items)} 条")
        time.sleep(0.3)
    
    print(f"\n共获取: {len(all_items)} 条")
    
    if incremental:
        r = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", SEARCH_DB], 
            input=f"SELECT page_url FROM gov_raw WHERE site_name='{SITE_NAME}';",
            capture_output=True, text=True, timeout=30)
        existing = set(r.stdout.strip().split("\n")) if r.stdout.strip() else set()
        all_items = [it for it in all_items if it["url"] not in existing]
        print(f"增量过滤后: {len(all_items)} 条")
    
    if skip_details:
        print("跳过详情 (--no-detail)")
        n, err = push_to_searchdb(all_items, SITE_NAME)
        print(f"入库: {n} 条 (仅列表数据)")
        return
    
    # Crawl details with Playwright
    new_count = 0
    batch = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:30]}...", end=" ", flush=True)
        try:
            detail = parse_detail_playwright(item["url"])
            item["pub_date"] = item["pub_date"] or detail["pub_date"]
            item["content"] = detail["content"]
            item["summary"] = detail["summary"][:200]
            batch.append(item)
            print(f"✅ ({len(detail.get('content',''))}b)")
        except Exception as e:
            print(f"❌ {e}")
        
        if len(batch) >= 10:
            n, err = push_to_searchdb(batch, SITE_NAME)
            new_count += n
            sync_fts([it["url"] for it in batch])
            if err: print(f"  DB Error: {err[:100]}")
            batch = []
    
    if batch:
        n, err = push_to_searchdb(batch, SITE_NAME)
        new_count += n
        sync_fts([it["url"] for it in batch])
        if err: print(f"  DB Error: {err[:100]}")
    
    print(f"\n✅ 完成！新增入库: {new_count} 条")

if __name__ == "__main__":
    main()
