#!/usr/bin/env python3
import os
"""
Fix eiafans.com empty content records
Login with real credentials, fetch detail pages, extract content, update search.db
"""
import requests, re, sqlite3, sys, time

# ─── Config ───
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
BASE = "http://www.eiafans.com"
CREDS = {"username": "Azure_2025", "password": "Ef6cHyRr81m!"}

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0"
}

def clean_html(html):
    """Extract meaningful text from Discuz! post content, keeping HTML structure"""
    if not html:
        return ""
    
    # Remove common empty/trash patterns
    html = re.sub(r'<script[^>]*>.*?</script>', '', html, flags=re.DOTALL)
    html = re.sub(r'<style[^>]*>.*?</style>', '', html, flags=re.DOTALL)
    html = re.sub(r'<br\s*/?>', '\n', html)
    
    # Remove Discuz! specific elements
    html = re.sub(r'<div class="pats"[^>]*>.*?</div>', '', html, flags=re.DOTALL)
    html = re.sub(r'<div class="locked"[^>]*>.*?</div>', '', html, flags=re.DOTALL)
    html = re.sub(r'<div class="mbn"[^>]*>.*?</div>', '', html, flags=re.DOTALL)
    html = re.sub(r'<div class="tip"[^>]*>.*?</div>', '', html, flags=re.DOTALL)
    
    # Remove attachment icons
    html = re.sub(r'<img[^>]*src="[^"]*attachment[^"]*"[^>]*>', '', html)
    
    # Clean up excessive whitespace
    html = re.sub(r'\n{3,}', '\n\n', html)
    
    return html.strip()

def get_post_content(session, tid, retries=3):
    """Fetch detail page and extract post content"""
    url = f"{BASE}/thread-{tid}-1-1.html"
    
    for attempt in range(retries):
        try:
            r = session.get(url, timeout=30)
            r.encoding = "gbk"
            
            # Check login state
            uid_m = re.search(r"discuz_uid\s*=\s*'(\d+)'", r.text)
            uid = uid_m.group(1) if uid_m else "0"
            
            if uid == "0":
                print(f"    ❌ 未登录状态，重新登录...")
                return None, "login_expired"
            
            # Check if paywalled
            if "购买主题" in r.text or "需向作者支付" in r.text:
                print(f"    🔒 需付费(金币)查看")
                return None, "paywalled"
            
            # Extract content from postmessage div
            # Pattern: <td class="t_f" id="postmessage_N"> ... content ... </td>
            content = None
            m = re.search(r'class="t_f"[^>]*id="postmessage_\d+"[^>]*>(.*?)</t[d]', r.text, re.DOTALL)
            if not m:
                m = re.search(r'id="postmessage_\d+"[^>]*class="t_f"[^>]*>(.*?)</t[d]', r.text, re.DOTALL)
            if not m:
                # Try just class=t_f
                m = re.search(r'class="t_f"[^>]*>(.*?)</td>', r.text, re.DOTALL)
            if not m:
                # Try the message area
                m = re.search(r'<div class="t_fsz"[^>]*>.*?<td[^>]*>(.*?)</t[d]', r.text, re.DOTALL)
            if not m:
                # Fallback: the whole post area
                m = re.search(r'<div class="pcb"[^>]*>(.*?)<div[^>]*class="notice', r.text, re.DOTALL)
            
            if m:
                content = m.group(1).strip()
            else:
                # Try to find any content at all
                # Check thread content area
                if "postmessage" in r.text:
                    idx = r.text.find("postmessage")
                    snippet = r.text[idx:idx+500]
                    print(f"    ❓ postmessage附近: {snippet[:200]}")
                else:
                    # Print page summary
                    body = re.sub(r'<[^>]+>', ' ', r.text[3000:5000])
                    body = re.sub(r'\s+', ' ', body)[:200]
                    print(f"    ❓ 页面内容: {body}")
                return None, "no_content"
            
            return content, "ok"
            
        except requests.exceptions.Timeout:
            print(f"    ⏱ 超时(重试{attempt+1}/{retries})...")
            time.sleep(3)
        except Exception as e:
            print(f"    ❌ 错误: {e}")
            time.sleep(2)
    
    return None, "timeout"

def login(session):
    """Login to eiafans.com"""
    r = session.get(f"{BASE}/forum-64-1.html", timeout=20)
    r.encoding = "gbk"
    fh_m = re.search(r"formhash=(\w+)", r.text)
    formhash = fh_m.group(1) if fh_m else ""
    
    r = session.post(
        f"{BASE}/member.php?mod=logging&action=login&loginsubmit=yes&handlekey=login&infloat=yes&formhash={formhash}",
        data={**CREDS, "cookietime": "2592000"},
        timeout=20,
    )
    r.encoding = "gbk"
    
    # Verify login
    cookies = session.cookies.get_dict()
    auth_keys = [k for k in cookies if "auth" in k.lower()]
    if auth_keys:
        print(f"  ✅ 登录成功 (auth cookie: {cookies[auth_keys[0]][:30]}...)")
        return True
    
    # Check response
    if "欢迎" in r.text:
        print("  ✅ 登录成功")
        return True
    if "密码错误" in r.text:
        print("  ❌ 密码错误")
        return False
    
    print(f"  ❓ 登录状态不确定, 响应: {r.text[:200]}")
    return False

def update_database(conn, tid, content, page_url):
    """Update gov_raw table with real content"""
    summary = content[:200].replace("'", "''") if content else ""
    content_clean = content.replace("'", "''") if content else ""
    
    sql = f"""
    UPDATE gov_raw 
    SET content = '{content_clean}',
        summary = '{summary}'
    WHERE page_url = '{page_url.replace("'", "''")}'
    """
    conn.execute(sql)
    conn.commit()

def update_fts(conn, tid, title, content, page_url):
    """Update FTS index"""
    title = title.replace("'", "''")
    content_text = re.sub(r'<[^>]+>', '', content).replace("'", "''")[:500] if content else ""
    
    # Get rowid from gov_raw
    row = conn.execute(f"SELECT rowid FROM gov_raw WHERE page_url = '{page_url.replace("'", "''")}'").fetchone()
    if row:
        rowid = row[0]
        # Delete old FTS entry and re-insert
        conn.execute(f"DELETE FROM gov_search WHERE rowid = {rowid}")
        conn.execute(f"INSERT INTO gov_search(rowid, title, site_name, summary) VALUES({rowid}, '{title}', '环评爱好者', '{content_text}')")
        conn.commit()

def main():
    print("=" * 60)
    print("环评爱好者 正文修复脚本")
    print("=" * 60)
    
    # Connect to DB
    conn = sqlite3.connect(SEARCH_DB)
    conn.execute("PRAGMA journal_mode=WAL")
    
    # Get records with empty content
    placeholder = "⚠️ 该数据来自环评爱好者论坛，详情页有反爬保护，请点击链接跳转原站查看正文内容。"
    rows = conn.execute(
        "SELECT rowid, page_url, title FROM gov_raw WHERE site_name='环评爱好者' "
        "AND (content IS NULL OR content = '' OR content = ?)",
        (placeholder,)
    ).fetchall()
    
    print(f"\n📊 待修复记录: {len(rows)} 条\n")
    
    if not rows:
        print("✅ 没有需要修复的记录")
        conn.close()
        return
    
    # Login
    session = requests.Session()
    session.headers.update(headers)
    
    print("🔑 登录...")
    if not login(session):
        print("❌ 登录失败，退出")
        conn.close()
        return
    
    # Process each record
    success = 0
    paywalled = 0
    failed = 0
    
    for i, (rowid, page_url, title) in enumerate(rows):
        tid_m = re.search(r"thread-(\d+)", page_url)
        if not tid_m:
            print(f"  [{i+1}/{len(rows)}] ⏭️ 无法提取TID: {page_url}")
            failed += 1
            continue
        
        tid = tid_m.group(1)
        print(f"  [{i+1}/{len(rows)}] TID {tid}: {title[:40]}...", end=" ")
        sys.stdout.flush()
        
        content, status = get_post_content(session, tid)
        
        if status == "ok" and content:
            # Update database
            update_database(conn, tid, content, page_url)
            update_fts(conn, tid, title, content, page_url)
            print(f"✅ 已更新 ({len(content)} 字符)")
            success += 1
        elif status == "paywalled":
            paywalled += 1
        elif status == "login_expired":
            print(f"  ⚠️ 重新登录...")
            if login(session):
                # Retry once
                content2, status2 = get_post_content(session, tid)
                if status2 == "ok" and content2:
                    update_database(conn, tid, content2, page_url)
                    update_fts(conn, tid, title, content2, page_url)
                    print(f"✅ 已更新 ({len(content2)} 字符)")
                    success += 1
                else:
                    failed += 1
            else:
                failed += 1
        else:
            failed += 1
        
        # Be polite - delay between requests
        time.sleep(1.5)
    
    conn.close()
    
    print(f"\n{'='*60}")
    print(f"📊 结果汇总:")
    print(f"  ✅ 成功更新: {success}")
    print(f"  🔒 需付费: {paywalled}")
    print(f"  ❌ 失败: {failed}")
    print(f"  📝 合计: {len(rows)}")
    print(f"{'='*60}")

if __name__ == "__main__":
    main()
