#!/usr/bin/env python3
"""
crawl_hualunchem.py — 华伦新材料 - 公示公告 (万网旺 CMS)
List: static HTML (page 1) + AJAX API for remaining pages
Detail: Body.js JS-rendered content extraction
"""
import re, sys, json, time, os, urllib.parse, warnings
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests
from bs4 import BeautifulSoup

BASE_URL = "https://www.hualunchem.com"
API_URL = f"{BASE_URL}/Designer/Common/GetData"
SITE_NAME = "华伦新材料-公示公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
}

from datetime import datetime, timedelta
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def unescape_js_str(s):
    """Unescape JavaScript string escape sequences"""
    result = []
    i = 0
    while i < len(s):
        if s[i] == '\\' and i + 1 < len(s):
            c = s[i+1]
            if c == 'n':
                result.append('\n')
                i += 2
            elif c == 'r':
                result.append('\r')
                i += 2
            elif c == 't':
                result.append('\t')
                i += 2
            elif c == '"' or c == "'" or c == '\\' or c == '/':
                result.append(c)
                i += 2
            elif c == 'u' and i + 5 < len(s):
                hex_str = s[i+2:i+6]
                try:
                    result.append(chr(int(hex_str, 16)))
                except:
                    result.append('\\' + c + hex_str)
                i += 6
            else:
                result.append(s[i])
                i += 1
        else:
            result.append(s[i])
            i += 1
    return ''.join(result)

def get_verification_token():
    """Get __RequestVerificationToken from the list page"""
    try:
        r = requests.get(f"{BASE_URL}/NewsInfoCategory?categoryId=398052", 
                        headers=HEADERS, timeout=15, verify=False)
        r.encoding = 'utf-8'
        m = re.search(r"name='__RequestVerificationToken'[^>]*value='([^']+)'", r.text)
        return m.group(1) if m else ''
    except:
        return ''

def fetch_list_items():
    """Fetch all list items from page 1 (HTML) + page 2 (AJAX)"""
    all_items = []
    seen_urls = set()
    
    # Page 1: Parse HTML
    try:
        r = requests.get(f"{BASE_URL}/NewsInfoCategory?categoryId=398052", 
                        headers=HEADERS, timeout=15, verify=False)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'html.parser')
        for li in soup.find_all('li', class_='w-list-item'):
            a_tag = li.find('a', class_='w-list-titlelink', href=True)
            if not a_tag:
                continue
            url = a_tag['href']
            if not url.startswith('/newsinfo/'):
                continue
            full_url = urllib.parse.urljoin(BASE_URL, url)
            title = a_tag.get_text(strip=True)
            date_span = li.find('span', class_='w-list-date')
            date_str = date_span.get_text(strip=True) if date_span else ''
            if full_url not in seen_urls:
                seen_urls.add(full_url)
                all_items.append((title, full_url, date_str))
    except Exception as e:
        print(f"  [ERROR] fetch page 1: {e}")
    
    # Page 2+: AJAX API
    token = get_verification_token()
    page_idx = 1  # 0-based index for page 2
    while True:
        try:
            data = {
                'dataType': 'news',
                'key': '',
                'pageIndex': str(page_idx),
                'pageSize': '5',
                'selectCategory': '398052',
                'selectId': '',
                'dateFormater': 'yyyy-MM-dd',
                'orderByField': 'createtime',
                'orderByType': 'desc',
                'templateId': '0',
                'postData': '{}',
                'es': 'false',
                'setTop': 'true',
                '__RequestVerificationToken': token
            }
            r = requests.post(API_URL, data=data, headers=HEADERS, 
                            timeout=15, verify=False)
            resp = r.json()
            if not resp.get('IsSuccess'):
                break
            data_items = resp.get('Data', [])
            if not data_items:
                break
            for item in data_items:
                url = item.get('LinkUrl', '')
                if not url or not url.startswith('/newsinfo/'):
                    continue
                full_url = urllib.parse.urljoin(BASE_URL, url)
                if full_url not in seen_urls:
                    seen_urls.add(full_url)
                    title = item.get('Name', '')
                    date_str = item.get('QTime', '')
                    all_items.append((title, full_url, date_str))
            page_idx += 1
        except Exception as e:
            print(f"  [ERROR] AJAX page {page_idx+1}: {e}")
            break
    
    # Filter by 3-year cutoff
    filtered = [(t, u, d) for t, u, d in all_items if d and d >= CUTOFF]
    return filtered

def fetch_detail(url):
    """Fetch detail page via Body.js extraction"""
    try:
        # Step 1: Get the page to find Body.js URL
        r = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        r.encoding = 'utf-8'
        
        # Step 2: Find the Body.js script src
        m = re.search(r"src='([^']+nocookie\.html\.Body\.js[^']*)'", r.text)
        if not m:
            m = re.search(r'src="([^"]+nocookie\.html\.Body\.js[^"]*)"', r.text)
        if not m:
            return ("", "", "", "")
        
        body_js_url = m.group(1)
        if body_js_url.startswith('//'):
            body_js_url = 'https:' + body_js_url
        
        # Step 3: Fetch Body.js with Referer
        headers = HEADERS.copy()
        headers['Referer'] = url
        r2 = requests.get(body_js_url, headers=headers, timeout=15, verify=False)
        r2.encoding = 'utf-8'
        
        # Step 4: Extract document.write content
        m2 = re.search(r"document\.write\('((?:[^'\\]|\\.)*)'\)", r2.text, re.S)
        if not m2:
            return ("", "", "", "")
        
        # Step 5: Unescape JS string → HTML
        html = unescape_js_str(m2.group(1))
        
        # Step 6: Parse with BeautifulSoup
        soup = BeautifulSoup(html, 'html.parser')
        
        title = ""
        h1 = soup.find('h1', class_='w-title')
        if h1:
            title = h1.get_text(strip=True)
        
        content = ""
        detail = soup.find('div', class_='w-detail')
        if detail:
            content = str(detail)
        
        # Date from pageinfo may not be reliable; use list date instead
        
        return (content, title, '', '')
        
    except Exception as e:
        print(f"  [ERROR] detail {url}: {e}")
        return ("", "", "", "")

def push_to_searchdb(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    cur = conn.cursor()
    new_count = 0
    for title, url, pub_date, source, content in items:
        try:
            cur.execute("""
                INSERT OR IGNORE INTO gov_raw 
                    (site_name, title, page_url, publish_date, source_url, content, category)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (SITE_NAME, title, url, pub_date, source, content, '公示公告'))
            if cur.rowcount > 0:
                new_count += 1
        except Exception as e:
            print(f"  [DB ERROR] {url}: {e}")
    conn.commit()
    conn.close()
    return new_count

def main():
    warnings.filterwarnings("ignore", message="Unverified HTTPS request")
    print(f"[{datetime.now().strftime('%H:%M:%S')}] {SITE_NAME}")
    
    # Step 1: Get all list items
    items = fetch_list_items()
    print(f"  Items within 3 years: {len(items)}")
    for t, u, d in items:
        print(f"    [{d}] {t[:50]}")
    
    if not items:
        print("  No items found!")
        return 0
    
    # Step 2: Fetch details concurrently
    details = []
    with ThreadPoolExecutor(max_workers=5) as executor:
        future_map = {executor.submit(fetch_detail, url): (title, url, date_str) 
                      for title, url, date_str in items}
        for i, future in enumerate(as_completed(future_map), 1):
            title, url, date_str = future_map[future]
            try:
                content, det_title, pub_date, source = future.result()
                final_title = det_title or title
                final_date = pub_date or date_str
                if content:
                    details.append((final_title, url, final_date, source, content))
                    print(f"  [{i}/{len(items)}] {final_title[:50]}... OK ({len(content)}B)")
                else:
                    print(f"  [{i}/{len(items)}] {final_title[:50]}... no content")
            except Exception as e:
                print(f"  [{i}/{len(items)}] {title[:40]}... ERROR: {e}")
    
    # Step 3: Insert into DB
    if details:
        new_count = push_to_searchdb(details)
        print(f"\n  === Done === New: {new_count}, Errors: 0")
    else:
        print(f"\n  === Done === No items")
    
    return len(details)

if __name__ == '__main__':
    main()
