#!/usr/bin/env python3
"""Crawl haihua.com.cn enterprise announcements (企业公告)"""
import asyncio, re, sys, os, sqlite3, traceback
from datetime import datetime
from playwright.async_api import async_playwright

DB_PATH = "/root/search.db"
SITE_NAME = "haihua"
BASE_URL = "https://www.haihua.com.cn"

async def parse_list_page(browser, page_num):
    """Get all news_detail URLs from a list page"""
    page = await browser.new_page(user_agent="Mozilla/5.0")
    url = f"{BASE_URL}/news_list1/{page_num}.html"
    await page.goto(url, timeout=30000, wait_until="domcontentloaded")
    await page.wait_for_timeout(5000)
    
    links = await page.query_selector_all('a[href*="news_detail"]')
    results = []
    for link in links:
        href = await link.get_attribute("href")
        text = await link.inner_text()
        text = text.strip()
        if href and text:
            full_url = href if href.startswith("http") else BASE_URL + href
            results.append((full_url, text))
    
    await page.close()
    return results

async def parse_detail_page(browser, detail_url):
    """Get title, date, content from a detail page"""
    page = await browser.new_page(user_agent="Mozilla/5.0")
    await page.goto(detail_url, timeout=30000, wait_until="domcontentloaded")
    await page.wait_for_timeout(5000)
    
    text = await page.inner_text("body")
    
    # Extract title - first h1 or the first significant text
    title = ""
    h1 = await page.query_selector("h1")
    if h1:
        title = (await h1.inner_text()).strip()
    else:
        # Find the first large text block
        lines = [l.strip() for l in text.split("\n") if l.strip() and len(l.strip()) > 10]
        title = lines[0] if lines else ""
    
    # Extract dates
    dates = re.findall(r"(\d{4}-\d{2}-\d{2})", text)
    publish_date = dates[0] if dates else ""
    
    # Extract content - look for the main article body
    # Try to find the content container
    content_parts = []
    # Look for content after the date
    if dates:
        date_pos = text.find(dates[0])
        if date_pos > 0:
            # Get text after the date
            after_date = text[date_pos + 10:]
            # Clean up
            lines = [l.strip() for l in after_date.split("\n") if l.strip() and len(l.strip()) > 20]
            # Filter out navigation/footer text
            skip_words = ["网站首页","关于海化","资讯中心","产品营销","企业文化","人才招聘",
                         "联系我们","集团新闻","媒体聚焦","视频资讯","企业公告","Copyright",
                         "鲁ICP","营业执照","网站建设","中企动力","扫一扫","版权所有"]
            for l in lines:
                if not any(kw in l for kw in skip_words):
                    content_parts.append(l)
    
    content = "\n".join(content_parts[:20])
    
    await page.close()
    return {
        "title": title,
        "publish_date": publish_date,
        "content": content,
        "source_url": detail_url,
    }

def save_to_db(records):
    """Insert records into search.db"""
    if not records:
        print("No records to save")
        return 0
    
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    
    count = 0
    for r in records:
        # Check existing by source_url
        existing = c.execute("SELECT id FROM gov_raw WHERE source_url=?", (r["source_url"],)).fetchone()
        if existing:
            print(f"  SKIP (exists): {r['title'][:40]}")
            continue
        
        c.execute("""
            INSERT INTO gov_raw (site_name, source_url, page_url, title, publish_date, date_rank, summary, content)
            VALUES (?, ?, ?, ?, ?, ?, ?, ?)
        """, (
            SITE_NAME,
            r["source_url"],
            r["source_url"],
            r["title"],
            r["publish_date"],
            int(datetime.strptime(r["publish_date"], "%Y-%m-%d").timestamp()) if r["publish_date"] else 0,
            r["content"][:200] if r["content"] else "",
            r["content"],
        ))
        conn.commit()
        count += 1
        print(f"  OK ({count}): {r['title'][:50]}")
    
    conn.close()
    return count

async def main():
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=["--no-sandbox"])
        
        all_items = []
        for page_num in [2]:  # Only page 2 has 企业公告
            items = await parse_list_page(browser, page_num)
            print(f"Page {page_num}: {len(items)} items")
            all_items.extend(items)
        
        print(f"\nTotal items found: {len(all_items)}")
        
        records = []
        for url, title in all_items:
            print(f"\nFetching: {title[:50]}...")
            try:
                detail = await parse_detail_page(browser, url)
                detail["title"] = title  # Use the list page title (more accurate)
                detail["source_url"] = url
                records.append(detail)
            except Exception as e:
                print(f"  ERROR: {e}")
                traceback.print_exc()
        
        await browser.close()
        
        # Save to DB
        saved = save_to_db(records)
        print(f"\n=== Complete: {saved}/{len(all_items)} new records ===")

asyncio.run(main())
