#!/usr/bin/env python3
"""
弘润石化（潍坊）有限责任公司 - 公示信息 (wfhrcn.com)
WAF: 宝塔防火墙 JS 挑战（Playwright 浏览器自动解密）
仅2页 ≈ 20条
"""
import sys, os, re, time, json
from datetime import datetime, timezone, timedelta
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "弘润石化（潍坊）有限责任公司"
BASE_URL = "https://wfhrcn.com"
LIST_URL = "https://wfhrcn.com/index.php?m=list&a=lists&id=hhE3TYUILt0="
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")

def crawl(incremental=False, limit=None):
    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")
    print("⏳ 启动 Playwright...")

    from playwright.sync_api import sync_playwright

    all_items = []
    seen_urls = set()

    with sync_playwright() as pw:
        browser = pw.chromium.launch(headless=True)
        page = browser.new_page(
            user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        )

        # 爬 2 页
        for pg in range(1, 3):
            url = f"{LIST_URL}&page={pg}" if pg > 1 else LIST_URL
            print(f"\n  📄 第{pg}页: ", end="", flush=True)
            page.goto(url, timeout=30000, wait_until="domcontentloaded")

            # 等待 WAF 解谜
            try:
                page.wait_for_selector('.right-side .list-info a', timeout=30000)
            except:
                print("WAF超时")
                break
            time.sleep(2)

            # 获取列表
            items = page.evaluate("""() => {
                const seen = new Set();
                return Array.from(document.querySelectorAll('.right-side .list-info a'))
                    .filter(a => { const t = a.textContent.trim(); return t.length > 5 && !seen.has(a.href) ? (seen.add(a.href), true) : false; })
                    .map(a => ({ title: a.textContent.trim(), href: a.href }));
            }""")

            if not items:
                print("0条")
                break

            page_new = 0
            for item in items:
                if item['href'] in seen_urls:
                    continue
                seen_urls.add(item['href'])
                title = item['title'].strip()
                print(f"\n    [{len(all_items)+1}] {title[:45]}...", end=" ", flush=True)

                # 详情页
                try:
                    detail_page = browser.new_page()
                    detail_page.goto(item['href'], timeout=20000, wait_until="domcontentloaded")
                    try:
                        detail_page.wait_for_selector('h2', timeout=15000)
                    except:
                        pass
                    time.sleep(1.5)

                    # 标题（净化）
                    h2 = detail_page.query_selector('h2')
                    final_title = title
                    pub_date = ""
                    if h2:
                        h2_text = h2.text_content().strip()
                        # 去掉日期
                        m = re.search(r'^(.*?)\s*(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2}:\d{2})$', h2_text)
                        if m:
                            final_title = m.group(1).strip()
                            pub_date = m.group(2)
                        # 备用 date 提取
                        if not pub_date:
                            dm = re.search(r'(\d{4}-\d{2}-\d{2})', h2_text)
                            if dm:
                                pub_date = dm.group(1)

                    # 正文 — 保留 HTML 结构
                    content = detail_page.evaluate("""() => {
                        const article = document.querySelector('#article_content');
                        if (!article) return '';
                        return article.innerHTML;
                    }""") or ""

                    if not content:
                        content = f'<p>{final_title}</p>'

                    detail_page.close()

                    summary = re.sub(r"<[^>]+>", " ", content).strip()[:300]
                    summary = re.sub(r"\s+", " ", summary) if summary else final_title
                    date_only = pub_date[:10] if pub_date else ''

                    all_items.append({
                        "site_name": SITE_NAME,
                        "title": final_title,
                        "url": item['href'],
                        "content": content,
                        "pub_date": date_only,
                        "summary": summary,
                        "tags": SITE_NAME,
                    })
                    page_new += 1
                    print(f"✅ ({(date_only + ' ' + str(len(content)) + ' chars') if date_only else str(len(content)) + ' chars'})", flush=True)

                except Exception as e:
                    print(f"❌ {str(e)[:60]}", flush=True)
                    continue

                if limit and len(all_items) >= limit:
                    break

            print(f" +{page_new}条")

            if limit and len(all_items) >= limit:
                break

        browser.close()

    if all_items:
        push_to_searchdb(all_items, "wfhr")
        print(f"\n✅ 完成! 共 {len(all_items)} 条")
    else:
        print("\n⏭ 无新数据")

if __name__ == '__main__':
    incremental = len(sys.argv) > 1 and sys.argv[1] in ('1', '--incremental')
    limit = None
    for i, a in enumerate(sys.argv):
        if a == '--limit' and i+1 < len(sys.argv):
            limit = int(sys.argv[i+1])
    t0 = time.time()
    crawl(incremental=incremental, limit=limit)
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
