#!/usr/bin/env python3
"""
荣盛石化-新闻专区 (www.cnrspc.com/xwzq)
万网/阿里云建站 CMS，需 Playwright 渲染
列表: <a href="/newsinfo/XXX"> 标题
日期: .w-al-date
详情: .w-detail 正文, .w-createtime-date 日期
分页: <a href="javascript:void(0);">N</a> 点击翻页，共~10页
"""
import sys, os, re, time, json
from datetime import datetime, timezone, timedelta
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "荣盛石化-新闻专区"
LIST_URL = "https://www.cnrspc.com/xwzq"
DOMAIN = "www.cnrspc.com"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 4  # 近3年只需前4页

def crawl(incremental=False, limit=None):
    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")
    print("⏳ 启动 Playwright...")

    from playwright.sync_api import sync_playwright

    all_items = []
    seen_urls = set()

    with sync_playwright() as pw:
        browser = pw.chromium.launch(headless=True)
        page = browser.new_page()
        page.goto(LIST_URL, wait_until="networkidle", timeout=30000)
        time.sleep(3)

        total_pages = MAX_PAGES
        stop = False

        for pg in range(1, total_pages + 1):
            if stop:
                break
            print(f"\n  📄 第{pg}页: ", end="", flush=True)

            # 提取当前页列表
            items = page.evaluate("""() => {
                const results = [];
                const links = document.querySelectorAll('a[href*="/newsinfo/"]');
                const seen = new Set();
                for (const a of links) {
                    const href = a.href;
                    const title = a.textContent.trim();
                    if (!title || title.length < 5 || seen.has(href)) continue;
                    seen.add(href);
                    const itemEl = a.closest('li, div');
                    const dateEl = itemEl ? itemEl.querySelector('.w-al-date') : null;
                    const date = dateEl ? dateEl.textContent.trim() : '';
                    results.push({href, title, date});
                }
                return results;
            }""")

            # 日期过滤 & 去重
            page_new = 0
            for item in items:
                if item['href'] in seen_urls:
                    continue
                if item['date'] and item['date'] < THREE_YEARS_AGO:
                    stop = True
                    break

                seen_urls.add(item['href'])
                title = item['title'].strip()
                print(f"\n    [{len(all_items)+1}] {title[:45]}...", end=" ", flush=True)

                # 打开详情页
                try:
                    detail_page = browser.new_page()
                    detail_page.goto(item['href'], wait_until="networkidle", timeout=30000)
                    time.sleep(2)

                    # 标题 —— 从 <title> 获取净化
                    page_title = detail_page.title()
                    final_title = re.sub(r'\s*[-—|]\s*荣盛石化股份有限公司.*', '', page_title).strip() if page_title else title

                    # 日期
                    pub_date = item['date']
                    try:
                        date_el = detail_page.query_selector('.w-createtime-date')
                        time_el = detail_page.query_selector('.w-createtime-time')
                        if date_el:
                            d = date_el.text_content().strip()
                            t = time_el.text_content().strip() if time_el else ''
                            pub_date = f"{d} {t}".strip()
                    except:
                        pass

                    # 正文
                    content = ''
                    try:
                        detail_el = detail_page.query_selector('.w-detail')
                        if detail_el:
                            html = detail_el.inner_html()
                            # 补全图片src为绝对路径
                            html = re.sub(r'src="(?!https?://)(/[^"]+)"', rf'src="https://{DOMAIN}\1"', html)
                            html = re.sub(r'src="(?!https?://)([^/][^"]*)"', '', html)
                            html = re.sub(r' style="[^"]*"', '', html)
                            html = re.sub(r'\s+height="[^"]*"', '', html)
                            html = re.sub(r' width="100%"', ' style="width:100%;height:auto;display:block"', html)
                            content = html.strip()
                    except:
                        pass

                    detail_page.close()

                    if not content:
                        content = f'<p>{title}</p>'

                    summary = re.sub(r"<[^>]+>", " ", content).strip()[:300]
                    summary = re.sub(r"\s+", " ", summary) if summary else final_title

                    all_items.append({
                        "site_name": SITE_NAME,
                        "title": final_title,
                        "url": item['href'],
                        "content": content,
                        "pub_date": pub_date[:10] if pub_date else '',
                        "summary": summary,
                        "tags": SITE_NAME,
                    })
                    page_new += 1
                    print(f"✅ ({len(content)} chars)", flush=True)

                except Exception as e:
                    print(f"❌ {str(e)[:60]}", flush=True)
                    continue

                if limit and len(all_items) >= limit:
                    stop = True
                    break

            print(f" +{page_new}条", end="", flush=True)

            if stop:
                break

            # 翻页
            if pg < total_pages:
                next_pn = pg + 1
                clicked = page.evaluate(f"""pn => {{
                    const links = document.querySelectorAll('a');
                    for (const a of links) {{
                        if (a.textContent.trim() === String(pn) && a.href === 'javascript:void(0);') {{
                            a.click();
                            return true;
                        }}
                    }}
                    return false;
                }}""", next_pn)
                if not clicked:
                    print(" (无下一页)")
                    break
                time.sleep(3)

        browser.close()

    if all_items:
        push_to_searchdb(all_items, "cnrspc")
        print(f"\n✅ 完成! 共 {len(all_items)} 条")
    else:
        print("\n⏭ 无新数据")

if __name__ == '__main__':
    incremental = len(sys.argv) > 1 and sys.argv[1] in ('1', '--incremental')
    limit = None
    for i, a in enumerate(sys.argv):
        if a == '--limit' and i+1 < len(sys.argv):
            limit = int(sys.argv[i+1])
    t0 = time.time()
    crawl(incremental=incremental, limit=limit)
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
