#!/usr/bin/env python3
"""
合肥经济技术开发区 - 公示公告 爬虫
hetda.hefei.gov.cn/xwzx/gsgg/index.html
WAF: 瑞数4代 JS challenge (全站 521) -> 必须 playwright chromium headless
CMS: 龙讯 Lonsun (doc_list list-6796431 + Ls.pagination + /content/column/6796431?pageIndex=N)
列表: playwright 渲染; 详情: playwright 同 context 抓取 (requests 被拦)
"""
import os, sys, re, time, json, asyncio
from urllib.parse import urljoin
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE_URL = "https://hetda.hefei.gov.cn"
LIST_URL = BASE_URL + "/xwzx/gsgg/index.html"
PAGE_URL = BASE_URL + "/content/column/6796431?pageIndex={}"
SITE_NAME = "合肥经开区-公示公告"
COLUMN_ID = 6796431
PAGE_SIZE = 20
MAX_PAGES = 25  # pageCount:25
CUTOFF = "2023-08-10"  # 3 年窗口

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"

def parse_pages_arg():
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            return int(a.split("=")[1])
        if a.isdigit():
            return int(a)
    return None

def clean_title(title):
    import html as html_mod
    title = html_mod.unescape(title or "")
    title = re.sub(r"&middot;|&nbsp;", "", title)
    title = re.sub(r"[\u200b\ufeff]", "", title)
    title = re.sub(r"^[•·]\s*", "", title)
    return title.strip()

async def fetch_list_page(page, page_num):
    """渲染列表页, 返回 (items, html)"""
    if page_num == 1:
        url = LIST_URL
    else:
        url = PAGE_URL.format(page_num)
    try:
        await page.goto(url, wait_until="domcontentloaded", timeout=45000)
        await page.wait_for_timeout(3000)
    except Exception as e:
        print(f"  [WARN] goto page {page_num} error: {e}")
        return []
    items = await page.evaluate("""() => {
        const lis = document.querySelectorAll("ul.doc_list li");
        const seen = new Set();
        const out = [];
        lis.forEach(li => {
            const a = li.querySelector("a[href*='/xwzx/gsgg/']");
            const sp = li.querySelector("span.right.date");
            if (a) {
                const href = a.href.split('?')[0];
                if (!seen.has(href)) {
                    seen.add(href);
                    out.push({
                        url: href,
                        title: (a.getAttribute('title') || a.textContent || '').trim(),
                        date: sp ? sp.textContent.trim() : ''
                    });
                }
            }
        });
        return out;
    }""")
    return items

async def fetch_detail(page, url):
    """渲染详情页, 提取 (title, date, content, attachments)"""
    try:
        await page.goto(url, wait_until="domcontentloaded", timeout=45000)
        await page.wait_for_timeout(2500)
        html = await page.content()
    except Exception as e:
        print(f"  [WARN] detail goto error: {e}")
        return None, None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # 标题: h1 / meta ArticleTitle / title
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = clean_title(h1.get_text(strip=True))
    if not title:
        m = re.search(r'<meta[^>]*name=["\']ArticleTitle["\'][^>]*content=["\']([^"\']*)["\']', html, re.I)
        if m:
            title = clean_title(m.group(1))
    if not title:
        m = re.search(r'<title>(.*?)</title>', html, re.S)
        if m:
            title = clean_title(m.group(1).split("_")[0])

    # 日期: meta PubDate / 发布时间
    date = ""
    m = re.search(r'<meta[^>]*name=["\']PubDate["\'][^>]*content=["\'](\d{4}-\d{2}-\d{2})', html, re.I)
    if m:
        date = m.group(1)
    if not date:
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
        if m:
            date = m.group(1)

    # 正文容器: 合肥经开 Lonsun 正文 = div.wzcon (分享条 weixin-share-open 在正文外, 勿误抓)
    body_el = None
    for sel in ["div.wzcon", "div.article-content", "div.content", "div.xl_text", "#zoom", "div.view", "div.newscontnet", "div.TRS_Editor"]:
        body_el = soup.select_one(sel)
        if body_el and body_el.get_text(strip=True):
            break
    if body_el is None:
        # 兜底: h1 后兄弟
        h1_el = soup.find("h1")
        if h1_el:
            parent = h1_el.find_parent("div")
            if parent:
                body_el = parent

    if body_el is None:
        return title, date, "", []

    # 附件收集 (占位符法)
    attachments = []
    body_html = str(body_el)
    # 先收集附件链接
    for a in body_el.find_all("a", href=True):
        href = a["href"].strip()
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href.lower()) or "download" in href.lower():
            full = urljoin(BASE_URL, href)
            name = a.get_text(strip=True) or href.split("/")[-1]
            attachments.append({"name": name, "url": full})

    # 构建正文: 遍历 p/table/img
    parts = []
    for el in body_el.find_all(["p", "table", "img"]):
        if el.name == "p":
            if el.find_parent("table") or el.find("table") or el.find("p"):
                continue
            # 段落内附件 a 标签 -> 绝对化
            for a in el.find_all("a", href=True):
                href = a["href"].strip()
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href.lower()) or "download" in href.lower():
                    a["href"] = urljoin(BASE_URL, href)
            txt = el.get_text(" ", strip=True)
            txt = re.sub(r"\s+", " ", txt)
            if txt:
                # 附件图标 img 已剥, 保留文本链接
                parts.append(f"<p>{txt}</p>")
        elif el.name == "table":
            parts.append(str(el))
        elif el.name == "img":
            src = el.get("src", "")
            if src and "files2/" not in src.lower():  # 跳过附件小图标
                full = urljoin(BASE_URL, src)
                alt = el.get("alt", "")
                parts.append(f'<p><img src="{full}" alt="{alt}"></p>')

    # 附件段落 (独立成段, 名称内嵌 URL)
    for att in attachments:
        parts.append(f'<p><a href="{att["url"]}">{att["name"]}</a></p>')

    content = "\n".join(parts)
    # 提取文本过短(<50)且容器无表格时, fallback 容器整体文本
    extracted_text = re.sub(r"<[^>]+>", "", content)
    extracted_text = re.sub(r"\s+", "", extracted_text)
    if (not content or len(extracted_text) < 50) and not body_el.find("table"):
        txt = body_el.get_text(" ", strip=True)
        if txt:
            content = f"<p>{txt}</p>"

    return title, date, content, attachments

async def main():
    max_pages = parse_pages_arg()
    if max_pages is None:
        max_pages = MAX_PAGES

    print(f"[AutoPg] max_pages={max_pages} (SITE={SITE_NAME})")

    from playwright.async_api import async_playwright
    all_items = []
    seen_urls = set()

    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            args=["--disable-blink-features=AutomationControlled", "--no-sandbox"]
        )
        ctx = await browser.new_context(user_agent=UA, locale="zh-CN")
        page = await ctx.new_page()

        for pg in range(1, max_pages + 1):
            items = await fetch_list_page(page, pg)
            new_items = [x for x in items if x["url"] not in seen_urls]
            for x in new_items:
                seen_urls.add(x["url"])
            all_items.extend(new_items)
            print(f"  Page {pg}: {len(items)}条(新{len(new_items)}) 累计{len(all_items)}")
            # CUTOFF
            if new_items:
                dates = [x["date"] for x in new_items if x["date"]]
                if dates and min(dates) < CUTOFF:
                    print(f"  已达 CUTOFF {CUTOFF}, 停止翻页")
                    break
            if not new_items:
                print("  无新增, 停止")
                break
            time.sleep(1)

        print(f"\n共 {len(all_items)} 条列表项, 开始抓详情...")
        results = []
        for i, item in enumerate(all_items):
            title, date, content, attachments = await fetch_detail(page, item["url"])
            if not title:
                title = item["title"]
            if not date:
                date = item["date"]
            if not content:
                print(f"  [WARN] 空正文: {title[:50]}")

            results.append({
                "site_name": SITE_NAME,
                "title": title,
                "pub_date": date,
                "content": content,
                "source_url": item["url"],
                "url": item["url"],
                "category": "公示公告",
                "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
            })
            if (i + 1) % 20 == 0:
                print(f"  [{i+1}/{len(all_items)}] 最新: {title[:40]}")
            time.sleep(0.5)

        await browser.close()

    push_to_searchdb(results, batch_label=SITE_NAME)
    print(f"\n{'='*50}")
    print(f"站点: {SITE_NAME}")
    print(f"采集条数: {len(results)}")
    dates = [r["pub_date"] for r in results if r["pub_date"]]
    print(f"日期范围: {min(dates) if dates else '-'} ~ {max(dates) if dates else '-'}")
    print(f"{'='*50}")

if __name__ == "__main__":
    asyncio.run(main())
