#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""抓详情页存HTML, 提取NewsContent原始结构"""
import re, time
from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

IP = "171.221.172.137"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
DURL = "http://www.pujiang.gov.cn/pjxrmzf/c160163/2026-08/17/content_8ce7c1a807084350a7dd43bcd43fdd70.shtml"

with sync_playwright() as p:
    browser = p.chromium.launch(headless=True, args=[
        "--disable-blink-features=AutomationControlled", "--no-sandbox",
        f"--host-resolver-rules=MAP www.pujiang.gov.cn {IP}, MAP pujiang.gov.cn {IP}",
    ])
    ctx = browser.new_context(user_agent=UA, locale="zh-CN", viewport={"width": 1920, "height": 1080})
    page = ctx.new_page()
    try:
        page.goto(DURL, timeout=10000, wait_until="domcontentloaded")
    except Exception:
        pass
    for i in range(6):
        time.sleep(2)
        try:
            html = page.content()
        except Exception:
            continue
        if "_ts" not in html and len(html) > 5000:
            break
    soup = BeautifulSoup(html, "html.parser")
    nc = soup.find("div", id="NewsContent")
    print("NewsContent 存在:", nc is not None)
    if nc:
        print("NewsContent 原始HTML长度:", len(str(nc)))
        # 统计内部标签
        from collections import Counter
        tags = Counter(t.name for t in nc.find_all(True))
        print("标签分布:", dict(tags.most_common(10)))
        imgs = nc.find_all("img")
        print("img数:", len(imgs))
        tables = nc.find_all("table")
        print("table数:", len(tables))
        # 文本
        txt = nc.get_text(" ", strip=True)
        print("纯文本:", txt[:300])
    browser.close()
