#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""check_yingquan.py —— step4 数据质量细查"""
import json
import re

rows = [json.loads(l) for l in open("/tmp/yingquan_p1.jsonl", encoding="utf-8")]
print("条数:", len(rows))
print("标题长度 min/max:", min(len(r["title"]) for r in rows), "/", max(len(r["title"]) for r in rows))
print("省略号截断标题:", [r["title"] for r in rows if "…" in r["title"] or r["title"].endswith("...")])
print("实体残留:", [r["title"][:22] for r in rows if re.search(r"&(?:amp|#\d+|nbsp)", r["content"])])
print("空正文:", sum(1 for r in rows if len(re.sub(r"<[^>]+>", "", r["content"]).strip()) < 10))
print("含 <p>:", sum(1 for r in rows if "<p" in r["content"]), "/", len(rows))
print("含 <table>:", sum(1 for r in rows if "<table" in r["content"]))
print("有附件:", sum(1 for r in rows if r["attachments"]))
print("正文长度 min/avg/max:", min(len(r["content"]) for r in rows),
      int(sum(len(r["content"]) for r in rows) / len(rows)), max(len(r["content"]) for r in rows))
print()

print("=== 附件 URL 绝对性 + 内嵌成段检查 ===")
bad = 0
for r in rows:
    if not r["attachments"]:
        continue
    for a in json.loads(r["attachments"]):
        u = a["url"]
        if not u.startswith("http"):
            bad += 1
            print("  ❌ 非绝对:", u[:80])
        # 正文里必须有对应 <a>
        if u not in r["content"]:
            bad += 1
            print("  ❌ 正文未内嵌:", u[:70])
    if bad == 0:
        break
print("  非绝对/未内嵌 问题数(前三例内):", bad)
print("  附件样例:")
for r in rows:
    if r["attachments"]:
        for a in json.loads(r["attachments"])[:2]:
            print("    ", a["title"][:60], "→", a["url"][:70])
        break

print("\n=== 污染词扫描 ===")
SUS = ["扫一扫", "顶部", "打印", "关闭", "j-goTop", "qrcode", "党委群体", "政府部门", "网站地图",
       "主办单位", "版权所有", "ICP备", "颍泉先锋网", "清风颍泉", "新浪微博", "无障碍", "字号",
       "上一篇", "下一篇", "当前位置", "网站首页", "m-dtdownload", "pagination"]
for s in SUS:
    hit = [r["title"][:18] for r in rows if s in r["content"]]
    if hit:
        print("  ⚠️ %-12s 命中 %d: %s" % (s, len(hit), hit[:2]))
print("  扫描完毕")

print("\n=== 首条正文全文 ===")
print(rows[0]["content"][:900])
print("\n--- 末 300 字（看有无页脚残留）---")
print(rows[0]["content"][-300:])
