#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""check_jcx.py —— step4 数据质量细查"""
import json
import re

rows = [json.loads(l) for l in open("/tmp/jcx_p1.jsonl", encoding="utf-8")]
print("条数:", len(rows))
print("标题长度 min/max:", min(len(r["title"]) for r in rows), "/", max(len(r["title"]) for r in rows))
print("省略号截断:", [r["title"] for r in rows if "…" in r["title"] or r["title"].endswith("...")])
print("实体残留:", [r["title"][:22] for r in rows if re.search(r"&(?:amp|#\d+|nbsp)", r["content"])])
print("空正文:", sum(1 for r in rows if len(re.sub(r"<[^>]+>", "", r["content"]).strip()) < 10))
print("含 <p>:", sum(1 for r in rows if "<p" in r["content"]), "/", len(rows))
print("含 <table>:", sum(1 for r in rows if "<table" in r["content"]))
print("有附件:", sum(1 for r in rows if r["attachments"]))
print("正文长度 min/avg/max:", min(len(r["content"]) for r in rows),
      int(sum(len(r["content"]) for r in rows) / len(rows)), max(len(r["content"]) for r in rows))
print()
print("=== 附件绝对化 + 内嵌 + 名称质量 ===")
prob = 0
for r in rows:
    for a in (json.loads(r["attachments"]) if r["attachments"] else []):
        if not a["url"].startswith("http"):
            print("  ❌ 非绝对:", a["url"][:70]); prob += 1
        if a["url"] not in r["content"]:
            print("  ❌ 未内嵌:", a["url"][:70]); prob += 1
        if not a["title"] or a["title"] in ("附件",):
            print("  ⚠️ 附件名为空/占位:", r["title"][:30], "|", a["url"][:50]); prob += 1
print("  问题数:", prob)
print("  附件样例:")
for r in rows:
    if r["attachments"]:
        for a in json.loads(r["attachments"])[:3]:
            print("    ", a["title"][:60], "→", a["url"][:66])
        break
print()
print("=== 污染词扫描 ===")
SUS = ["已下载", "下一条", "上一条", "nextList", "打印本页", "返回顶部", "扫一扫", "字号",
       "分享到", "下载Word", "html2canvas", "上一篇", "下一篇", "当前位置", "网站首页",
       "主办单位", "版权所有", "ICP备", "dynclicks", "line_u9"]
hits = 0
for s in SUS:
    h2 = [r["title"][:18] for r in rows if s in r["content"]]
    if h2:
        hits += 1
        print("  ⚠️ %-12s 命中 %d: %s" % (s, len(h2), h2[:2]))
print("  命中种类:", hits)
print()
print("=== 首条正文全文（含附件段）===")
for r in rows:
    if r["attachments"]:
        print(r["content"][:1500])
        break
