#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
批量质检 crawler JSONL 输出（--pages=1 干跑 / JSONL_PATH 模式）。

失败级（❌ → 退出码 1）:
  empty     空正文
  badtitle  标题污染黑名单命中
  atturl    attachments 列出现非 http URL（mailto:/tel: 被误当附件）
            —— 2026-09-11 大丰实测：正文尾部「邮箱：x@163.com」被 _link_repl 当附件，122/249 行中招，
               日志显示「跳过 0」完全正常，只有本检查能发现。
  count     条数对账不符（仅当传 --list-total / --cutoff 时校验）
            —— 2026-09-11 大丰实测：dataproxy 253 条 − CUTOFF 4 = 249，实际只入 248（「声 明」3 字符
               被 len(txt)<4 静默丢弃），日志同样显示「跳过 0」。对账是唯一能发现静默漏抓的手段。

提示级（⚠️ 不改变退出码）:
  nopara    正文既无 <p> 也无 <table> → search_app 按 Markdown 渲染，段落会挤成一坨
  notin     attachments 有 http 附件，但正文没有内嵌 <a href="http...">（详情页看不到附件）
  short     短正文（<30 字；纯附件公告/产权声明类本身就只有一句话，属正常）
  attnomd   正文出现 markdown 链接（[text](http…)）→ 违反用户铁律，应转 <p><a href>

用法:
  python3 audit_jsonl.py /tmp/c_p1.jsonl
  python3 audit_jsonl.py /tmp/c_p1.jsonl --list-total 253 --cutoff 4
  python3 audit_jsonl.py /tmp/c_p1/*.jsonl
退出码: 0=通过（无失败级）, 1=有失败级, 2=用法错误
"""
import json, glob, os, re, sys

# 标题污染黑名单：站点导航词 + 未剥离的后缀变体
BAD_WORDS = [
    '网站支持IPv6', '无障碍浏览', '当前位置', 'Copyright', '版权所有', '主办单位',
    '热点推荐', '404', '出错了', '| 崇义', '| 崇义县人民政府',
    '_通知公告', '_药都资讯', '_建设项目', '_公示公告', '_政务公开',
    '_主动公开', '_工作动态', '_部门文件', '_双公示', '_人事信息', '_组织机构',
    ' - 通知', ' - 建设', '-赤壁', '-赤壁市政府网',
]


def _att_urls(raw):
    """attachments 列 → URL 列表。兼容 JSON 数组（[{"title","url"}]）与管道串（url|name; url2|name2）。"""
    if not raw:
        return []
    s = raw.strip()
    if s.startswith('['):
        try:
            out = []
            for a in json.loads(s):
                if isinstance(a, dict):
                    out.append(a.get('url') or a.get('href') or '')
                elif isinstance(a, str):
                    out.append(a)
            return [u for u in out if u]
        except Exception:
            return []
    return [p.split('|', 1)[0].strip() for p in re.split(r';', s) if p.strip()]


def audit(path, verbose=True, list_total=None, cutoff=0):
    lines = [l for l in open(path, encoding='utf-8').read().strip().split('\n') if l.strip()]
    n = len(lines)
    empty = short = bad = attnomd = nopara = notin = atturl = 0
    bad_samples, atturl_samples = [], []
    for l in lines:
        try:
            d = json.loads(l)
        except Exception:
            continue
        t = d.get('title', '')
        c = d.get('content', '') or ''
        if not c.strip():
            empty += 1
        elif len(re.sub(r'<[^>]+>', '', c).strip()) < 30:
            short += 1

        if any(w in t for w in BAD_WORDS):
            bad += 1
            if len(bad_samples) < 3:
                bad_samples.append(t[:70])

        # 失败级：非 http 附件（mailto/tel/相对路径）
        urls = _att_urls(d.get('attachments'))
        non_http = [u for u in urls if not u.startswith('http')]
        if non_http:
            atturl += 1
            if len(atturl_samples) < 3:
                atturl_samples.append('%s -> %s' % (t[:34], non_http[:3]))

        # 提示级：段落承载 / 附件内嵌 / markdown 残留
        if c.strip() and not re.search(r'<p[ >]', c, re.I) and not re.search(r'<table', c, re.I):
            nopara += 1
        http_atts = [u for u in urls if u.startswith('http')]
        if http_atts and not re.search(r'<a[^>]*href="http[^"]*"[^>]*>[^<]+</a>', c):
            notin += 1
        if re.search(r'\[[^\]]+\]\(http', c):
            attnomd += 1

    counted_total = n + cutoff
    count_bad = list_total is not None and counted_total != list_total
    if list_total is not None and count_bad:
        print('   ❌ 条数对账: 入库 %d + CUTOFF %d = %d , 列表 %d , 差 %d'
              % (n, cutoff, counted_total, list_total, list_total - counted_total))

    fail = (empty or bad or atturl or count_bad)
    flag = '❌' if fail else '✅'
    print('%s %s: n=%d empty=%d badtitle=%d atturl=%d'
          % (flag, os.path.basename(path), n, empty, bad, atturl))
    print('   ⚠️ nopara=%d notin=%d markdown_link=%d short=%d'
          % (nopara, notin, attnomd, short))
    for s in bad_samples:
        print('   BAD TITLE: %s' % s)
    for s in atturl_samples:
        print('   非 http 附件（mailto/tel 被当附件?）: %s' % s)
    if notin:
        print('   ⚠️ 有附件但正文未内嵌 <p><a href=绝对URL>附件名</a></p> → 详情页看不到附件')
    if nopara:
        print('   ⚠️ 正文无 <p>/<table> → search_app 会按 Markdown 渲染，段落挤成一坨')
    return not fail


if __name__ == '__main__':
    args = sys.argv[1:]
    list_total, cutoff, paths = None, 0, []
    i = 0
    while i < len(args):
        a = args[i]
        if a == '--list-total' and i + 1 < len(args):
            list_total = int(args[i + 1]); i += 2; continue
        if a.startswith('--list-total='):
            list_total = int(a.split('=', 1)[1]); i += 1; continue
        if a == '--cutoff' and i + 1 < len(args):
            cutoff = int(args[i + 1]); i += 2; continue
        if a.startswith('--cutoff='):
            cutoff = int(a.split('=', 1)[1]); i += 1; continue
        paths.append(a); i += 1
    if not paths:
        paths = sorted(glob.glob('/tmp/c_p1/*.jsonl'))
    if not paths:
        print(__doc__)
        sys.exit(2)
    ok = all(audit(p, list_total=list_total, cutoff=cutoff) for p in paths)
    sys.exit(0 if ok else 1)
