#!/usr/bin/env python3
"""
fix_md_after_crawl.py - 爬虫日跑后统一清理 MD 格式残留
扫描最近 N 天插入的记录，把 content 里的 MD 图片/链接转成 HTML 内嵌格式。
用法: python3 fix_md_after_crawl.py [--days 1]
"""
import sqlite3
import re
import sys
import time

DB = '/mnt/data/search.db'

MD_IMG = re.compile(r'!\[((?:[^\[\]]|\[[^\[\]]*\])*)\]\(([^)]+)\)')
MD_LINK = re.compile(r'\[((?:[^\[\]]|\[[^\[\]]*\])*)\]\(([^)]+)\)')
ATT_LINE = re.compile(r'^附件\s*[:：]?\s*\[((?:[^\[\]]|\[[^\[\]]*\])*)\]\(([^)]+)\)\s*$')
MD_ATT = re.compile(r'\[((?:[^\[\]]|\[[^\[\]]*\])*)\]\(([^)]+)\)')

# ── 正文噪声块（2026-09-11）──────────────────────────────────────────
# 政府站正文常整段内联 CSS（绍兴上虞某条 content 9853 字里 9300 字是 3 个 <style>）。
# 不剥的后果：search_app 列表摘要变成 CSS 文本、详情页 <style> 被原样注入污染全页样式。
# crawler_lib.push_to_searchdb 已在入库时剥（覆盖 561 个脚本），但直写 sqlite 的脚本绕过它
# → 这里按天兜底，覆盖全部脚本（与 fix_content 同一个「调度器末尾扫最近 N 天」模式）。
NOISE_BLOCK = re.compile(r'<(script|style|noscript|template)\b[^>]*>.*?</\1\s*>', re.I | re.S)
NOISE_OPEN = re.compile(r'<(?:script|style|noscript|template)\b[^>]*>.*\Z', re.I | re.S)
# CMS 模板标记（ZJEG/TRS 系）：有的站爬虫把注释符去掉了 → 裸文本进正文（高青县-通知公告 762 处）
CMS_MARKER = re.compile(
    r'(?:<!--\s*)?(?:<\$\[[^\]]*\]>|ZJEG_RSS\.[\w.]*)\s*(?:begin|end)?\s*(?:-->)?', re.I)


def strip_noise_blocks(html):
    """剥掉 <script>/<style>/<noscript>/<template> 块（连块内文本）、注释、<link>、CMS 模板标记。

    ⚠️ 只删标签不够：标签**之间**的 CSS 文本会原样留下。
    ⚠️ 未闭合兜底：内容被截断时切在 </style> 之前 → 未闭合开标签整段截到末尾。
    """
    if not html:
        return html
    s = html
    prev = None
    while prev != s:
        prev = s
        s = NOISE_BLOCK.sub(' ', s)
    s = NOISE_OPEN.sub(' ', s)
    s = re.sub(r'<!--.*?-->', ' ', s, flags=re.S)
    s = re.sub(r'<link\b[^>]*>', ' ', s, flags=re.I)
    s = CMS_MARKER.sub(' ', s)
    return s


def fix_noise(conn, cutoff):
    """剥掉最近 N 天新记录里的 <style>/<script> 块。返回 (扫描数, 修复数)。"""
    rows = conn.execute("""
        SELECT id, content FROM gov_raw
        WHERE inserted_at >= ?
          AND (content LIKE '%<style%' OR content LIKE '%<script%'
               OR content LIKE '%<noscript%' OR content LIKE '%<link%'
               OR content LIKE '%ZJEG_RSS%' OR content LIKE '%<$[%')
    """, (cutoff,)).fetchall()
    n_fixed = 0
    for rid, content in rows:
        new_content = strip_noise_blocks(content)
        if new_content != content:
            conn.execute('UPDATE gov_raw SET content=? WHERE id=?', (new_content, rid))
            n_fixed += 1
    return len(rows), n_fixed


def is_icon_url(url):
    return (url.endswith('.gif') or 'fileTypeImages' in url
            or 'icon_' in url or 'files2/' in url or '/ico_' in url)

def fix_content(content):
    """纯 MD/文本 content → HTML"""
    if not content:
        return content, 0, 0, 0
    lines = [l.strip() for l in content.split('\n') if l.strip()]
    parts = []
    n_img = n_link = n_att = 0
    for line in lines:
        m = ATT_LINE.fullmatch(line)
        if m:
            text, url = m.group(1).strip(), m.group(2).strip()
            if url.startswith('http'):
                n_att += 1
                parts.append(f'<p><a href="{url}">{text}</a></p>')
                continue
        m = MD_IMG.fullmatch(line)
        if m:
            url = m.group(2).strip()
            if is_icon_url(url):
                continue
            n_img += 1
            parts.append(f'<p><a href="{url}">查看图片</a></p>')
            continue
        m = MD_LINK.fullmatch(line)
        if m:
            text, url = m.group(1).strip(), m.group(2).strip()
            if url.startswith('http'):
                n_link += 1
                parts.append(f'<p><a href="{url}">{text}</a></p>')
                continue
        def img_repl(m):
            nonlocal n_img
            alt, url = m.group(1), m.group(2).strip()
            if is_icon_url(url):
                return ''
            n_img += 1
            return f'<p><a href="{url}">查看图片</a></p>'
        def link_repl(m):
            nonlocal n_link
            text, url = m.group(1), m.group(2).strip()
            if not url.startswith('http'):
                return m.group(0)
            n_link += 1
            return f'<a href="{url}">{text}</a>'
        line = MD_IMG.sub(img_repl, line)
        line = MD_LINK.sub(link_repl, line)
        if line.strip():
            parts.append(f'<p>{line}</p>')
    return '\n\n'.join(parts), n_img, n_link, n_att

def fix_attachments(att):
    if not att:
        return att, 0
    if not MD_ATT.search(att):
        return att, 0
    parts = []
    for m in MD_ATT.finditer(att):
        text, url = m.group(1).strip(), m.group(2).strip()
        if url.startswith('http'):
            parts.append(f'<p><a href="{url}">{text}</a></p>')
    if not parts:
        return att, 0
    return '\n'.join(parts), len(parts)

def main():
    days = 1
    for a in sys.argv:
        if a.startswith('--days='):
            try:
                days = int(a.split('=')[1])
            except ValueError:
                pass

    cutoff = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime(time.time() - days * 86400))
    conn = sqlite3.connect(DB, timeout=180)
    conn.execute('PRAGMA busy_timeout=180000')

    rows = conn.execute("""
        SELECT id, content, attachments, site_name FROM gov_raw
        WHERE inserted_at >= ?
          AND ((content NOT LIKE '%<p%' AND content NOT LIKE '%<table%' AND content NOT LIKE '%<div%'
                AND (content LIKE '%![%' OR content LIKE '%](http%'))
            OR (attachments LIKE '[%' AND attachments NOT LIKE '[{%'))
    """, (cutoff,)).fetchall()

    fixed_c = fixed_a = 0
    for rid, content, att, site in rows:
        new_content, _, _, _ = fix_content(content)
        new_att, _ = fix_attachments(att)
        if new_content != content or new_att != att:
            conn.execute(
                'UPDATE gov_raw SET content=?, attachments=? WHERE id=?',
                (new_content, new_att, rid)
            )
            if new_content != content:
                fixed_c += 1
            if new_att != att:
                fixed_a += 1

    # 噪声块剥离（独立 pass，覆盖所有脚本，含绕过 crawler_lib 直写 sqlite 的）
    n_scan, n_noise = fix_noise(conn, cutoff)

    conn.commit()
    conn.close()
    print(f'[FIX-MD] 扫描 {len(rows)} 条, content修复 {fixed_c}, attachments修复 {fixed_a}')
    print(f'[FIX-NOISE] 扫描 {n_scan} 条含 <style>/<script>/<link>, 剥离 {n_noise} 条')

if __name__ == '__main__':
    main()
