#!/usr/bin/env python3
"""
fix_md_after_crawl.py - 爬虫日跑后统一清理 MD 格式残留
扫描最近 N 天插入的记录，把 content 里的 MD 图片/链接转成 HTML 内嵌格式。
用法: python3 fix_md_after_crawl.py [--days 1]
"""
import sqlite3
import re
import sys
import time

DB = '/mnt/data/search.db'

MD_IMG = re.compile(r'!\[((?:[^\[\]]|\[[^\[\]]*\])*)\]\(([^)]+)\)')
MD_LINK = re.compile(r'\[((?:[^\[\]]|\[[^\[\]]*\])*)\]\(([^)]+)\)')
ATT_LINE = re.compile(r'^附件\s*[:：]?\s*\[((?:[^\[\]]|\[[^\[\]]*\])*)\]\(([^)]+)\)\s*$')
MD_ATT = re.compile(r'\[((?:[^\[\]]|\[[^\[\]]*\])*)\]\(([^)]+)\)')

def is_icon_url(url):
    return (url.endswith('.gif') or 'fileTypeImages' in url
            or 'icon_' in url or 'files2/' in url or '/ico_' in url)

def fix_content(content):
    """纯 MD/文本 content → HTML"""
    if not content:
        return content, 0, 0, 0
    lines = [l.strip() for l in content.split('\n') if l.strip()]
    parts = []
    n_img = n_link = n_att = 0
    for line in lines:
        m = ATT_LINE.fullmatch(line)
        if m:
            text, url = m.group(1).strip(), m.group(2).strip()
            if url.startswith('http'):
                n_att += 1
                parts.append(f'<p><a href="{url}">{text}</a></p>')
                continue
        m = MD_IMG.fullmatch(line)
        if m:
            url = m.group(2).strip()
            if is_icon_url(url):
                continue
            n_img += 1
            parts.append(f'<p><a href="{url}">查看图片</a></p>')
            continue
        m = MD_LINK.fullmatch(line)
        if m:
            text, url = m.group(1).strip(), m.group(2).strip()
            if url.startswith('http'):
                n_link += 1
                parts.append(f'<p><a href="{url}">{text}</a></p>')
                continue
        def img_repl(m):
            nonlocal n_img
            alt, url = m.group(1), m.group(2).strip()
            if is_icon_url(url):
                return ''
            n_img += 1
            return f'<p><a href="{url}">查看图片</a></p>'
        def link_repl(m):
            nonlocal n_link
            text, url = m.group(1), m.group(2).strip()
            if not url.startswith('http'):
                return m.group(0)
            n_link += 1
            return f'<a href="{url}">{text}</a>'
        line = MD_IMG.sub(img_repl, line)
        line = MD_LINK.sub(link_repl, line)
        if line.strip():
            parts.append(f'<p>{line}</p>')
    return '\n\n'.join(parts), n_img, n_link, n_att

def fix_attachments(att):
    if not att:
        return att, 0
    if not MD_ATT.search(att):
        return att, 0
    parts = []
    for m in MD_ATT.finditer(att):
        text, url = m.group(1).strip(), m.group(2).strip()
        if url.startswith('http'):
            parts.append(f'<p><a href="{url}">{text}</a></p>')
    if not parts:
        return att, 0
    return '\n'.join(parts), len(parts)

def main():
    days = 1
    for a in sys.argv:
        if a.startswith('--days='):
            try:
                days = int(a.split('=')[1])
            except ValueError:
                pass

    cutoff = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime(time.time() - days * 86400))
    conn = sqlite3.connect(DB, timeout=180)
    conn.execute('PRAGMA busy_timeout=180000')

    rows = conn.execute("""
        SELECT id, content, attachments, site_name FROM gov_raw
        WHERE inserted_at >= ?
          AND ((content NOT LIKE '%<p%' AND content NOT LIKE '%<table%' AND content NOT LIKE '%<div%'
                AND (content LIKE '%![%' OR content LIKE '%](http%'))
            OR (attachments LIKE '[%' AND attachments NOT LIKE '[{%'))
    """, (cutoff,)).fetchall()

    fixed_c = fixed_a = 0
    for rid, content, att, site in rows:
        new_content, _, _, _ = fix_content(content)
        new_att, _ = fix_attachments(att)
        if new_content != content or new_att != att:
            conn.execute(
                'UPDATE gov_raw SET content=?, attachments=? WHERE id=?',
                (new_content, new_att, rid)
            )
            if new_content != content:
                fixed_c += 1
            if new_att != att:
                fixed_a += 1

    conn.commit()
    conn.close()
    print(f'[FIX-MD] 扫描 {len(rows)} 条, content修复 {fixed_c}, attachments修复 {fixed_a}')

if __name__ == '__main__':
    main()
