#!/usr/bin/env python3
"""修复器 v7：处理 25 个剩余 md 表格脚本变体。
模式:
  A) BS children 型: if name=='table' / elif el.name=="table" / elif tag.name=='table':
     -> 替换整块为 parts.append(html_table_to_html(X, url_var))
  B) re.finditer 正则型: if elem.group(0).startswith('<table'):
     -> 替换整块为 parts.append(elem.group(0))
  C) 死代码型: md_rows.append(elem.group(0)) (md_rows 未初始化)
     -> 改为 parts.append(elem.group(0))
  D) 独立函数型: html_to_markdown / table_to_markdown 函数体保留 HTML
     -> 函数体改为 return html
"""
import re
import sys
import os

HTML_TABLE_FN = '''def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
'''

# 模式 A: 表格对象变量名 -> 该文件 url 变量（猜测列表，验证存在）
def find_url_var(lines):
    cands = ["page_url", "detail_url", "source_url", "url", "base_url"]
    for c in cands:
        if any(re.search(r"\b%s\b" % c, ln) for ln in lines):
            return c
    return ""

def process_file(path):
    src = open(path, encoding="utf-8", errors="replace").read()
    lines = src.split("\n")
    orig = lines[:]
    changed = False
    has_fn = "def html_table_to_html" in src

    # 找所有表格处理块（模式 A/B）
    i = 0
    new_lines = []
    skip_until = -1
    while i < len(lines):
        ln = lines[i]
        stripped = ln.strip()
        # 模式 A: BS children 型
        m_a = re.match(r"^(\s*)(?:if|elif)\s+(?:name|el\.name|tag\.name|child\.name)\s*==\s*['\"]table['\"]\s*:", ln)
        # 模式 B: re.finditer 正则型
        m_b = re.match(r"^(\s*)if\s+elem\.group\(0\)\.startswith\(['\"]<table['\"]\)\s*:", ln)
        # 模式 C: 死代码型 md_rows.append(elem.group(0)) 或 tag.group(0)
        m_c = re.match(r"^(\s*)md_rows\.append\((elem|tag)\.group\(0\)\)\s*$", ln)

        if m_a:
            indent = m_a.group(1)
            # 找块结束：下一个缩进 <= indent 的行
            j = i + 1
            while j < len(lines) and (not lines[j].strip() or lines[j].startswith(indent) and len(lines[j]) > len(indent) and not lines[j][len(indent):].strip().startswith("#")):
                # 只有当行有内容且缩进更深才算块内
                if lines[j].strip() and not (lines[j].startswith(indent + " ") or lines[j].startswith(indent + "\t")):
                    break
                j += 1
            # 表格对象变量
            obj = "el" if "el.name" in ln else ("tag" if "tag.name" in ln else ("child" if "child.name" in ln else "child"))
            url_var = find_url_var(lines)
            url_arg = ", %s" % url_var if url_var else ""
            new_lines.append("%sif %s.name == 'table':" % (indent, obj))
            new_lines.append("%s    parts.append(html_table_to_html(%s%s))" % (indent, obj, url_arg))
            changed = True
            i = j
            continue
        elif m_b:
            indent = m_b.group(1)
            # 块内包含 md_rows 构建 -> 替换为 parts.append(elem.group(0))
            j = i + 1
            while j < len(lines) and (not lines[j].strip() or (lines[j].startswith(indent + " ") or lines[j].startswith(indent + "\t"))):
                j += 1
            new_lines.append("%sif elem.group(0).startswith('<table'):" % indent)
            new_lines.append("%s    parts.append(elem.group(0))" % indent)
            changed = True
            i = j
            continue
        elif m_c:
            indent = m_c.group(1)
            var = m_c.group(2)
            new_lines.append("%sparts.append(%s.group(0))" % (indent, var))
            changed = True
            i += 1
            continue
        else:
            new_lines.append(ln)
            i += 1

    out = "\n".join(new_lines)

    # 注入 html_table_to_html 函数（如缺失且用到了）
    if changed and not has_fn and "html_table_to_html(" in out:
        # 在第一个 def 前插入
        m = re.search(r"\n(def |\nclass )", out)
        if m:
            pos = out.find(m.group(0).strip())
            # 找到第一个顶层 def
            def_m = re.search(r"^def ", out, re.M)
            if def_m:
                pos = def_m.start()
                # 检查 import urllib.parse
                if "import urllib.parse" not in out and "from urllib.parse" not in out:
                    # 找 import 行后插入
                    last_import = 0
                    for mm in re.finditer(r"^import .*$|^from .* import .*$", out, re.M):
                        last_import = mm.end()
                    if last_import:
                        out = out[:last_import] + "\nimport urllib.parse" + out[last_import:]
                        # 重新定位 def 位置
                        def_m = re.search(r"^def ", out, re.M)
                        pos = def_m.start()
                out = out[:pos] + HTML_TABLE_FN + "\n" + out[pos:]
            else:
                out = HTML_TABLE_FN + "\n" + out
        else:
            out = HTML_TABLE_FN + "\n" + out

    if out != src:
        open(path, "w", encoding="utf-8").write(out)
        return True, "changed"
    return False, "no-change"

if __name__ == "__main__":
    files = sys.argv[1:]
    if not files:
        files = [l.strip() for l in open("/tmp/real_remnants.txt") if l.strip()]
    ok, fail = 0, 0
    for f in files:
        if not os.path.exists(f):
            print("MISS %s" % f)
            continue
        try:
            ch, st = process_file(f)
            # 语法校验
            import py_compile
            py_compile.compile(f, doraise=True)
            print("%s %s %s" % ("OK " if ch else "SAME", st, f))
            if ch:
                ok += 1
        except Exception as e:
            print("FAIL %s: %s" % (f, e))
            fail += 1
    print("changed=%d fail=%d" % (ok, fail))
