#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""fix_paragraph_div.py —— 修「源站用 <div> 分段 → div 被 unwrap → 段落糊成一坨」

根因：很多站（安徽 showList 系等）正文每个段落是一个 <div>，而 html_to_text 把 <div>
      全部 unwrap → 段落边界丢失 → content 里只剩裸文本行，search_app 按 HTML 渲染时
      \n 被浏览器折叠 → 段落糊成一坨（实测颍泉《氨基葡萄糖…公示》：33 段只有 1 个 <p>）。

三处改动（对3个同日新脚本一致施加）：
  A. 叶子 <div>（不含 div/table 子元素）→ 改名成 <p>，保住段落语义
  B. parts 循环：凡未被 <p> 包裹、且不是 table/ul/ol/h* 的块 → 一律补 <p>
  C. 最终兜底：去掉嵌套/空的 <p>

安全：逐文件逐锚点，命中数必须为 1，任一失败则该文件不动；改后 ast 校验；先备份。
"""
import ast
import io
import os
import shutil
import sys

FILES = [
    "/root/gov_crawler/crawl_yingquan_gggs.py",
    "/root/gov_crawler/crawl_danzhai_hjzf.py",
    "/root/gov_crawler/crawl_jcx_tzgg.py",
]

# ── A. 叶子 div → <p> ───────────────────────────────────────────────
A_OLD = """            for d in _soup.find_all("div"):
                if not d.get_text(strip=True) and not d.find("table") \\
                        and not d.find("img") and not d.find("a"):
                    d.decompose()
                else:
                    d.unwrap()"""
A_NEW = """            for d in _soup.find_all("div"):
                # 叶子 div（不含 div/table 子块）= 一个段落（安徽 showList 系等用 <div> 分段）
                #   → 改名 <p> 保住分段；否则 unwrap 会丢段落边界，正文糊成一坨
                if d.find("div") is None and d.find("table") is None:
                    if not d.get_text(strip=True) and not d.find("img") and not d.find("a"):
                        d.decompose()
                    else:
                        d.name = "p"
                elif not d.get_text(strip=True) and not d.find("table") \\
                        and not d.find("img") and not d.find("a"):
                    d.decompose()
                else:
                    d.unwrap()"""

# ── B. parts 循环兜底补 <p> ─────────────────────────────────────────
B_OLD = """        if not re.search(r"<p[ >]", b, re.I) and re.search(r"<a[ >]", b, re.I):
            b = "<p>%s</p>" % b"""
B_NEW = """        # 未被 <p> 包裹、且不是块级内容 → 一律补 <p>（否则裸文本会被浏览器折叠掉换行）
        if (not re.search(r"<p[ >]", b, re.I)
                and not re.search(r"<(?:table|ul|ol|h[1-6])\\b", b, re.I)):
            b = "<p>%s</p>" % b"""

# ── C. 最终去嵌套/空 <p> ────────────────────────────────────────────
C_OLD = """    seen, final = set(), []
    for p in out.split("\\n\\n"):"""
C_NEW = """    # 最终兜底：去嵌套/空的 <p>（源站 <div><div> 常见的产物）
    for _ in range(3):
        out = re.sub(r"<p\\b[^>]*>\\s*<p\\b[^>]*>", "<p>", out, flags=re.I)
        out = re.sub(r"</p\\s*>\\s*</p\\s*>", "</p>", out, flags=re.I)
    out = re.sub(r"<p\\b[^>]*>\\s*</p\\s*>", "", out, flags=re.I)

    seen, final = set(), []
    for p in out.split("\\n\\n"):"""

for P in FILES:
    name = os.path.basename(P)
    src = io.open(P, encoding="utf-8").read()
    orig = src
    ok = True
    for tag, old, new in (("A 叶子div→p", A_OLD, A_NEW), ("B 补<p>", B_OLD, B_NEW), ("C 去嵌套", C_OLD, C_NEW)):
        n = src.count(old)
        if n != 1:
            print("  %-26s %-12s ❌ 命中 %d 次（期望 1）" % (name, tag, n))
            ok = False
            break
        src = src.replace(old, new)
    if not ok:
        print("  %-26s → 本文件不动" % name)
        continue
    try:
        ast.parse(src)
    except SyntaxError as e:
        print("  %-26s ❌ 语法错误，不动：%s" % (name, e))
        continue
    bak = "/root/gov_crawler/Archive/%s.bak_20260924_para" % name
    os.makedirs(os.path.dirname(bak), exist_ok=True)
    if not os.path.exists(bak):
        shutil.copy2(P, bak)
    io.open(P, "w", encoding="utf-8").write(src)
    print("  %-26s ✅ A/B/C 全部施加（%d → %d 行）  备份 %s" % (
        name, orig.count("\n") + 1, src.count("\n") + 1, os.path.basename(bak)))
