import re

with open("/root/gov_crawler/crawl_luanxian_854.py", "r") as f:
    content = f.read()

old1 = """    # 正文容器: <div id="conN"> - 通常只有附件链接
    m = re.search(r'<div[^>]*id="conN"[^>]*>(.*?)</div>\\s*<div[^>]*class="clearfloat"', html, re.DOTALL)
    if not m:
        m = re.search(r'<div[^>]*id="conN"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not m:
        return "", []"""

new1 = """    # 正文容器: <div id="conN">
    m = re.search(r'<div[^>]*id="conN"[^>]*>(.*?)</div>\\s*<div[^>]*class="clearfloat"', html, re.DOTALL)
    if not m:
        m = re.search(r'<div[^>]*id="conN"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not m:
        for pat in [r'<div[^>]*class="ftext"[^>]*>(.*?)</div>',
                    r'<div[^>]*class="content"[^>]*>(.*?)</div>',
                    r'<div[^>]*id="zoom"[^>]*>(.*?)</div>',
                    r'<div[^>]*class="article"[^>]*>(.*?)</div>',
                    r'<div[^>]*class="conTxt"[^>]*>(.*?)</div>']:
            m = re.search(pat, html, re.DOTALL)
            if m:
                break
    if not m:
        return "", []"""

if old1 in content:
    content = content.replace(old1, new1, 1)
    print("Patch 1 OK", flush=True)
else:
    print("Patch 1 FAILED", flush=True)
    idx = content.find("conN")
    if idx >= 0:
        print("Context:", repr(content[idx-30:idx+200]), flush=True)

old2 = """    # 附件（全文查找，不限于#conN）
    for a_href, a_text in re.findall(
        r'<a[^>]*href="([^"]+\\.(?:doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar|7z|txt|wps|et|DOC))"[^>]*>([^<]+)</a>',
        html, re.DOTALL
    ):
        attachments.append({"title": a_text.strip(), "url": urllib.parse.urljoin(url, a_href)})
    
    return content, attachments"""

new2 = """    # 附件（全文查找，不限于#conN）
    for a_href, a_text in re.findall(
        r'<a[^>]*href="([^"]+\\.(?:doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar|7z|txt|wps|et|DOC))"[^>]*>([^<]+)</a>',
        html, re.DOTALL
    ):
        attachments.append({"title": a_text.strip(), "url": urllib.parse.urljoin(url, a_href)})
    
    # 正文为空但有附件时，加PDF占位提示
    if (not content or not content.strip()) and attachments:
        content = "【该文档内容详见附件PDF/文档】\\n"
        for att in attachments:
            content += "附件: " + att["title"] + " - " + att["url"] + "\\n"
    
    return content, attachments"""

if old2 in content:
    content = content.replace(old2, new2, 1)
    print("Patch 2 OK", flush=True)
else:
    print("Patch 2 FAILED", flush=True)

with open("/root/gov_crawler/crawl_luanxian_854.py", "w") as f:
    f.write(content)
print("File saved", flush=True)
