"""PDF OCR处理后上传到服务器展示"""
import sys, os, re, json, base64
sys.path.insert(0, os.path.expanduser("~/gov_crawler/venv/lib/python3.9/site-packages"))

import pymupdf
from rapidocr_onnxruntime import RapidOCR

PDF_DIR = os.path.expanduser("~/Nutstore Files/Downloads_nuc")
OUTPUT_DIR = "/tmp/cceup_ocr_preview"

def ocr_pdf_to_html(pdf_path):
    """OCR PDF并生成预览HTML"""
    fname = os.path.basename(pdf_path).replace('.pdf', '')
    doc = pymupdf.open(pdf_path)
    ocr = RapidOCR()

    pages_html = []
    total_ocr_text = []

    for i, page in enumerate(doc):
        # 渲染页面为图片（200dpi）
        pix = page.get_pixmap(dpi=200)
        img_bytes = pix.tobytes("png")
        img_b64 = base64.b64encode(img_bytes).decode()

        # OCR识别
        temp_path = f"/tmp/ocr_preview_{os.getpid()}_{i}.png"
        with open(temp_path, "wb") as f:
            f.write(img_bytes)
        result, _ = ocr(temp_path)
        os.remove(temp_path)

        # 收集OCR文本
        ocr_lines = []
        if result:
            for box, text, score in result:
                if score > 0.8:
                    ocr_lines.append(f"[{score:.2f}] {text}")

        page_text = '\n'.join(ocr_lines)
        total_ocr_text.append(page_text)

        pages_html.append(f'''
        <div class="page">
            <h3>第 {i+1} 页</h3>
            <div class="page-content">
                <div class="page-image">
                    <img src="data:image/png;base64,{img_b64}" alt="第{i+1}页">
                </div>
                <div class="page-ocr">
                    <pre>{"<br>".join(ocr_lines) if ocr_lines else "(无识别结果)"}</pre>
                </div>
            </div>
        </div>
        ''')

    doc.close()

    # 提取联系方式
    full_text = '\n'.join(total_ocr_text)
    phones = re.findall(r'1[3-9]\d{9}', full_text)
    landlines = re.findall(r'0[1-9]\d{1,2}-?\d{7,8}', full_text)
    names = re.findall(
        r'(?:联系人|联\s*系\s*人|项目负责人|法人|法人代表|法定代表人|代表|业主|经办人|'
        r'负责人|项目经理|采购人|招标人|姓名|姓\s*名)[：:\s]*([\u4e00-\u9fff·]{2,6})',
        full_text
    )
    # 黑名单过滤
    blacklist = {'手机','代表','单位注册地址','为外资','姓名及号码','姓名'}
    names = [n for n in names if n not in blacklist and len(n) >= 2]

    contacts_html = ''
    if names or phones or landlines:
        contacts_html = f'''
        <div class="contacts">
            <h2>📞 提取到的联系方式</h2>
            <p>姓名: {"、".join(set(names)) or "未识别"}</p>
            <p>手机: {"、".join(set(phones)) or "未识别"}</p>
            <p>座机: {"、".join(set(landlines)) or "未识别"}</p>
        </div>'''

    html = f'''<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>📄 {html_escape(fname)}</title>
<style>
* {{ margin:0; padding:0; box-sizing:border-box; }}
body {{ font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,sans-serif; background:#1a1a2e; color:#e0e0e0; }}
.container {{ max-width:1400px; margin:0 auto; padding:20px; }}
h1 {{ font-size:20px; margin-bottom:20px; color:#e94560; }}
h2 {{ font-size:18px; margin-bottom:12px; color:#e94560; }}
h3 {{ font-size:15px; margin-bottom:10px; color:#aaa; }}
.page {{ background:#16213e; border-radius:10px; padding:20px; margin-bottom:20px; }}
.page-content {{ display:flex; gap:20px; }}
.page-image {{ flex:1; }}
.page-image img {{ width:100%; border-radius:6px; border:1px solid #333; }}
.page-ocr {{ flex:1; background:#0f3460; border-radius:6px; padding:15px; overflow:auto; max-height:600px; font-size:13px; line-height:1.6; }}
.page-ocr pre {{ font-family:monospace; color:#2ecc71; }}
.contacts {{ background:#16213e; border-radius:10px; padding:20px; margin-bottom:20px; }}
.contacts p {{ margin:6px 0; font-size:15px; }}
.contacts p:nth-child(2) {{ color:#e94560; font-weight:bold; }}
.contacts p:nth-child(3) {{ color:#2ecc71; font-family:monospace; }}
.contacts p:nth-child(4) {{ color:#3498db; font-family:monospace; }}
.back {{ display:inline-block; margin-bottom:16px; color:#e94560; text-decoration:none; font-size:14px; }}
.back:hover {{ text-decoration:underline; }}
</style>
</head>
<body>
<div class="container">
    <a href="index.html" class="back">← 返回列表</a>
    <h1>📄 {html_escape(fname)}</h1>
    <div class="info">共 {len(pages_html)} 页</div>
    {contacts_html}
    {"".join(pages_html)}
</div>
</body>
</html>'''

    return html, fname


def html_escape(s):
    return s.replace('&', '&amp;').replace('<', '&lt;').replace('>', '&gt;').replace('"', '&quot;')


if __name__ == "__main__":
    os.makedirs(OUTPUT_DIR, exist_ok=True)

    # 选5个样本
    all_pdfs = sorted([f for f in os.listdir(PDF_DIR) if f.endswith('.pdf')])
    # 避开头尾，选中间几个不同类型的
    samples = [all_pdfs[2], all_pdfs[10], all_pdfs[20], all_pdfs[35], all_pdfs[55]]

    index_items = []

    for fname in samples:
        fpath = os.path.join(PDF_DIR, fname)
        print(f"处理: {fname}...")
        html, short_name = ocr_pdf_to_html(fpath)
        safe_name = fname.replace('.pdf', '').replace('/', '_')
        out_path = os.path.join(OUTPUT_DIR, f"{safe_name}.html")
        with open(out_path, 'w', encoding='utf-8') as f:
            f.write(html)
        print(f"  → {out_path}")
        index_items.append((safe_name, fname))

    # 生成索引页
    index_html = '''<!DOCTYPE html>
<html lang="zh-CN">
<head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
<title>📄 CCEUP PDF OCR 预览</title>
<style>
*{margin:0;padding:0;box-sizing:border-box;}
body{font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,sans-serif;background:#1a1a2e;color:#e0e0e0;}
.container{max-width:800px;margin:0 auto;padding:40px 20px;}
h1{font-size:24px;color:#e94560;margin-bottom:8px;}
p{color:#888;margin-bottom:24px;font-size:14px;}
ul{list-style:none;}
li{background:#16213e;border-radius:8px;margin-bottom:12px;padding:16px 20px;}
li a{color:#e94560;text-decoration:none;font-size:15px;display:block;}
li a:hover{color:#ff6b6b;}
li .desc{color:#666;font-size:12px;margin-top:4px;}
</style>
</head>
<body>
<div class="container">
    <h1>📄 CCEUP PDF OCR 预览</h1>
    <p>5 个样本 PDF — 左侧为原始页面截图，右侧为 RapidOCR 识别结果</p>
    <ul>
'''
    for safe_name, orig_name in index_items:
        display = orig_name.replace('.pdf', '')
        index_html += f'        <li><a href="{safe_name}.html">{html_escape(display)}</a><div class="desc">点击查看OCR效果</div></li>\n'

    index_html += '''    </ul>
</div>
</body>
</html>'''

    with open(os.path.join(OUTPUT_DIR, "index.html"), 'w', encoding='utf-8') as f:
        f.write(index_html)

    print(f"\n完成！共 {len(samples)} 个PDF")
    print(f"预览文件: {OUTPUT_DIR}/index.html")
