"""重新生成轻量版OCR预览 — 图片外链"""
import sys, os, re, json
sys.path.insert(0, os.path.expanduser("~/gov_crawler/venv/lib/python3.9/site-packages"))

import pymupdf
from rapidocr_onnxruntime import RapidOCR

PDF_DIR = os.path.expanduser("~/Nutstore Files/Downloads_nuc")
OUTPUT_DIR = "/tmp/cceup_ocr_lite"

def html_escape(s):
    return s.replace('&','&amp;').replace('<','&lt;').replace('>','&gt;').replace('"','&quot;')

def process_one(pdf_path, out_dir):
    fname = os.path.basename(pdf_path).replace('.pdf', '')
    safe = re.sub(r'[^\w\u4e00-\u9fff]', '_', fname)[:50]

    doc = pymupdf.open(pdf_path)
    ocr = RapidOCR()

    pages_html = []
    contact_names = []
    contact_phones = []

    for i, page in enumerate(doc):
        pix = page.get_pixmap(dpi=200)
        img_path = os.path.join(out_dir, f"{safe}_p{i}.png")
        pix.save(img_path)

        # OCR
        result, _ = ocr(img_path)
        ocr_lines = []
        if result:
            for _, text, score in result:
                if score > 0.8:
                    ocr_lines.append(text)

        page_text = '\n'.join(ocr_lines)
        pages_html.append(f'''<div class="page">
    <h3>第 {i+1} 页</h3>
    <div class="page-content">
        <div class="page-image"><img src="{safe}_p{i}.png" alt="第{i+1}页"></div>
        <div class="page-ocr"><pre>{"<br>".join(ocr_lines[:60]) if ocr_lines else "(无识别结果)"}</pre></div>
    </div>
</div>''')

        # 提取联系信息
        phones = re.findall(r'1[3-9]\d{9}', page_text)
        contact_phones.extend(phones)

    doc.close()

    # 联系人归纳
    blacklist = {'手机','代表','单位注册地址','为外资','姓名及号码','姓名'}
    names = re.findall(
        r'(?:联系人|联系人|项目负责人|法人|法定代表人|代表|业主|经办人|'
        r'负责人|项目经理|采购人|招标人|姓名)[：:\s]*([\u4e00-\u9fff·]{2,6})',
        '\n'.join(pages_html)
    )
    names = [n for n in names if n not in blacklist and len(n) >= 2]

    contacts = ''
    if names or contact_phones:
        contacts = f'''<div class="contacts">
    <h2>📞 提取到的联系方式</h2>
    <p>联系人: {"、".join(set(names)) or "未识别"}</p>
    <p>电话: {"、".join(set(contact_phones)) or "未识别"}</p>
</div>'''

    html = f'''<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>📄 {html_escape(fname)}</title>
<style>
*{{margin:0;padding:0;box-sizing:border-box;}}
body{{font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,sans-serif;background:#1a1a2e;color:#e0e0e0;}}
.container{{max-width:1400px;margin:0 auto;padding:20px;}}
h1{{font-size:20px;color:#e94560;margin-bottom:16px;}}
h2{{font-size:18px;color:#e94560;margin-bottom:8px;}}
h3{{font-size:15px;color:#aaa;margin-bottom:8px;}}
.page{{background:#16213e;border-radius:10px;padding:20px;margin-bottom:16px;}}
.page-content{{display:flex;gap:16px;}}
.page-image{{flex:1;}}
.page-image img{{width:100%;border-radius:6px;border:1px solid #333;}}
.page-ocr{{flex:1;background:#0f3460;border-radius:6px;padding:15px;max-height:500px;overflow:auto;font-size:13px;line-height:1.6;}}
.page-ocr pre{{font-family:monospace;color:#2ecc71;}}
.contacts{{background:#16213e;border-radius:10px;padding:20px;margin-bottom:16px;}}
.contacts p{{margin:6px 0;font-size:15px;font-weight:bold;}}
.contacts p:last-child{{color:#2ecc71;font-family:monospace;}}
.back{{display:inline-block;margin-bottom:12px;color:#e94560;text-decoration:none;font-size:14px;}}
</style>
</head>
<body>
<div class="container">
    <a href="index.html" class="back">← 返回列表</a>
    <h1>📄 {html_escape(fname)}</h1>
    {contacts}
    {"".join(pages_html)}
</div>
</body>
</html>'''

    return html, safe

# 主流程
os.makedirs(OUTPUT_DIR, exist_ok=True)
all_pdfs = sorted([f for f in os.listdir(PDF_DIR) if f.endswith('.pdf')])
samples = [all_pdfs[2], all_pdfs[10], all_pdfs[20], all_pdfs[35], all_pdfs[55]]

index_items = []

for fname in samples:
    fpath = os.path.join(PDF_DIR, fname)
    print(f"处理: {fname}...")
    html, safe = process_one(fpath, OUTPUT_DIR)
    with open(os.path.join(OUTPUT_DIR, f"{safe}.html"), 'w', encoding='utf-8') as f:
        f.write(html)
    index_items.append((safe, fname))
    print(f"  OK")

# 索引页
index_html = '''<!DOCTYPE html>
<html lang="zh-CN">
<head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
<title>📄 CCEUP PDF OCR 预览</title>
<style>
*{margin:0;padding:0;box-sizing:border-box;}
body{font-family:-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,sans-serif;background:#1a1a2e;color:#e0e0e0;}
.container{max-width:800px;margin:0 auto;padding:40px 20px;}
h1{font-size:24px;color:#e94560;margin-bottom:8px;}
p{color:#888;margin-bottom:24px;font-size:14px;}
ul{list-style:none;}
li{background:#16213e;border-radius:8px;margin-bottom:12px;padding:16px 20px;}
li a{color:#e94560;text-decoration:none;font-size:15px;display:block;}
li a:hover{color:#ff6b6b;}
li .desc{color:#666;font-size:12px;margin-top:4px;}
</style>
</head>
<body>
<div class="container">
    <h1>📄 CCEUP PDF OCR 预览</h1>
    <p>5 个样本 — 左侧为PDF截图，右侧为RapidOCR识别结果</p>
    <ul>
'''
for safe, orig in index_items:
    display = orig.replace('.pdf','')[:60]
    index_html += f'<li><a href="{safe}.html">{html_escape(display)}</a><div class="desc">点击查看</div></li>\n'
index_html += '</ul></div></body></html>'

with open(os.path.join(OUTPUT_DIR, "index.html"), 'w', encoding='utf-8') as f:
    f.write(index_html)

print(f"\n完成! 输出: {OUTPUT_DIR}/")
print(f"文件列表: {os.listdir(OUTPUT_DIR)}")
