"""测试 RapidOCR（PaddleOCR ONNX 版）识图效果"""
import sys, os, re
sys.path.insert(0, os.path.expanduser("~/gov_crawler/venv/lib/python3.9/site-packages"))

# 1. 生成测试图：中文+电话
from PIL import Image, ImageDraw, ImageFont

def make_test_image(texts, save_path="/tmp/ocr_test.png"):
    """画一张模拟PDF截图的图片"""
    line_h = 50
    margin = 30
    font_size = 28
    
    # macOS 中文字体
    font_paths = [
        "/System/Library/Fonts/PingFang.ttc",
        "/System/Library/Fonts/STHeiti Light.ttc",
        "/System/Library/Fonts/Supplemental/Songti.ttc",
    ]
    font = None
    for p in font_paths:
        if os.path.exists(p):
            font = ImageFont.truetype(p, font_size)
            break
    if not font:
        # fallback — 用默认字体
        font = ImageFont.load_default()
    
    w = 600
    h = margin * 2 + line_h * len(texts)
    img = Image.new("RGB", (w, h), "white")
    draw = ImageDraw.Draw(img)
    
    for i, text in enumerate(texts):
        y = margin + i * line_h
        draw.text((margin, y), text, fill="black", font=font)
    
    img.save(save_path)
    print(f"[生成] {save_path} ({w}x{h})")
    return save_path

# 2. OCR识别
from rapidocr_onnxruntime import RapidOCR

def test_ocr(img_path):
    ocr = RapidOCR()
    result, elapse = ocr(img_path)
    
    if result is None:
        print("[OCR] 未识别到任何文字")
        return
    
    elapse_total = sum(elapse) if elapse else 0
    print(f"\n[OCR 结果] ({len(result)} 个文本块, 耗时 {elapse_total:.2f}s)")
    all_text = []
    for box, text, score in result:
        all_text.append(text)
        print(f"  [{score:.3f}] {text}")
    
    full_text = " ".join(all_text)
    # 提取电话
    phones = re.findall(r'1[3-9]\d{9}', full_text)
    if phones:
        print(f"\n📞 识别到电话: {', '.join(phones)}")
    else:
        print("\n📞 未识别到电话")
    
    return all_text

if __name__ == "__main__":
    # 模拟政府网站PDF上常见的联系方式
    test_texts = [
        "建设项目环境影响评价信息公示",
        "建设单位：浙江华海药业股份有限公司",
        "联系人：王经理",
        "联系电话：13800138000",
        "邮箱：wang@example.com",
        "地址：浙江省台州市临海市杜桥镇",
    ]
    img = make_test_image(test_texts)
    test_ocr(img)
