Spaces:
Running
Running
File size: 4,085 Bytes
ba43ff8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 | """
药店拜访助手 - Hugging Face Spaces 版本
使用 Gradio SDK(免费)+ FastAPI 自定义路由
"""
import gradio as gr
from fastapi import Request, UploadFile, File
from fastapi.responses import HTMLResponse, JSONResponse, Response
from fastapi.middleware.cors import CORSMiddleware
import io
import json
import os
import sys
import platform
import shutil
# 确保能导入同目录模块
WORK_DIR = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, WORK_DIR)
# 导入导出服务核心函数
from export_server import generate_docx
# 导入OCR
import pytesseract
from PIL import Image
# 跨平台 Tesseract 配置
if platform.system() == 'Windows':
pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'
os.environ['TESSDATA_PREFIX'] = os.path.expandvars(r'%USERPROFILE%\.tesseract\tessdata')
else:
tesseract_bin = shutil.which('tesseract')
if tesseract_bin:
pytesseract.pytesseract.tesseract_cmd = tesseract_bin
for _tessdata in ['/usr/share/tesseract-ocr/5/tessdata',
'/usr/share/tesseract-ocr/4.00/tessdata',
'/usr/share/tessdata']:
if os.path.isdir(_tessdata):
os.environ.setdefault('TESSDATA_PREFIX', _tessdata)
break
# 读取H5页面内容
H5_FILE = os.path.join(WORK_DIR, 'test-h5.html')
with open(H5_FILE, 'r', encoding='utf-8') as f:
h5_content = f.read()
# 创建 Gradio Blocks(HF Spaces 要求 demo 变量)
demo = gr.Blocks(css="footer{display:none !important}")
with demo:
gr.HTML('<script>window.location.href="/app";</script>')
# 启动 Gradio 获取底层 FastAPI 应用
app, _, _ = demo.launch(
prevent_thread_lock=True,
server_name="0.0.0.0",
server_port=7860,
quiet=True,
show_error=True
)
# 添加 CORS 支持
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
allow_methods=["*"],
allow_headers=["*"],
)
# ===== 自定义路由 =====
@app.get("/app", response_class=HTMLResponse)
async def serve_h5():
"""返回 H5 页面"""
return HTMLResponse(content=h5_content)
@app.get("/health")
async def health():
"""健康检查"""
return {"status": "ok"}
@app.post("/ocr")
async def ocr_endpoint(file: UploadFile = File(...)):
"""OCR 识别接口"""
try:
image_data = await file.read()
img = Image.open(io.BytesIO(image_data))
w, h = img.size
max_dim = 2000
if w > max_dim or h > max_dim:
ratio = min(max_dim / w, max_dim / h)
img = img.resize((int(w * ratio), int(h * ratio)), Image.LANCZOS)
text = pytesseract.image_to_string(img, lang='chi_sim', config='--psm 6')
result = {"text": text.strip(), "length": len(text.strip())}
print(f"[OCR] 识别成功,提取 {len(text.strip())} 个字符")
return JSONResponse(content=result)
except Exception as e:
print(f"[OCR] 识别失败: {e}")
return JSONResponse(content={"error": f"OCR识别失败: {str(e)}"}, status_code=500)
@app.post("/export")
async def export_endpoint(request: Request):
"""Word 文档导出接口"""
try:
body = await request.body()
visit_data = json.loads(body)
buffer = generate_docx(visit_data)
filename = 'visit_%s_%sstores.docx' % (
visit_data.get('visitDateDisplay', '').replace('.', '-'),
visit_data.get('totalCount', 0)
)
print(f"[Export] 导出成功: {filename}")
return Response(
content=buffer.getvalue(),
media_type="application/vnd.openxmlformats-officedocument.wordprocessingml.document",
headers={"Content-Disposition": f'attachment; filename="{filename}"'}
)
except Exception as e:
print(f"[Export] 导出失败: {e}")
return JSONResponse(content={"error": str(e)}, status_code=500)
print("=" * 50)
print(" 药店拜访助手 - HF Spaces 版本已启动")
print(" H5 页面: /app")
print(" OCR 接口: /ocr")
print(" 导出接口: /export")
print("=" * 50)
|