第一次提交
This commit is contained in:
@@ -0,0 +1,15 @@
|
||||
"""多模态大模型 OCR 引擎:直接看图转写题目为 Markdown。"""
|
||||
from app.services.llm.client import LLMClient
|
||||
from app.services.llm.prompts import VLM_OCR_SYSTEM, VLM_OCR_USER
|
||||
from app.services.ocr.base import OcrResult
|
||||
|
||||
|
||||
class VlmOcrEngine:
|
||||
def __init__(self, llm: LLMClient):
|
||||
self._llm = llm
|
||||
|
||||
async def recognize(self, image_bytes: bytes, mime: str) -> OcrResult:
|
||||
result = await self._llm.chat_vision(
|
||||
VLM_OCR_SYSTEM, VLM_OCR_USER, image_bytes, mime=mime
|
||||
)
|
||||
return OcrResult(markdown=result.content, confidence=1.0, engine="vlm")
|
||||
Reference in New Issue
Block a user