Suya OCR API — vLLM-backed, OpenAI-compatible OCR service

FastAPI service wrapping the Surya-OCR-2 model (datalab-to) served through vLLM:
legacy /v1/api/ai/* endpoints, an OpenAI-compatible /v1/chat/completions endpoint,
a coalescing request batcher, a local OCR CLI, Docker packaging, multilingual
example outputs, and quantization/concurrency benchmarks.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Fu Dai
2026-06-17 10:20:02 +04:00
co-authored by Claude Opus 4.8
commit 1a585693be
147 changed files with 13827 additions and 0 deletions
+48
View File
@@ -0,0 +1,48 @@
from typing import Optional
from surya.common.load import ModelLoader
from surya.logging import get_logger
from surya.ocr_error.model.config import DistilBertConfig
from surya.ocr_error.model.encoder import DistilBertForSequenceClassification
from surya.ocr_error.tokenizer import DistilBertTokenizer
from surya.settings import settings
logger = get_logger()
class OCRErrorModelLoader(ModelLoader):
def __init__(self, checkpoint: Optional[str] = None):
super().__init__(checkpoint)
if self.checkpoint is None:
self.checkpoint = settings.OCR_ERROR_MODEL_CHECKPOINT
def model(
self,
device=settings.TORCH_DEVICE_MODEL,
dtype=settings.MODEL_DTYPE,
attention_implementation: Optional[str] = None,
) -> DistilBertForSequenceClassification:
if device is None:
device = settings.TORCH_DEVICE_MODEL
if dtype is None:
dtype = settings.MODEL_DTYPE
config = DistilBertConfig.from_pretrained(self.checkpoint)
model = (
DistilBertForSequenceClassification.from_pretrained(
self.checkpoint,
dtype=dtype,
config=config,
)
.to(device)
.eval()
)
return model
def processor(
self, device=settings.TORCH_DEVICE_MODEL, dtype=settings.MODEL_DTYPE
) -> DistilBertTokenizer:
return DistilBertTokenizer.from_pretrained(self.checkpoint)