FastAPI service wrapping the Surya-OCR-2 model (datalab-to) served through vLLM: legacy /v1/api/ai/* endpoints, an OpenAI-compatible /v1/chat/completions endpoint, a coalescing request batcher, a local OCR CLI, Docker packaging, multilingual example outputs, and quantization/concurrency benchmarks. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
14 lines
379 B
Python
14 lines
379 B
Python
from surya.inference.backends.openai_client import resolve_max_workers
|
|
|
|
|
|
def test_caps_at_max_inflight():
|
|
assert resolve_max_workers(batch_len=160, max_inflight=16) == 16
|
|
|
|
|
|
def test_uses_batch_len_when_smaller():
|
|
assert resolve_max_workers(batch_len=4, max_inflight=16) == 4
|
|
|
|
|
|
def test_never_below_one():
|
|
assert resolve_max_workers(batch_len=0, max_inflight=16) == 1
|