FROM vllm/vllm-openai:v0.20.1 ARG DEBIAN_FRONTEND=noninteractive RUN apt-get update && \ apt-get install --no-install-recommends -y \ ffmpeg \ libsm6 \ libxext6 \ libcairo2 \ libgirepository1.0-dev \ libdbus-1-3 \ && rm -rf /var/lib/apt/lists/* ENV PYTHONUNBUFFERED=1 ENV PYTHONPATH=/opt/suya-ocr ENV SURYA_MODEL_CHECKPOINT=datalab-to/surya-ocr-2 ENV SURYA_INFERENCE_BACKEND=vllm ENV SURYA_INFERENCE_URL=http://127.0.0.1:8000/v1 ENV SURYA_INFERENCE_AUTOSTART=false ENV SURYA_INFERENCE_KEEP_ALIVE=false ENV SURYA_INFERENCE_PARALLEL=8 ENV SURYA_INFERENCE_LOGPROBS=false ENV SURYA_INFERENCE_MAX_RETRIES=1 # Cap concurrent in-flight chat-completion requests to vLLM per coalesced batch. # Must track VLLM_MAX_NUM_SEQS below: the request batcher otherwise ran only # SURYA_INFERENCE_PARALLEL(=8) wide, leaving half of vLLM's 16 sequence slots # idle. Setting this to 16 closed a 33% mean / 32% p95 regression (see # docs/diagnosis_baseline.md). Do not exceed VLLM_MAX_NUM_SEQS (overcommit only # queues). ENV SURYA_INFERENCE_MAX_INFLIGHT=16 ENV SUYA_OCR_MODE=block ENV SUYA_MAX_BATCH_SIZE=8 ENV SUYA_BATCH_WAIT_MS=25 ENV SUYA_MAX_QUEUE_SIZE=128 ENV SUYA_VLLM_IMAGE_FORMAT=JPEG ENV SUYA_VLLM_JPEG_QUALITY=92 ENV SURYA_MAX_BLOCKS_PER_PAGE=80 ENV SURYA_MAX_TOKENS_FULL_PAGE=6144 # GPU/dtype defaults target T4 (deployment hardware). Do not change these # without confirming the deployment target. ENV VLLM_GPU_TYPE=t4 ENV VLLM_DTYPE=float16 ENV VLLM_GPU_MEMORY_UTILIZATION=0.85 ENV VLLM_MAX_MODEL_LEN=18000 ENV VLLM_MAX_NUM_SEQS=16 ENV VLLM_MAX_BATCHED_TOKENS=4096 # Latency-tuned vLLM serve flags (CUDA graph, prefix caching, chunked prefill) # are applied in scripts/start_single_container.sh — see that file and # docs/quantization_benchmark_results.md §5 for the measured rationale. WORKDIR /opt/suya-ocr COPY requirements.txt pyproject.toml ./ RUN pip install --no-cache-dir uv==0.8.15 && \ python3 -m uv pip install --system -r requirements.txt COPY . . EXPOSE 5002 ENTRYPOINT [] CMD ["/opt/suya-ocr/scripts/start_single_container.sh"]