FROM vllm/vllm-openai:v0.20.1

ARG DEBIAN_FRONTEND=noninteractive

RUN apt-get update && \
  apt-get install --no-install-recommends -y \
    ffmpeg \
    libsm6 \
    libxext6 \
    libcairo2 \
    libgirepository1.0-dev \
    libdbus-1-3 \
  && rm -rf /var/lib/apt/lists/*

ENV PYTHONUNBUFFERED=1
ENV PYTHONPATH=/opt/suya-ocr

ENV SURYA_MODEL_CHECKPOINT=datalab-to/surya-ocr-2
ENV SURYA_INFERENCE_BACKEND=vllm
ENV SURYA_INFERENCE_URL=http://127.0.0.1:8000/v1
ENV SURYA_INFERENCE_AUTOSTART=false
ENV SURYA_INFERENCE_KEEP_ALIVE=false
ENV SURYA_INFERENCE_PARALLEL=8
ENV SURYA_INFERENCE_LOGPROBS=false
ENV SURYA_INFERENCE_MAX_RETRIES=1
# Cap concurrent in-flight chat-completion requests to vLLM per coalesced batch.
# Must track VLLM_MAX_NUM_SEQS below: the request batcher otherwise ran only
# SURYA_INFERENCE_PARALLEL(=8) wide, leaving half of vLLM's 16 sequence slots
# idle. Setting this to 16 closed a 33% mean / 32% p95 regression (see
# docs/diagnosis_baseline.md). Do not exceed VLLM_MAX_NUM_SEQS (overcommit only
# queues).
ENV SURYA_INFERENCE_MAX_INFLIGHT=16

ENV SUYA_OCR_MODE=block
ENV SUYA_MAX_BATCH_SIZE=8
ENV SUYA_BATCH_WAIT_MS=25
ENV SUYA_MAX_QUEUE_SIZE=128
ENV SUYA_VLLM_IMAGE_FORMAT=JPEG
ENV SUYA_VLLM_JPEG_QUALITY=92
ENV SURYA_MAX_BLOCKS_PER_PAGE=80
ENV SURYA_MAX_TOKENS_FULL_PAGE=6144

# GPU/dtype defaults target T4 (deployment hardware). Do not change these
# without confirming the deployment target.
ENV VLLM_GPU_TYPE=t4
ENV VLLM_DTYPE=float16
ENV VLLM_GPU_MEMORY_UTILIZATION=0.85
ENV VLLM_MAX_MODEL_LEN=18000
ENV VLLM_MAX_NUM_SEQS=16
ENV VLLM_MAX_BATCHED_TOKENS=4096
# Latency-tuned vLLM serve flags (CUDA graph, prefix caching, chunked prefill)
# are applied in scripts/start_single_container.sh — see that file and
# docs/quantization_benchmark_results.md §5 for the measured rationale.

WORKDIR /opt/suya-ocr

COPY requirements.txt pyproject.toml ./
RUN pip install --no-cache-dir uv==0.8.15 && \
  python3 -m uv pip install --system -r requirements.txt

COPY . .

EXPOSE 5002

ENTRYPOINT []
CMD ["/opt/suya-ocr/scripts/start_single_container.sh"]
