Suya OCR API — vLLM-backed, OpenAI-compatible OCR service
FastAPI service wrapping the Surya-OCR-2 model (datalab-to) served through vLLM: legacy /v1/api/ai/* endpoints, an OpenAI-compatible /v1/chat/completions endpoint, a coalescing request batcher, a local OCR CLI, Docker packaging, multilingual example outputs, and quantization/concurrency benchmarks. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,96 @@
|
||||
from typing import Tuple
|
||||
|
||||
from PIL import Image
|
||||
|
||||
|
||||
def scale_to_fit(
|
||||
img: Image.Image,
|
||||
max_size: Tuple[int, int] = (3072, 2048),
|
||||
min_size: Tuple[int, int] = (1792, 28),
|
||||
grid_size: int = 28,
|
||||
) -> Image.Image:
|
||||
resample_method = Image.Resampling.LANCZOS
|
||||
|
||||
width, height = img.size
|
||||
|
||||
if width <= 0 or height <= 0:
|
||||
return img
|
||||
|
||||
original_ar = width / height
|
||||
current_pixels = width * height
|
||||
max_pixels = max_size[0] * max_size[1]
|
||||
min_pixels = min_size[0] * min_size[1]
|
||||
|
||||
scale = 1.0
|
||||
if current_pixels > max_pixels:
|
||||
scale = (max_pixels / current_pixels) ** 0.5
|
||||
elif current_pixels < min_pixels:
|
||||
scale = (min_pixels / current_pixels) ** 0.5
|
||||
|
||||
w_blocks = max(1, round((width * scale) / grid_size))
|
||||
h_blocks = max(1, round((height * scale) / grid_size))
|
||||
|
||||
while (w_blocks * h_blocks * grid_size * grid_size) > max_pixels:
|
||||
if w_blocks == 1 and h_blocks == 1:
|
||||
break
|
||||
|
||||
if w_blocks == 1:
|
||||
h_blocks -= 1
|
||||
continue
|
||||
if h_blocks == 1:
|
||||
w_blocks -= 1
|
||||
continue
|
||||
|
||||
ar_w_loss = abs(((w_blocks - 1) / h_blocks) - original_ar)
|
||||
ar_h_loss = abs((w_blocks / (h_blocks - 1)) - original_ar)
|
||||
|
||||
if ar_w_loss < ar_h_loss:
|
||||
w_blocks -= 1
|
||||
else:
|
||||
h_blocks -= 1
|
||||
|
||||
new_width = w_blocks * grid_size
|
||||
new_height = h_blocks * grid_size
|
||||
|
||||
if (new_width, new_height) == (width, height):
|
||||
return img
|
||||
|
||||
return img.resize((new_width, new_height), resample=resample_method)
|
||||
|
||||
|
||||
def detect_repeat_token(
|
||||
predicted_tokens: str,
|
||||
base_max_repeats: int = 4,
|
||||
window_size: int = 500,
|
||||
cut_from_end: int = 0,
|
||||
scaling_factor: float = 3.0,
|
||||
) -> bool:
|
||||
if cut_from_end > 0:
|
||||
predicted_tokens = predicted_tokens[:-cut_from_end]
|
||||
|
||||
for seq_len in range(1, window_size // 2 + 1):
|
||||
candidate_seq = predicted_tokens[-seq_len:]
|
||||
|
||||
max_repeats = int(base_max_repeats * (1 + scaling_factor / seq_len))
|
||||
|
||||
repeat_count = 0
|
||||
pos = len(predicted_tokens) - seq_len
|
||||
if pos < 0:
|
||||
continue
|
||||
|
||||
while pos >= 0:
|
||||
if predicted_tokens[pos : pos + seq_len] == candidate_seq:
|
||||
repeat_count += 1
|
||||
pos -= seq_len
|
||||
else:
|
||||
break
|
||||
|
||||
if repeat_count > max_repeats:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def image_token_budget(block_count: int, ceiling: int = 4096, floor: int = 64) -> int:
|
||||
"""Per-block max_tokens: count + 100, clamped to [floor, ceiling]."""
|
||||
return min(max(block_count + 100, floor), ceiling)
|
||||
Reference in New Issue
Block a user