fix(veria): scope reducto file IDs to current request + register pricing

- Reject reducto:// file IDs sent through the proxy /v1/ocr JSON API.
  The IDs are not bound to a LiteLLM key, so an authenticated user
  could submit another user's file ID and receive OCR text via the
  proxy's shared Reducto credentials. Force fresh uploads (multipart
  form or inline base64 data URI) so every OCR call is server-mediated
  and implicitly bound to the originating request.

- Add ocr_cost_per_credit=0.015 to reducto/parse-v3 and
  reducto/parse-legacy in both pricing JSONs so successful Reducto OCR
  calls debit key/team spend instead of recording zero.
This commit is contained in:
mateo-berri 2026-05-20 22:14:33 +00:00
parent 9ee908089b
commit 2206105ec6
No known key found for this signature in database
4 changed files with 81 additions and 9 deletions

View file

@ -29103,6 +29103,8 @@
"reducto/parse-legacy": {
"litellm_provider": "reducto",
"mode": "ocr",
"ocr_cost_per_credit": 0.015,
"source": "https://reducto.ai/pricing",
"supported_endpoints": [
"/v1/ocr"
]
@ -29110,6 +29112,8 @@
"reducto/parse-v3": {
"litellm_provider": "reducto",
"mode": "ocr",
"ocr_cost_per_credit": 0.015,
"source": "https://reducto.ai/pricing",
"supported_endpoints": [
"/v1/ocr"
]

View file

@ -178,6 +178,24 @@ async def _parse_ocr_request(request: Request) -> Dict[str, Any]:
"For JSON requests, use 'document_url' or 'image_url' document types."
)
# Security: reject provider-native file IDs (e.g. reducto://) received via
# JSON. These IDs are not scoped to the LiteLLM proxy user/key, so an
# authenticated user who obtains another user's file ID could submit it
# here and receive the OCR result using the proxy's shared provider
# credentials. Force callers to upload fresh content per request via
# multipart/form-data or an inline base64 data URI, both of which produce
# a server-mediated upload bound to the current request.
if isinstance(doc, dict):
for url_field in ("document_url", "image_url"):
url_value = doc.get(url_field)
if isinstance(url_value, str) and url_value.startswith("reducto://"):
raise ValueError(
"reducto:// file IDs are not accepted through the proxy "
"OCR API; upload the file in the same request via "
"multipart/form-data with a 'file' field, or pass an "
"inline base64 data URI as the document URL."
)
return data

View file

@ -29123,6 +29123,8 @@
"reducto/parse-legacy": {
"litellm_provider": "reducto",
"mode": "ocr",
"ocr_cost_per_credit": 0.015,
"source": "https://reducto.ai/pricing",
"supported_endpoints": [
"/v1/ocr"
]
@ -29130,6 +29132,8 @@
"reducto/parse-v3": {
"litellm_provider": "reducto",
"mode": "ocr",
"ocr_cost_per_credit": 0.015,
"source": "https://reducto.ai/pricing",
"supported_endpoints": [
"/v1/ocr"
]

View file

@ -51,16 +51,10 @@ def client_no_auth(fake_env_vars):
litellm.in_memory_llm_clients_cache.flush_cache()
def test_proxy_reducto_ocr_json_passthrough(client_no_auth):
mocked_response = OCRResponse(
pages=[OCRPage(index=0, markdown="Proxy OCR")],
model="parse-v3",
usage_info=OCRUsageInfo(pages_processed=1, credits=1),
)
def test_proxy_reducto_ocr_json_rejects_reducto_id(client_no_auth):
with patch(
"litellm.proxy.proxy_server.llm_router.aocr",
new=AsyncMock(return_value=mocked_response),
new=AsyncMock(),
) as mock_aocr:
response = client_no_auth.post(
"/v1/ocr",
@ -75,12 +69,64 @@ def test_proxy_reducto_ocr_json_passthrough(client_no_auth):
},
)
assert response.status_code >= 400
assert "reducto://" in response.text
assert mock_aocr.await_count == 0
def test_proxy_reducto_ocr_json_rejects_reducto_id_in_image_url(client_no_auth):
with patch(
"litellm.proxy.proxy_server.llm_router.aocr",
new=AsyncMock(),
) as mock_aocr:
response = client_no_auth.post(
"/v1/ocr",
json={
"model": "reducto/parse-v3",
"document": {
"type": "image_url",
"image_url": "reducto://proxy.png",
},
},
)
assert response.status_code >= 400
assert "reducto://" in response.text
assert mock_aocr.await_count == 0
def test_proxy_reducto_ocr_json_passthrough_data_uri(client_no_auth):
mocked_response = OCRResponse(
pages=[OCRPage(index=0, markdown="Proxy OCR")],
model="parse-v3",
usage_info=OCRUsageInfo(pages_processed=1, credits=1),
)
data_uri = "data:application/pdf;base64,JVBERi0xLjQK"
with patch(
"litellm.proxy.proxy_server.llm_router.aocr",
new=AsyncMock(return_value=mocked_response),
) as mock_aocr:
response = client_no_auth.post(
"/v1/ocr",
json={
"model": "reducto/parse-v3",
"document": {
"type": "document_url",
"document_url": data_uri,
},
"api_key": "proxy-key",
"api_base": "https://platform.reducto.ai",
},
)
assert response.status_code == 200
assert mock_aocr.await_count == 1
assert mock_aocr.await_args.kwargs["model"] == "reducto/parse-v3"
assert mock_aocr.await_args.kwargs["document"] == {
"type": "document_url",
"document_url": "reducto://proxy.pdf",
"document_url": data_uri,
}
assert mock_aocr.await_args.kwargs["api_key"] == "proxy-key"
assert mock_aocr.await_args.kwargs["api_base"] == "https://platform.reducto.ai"