revert(ocr): remove implementation and reducto test changes

This commit is contained in:
Yujong Lee 2026-08-29 12:07:14 -07:00 • committed by GitHub
parent 78a195b147
commit bff6268ee2
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 38 additions and 50 deletions

View file

@ -1,4 +1,4 @@
from typing import TYPE_CHECKING, Final
from typing import TYPE_CHECKING, Any, Final
import httpx
@ -150,8 +150,8 @@ class _BaseReductoOCRConfig(BaseOCRConfig):
class ReductoParseV3Config(_BaseReductoOCRConfig):
def get_supported_ocr_params(self, model: str) -> list[str]:
return ["enhance", "retrieval", "formatting", "spreadsheet", "settings"]
def get_supported_ocr_params(self, model: str) -> list:
return ["formatting", "retrieval", "settings"]
def transform_ocr_request(
self,
@ -187,11 +187,15 @@ class ReductoParseV3Config(_BaseReductoOCRConfig):
class ReductoParseLegacyConfig(_BaseReductoOCRConfig):
def get_supported_ocr_params(self, model: str) -> list[str]:
return ["options", "advanced_options", "experimental_options", "priority"]
def get_supported_ocr_params(self, model: str) -> list:
return ["enhance"]
def _build_legacy_body(self, file_id: str, optional_params: dict[str, object]) -> dict[str, object]:
return {"document_url": file_id, **optional_params}
def _build_legacy_body(self, file_id: str, optional_params: dict) -> dict[str, Any]:
body: Final[dict[str, Any]] = {"document_url": file_id}
enhance: Final = optional_params.get("enhance")
if enhance is not None:
body["options"] = {"enhance": enhance}
return body
def transform_ocr_request(
self,

View file

@ -62,7 +62,8 @@ class _PreparedRustOCRCall:
_RUST_OCR_PROVIDERS: Final = {
"mistral",
"reducto",
"azure_ai",
"vertex_ai",
}
@ -190,11 +191,6 @@ def _prepare_ocr_request(
def _rust_ocr_supported(prepared_request: _PreparedOCRRequest) -> bool:
if prepared_request.optional_params.get(OCR_REQUEST_FORMAT_PARAM) == "native":
return False
if prepared_request.custom_llm_provider == "reducto":
source: Final = prepared_request.document.get(
"document_url"
) or prepared_request.document.get("image_url")
return prepared_request.model == "parse-v3" and isinstance(source, str) and source.startswith("reducto://")
return prepared_request.custom_llm_provider in _RUST_OCR_PROVIDERS
@ -304,17 +300,7 @@ def _run_rust_ocr(
)
if rust_response is None:
return None
return _ocr_response_from_rust(rust_response)
def _ocr_response_from_rust(rust_response: dict[str, object]) -> OCRResponse:
reducto_raw: Final = rust_response.get("reducto_raw")
response: Final = OCRResponse.model_validate(
{key: value for key, value in rust_response.items() if key != "reducto_raw"}
)
if isinstance(reducto_raw, dict):
response._hidden_params["reducto_raw"] = reducto_raw
return response
return OCRResponse.model_validate(rust_response)
async def _run_rust_aocr(
@ -339,7 +325,7 @@ async def _run_rust_aocr(
)
if rust_response is None:
return None
return _ocr_response_from_rust(rust_response)
return OCRResponse.model_validate(rust_response)
@client

View file

@ -1,8 +1,7 @@
import json
import pytest
import litellm
import pytest
@pytest.fixture()
@ -18,7 +17,9 @@ def disable_aiohttp_transport():
@pytest.mark.asyncio
async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transport, respx_mock):
async def test_parse_legacy_wraps_enhance_under_options(
disable_aiohttp_transport, respx_mock
):
upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond(
json={"file_id": "reducto://legacy.pdf"}
)
@ -45,10 +46,7 @@ async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transp
},
api_key="legacy-key",
api_base="https://platform.reducto.ai",
options={"ocr_mode": "agentic", "chunking": {"chunk_mode": "section"}},
advanced_options={"table_output_format": "html", "ocr_system": "highres"},
experimental_options={"enable_checkboxes": True},
priority=False,
enhance={"agentic": [{"type": "table"}]},
)
assert upload_route.called
@ -56,9 +54,6 @@ async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transp
request_body = json.loads(parse_route.calls[0].request.read())
assert request_body == {
"document_url": "reducto://legacy.pdf",
"options": {"ocr_mode": "agentic", "chunking": {"chunk_mode": "section"}},
"advanced_options": {"table_output_format": "html", "ocr_system": "highres"},
"experimental_options": {"enable_checkboxes": True},
"priority": False,
"options": {"enhance": {"agentic": [{"type": "table"}]}},
}
assert response.pages[0].markdown == "Legacy parse"

View file

@ -1,8 +1,7 @@
import json
import pytest
import litellm
import pytest
def _reducto_parse_response() -> dict:
@ -69,11 +68,15 @@ def disable_aiohttp_transport():
@pytest.mark.asyncio
async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transport, respx_mock):
async def test_parse_v3_file_upload_and_response_mapping(
disable_aiohttp_transport, respx_mock
):
upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond(
json={"file_id": "reducto://uploaded.pdf"}
)
parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(json=_reducto_parse_response())
parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(
json=_reducto_parse_response()
)
response = await litellm.aocr(
model="reducto/parse-v3",
@ -84,10 +87,8 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo
},
api_key="test-key",
api_base="https://platform.reducto.ai",
enhance={"agentic": [{"scope": "table", "mode": "max"}]},
formatting={"table_output_format": "html"},
retrieval={"chunking": {"chunk_mode": "section"}},
spreadsheet={"clustering": "fast"},
retrieval={"chunk_mode": "section"},
settings={"ocr_system": "standard"},
)
@ -105,10 +106,8 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo
parse_request_body = json.loads(parse_route.calls[0].request.read())
assert parse_request_body["input"] == "reducto://uploaded.pdf"
assert parse_request_body["enhance"] == {"agentic": [{"scope": "table", "mode": "max"}]}
assert parse_request_body["formatting"] == {"table_output_format": "html"}
assert parse_request_body["retrieval"] == {"chunking": {"chunk_mode": "section"}}
assert parse_request_body["spreadsheet"] == {"clustering": "fast"}
assert parse_request_body["retrieval"] == {"chunk_mode": "section"}
assert parse_request_body["settings"] == {"ocr_system": "standard"}
assert response.usage_info is not None
@ -124,11 +123,15 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo
@pytest.mark.asyncio
async def test_parse_v3_reducto_id_passthrough_skips_upload(disable_aiohttp_transport, respx_mock):
async def test_parse_v3_reducto_id_passthrough_skips_upload(
disable_aiohttp_transport, respx_mock
):
upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond(
json={"file_id": "reducto://should-not-upload.pdf"}
)
parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(json=_reducto_parse_response())
parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(
json=_reducto_parse_response()
)
response = await litellm.aocr(
model="reducto/parse-v3",
@ -138,12 +141,12 @@ async def test_parse_v3_reducto_id_passthrough_skips_upload(disable_aiohttp_tran
},
api_key="test-key",
api_base="https://platform.reducto.ai",
retrieval={"chunking": {"chunk_mode": "section"}},
retrieval={"chunk_mode": "section"},
)
assert not upload_route.called
assert parse_route.called
parse_request_body = json.loads(parse_route.calls[0].request.read())
assert parse_request_body["input"] == "reducto://already-uploaded.pdf"
assert parse_request_body["retrieval"]["chunking"]["chunk_mode"] == "section"
assert parse_request_body["retrieval"]["chunk_mode"] == "section"
assert response.pages[0].markdown.startswith("Page 1 block A")