diff --git a/litellm/llms/reducto/ocr/transformation.py b/litellm/llms/reducto/ocr/transformation.py index 109875c8999..a7216e4ec40 100644 --- a/litellm/llms/reducto/ocr/transformation.py +++ b/litellm/llms/reducto/ocr/transformation.py @@ -1,4 +1,4 @@ -from typing import TYPE_CHECKING, Final +from typing import TYPE_CHECKING, Any, Final import httpx @@ -150,8 +150,8 @@ class _BaseReductoOCRConfig(BaseOCRConfig): class ReductoParseV3Config(_BaseReductoOCRConfig): - def get_supported_ocr_params(self, model: str) -> list[str]: - return ["enhance", "retrieval", "formatting", "spreadsheet", "settings"] + def get_supported_ocr_params(self, model: str) -> list: + return ["formatting", "retrieval", "settings"] def transform_ocr_request( self, @@ -187,11 +187,15 @@ class ReductoParseV3Config(_BaseReductoOCRConfig): class ReductoParseLegacyConfig(_BaseReductoOCRConfig): - def get_supported_ocr_params(self, model: str) -> list[str]: - return ["options", "advanced_options", "experimental_options", "priority"] + def get_supported_ocr_params(self, model: str) -> list: + return ["enhance"] - def _build_legacy_body(self, file_id: str, optional_params: dict[str, object]) -> dict[str, object]: - return {"document_url": file_id, **optional_params} + def _build_legacy_body(self, file_id: str, optional_params: dict) -> dict[str, Any]: + body: Final[dict[str, Any]] = {"document_url": file_id} + enhance: Final = optional_params.get("enhance") + if enhance is not None: + body["options"] = {"enhance": enhance} + return body def transform_ocr_request( self, diff --git a/litellm/ocr/main.py b/litellm/ocr/main.py index 96836937817..b918f013700 100644 --- a/litellm/ocr/main.py +++ b/litellm/ocr/main.py @@ -62,7 +62,8 @@ class _PreparedRustOCRCall: _RUST_OCR_PROVIDERS: Final = { "mistral", - "reducto", + "azure_ai", + "vertex_ai", } @@ -190,11 +191,6 @@ def _prepare_ocr_request( def _rust_ocr_supported(prepared_request: _PreparedOCRRequest) -> bool: if prepared_request.optional_params.get(OCR_REQUEST_FORMAT_PARAM) == "native": return False - if prepared_request.custom_llm_provider == "reducto": - source: Final = prepared_request.document.get( - "document_url" - ) or prepared_request.document.get("image_url") - return prepared_request.model == "parse-v3" and isinstance(source, str) and source.startswith("reducto://") return prepared_request.custom_llm_provider in _RUST_OCR_PROVIDERS @@ -304,17 +300,7 @@ def _run_rust_ocr( ) if rust_response is None: return None - return _ocr_response_from_rust(rust_response) - - -def _ocr_response_from_rust(rust_response: dict[str, object]) -> OCRResponse: - reducto_raw: Final = rust_response.get("reducto_raw") - response: Final = OCRResponse.model_validate( - {key: value for key, value in rust_response.items() if key != "reducto_raw"} - ) - if isinstance(reducto_raw, dict): - response._hidden_params["reducto_raw"] = reducto_raw - return response + return OCRResponse.model_validate(rust_response) async def _run_rust_aocr( @@ -339,7 +325,7 @@ async def _run_rust_aocr( ) if rust_response is None: return None - return _ocr_response_from_rust(rust_response) + return OCRResponse.model_validate(rust_response) @client diff --git a/tests/test_litellm/llms/reducto/test_parse_legacy.py b/tests/test_litellm/llms/reducto/test_parse_legacy.py index 823b7dff1f7..db19460baa3 100644 --- a/tests/test_litellm/llms/reducto/test_parse_legacy.py +++ b/tests/test_litellm/llms/reducto/test_parse_legacy.py @@ -1,8 +1,7 @@ import json -import pytest - import litellm +import pytest @pytest.fixture() @@ -18,7 +17,9 @@ def disable_aiohttp_transport(): @pytest.mark.asyncio -async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transport, respx_mock): +async def test_parse_legacy_wraps_enhance_under_options( + disable_aiohttp_transport, respx_mock +): upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond( json={"file_id": "reducto://legacy.pdf"} ) @@ -45,10 +46,7 @@ async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transp }, api_key="legacy-key", api_base="https://platform.reducto.ai", - options={"ocr_mode": "agentic", "chunking": {"chunk_mode": "section"}}, - advanced_options={"table_output_format": "html", "ocr_system": "highres"}, - experimental_options={"enable_checkboxes": True}, - priority=False, + enhance={"agentic": [{"type": "table"}]}, ) assert upload_route.called @@ -56,9 +54,6 @@ async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transp request_body = json.loads(parse_route.calls[0].request.read()) assert request_body == { "document_url": "reducto://legacy.pdf", - "options": {"ocr_mode": "agentic", "chunking": {"chunk_mode": "section"}}, - "advanced_options": {"table_output_format": "html", "ocr_system": "highres"}, - "experimental_options": {"enable_checkboxes": True}, - "priority": False, + "options": {"enhance": {"agentic": [{"type": "table"}]}}, } assert response.pages[0].markdown == "Legacy parse" diff --git a/tests/test_litellm/llms/reducto/test_parse_v3.py b/tests/test_litellm/llms/reducto/test_parse_v3.py index 07b8a732d7c..140b9737dc0 100644 --- a/tests/test_litellm/llms/reducto/test_parse_v3.py +++ b/tests/test_litellm/llms/reducto/test_parse_v3.py @@ -1,8 +1,7 @@ import json -import pytest - import litellm +import pytest def _reducto_parse_response() -> dict: @@ -69,11 +68,15 @@ def disable_aiohttp_transport(): @pytest.mark.asyncio -async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transport, respx_mock): +async def test_parse_v3_file_upload_and_response_mapping( + disable_aiohttp_transport, respx_mock +): upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond( json={"file_id": "reducto://uploaded.pdf"} ) - parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(json=_reducto_parse_response()) + parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond( + json=_reducto_parse_response() + ) response = await litellm.aocr( model="reducto/parse-v3", @@ -84,10 +87,8 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo }, api_key="test-key", api_base="https://platform.reducto.ai", - enhance={"agentic": [{"scope": "table", "mode": "max"}]}, formatting={"table_output_format": "html"}, - retrieval={"chunking": {"chunk_mode": "section"}}, - spreadsheet={"clustering": "fast"}, + retrieval={"chunk_mode": "section"}, settings={"ocr_system": "standard"}, ) @@ -105,10 +106,8 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo parse_request_body = json.loads(parse_route.calls[0].request.read()) assert parse_request_body["input"] == "reducto://uploaded.pdf" - assert parse_request_body["enhance"] == {"agentic": [{"scope": "table", "mode": "max"}]} assert parse_request_body["formatting"] == {"table_output_format": "html"} - assert parse_request_body["retrieval"] == {"chunking": {"chunk_mode": "section"}} - assert parse_request_body["spreadsheet"] == {"clustering": "fast"} + assert parse_request_body["retrieval"] == {"chunk_mode": "section"} assert parse_request_body["settings"] == {"ocr_system": "standard"} assert response.usage_info is not None @@ -124,11 +123,15 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo @pytest.mark.asyncio -async def test_parse_v3_reducto_id_passthrough_skips_upload(disable_aiohttp_transport, respx_mock): +async def test_parse_v3_reducto_id_passthrough_skips_upload( + disable_aiohttp_transport, respx_mock +): upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond( json={"file_id": "reducto://should-not-upload.pdf"} ) - parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(json=_reducto_parse_response()) + parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond( + json=_reducto_parse_response() + ) response = await litellm.aocr( model="reducto/parse-v3", @@ -138,12 +141,12 @@ async def test_parse_v3_reducto_id_passthrough_skips_upload(disable_aiohttp_tran }, api_key="test-key", api_base="https://platform.reducto.ai", - retrieval={"chunking": {"chunk_mode": "section"}}, + retrieval={"chunk_mode": "section"}, ) assert not upload_route.called assert parse_route.called parse_request_body = json.loads(parse_route.calls[0].request.read()) assert parse_request_body["input"] == "reducto://already-uploaded.pdf" - assert parse_request_body["retrieval"]["chunking"]["chunk_mode"] == "section" + assert parse_request_body["retrieval"]["chunk_mode"] == "section" assert response.pages[0].markdown.startswith("Page 1 block A")