mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
revert(ocr): remove implementation and reducto test changes
This commit is contained in:
parent
78a195b147
commit
bff6268ee2
4 changed files with 38 additions and 50 deletions
|
|
@ -1,4 +1,4 @@
|
|||
from typing import TYPE_CHECKING, Final
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
|
||||
import httpx
|
||||
|
||||
|
|
@ -150,8 +150,8 @@ class _BaseReductoOCRConfig(BaseOCRConfig):
|
|||
|
||||
|
||||
class ReductoParseV3Config(_BaseReductoOCRConfig):
|
||||
def get_supported_ocr_params(self, model: str) -> list[str]:
|
||||
return ["enhance", "retrieval", "formatting", "spreadsheet", "settings"]
|
||||
def get_supported_ocr_params(self, model: str) -> list:
|
||||
return ["formatting", "retrieval", "settings"]
|
||||
|
||||
def transform_ocr_request(
|
||||
self,
|
||||
|
|
@ -187,11 +187,15 @@ class ReductoParseV3Config(_BaseReductoOCRConfig):
|
|||
|
||||
|
||||
class ReductoParseLegacyConfig(_BaseReductoOCRConfig):
|
||||
def get_supported_ocr_params(self, model: str) -> list[str]:
|
||||
return ["options", "advanced_options", "experimental_options", "priority"]
|
||||
def get_supported_ocr_params(self, model: str) -> list:
|
||||
return ["enhance"]
|
||||
|
||||
def _build_legacy_body(self, file_id: str, optional_params: dict[str, object]) -> dict[str, object]:
|
||||
return {"document_url": file_id, **optional_params}
|
||||
def _build_legacy_body(self, file_id: str, optional_params: dict) -> dict[str, Any]:
|
||||
body: Final[dict[str, Any]] = {"document_url": file_id}
|
||||
enhance: Final = optional_params.get("enhance")
|
||||
if enhance is not None:
|
||||
body["options"] = {"enhance": enhance}
|
||||
return body
|
||||
|
||||
def transform_ocr_request(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -62,7 +62,8 @@ class _PreparedRustOCRCall:
|
|||
|
||||
_RUST_OCR_PROVIDERS: Final = {
|
||||
"mistral",
|
||||
"reducto",
|
||||
"azure_ai",
|
||||
"vertex_ai",
|
||||
}
|
||||
|
||||
|
||||
|
|
@ -190,11 +191,6 @@ def _prepare_ocr_request(
|
|||
def _rust_ocr_supported(prepared_request: _PreparedOCRRequest) -> bool:
|
||||
if prepared_request.optional_params.get(OCR_REQUEST_FORMAT_PARAM) == "native":
|
||||
return False
|
||||
if prepared_request.custom_llm_provider == "reducto":
|
||||
source: Final = prepared_request.document.get(
|
||||
"document_url"
|
||||
) or prepared_request.document.get("image_url")
|
||||
return prepared_request.model == "parse-v3" and isinstance(source, str) and source.startswith("reducto://")
|
||||
return prepared_request.custom_llm_provider in _RUST_OCR_PROVIDERS
|
||||
|
||||
|
||||
|
|
@ -304,17 +300,7 @@ def _run_rust_ocr(
|
|||
)
|
||||
if rust_response is None:
|
||||
return None
|
||||
return _ocr_response_from_rust(rust_response)
|
||||
|
||||
|
||||
def _ocr_response_from_rust(rust_response: dict[str, object]) -> OCRResponse:
|
||||
reducto_raw: Final = rust_response.get("reducto_raw")
|
||||
response: Final = OCRResponse.model_validate(
|
||||
{key: value for key, value in rust_response.items() if key != "reducto_raw"}
|
||||
)
|
||||
if isinstance(reducto_raw, dict):
|
||||
response._hidden_params["reducto_raw"] = reducto_raw
|
||||
return response
|
||||
return OCRResponse.model_validate(rust_response)
|
||||
|
||||
|
||||
async def _run_rust_aocr(
|
||||
|
|
@ -339,7 +325,7 @@ async def _run_rust_aocr(
|
|||
)
|
||||
if rust_response is None:
|
||||
return None
|
||||
return _ocr_response_from_rust(rust_response)
|
||||
return OCRResponse.model_validate(rust_response)
|
||||
|
||||
|
||||
@client
|
||||
|
|
|
|||
|
|
@ -1,8 +1,7 @@
|
|||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
|
|
@ -18,7 +17,9 @@ def disable_aiohttp_transport():
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transport, respx_mock):
|
||||
async def test_parse_legacy_wraps_enhance_under_options(
|
||||
disable_aiohttp_transport, respx_mock
|
||||
):
|
||||
upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond(
|
||||
json={"file_id": "reducto://legacy.pdf"}
|
||||
)
|
||||
|
|
@ -45,10 +46,7 @@ async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transp
|
|||
},
|
||||
api_key="legacy-key",
|
||||
api_base="https://platform.reducto.ai",
|
||||
options={"ocr_mode": "agentic", "chunking": {"chunk_mode": "section"}},
|
||||
advanced_options={"table_output_format": "html", "ocr_system": "highres"},
|
||||
experimental_options={"enable_checkboxes": True},
|
||||
priority=False,
|
||||
enhance={"agentic": [{"type": "table"}]},
|
||||
)
|
||||
|
||||
assert upload_route.called
|
||||
|
|
@ -56,9 +54,6 @@ async def test_parse_legacy_forwards_legacy_option_groups(disable_aiohttp_transp
|
|||
request_body = json.loads(parse_route.calls[0].request.read())
|
||||
assert request_body == {
|
||||
"document_url": "reducto://legacy.pdf",
|
||||
"options": {"ocr_mode": "agentic", "chunking": {"chunk_mode": "section"}},
|
||||
"advanced_options": {"table_output_format": "html", "ocr_system": "highres"},
|
||||
"experimental_options": {"enable_checkboxes": True},
|
||||
"priority": False,
|
||||
"options": {"enhance": {"agentic": [{"type": "table"}]}},
|
||||
}
|
||||
assert response.pages[0].markdown == "Legacy parse"
|
||||
|
|
|
|||
|
|
@ -1,8 +1,7 @@
|
|||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
import pytest
|
||||
|
||||
|
||||
def _reducto_parse_response() -> dict:
|
||||
|
|
@ -69,11 +68,15 @@ def disable_aiohttp_transport():
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transport, respx_mock):
|
||||
async def test_parse_v3_file_upload_and_response_mapping(
|
||||
disable_aiohttp_transport, respx_mock
|
||||
):
|
||||
upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond(
|
||||
json={"file_id": "reducto://uploaded.pdf"}
|
||||
)
|
||||
parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(json=_reducto_parse_response())
|
||||
parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(
|
||||
json=_reducto_parse_response()
|
||||
)
|
||||
|
||||
response = await litellm.aocr(
|
||||
model="reducto/parse-v3",
|
||||
|
|
@ -84,10 +87,8 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo
|
|||
},
|
||||
api_key="test-key",
|
||||
api_base="https://platform.reducto.ai",
|
||||
enhance={"agentic": [{"scope": "table", "mode": "max"}]},
|
||||
formatting={"table_output_format": "html"},
|
||||
retrieval={"chunking": {"chunk_mode": "section"}},
|
||||
spreadsheet={"clustering": "fast"},
|
||||
retrieval={"chunk_mode": "section"},
|
||||
settings={"ocr_system": "standard"},
|
||||
)
|
||||
|
||||
|
|
@ -105,10 +106,8 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo
|
|||
|
||||
parse_request_body = json.loads(parse_route.calls[0].request.read())
|
||||
assert parse_request_body["input"] == "reducto://uploaded.pdf"
|
||||
assert parse_request_body["enhance"] == {"agentic": [{"scope": "table", "mode": "max"}]}
|
||||
assert parse_request_body["formatting"] == {"table_output_format": "html"}
|
||||
assert parse_request_body["retrieval"] == {"chunking": {"chunk_mode": "section"}}
|
||||
assert parse_request_body["spreadsheet"] == {"clustering": "fast"}
|
||||
assert parse_request_body["retrieval"] == {"chunk_mode": "section"}
|
||||
assert parse_request_body["settings"] == {"ocr_system": "standard"}
|
||||
|
||||
assert response.usage_info is not None
|
||||
|
|
@ -124,11 +123,15 @@ async def test_parse_v3_file_upload_and_response_mapping(disable_aiohttp_transpo
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_parse_v3_reducto_id_passthrough_skips_upload(disable_aiohttp_transport, respx_mock):
|
||||
async def test_parse_v3_reducto_id_passthrough_skips_upload(
|
||||
disable_aiohttp_transport, respx_mock
|
||||
):
|
||||
upload_route = respx_mock.post("https://platform.reducto.ai/upload").respond(
|
||||
json={"file_id": "reducto://should-not-upload.pdf"}
|
||||
)
|
||||
parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(json=_reducto_parse_response())
|
||||
parse_route = respx_mock.post("https://platform.reducto.ai/parse").respond(
|
||||
json=_reducto_parse_response()
|
||||
)
|
||||
|
||||
response = await litellm.aocr(
|
||||
model="reducto/parse-v3",
|
||||
|
|
@ -138,12 +141,12 @@ async def test_parse_v3_reducto_id_passthrough_skips_upload(disable_aiohttp_tran
|
|||
},
|
||||
api_key="test-key",
|
||||
api_base="https://platform.reducto.ai",
|
||||
retrieval={"chunking": {"chunk_mode": "section"}},
|
||||
retrieval={"chunk_mode": "section"},
|
||||
)
|
||||
|
||||
assert not upload_route.called
|
||||
assert parse_route.called
|
||||
parse_request_body = json.loads(parse_route.calls[0].request.read())
|
||||
assert parse_request_body["input"] == "reducto://already-uploaded.pdf"
|
||||
assert parse_request_body["retrieval"]["chunking"]["chunk_mode"] == "section"
|
||||
assert parse_request_body["retrieval"]["chunk_mode"] == "section"
|
||||
assert response.pages[0].markdown.startswith("Page 1 block A")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue