From 62e83714fcf440bd127e685d8fc8624cd4a0e1fe Mon Sep 17 00:00:00 2001 From: James Liounis Date: Fri, 21 Aug 2026 16:49:48 -0400 Subject: [PATCH] fix(parallel_ai): stop a caller from pricing its own search request MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `_parallel_ai_usage` carries the provider's reported usage into cost calculation. It was only written when the response contained a usage block, so a caller could pass `_parallel_ai_usage=[{"name": "sku_search", "count": 0}]` and, whenever the provider omitted usage, bill $0.00 instead of $0.005 — the value also reached the upstream request body as an unknown field. The key is now stripped from inbound params and written unconditionally from the parsed response, so only the provider can populate it. --- .../llms/parallel_ai/search/transformation.py | 15 ++++++++---- .../parallel_ai/test_parallel_ai_search.py | 24 +++++++++++++++++++ 2 files changed, 34 insertions(+), 5 deletions(-) diff --git a/litellm/llms/parallel_ai/search/transformation.py b/litellm/llms/parallel_ai/search/transformation.py index 3092674363b..bf4b19b1d09 100644 --- a/litellm/llms/parallel_ai/search/transformation.py +++ b/litellm/llms/parallel_ai/search/transformation.py @@ -214,6 +214,10 @@ class ParallelAISearchConfig(BaseSearchConfig): # unified-spec param with no v1 equivalent params.pop("max_tokens_per_page", None) + # reserved for the provider's own reported usage, which prices the request; + # a caller-supplied value would otherwise set its own cost + params.pop(PARALLEL_AI_USAGE_PARAM, None) + return {**request_data, **params} def transform_search_response( @@ -237,11 +241,12 @@ class ParallelAISearchConfig(BaseSearchConfig): """ parsed: Final = _ParallelAIV1SearchResponse.model_validate(raw_response.json()) - if parsed.usage is not None: - logging_obj.optional_params = { - **logging_obj.optional_params, - PARALLEL_AI_USAGE_PARAM: parsed.usage, - } + # written unconditionally: leaving a caller-supplied value in place when the + # provider reports no usage would let the caller price its own request + logging_obj.optional_params = { + **logging_obj.optional_params, + PARALLEL_AI_USAGE_PARAM: parsed.usage, + } results: Final = tuple( SearchResult.model_validate( diff --git a/tests/test_litellm/llms/parallel_ai/test_parallel_ai_search.py b/tests/test_litellm/llms/parallel_ai/test_parallel_ai_search.py index 38846c741cc..337d2314831 100644 --- a/tests/test_litellm/llms/parallel_ai/test_parallel_ai_search.py +++ b/tests/test_litellm/llms/parallel_ai/test_parallel_ai_search.py @@ -472,3 +472,27 @@ class TestParallelAISearch: ) assert response._hidden_params["response_cost"] == pytest.approx(0.005) + + @pytest.mark.asyncio + async def test_caller_cannot_supply_provider_usage(self, bundled_cost_map): + """`_parallel_ai_usage` prices the request, so a caller must not be able to set it. + + The provider reports no usage here, which is the case where a caller-supplied + value would otherwise survive into the cost calculation. + """ + response_payload = {k: v for k, v in MOCK_V1_RESPONSE.items() if k != "usage"} + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + new_callable=AsyncMock, + ) as mock_post: + mock_post.return_value = _mock_response(response_payload) + + response = await litellm.asearch( + query="AI developments", + search_provider="parallel_ai", + mode="basic", + _parallel_ai_usage=[{"name": "sku_search", "count": 0}], + ) + + assert response._hidden_params["response_cost"] == pytest.approx(0.005) + assert "_parallel_ai_usage" not in mock_post.call_args.kwargs["json"]