diff --git a/litellm/batches/batch_utils.py b/litellm/batches/batch_utils.py index eef4cf8d87f..60bc1ccf98a 100644 --- a/litellm/batches/batch_utils.py +++ b/litellm/batches/batch_utils.py @@ -432,6 +432,10 @@ def _get_batch_job_usage_from_response_body(response_body: dict, custom_llm_prov reasoning_content=None, ) _usage_dict = response_body.get("usage", None) or {} + from litellm.responses.utils import ResponseAPILoggingUtils + + if ResponseAPILoggingUtils._is_response_api_usage(_usage_dict): + return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_usage_dict) usage: Usage = Usage(**_usage_dict) return usage diff --git a/litellm/batches/main.py b/litellm/batches/main.py index 3a2d9e13f77..073f25d19b8 100644 --- a/litellm/batches/main.py +++ b/litellm/batches/main.py @@ -103,7 +103,7 @@ def _resolve_timeout( @client async def acreate_batch( completion_window: Literal["24h"], - endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions"], + endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions", "/v1/responses"], input_file_id: str, custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"] = "openai", metadata: Optional[Dict[str, str]] = None, @@ -153,7 +153,7 @@ async def acreate_batch( @client def create_batch( completion_window: Literal["24h"], - endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions"], + endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions", "/v1/responses"], input_file_id: str, custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"] = "openai", metadata: Optional[Dict[str, str]] = None, diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index 314bb653196..8e3b92b50f0 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -431,7 +431,7 @@ class CreateBatchRequest(TypedDict, total=False): """ completion_window: Literal["24h"] - endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions"] + endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions", "/v1/responses"] input_file_id: str metadata: Optional[Dict[str, str]] output_expires_after: FileExpiresAfter diff --git a/tests/batches_tests/test_batch_custom_pricing.py b/tests/batches_tests/test_batch_custom_pricing.py index c2159b564a8..01a3e44a496 100644 --- a/tests/batches_tests/test_batch_custom_pricing.py +++ b/tests/batches_tests/test_batch_custom_pricing.py @@ -128,6 +128,29 @@ def test_aggregate_batch_cost_uses_custom_model_info(): ), f"Expected total cost {expected}, got {cost}" +def test_aggregate_batch_cost_normalizes_mixed_responses_and_chat_usage(): + responses_line = _make_batch_output_line(prompt_tokens=0, completion_tokens=0) + responses_line["response"]["body"]["usage"] = { + "input_tokens": 20, + "output_tokens": 7, + "total_tokens": 27, + "input_tokens_details": {"cached_tokens": 3}, + } + chat_line = _make_batch_output_line(prompt_tokens=10, completion_tokens=5) + + cost, usage, _ = _aggregate_batch_cost_usage_models( + entries=[responses_line, chat_line], + custom_llm_provider="openai", + model_info=CUSTOM_MODEL_INFO, + ) + + assert usage.prompt_tokens == 30 + assert usage.completion_tokens == 12 + assert usage.total_tokens == 42 + assert usage.cache_read_input_tokens == 3 + assert cost == pytest.approx((30 * 0.00125) + (12 * 0.005)) + + @pytest.mark.parametrize("data_residency", ["eu", "us"]) def test_batch_cost_calculator_applies_data_residency_uplift( data_residency, monkeypatch