fix(batches): account for Responses API usage

This commit is contained in:
Rithvik Mysore Suresh 2026-07-31 10:07:10 -04:00
parent 3c2264cfac
commit cdd0639efa
4 changed files with 30 additions and 3 deletions

View file

@ -432,6 +432,10 @@ def _get_batch_job_usage_from_response_body(response_body: dict, custom_llm_prov
reasoning_content=None,
)
_usage_dict = response_body.get("usage", None) or {}
from litellm.responses.utils import ResponseAPILoggingUtils
if ResponseAPILoggingUtils._is_response_api_usage(_usage_dict):
return ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(_usage_dict)
usage: Usage = Usage(**_usage_dict)
return usage

View file

@ -103,7 +103,7 @@ def _resolve_timeout(
@client
async def acreate_batch(
completion_window: Literal["24h"],
endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions"],
endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions", "/v1/responses"],
input_file_id: str,
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"] = "openai",
metadata: Optional[Dict[str, str]] = None,
@ -153,7 +153,7 @@ async def acreate_batch(
@client
def create_batch(
completion_window: Literal["24h"],
endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions"],
endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions", "/v1/responses"],
input_file_id: str,
custom_llm_provider: Literal["openai", "azure", "vertex_ai", "bedrock", "hosted_vllm"] = "openai",
metadata: Optional[Dict[str, str]] = None,

View file

@ -431,7 +431,7 @@ class CreateBatchRequest(TypedDict, total=False):
"""
completion_window: Literal["24h"]
endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions"]
endpoint: Literal["/v1/chat/completions", "/v1/embeddings", "/v1/completions", "/v1/responses"]
input_file_id: str
metadata: Optional[Dict[str, str]]
output_expires_after: FileExpiresAfter

View file

@ -128,6 +128,29 @@ def test_aggregate_batch_cost_uses_custom_model_info():
), f"Expected total cost {expected}, got {cost}"
def test_aggregate_batch_cost_normalizes_mixed_responses_and_chat_usage():
responses_line = _make_batch_output_line(prompt_tokens=0, completion_tokens=0)
responses_line["response"]["body"]["usage"] = {
"input_tokens": 20,
"output_tokens": 7,
"total_tokens": 27,
"input_tokens_details": {"cached_tokens": 3},
}
chat_line = _make_batch_output_line(prompt_tokens=10, completion_tokens=5)
cost, usage, _ = _aggregate_batch_cost_usage_models(
entries=[responses_line, chat_line],
custom_llm_provider="openai",
model_info=CUSTOM_MODEL_INFO,
)
assert usage.prompt_tokens == 30
assert usage.completion_tokens == 12
assert usage.total_tokens == 42
assert usage.cache_read_input_tokens == 3
assert cost == pytest.approx((30 * 0.00125) + (12 * 0.005))
@pytest.mark.parametrize("data_residency", ["eu", "us"])
def test_batch_cost_calculator_applies_data_residency_uplift(
data_residency, monkeypatch