fix: fix responses cost calc

This commit is contained in:
Krrish Dholakia 2026-03-18 19:52:59 -07:00
parent 488b93c477
commit 0ecced9780
3 changed files with 117 additions and 36 deletions

View file

@ -1455,6 +1455,20 @@ class Logging(LiteLLMLoggingBaseClass):
): # use model_id if not already set
router_model_id = hidden_params["model_id"]
# Fallback: extract router_model_id from litellm_params when not available
# from the result object. ResponsesAPIResponse objects (used by /v1/responses
# streaming) don't carry _hidden_params["model_id"] like ModelResponse does.
if router_model_id is None and hasattr(self, "litellm_params"):
for metadata_key in ("litellm_metadata", "metadata"):
_metadata: dict = (
self.litellm_params.get(metadata_key, {}) or {}
)
_model_info: dict = _metadata.get("model_info", {}) or {}
_model_id = _model_info.get("id")
if _model_id is not None:
router_model_id = _model_id
break
## RESPONSE COST ##
custom_pricing = use_custom_pricing_for_model(
litellm_params=(

View file

@ -1,41 +1,32 @@
model_list:
- model_name: gpt-3.5-turbo
litellm_params:
model: openai/gpt-3.5-turbo
api_key: os.environ/OPENAI_API_KEY
- model_name: gpt-4o
litellm_params:
model: openai/gpt-4o
api_key: os.environ/OPENAI_API_KEY
- model_name: claude-sonnet-4-5-20250929
litellm_params:
model: anthropic/claude-sonnet-4-5-20250929
- model_name: gpt-4.1-mini
# OpenAI model for /v1/chat/completions test — 200x custom pricing
- model_name: "gpt-4.1-mini"
litellm_params:
model: openai/gpt-4.1-mini
- model_name: gpt-5-mini
api_key: os.environ/OPENAI_API_KEY
model_info:
id: gpt-4.1-mini-custom-pricing
input_cost_per_token: 0.00004 # 100x standard ($0.40/1M = $0.0000004)
output_cost_per_token: 0.00016 # 100x standard ($1.60/1M = $0.0000016)
# OpenAI model for /v1/responses test — 100x custom pricing
- model_name: "gpt-5"
litellm_params:
model: openai/gpt-5-mini
- model_name: custom_litellm_model
model: openai/gpt-5
api_key: os.environ/OPENAI_API_KEY
model_info:
id: gpt-5-custom-pricing
mode: "chat"
input_cost_per_token: 125 # 100x standard ($1.25/1M = $0.00000125)
output_cost_per_token: 10 # 100x standard ($10.00/1M = $0.00001)
# Anthropic model for /v1/messages test — 100x custom pricing
- model_name: "claude-sonnet-4-20250514"
litellm_params:
model: litellm_agent/claude-sonnet-4-5-20250929
litellm_system_prompt: "Be a helpful assistant."
guardrails:
- guardrail_name: "tool_policy"
litellm_params:
guardrail: tool_policy
mode: [pre_call, post_call]
default_on: true
mcp_servers:
my_http_server:
url: "http://0.0.0.0:8001/mcp"
transport: "http"
description: "My custom MCP server"
available_on_public_internet: true
general_settings:
store_model_in_db: true
store_prompts_in_spend_logs: true
model: anthropic/claude-sonnet-4-20250514
api_key: os.environ/ANTHROPIC_API_KEY
model_info:
id: claude-sonnet-4-custom-pricing
input_cost_per_token: 0.0003 # 100x standard ($0.000003)
output_cost_per_token: 0.0015 # 100x standard ($0.000015)

View file

@ -180,6 +180,82 @@ def test_use_custom_pricing_not_detected_litellm_metadata_no_pricing():
assert use_custom_pricing_for_model(litellm_params) is False
def test_response_cost_calculator_uses_router_model_id_from_litellm_metadata():
"""_response_cost_calculator should extract router_model_id from
litellm_params.litellm_metadata.model_info.id when the result object
does not carry _hidden_params (e.g. ResponsesAPIResponse from /v1/responses
streaming). Regression test for custom pricing on streaming responses."""
import litellm
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.types.llms.openai import ResponsesAPIResponse
custom_model_id = "gpt-5-custom-pricing"
custom_input_cost = 125.0
custom_output_cost = 10.0
litellm.register_model(
model_cost={
custom_model_id: {
"input_cost_per_token": custom_input_cost,
"output_cost_per_token": custom_output_cost,
"max_tokens": 128000,
"max_input_tokens": 128000,
"max_output_tokens": 16384,
"litellm_provider": "openai",
}
}
)
try:
logging_obj = LiteLLMLoggingObj(
model="gpt-5",
messages=[{"role": "user", "content": "Hi"}],
stream=True,
call_type="aresponses",
start_time=time.time(),
litellm_call_id="test-123",
function_id="test-fn",
)
logging_obj.update_environment_variables(
model="gpt-5",
user="",
optional_params={},
litellm_params={
"api_base": "",
"litellm_metadata": {
"model_info": {
"id": custom_model_id,
"input_cost_per_token": custom_input_cost,
"output_cost_per_token": custom_output_cost,
},
},
},
)
response_obj = ResponsesAPIResponse(
id="resp_abc",
created_at=1234567890,
model="gpt-5",
output=[],
usage={
"input_tokens": 10,
"output_tokens": 5,
"total_tokens": 15,
},
)
cost = logging_obj._response_cost_calculator(result=response_obj)
assert cost is not None, "Cost should not be None"
expected_cost = (10 * custom_input_cost) + (5 * custom_output_cost)
assert cost == pytest.approx(
expected_cost
), f"Expected {expected_cost}, got {cost}"
finally:
litellm.model_cost.pop(custom_model_id, None)
class TestUpdateFromKwargs:
"""Tests for the update_from_kwargs convenience wrapper."""