mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
fix: fix responses cost calc
This commit is contained in:
parent
488b93c477
commit
0ecced9780
3 changed files with 117 additions and 36 deletions
|
|
@ -1455,6 +1455,20 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
): # use model_id if not already set
|
||||
router_model_id = hidden_params["model_id"]
|
||||
|
||||
# Fallback: extract router_model_id from litellm_params when not available
|
||||
# from the result object. ResponsesAPIResponse objects (used by /v1/responses
|
||||
# streaming) don't carry _hidden_params["model_id"] like ModelResponse does.
|
||||
if router_model_id is None and hasattr(self, "litellm_params"):
|
||||
for metadata_key in ("litellm_metadata", "metadata"):
|
||||
_metadata: dict = (
|
||||
self.litellm_params.get(metadata_key, {}) or {}
|
||||
)
|
||||
_model_info: dict = _metadata.get("model_info", {}) or {}
|
||||
_model_id = _model_info.get("id")
|
||||
if _model_id is not None:
|
||||
router_model_id = _model_id
|
||||
break
|
||||
|
||||
## RESPONSE COST ##
|
||||
custom_pricing = use_custom_pricing_for_model(
|
||||
litellm_params=(
|
||||
|
|
|
|||
|
|
@ -1,41 +1,32 @@
|
|||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-4o
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: claude-sonnet-4-5-20250929
|
||||
litellm_params:
|
||||
model: anthropic/claude-sonnet-4-5-20250929
|
||||
- model_name: gpt-4.1-mini
|
||||
|
||||
# OpenAI model for /v1/chat/completions test — 200x custom pricing
|
||||
- model_name: "gpt-4.1-mini"
|
||||
litellm_params:
|
||||
model: openai/gpt-4.1-mini
|
||||
- model_name: gpt-5-mini
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
model_info:
|
||||
id: gpt-4.1-mini-custom-pricing
|
||||
input_cost_per_token: 0.00004 # 100x standard ($0.40/1M = $0.0000004)
|
||||
output_cost_per_token: 0.00016 # 100x standard ($1.60/1M = $0.0000016)
|
||||
|
||||
# OpenAI model for /v1/responses test — 100x custom pricing
|
||||
- model_name: "gpt-5"
|
||||
litellm_params:
|
||||
model: openai/gpt-5-mini
|
||||
- model_name: custom_litellm_model
|
||||
model: openai/gpt-5
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
model_info:
|
||||
id: gpt-5-custom-pricing
|
||||
mode: "chat"
|
||||
input_cost_per_token: 125 # 100x standard ($1.25/1M = $0.00000125)
|
||||
output_cost_per_token: 10 # 100x standard ($10.00/1M = $0.00001)
|
||||
|
||||
# Anthropic model for /v1/messages test — 100x custom pricing
|
||||
- model_name: "claude-sonnet-4-20250514"
|
||||
litellm_params:
|
||||
model: litellm_agent/claude-sonnet-4-5-20250929
|
||||
litellm_system_prompt: "Be a helpful assistant."
|
||||
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "tool_policy"
|
||||
litellm_params:
|
||||
guardrail: tool_policy
|
||||
mode: [pre_call, post_call]
|
||||
default_on: true
|
||||
|
||||
mcp_servers:
|
||||
my_http_server:
|
||||
url: "http://0.0.0.0:8001/mcp"
|
||||
transport: "http"
|
||||
description: "My custom MCP server"
|
||||
available_on_public_internet: true
|
||||
|
||||
general_settings:
|
||||
store_model_in_db: true
|
||||
store_prompts_in_spend_logs: true
|
||||
model: anthropic/claude-sonnet-4-20250514
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
model_info:
|
||||
id: claude-sonnet-4-custom-pricing
|
||||
input_cost_per_token: 0.0003 # 100x standard ($0.000003)
|
||||
output_cost_per_token: 0.0015 # 100x standard ($0.000015)
|
||||
|
|
@ -180,6 +180,82 @@ def test_use_custom_pricing_not_detected_litellm_metadata_no_pricing():
|
|||
assert use_custom_pricing_for_model(litellm_params) is False
|
||||
|
||||
|
||||
def test_response_cost_calculator_uses_router_model_id_from_litellm_metadata():
|
||||
"""_response_cost_calculator should extract router_model_id from
|
||||
litellm_params.litellm_metadata.model_info.id when the result object
|
||||
does not carry _hidden_params (e.g. ResponsesAPIResponse from /v1/responses
|
||||
streaming). Regression test for custom pricing on streaming responses."""
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
custom_model_id = "gpt-5-custom-pricing"
|
||||
custom_input_cost = 125.0
|
||||
custom_output_cost = 10.0
|
||||
|
||||
litellm.register_model(
|
||||
model_cost={
|
||||
custom_model_id: {
|
||||
"input_cost_per_token": custom_input_cost,
|
||||
"output_cost_per_token": custom_output_cost,
|
||||
"max_tokens": 128000,
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 16384,
|
||||
"litellm_provider": "openai",
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
try:
|
||||
logging_obj = LiteLLMLoggingObj(
|
||||
model="gpt-5",
|
||||
messages=[{"role": "user", "content": "Hi"}],
|
||||
stream=True,
|
||||
call_type="aresponses",
|
||||
start_time=time.time(),
|
||||
litellm_call_id="test-123",
|
||||
function_id="test-fn",
|
||||
)
|
||||
|
||||
logging_obj.update_environment_variables(
|
||||
model="gpt-5",
|
||||
user="",
|
||||
optional_params={},
|
||||
litellm_params={
|
||||
"api_base": "",
|
||||
"litellm_metadata": {
|
||||
"model_info": {
|
||||
"id": custom_model_id,
|
||||
"input_cost_per_token": custom_input_cost,
|
||||
"output_cost_per_token": custom_output_cost,
|
||||
},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
response_obj = ResponsesAPIResponse(
|
||||
id="resp_abc",
|
||||
created_at=1234567890,
|
||||
model="gpt-5",
|
||||
output=[],
|
||||
usage={
|
||||
"input_tokens": 10,
|
||||
"output_tokens": 5,
|
||||
"total_tokens": 15,
|
||||
},
|
||||
)
|
||||
|
||||
cost = logging_obj._response_cost_calculator(result=response_obj)
|
||||
|
||||
assert cost is not None, "Cost should not be None"
|
||||
expected_cost = (10 * custom_input_cost) + (5 * custom_output_cost)
|
||||
assert cost == pytest.approx(
|
||||
expected_cost
|
||||
), f"Expected {expected_cost}, got {cost}"
|
||||
finally:
|
||||
litellm.model_cost.pop(custom_model_id, None)
|
||||
|
||||
|
||||
class TestUpdateFromKwargs:
|
||||
"""Tests for the update_from_kwargs convenience wrapper."""
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue