mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
Merge pull request #15448 from dhruvyad/main
Get completion cost directly from OpenRouter
This commit is contained in:
commit
2ea7005c40
2 changed files with 187 additions and 1 deletions
|
|
@ -147,8 +147,70 @@ class OpenrouterConfig(OpenAIGPTConfig):
|
|||
model, messages, optional_params, litellm_params, headers
|
||||
)
|
||||
response.update(extra_body)
|
||||
|
||||
# ALWAYS add usage parameter to get cost data from OpenRouter
|
||||
# This ensures cost tracking works for all OpenRouter models
|
||||
if "usage" not in response:
|
||||
response["usage"] = {"include": True}
|
||||
|
||||
return response
|
||||
|
||||
def transform_response(
|
||||
self,
|
||||
model: str,
|
||||
raw_response: httpx.Response,
|
||||
model_response: ModelResponse,
|
||||
logging_obj: Any,
|
||||
request_data: dict,
|
||||
messages: List[AllMessageValues],
|
||||
optional_params: dict,
|
||||
litellm_params: dict,
|
||||
encoding: Any,
|
||||
api_key: Optional[str] = None,
|
||||
json_mode: Optional[bool] = None,
|
||||
) -> ModelResponse:
|
||||
"""
|
||||
Transform the response from OpenRouter API.
|
||||
|
||||
Extracts cost information from response headers if available.
|
||||
|
||||
Returns:
|
||||
ModelResponse: The transformed response with cost information.
|
||||
"""
|
||||
# Call parent transform_response to get the standard ModelResponse
|
||||
model_response = super().transform_response(
|
||||
model=model,
|
||||
raw_response=raw_response,
|
||||
model_response=model_response,
|
||||
logging_obj=logging_obj,
|
||||
request_data=request_data,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
encoding=encoding,
|
||||
api_key=api_key,
|
||||
json_mode=json_mode,
|
||||
)
|
||||
|
||||
# Extract cost from OpenRouter response body
|
||||
# OpenRouter returns cost information in the usage object when usage.include=true
|
||||
try:
|
||||
response_json = raw_response.json()
|
||||
if "usage" in response_json and response_json["usage"]:
|
||||
response_cost = response_json["usage"].get("cost")
|
||||
if response_cost is not None:
|
||||
# Store cost in hidden params for the cost calculator to use
|
||||
if not hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = {}
|
||||
if "additional_headers" not in model_response._hidden_params:
|
||||
model_response._hidden_params["additional_headers"] = {}
|
||||
model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(response_cost)
|
||||
except Exception:
|
||||
# If we can't extract cost, continue without it - don't fail the response
|
||||
pass
|
||||
|
||||
return model_response
|
||||
|
||||
def get_error_class(
|
||||
self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers]
|
||||
) -> BaseLLMException:
|
||||
|
|
|
|||
|
|
@ -1,12 +1,14 @@
|
|||
import os
|
||||
import sys
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath("../../../../..")
|
||||
) # Adds the parent directory to the system path
|
||||
|
||||
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
from litellm.llms.openrouter.chat.transformation import (
|
||||
OpenRouterChatCompletionStreamingHandler,
|
||||
OpenrouterConfig,
|
||||
|
|
@ -364,4 +366,126 @@ def test_openrouter_transform_request_multiple_cache_controls():
|
|||
|
||||
assert system_message["content"][4]["cache_control"] == {"type": "ephemeral"}
|
||||
assert "cache_control" not in system_message
|
||||
|
||||
|
||||
|
||||
def test_openrouter_cost_tracking_non_streaming():
|
||||
"""
|
||||
Test OpenRouter cost tracking for non-streaming completions.
|
||||
|
||||
Verifies:
|
||||
1. Request includes usage.include=true to get cost data
|
||||
2. Response extracts cost from usage.cost and stores in _hidden_params
|
||||
"""
|
||||
from unittest.mock import Mock, patch
|
||||
from litellm.types.utils import ModelResponse, Choices, Message, Usage
|
||||
|
||||
config = OpenrouterConfig()
|
||||
|
||||
# Test request adds usage parameter
|
||||
transformed_request = config.transform_request(
|
||||
model="openrouter/anthropic/claude-sonnet-4.5",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert "usage" in transformed_request
|
||||
assert transformed_request["usage"] == {"include": True}
|
||||
|
||||
# Test response extracts cost
|
||||
mock_response = Mock(spec=httpx.Response)
|
||||
mock_response.json.return_value = {
|
||||
"id": "gen-123",
|
||||
"model": "openrouter/anthropic/claude-sonnet-4.5",
|
||||
"choices": [{"message": {"role": "assistant", "content": "Hello!"}, "finish_reason": "stop", "index": 0}],
|
||||
"usage": {"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30, "cost": 0.00015}
|
||||
}
|
||||
mock_response.headers = {}
|
||||
|
||||
model_response = ModelResponse(
|
||||
id="gen-123",
|
||||
choices=[Choices(finish_reason="stop", index=0, message=Message(content="Hello!", role="assistant"))],
|
||||
created=1234567890,
|
||||
model="openrouter/anthropic/claude-sonnet-4.5",
|
||||
object="chat.completion",
|
||||
usage=Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30)
|
||||
)
|
||||
|
||||
with patch.object(OpenAIGPTConfig, 'transform_response', return_value=model_response):
|
||||
result = config.transform_response(
|
||||
model="openrouter/anthropic/claude-sonnet-4.5",
|
||||
raw_response=mock_response,
|
||||
model_response=model_response,
|
||||
logging_obj=Mock(),
|
||||
request_data={},
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
encoding=None,
|
||||
)
|
||||
|
||||
assert hasattr(result, "_hidden_params")
|
||||
assert "llm_provider-x-litellm-response-cost" in result._hidden_params["additional_headers"]
|
||||
assert result._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] == 0.00015
|
||||
|
||||
|
||||
def test_openrouter_cost_tracking_streaming():
|
||||
"""
|
||||
Test OpenRouter cost tracking for streaming completions.
|
||||
|
||||
Verifies:
|
||||
1. Request includes usage.include=true (same as non-streaming)
|
||||
2. Streaming chunks preserve usage/cost data in the final chunk
|
||||
3. Cost field is accessible in the usage object
|
||||
"""
|
||||
config = OpenrouterConfig()
|
||||
|
||||
# Test request adds usage parameter for streaming
|
||||
transformed_request = config.transform_request(
|
||||
model="openrouter/anthropic/claude-sonnet-4.5",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert "usage" in transformed_request
|
||||
assert transformed_request["usage"] == {"include": True}
|
||||
|
||||
# Test streaming chunks preserve cost data
|
||||
handler = OpenRouterChatCompletionStreamingHandler(
|
||||
streaming_response=None, sync_stream=True
|
||||
)
|
||||
|
||||
# First chunk - content only
|
||||
chunk1 = {
|
||||
"id": "gen-stream-456",
|
||||
"created": 1234567890,
|
||||
"model": "openrouter/anthropic/claude-sonnet-4.5",
|
||||
"choices": [{"delta": {"content": "Hello", "reasoning": None}, "index": 0}],
|
||||
}
|
||||
|
||||
# Final chunk - usage and cost
|
||||
chunk2 = {
|
||||
"id": "gen-stream-456",
|
||||
"created": 1234567890,
|
||||
"model": "openrouter/anthropic/claude-sonnet-4.5",
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 10, "total_tokens": 15, "cost": 0.0001},
|
||||
"choices": [{"delta": {"content": "", "reasoning": None}, "finish_reason": "stop", "index": 0}],
|
||||
}
|
||||
|
||||
result1 = handler.chunk_parser(chunk1)
|
||||
result2 = handler.chunk_parser(chunk2)
|
||||
|
||||
# First chunk has content, no usage
|
||||
assert result1.choices[0]["delta"]["content"] == "Hello"
|
||||
assert result1.usage is None
|
||||
|
||||
# Final chunk has usage with cost preserved
|
||||
assert result2.choices[0]["finish_reason"] == "stop"
|
||||
assert result2.usage is not None
|
||||
assert result2.usage.prompt_tokens == 5
|
||||
assert result2.usage.completion_tokens == 10
|
||||
assert result2.usage.total_tokens == 15
|
||||
# Verify cost field is preserved in the Usage object - this is the key data for cost tracking
|
||||
# The chunk_parser converts the dict to a Usage Pydantic model which includes the cost field
|
||||
assert result2.usage.cost == 0.0001
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue