From 066424abc979f36db4513a8eafa870492c87ebb5 Mon Sep 17 00:00:00 2001 From: Dhruv Yadav Date: Sat, 11 Oct 2025 18:03:58 +0530 Subject: [PATCH 1/2] direct cost calculation from openrouter --- .../llms/openrouter/chat/transformation.py | 62 +++++++++++++++++++ 1 file changed, 62 insertions(+) diff --git a/litellm/llms/openrouter/chat/transformation.py b/litellm/llms/openrouter/chat/transformation.py index d5861c79b2d..f1eafe4e294 100644 --- a/litellm/llms/openrouter/chat/transformation.py +++ b/litellm/llms/openrouter/chat/transformation.py @@ -147,8 +147,70 @@ class OpenrouterConfig(OpenAIGPTConfig): model, messages, optional_params, litellm_params, headers ) response.update(extra_body) + + # ALWAYS add usage parameter to get cost data from OpenRouter + # This ensures cost tracking works for all OpenRouter models + if "usage" not in response: + response["usage"] = {"include": True} + return response + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: Any, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + json_mode: Optional[bool] = None, + ) -> ModelResponse: + """ + Transform the response from OpenRouter API. + + Extracts cost information from response headers if available. + + Returns: + ModelResponse: The transformed response with cost information. + """ + # Call parent transform_response to get the standard ModelResponse + model_response = super().transform_response( + model=model, + raw_response=raw_response, + model_response=model_response, + logging_obj=logging_obj, + request_data=request_data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + api_key=api_key, + json_mode=json_mode, + ) + + # Extract cost from OpenRouter response body + # OpenRouter returns cost information in the usage object when usage.include=true + try: + response_json = raw_response.json() + if "usage" in response_json and response_json["usage"]: + response_cost = response_json["usage"].get("cost") + if response_cost is not None: + # Store cost in hidden params for the cost calculator to use + if not hasattr(model_response, "_hidden_params"): + model_response._hidden_params = {} + if "additional_headers" not in model_response._hidden_params: + model_response._hidden_params["additional_headers"] = {} + model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(response_cost) + except Exception: + # If we can't extract cost, continue without it - don't fail the response + pass + + return model_response + def get_error_class( self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] ) -> BaseLLMException: From 84a65440c511e518a6caf56ae1a7d8430226bf72 Mon Sep 17 00:00:00 2001 From: Dhruv Yadav Date: Sun, 12 Oct 2025 12:36:24 +0530 Subject: [PATCH 2/2] add tests for openrouter cost tracking --- .../test_openrouter_chat_transformation.py | 126 +++++++++++++++++- 1 file changed, 125 insertions(+), 1 deletion(-) diff --git a/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py b/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py index 660ced53e5e..64ac299fd79 100644 --- a/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py +++ b/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py @@ -1,12 +1,14 @@ import os import sys +import httpx import pytest sys.path.insert( 0, os.path.abspath("../../../../..") ) # Adds the parent directory to the system path +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig from litellm.llms.openrouter.chat.transformation import ( OpenRouterChatCompletionStreamingHandler, OpenrouterConfig, @@ -364,4 +366,126 @@ def test_openrouter_transform_request_multiple_cache_controls(): assert system_message["content"][4]["cache_control"] == {"type": "ephemeral"} assert "cache_control" not in system_message - \ No newline at end of file + + +def test_openrouter_cost_tracking_non_streaming(): + """ + Test OpenRouter cost tracking for non-streaming completions. + + Verifies: + 1. Request includes usage.include=true to get cost data + 2. Response extracts cost from usage.cost and stores in _hidden_params + """ + from unittest.mock import Mock, patch + from litellm.types.utils import ModelResponse, Choices, Message, Usage + + config = OpenrouterConfig() + + # Test request adds usage parameter + transformed_request = config.transform_request( + model="openrouter/anthropic/claude-sonnet-4.5", + messages=[{"role": "user", "content": "Hello"}], + optional_params={}, + litellm_params={}, + headers={}, + ) + assert "usage" in transformed_request + assert transformed_request["usage"] == {"include": True} + + # Test response extracts cost + mock_response = Mock(spec=httpx.Response) + mock_response.json.return_value = { + "id": "gen-123", + "model": "openrouter/anthropic/claude-sonnet-4.5", + "choices": [{"message": {"role": "assistant", "content": "Hello!"}, "finish_reason": "stop", "index": 0}], + "usage": {"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30, "cost": 0.00015} + } + mock_response.headers = {} + + model_response = ModelResponse( + id="gen-123", + choices=[Choices(finish_reason="stop", index=0, message=Message(content="Hello!", role="assistant"))], + created=1234567890, + model="openrouter/anthropic/claude-sonnet-4.5", + object="chat.completion", + usage=Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30) + ) + + with patch.object(OpenAIGPTConfig, 'transform_response', return_value=model_response): + result = config.transform_response( + model="openrouter/anthropic/claude-sonnet-4.5", + raw_response=mock_response, + model_response=model_response, + logging_obj=Mock(), + request_data={}, + messages=[{"role": "user", "content": "Hello"}], + optional_params={}, + litellm_params={}, + encoding=None, + ) + + assert hasattr(result, "_hidden_params") + assert "llm_provider-x-litellm-response-cost" in result._hidden_params["additional_headers"] + assert result._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] == 0.00015 + + +def test_openrouter_cost_tracking_streaming(): + """ + Test OpenRouter cost tracking for streaming completions. + + Verifies: + 1. Request includes usage.include=true (same as non-streaming) + 2. Streaming chunks preserve usage/cost data in the final chunk + 3. Cost field is accessible in the usage object + """ + config = OpenrouterConfig() + + # Test request adds usage parameter for streaming + transformed_request = config.transform_request( + model="openrouter/anthropic/claude-sonnet-4.5", + messages=[{"role": "user", "content": "Hello"}], + optional_params={}, + litellm_params={}, + headers={}, + ) + assert "usage" in transformed_request + assert transformed_request["usage"] == {"include": True} + + # Test streaming chunks preserve cost data + handler = OpenRouterChatCompletionStreamingHandler( + streaming_response=None, sync_stream=True + ) + + # First chunk - content only + chunk1 = { + "id": "gen-stream-456", + "created": 1234567890, + "model": "openrouter/anthropic/claude-sonnet-4.5", + "choices": [{"delta": {"content": "Hello", "reasoning": None}, "index": 0}], + } + + # Final chunk - usage and cost + chunk2 = { + "id": "gen-stream-456", + "created": 1234567890, + "model": "openrouter/anthropic/claude-sonnet-4.5", + "usage": {"prompt_tokens": 5, "completion_tokens": 10, "total_tokens": 15, "cost": 0.0001}, + "choices": [{"delta": {"content": "", "reasoning": None}, "finish_reason": "stop", "index": 0}], + } + + result1 = handler.chunk_parser(chunk1) + result2 = handler.chunk_parser(chunk2) + + # First chunk has content, no usage + assert result1.choices[0]["delta"]["content"] == "Hello" + assert result1.usage is None + + # Final chunk has usage with cost preserved + assert result2.choices[0]["finish_reason"] == "stop" + assert result2.usage is not None + assert result2.usage.prompt_tokens == 5 + assert result2.usage.completion_tokens == 10 + assert result2.usage.total_tokens == 15 + # Verify cost field is preserved in the Usage object - this is the key data for cost tracking + # The chunk_parser converts the dict to a Usage Pydantic model which includes the cost field + assert result2.usage.cost == 0.0001