From 49f85239b1b5c9cb854ad906ad5e5f7e290dbc07 Mon Sep 17 00:00:00 2001 From: "liangjie.wanglj" <122603020@qq.com> Date: Tue, 16 Jun 2026 15:40:59 +0800 Subject: [PATCH 1/2] fix(openrouter): move usage accounting to extra_body for compatibility --- .../llms/openrouter/chat/transformation.py | 16 +++--- .../test_openrouter_chat_transformation.py | 51 ++++++++++++++++--- 2 files changed, 52 insertions(+), 15 deletions(-) diff --git a/litellm/llms/openrouter/chat/transformation.py b/litellm/llms/openrouter/chat/transformation.py index 107d5c25e6d..56eefec2e15 100644 --- a/litellm/llms/openrouter/chat/transformation.py +++ b/litellm/llms/openrouter/chat/transformation.py @@ -167,17 +167,19 @@ class OpenrouterConfig(OpenAIGPTConfig): messages = self._move_cache_control_to_content(messages) extra_body = optional_params.pop("extra_body", {}) - response = super().transform_request( + request = super().transform_request( model, messages, optional_params, litellm_params, headers ) - response.update(extra_body) + request.update(extra_body) - # ALWAYS add usage parameter to get cost data from OpenRouter - # This ensures cost tracking works for all OpenRouter models - if "usage" not in response: - response["usage"] = {"include": True} + # Usage accounting is an OpenRouter-specific flag, not a valid top-level + # OpenAI chat param. Sending it at the top level makes OpenAI-compatible + # clients reject the request ("unexpected keyword argument 'usage'"), so + # it must travel inside `extra_body` instead. + extra_body = request.setdefault("extra_body", {}) + extra_body.setdefault("usage", {"include": True}) - return response + return request def transform_response( self, diff --git a/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py b/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py index 8d1129cc5da..9d802893b4e 100644 --- a/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py +++ b/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py @@ -372,7 +372,7 @@ def test_openrouter_cost_tracking_non_streaming(): Test OpenRouter cost tracking for non-streaming completions. Verifies: - 1. Request includes usage.include=true to get cost data + 1. Request asks for usage accounting via extra_body, not a top-level usage field 2. Response extracts cost from usage.cost and stores in _hidden_params """ from unittest.mock import Mock, patch @@ -380,7 +380,6 @@ def test_openrouter_cost_tracking_non_streaming(): config = OpenrouterConfig() - # Test request adds usage parameter transformed_request = config.transform_request( model="openrouter/anthropic/claude-sonnet-4.5", messages=[{"role": "user", "content": "Hello"}], @@ -388,8 +387,8 @@ def test_openrouter_cost_tracking_non_streaming(): litellm_params={}, headers={}, ) - assert "usage" in transformed_request - assert transformed_request["usage"] == {"include": True} + assert "usage" not in transformed_request + assert transformed_request["extra_body"]["usage"] == {"include": True} # Test response extracts cost mock_response = Mock(spec=httpx.Response) @@ -460,13 +459,12 @@ def test_openrouter_cost_tracking_streaming(): Test OpenRouter cost tracking for streaming completions. Verifies: - 1. Request includes usage.include=true (same as non-streaming) + 1. Request asks for usage accounting via extra_body (same as non-streaming) 2. Streaming chunks preserve usage/cost data in the final chunk 3. Cost field is accessible in the usage object """ config = OpenrouterConfig() - # Test request adds usage parameter for streaming transformed_request = config.transform_request( model="openrouter/anthropic/claude-sonnet-4.5", messages=[{"role": "user", "content": "Hello"}], @@ -474,8 +472,8 @@ def test_openrouter_cost_tracking_streaming(): litellm_params={}, headers={}, ) - assert "usage" in transformed_request - assert transformed_request["usage"] == {"include": True} + assert "usage" not in transformed_request + assert transformed_request["extra_body"]["usage"] == {"include": True} # Test streaming chunks preserve cost data handler = OpenRouterChatCompletionStreamingHandler( @@ -528,6 +526,43 @@ def test_openrouter_cost_tracking_streaming(): assert result2.usage.cost == 0.0001 +def test_openrouter_usage_accounting_not_top_level(): + """ + Regression: usage accounting must be requested via extra_body, never as a + top-level field. A top-level `usage` is rejected by OpenAI-compatible clients + with "unexpected keyword argument 'usage'" (e.g. for OpenAI models served + through OpenRouter). + """ + transformed_request = OpenrouterConfig().transform_request( + model="openrouter/openai/gpt-4o", + messages=[{"role": "user", "content": "Hello"}], + optional_params={}, + litellm_params={}, + headers={}, + ) + + assert "usage" not in transformed_request + assert transformed_request["extra_body"]["usage"] == {"include": True} + + +def test_openrouter_usage_accounting_preserves_user_extra_body(): + """ + User-provided OpenRouter params still flatten to the top level (issue #8425) + while usage accounting is added under extra_body without clobbering them. + """ + transformed_request = OpenrouterConfig().transform_request( + model="openrouter/openai/gpt-4o", + messages=[{"role": "user", "content": "Hello"}], + optional_params={"extra_body": {"provider": {"order": ["OpenAI"]}}}, + litellm_params={}, + headers={}, + ) + + assert "usage" not in transformed_request + assert transformed_request["provider"]["order"] == ["OpenAI"] + assert transformed_request["extra_body"]["usage"] == {"include": True} + + def test_openrouter_reasoning_models_allow_reasoning_effort_param(): """ OpenRouter reasoning-capable models should accept the reasoning_effort param. From 88840fff5473669d2395e3a03a6c33bb81d048f7 Mon Sep 17 00:00:00 2001 From: "liangjie.wanglj" <122603020@qq.com> Date: Wed, 17 Jun 2026 18:45:28 +0800 Subject: [PATCH 2/2] test(openrouter): exercise real handler path for usage accounting The previous test called transform_request directly with optional_params={'extra_body': {...}}, but in the real completion path extra_body is popped in llm_http_handler before transform_request runs, so the user-param merge happens in the handler, not in transform_request. That test passed against a code path production never takes. Drive the full completion path with a mocked client and assert on the request body actually posted: the user-provided OpenRouter param lands at the top level (handler merge) and usage accounting reaches the shipped body. Both assertions fail if the usage accounting or the handler merge is removed. --- .../test_openrouter_chat_transformation.py | 62 ++++++++++++++++--- 1 file changed, 53 insertions(+), 9 deletions(-) diff --git a/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py b/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py index 9d802893b4e..2fa281ca6a4 100644 --- a/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py +++ b/tests/test_litellm/llms/openrouter/chat/test_openrouter_chat_transformation.py @@ -547,20 +547,64 @@ def test_openrouter_usage_accounting_not_top_level(): def test_openrouter_usage_accounting_preserves_user_extra_body(): """ - User-provided OpenRouter params still flatten to the top level (issue #8425) - while usage accounting is added under extra_body without clobbering them. + Regression for the real request path (issue #8425): user-provided OpenRouter + params (e.g. provider routing) must land at the top level of the body that + ships, and usage accounting must reach that body without clobbering them. + + This drives the full completion path on purpose. `extra_body` is popped in + llm_http_handler before `transform_request` runs, so the merge that lifts user + params to the top level happens in the handler, not in `transform_request`; + asserting on `transform_request` in isolation would test a path production + never takes. """ - transformed_request = OpenrouterConfig().transform_request( + import json + from unittest.mock import MagicMock + + import litellm + from litellm.llms.custom_httpx.http_handler import HTTPHandler + + payload = { + "id": "x", + "object": "chat.completion", + "model": "openai/gpt-4o", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "hi"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + } + mock_response = MagicMock(spec=httpx.Response) + mock_response.status_code = 200 + mock_response.headers = {} + mock_response.json.return_value = payload + mock_response.text = json.dumps(payload) + + mock_client = MagicMock(spec=HTTPHandler) + mock_client.post.return_value = mock_response + + litellm.completion( model="openrouter/openai/gpt-4o", messages=[{"role": "user", "content": "Hello"}], - optional_params={"extra_body": {"provider": {"order": ["OpenAI"]}}}, - litellm_params={}, - headers={}, + extra_body={"provider": {"order": ["OpenAI"]}}, + api_key="sk-fake", + client=mock_client, ) - assert "usage" not in transformed_request - assert transformed_request["provider"]["order"] == ["OpenAI"] - assert transformed_request["extra_body"]["usage"] == {"include": True} + shipped_body = json.loads(mock_client.post.call_args.kwargs["data"]) + + # User-provided OpenRouter param survives at the top level. It is merged by the + # handler, not by transform_request (which never sees extra_body in the real + # path), so dropping the handler merge would fail here. + assert shipped_body["provider"] == {"order": ["OpenAI"]} + + # Usage accounting reaches the shipped body without clobbering the user param. + usage_flag = shipped_body.get("usage") or shipped_body.get("extra_body", {}).get( + "usage" + ) + assert usage_flag == {"include": True} def test_openrouter_reasoning_models_allow_reasoning_effort_param():