From 062d8ceeed152b8ca46bb40212c17b19f2107552 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 26 Jun 2026 09:31:02 +0530 Subject: [PATCH] fix(vertex_ai): prevent stale Vertex bearer token causing /v1/messages 401 after token expiry (#31276) * fix(vertex_ai): prevent stale Vertex bearer token causing /v1/messages 401 after token expiry Router shallow-copies litellm_params so extra_headers is a shared reference. The chat/completions path was calling headers.update() on that shared dict, persisting the Vertex OAuth bearer. After ~1 h the token expired and /v1/messages kept reusing it (skipping refresh due to Authorization-already-present guard). - Build a new headers dict in the Claude partner-models completion path instead of mutating the shared extra_headers object. - Always call _ensure_access_token() in validate_anthropic_messages_environment regardless of an existing Authorization header; the token cache makes this cheap. Co-authored-by: Cursor * fix(vertex_ai): copy headers in validate_anthropic_messages_environment to prevent shared-dict mutation Co-authored-by: Cursor --------- Co-authored-by: Cursor --- .../transformation.py | 28 ++--- .../vertex_ai_partner_models/main.py | 8 +- ...artner_models_anthropic_messages_config.py | 113 ++++++++++++++---- 3 files changed, 108 insertions(+), 41 deletions(-) diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py index 8a92e7ec4a5..a633ef3298a 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py @@ -35,27 +35,21 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert Validate the environment for the request """ + # Work on a local copy — router shallow-copies litellm_params so the caller's + # headers dict may be the shared deployment extra_headers object. + headers = dict(headers) vertex_ai_project = VertexBase.safe_get_vertex_ai_project(litellm_params) vertex_ai_location = VertexBase.safe_get_vertex_ai_location(litellm_params) - project_id: Optional[str] = None - if "Authorization" not in headers: - vertex_credentials = VertexBase.safe_get_vertex_ai_credentials( - litellm_params - ) + vertex_credentials = VertexBase.safe_get_vertex_ai_credentials(litellm_params) + access_token, project_id = self._ensure_access_token( + credentials=vertex_credentials, + project_id=vertex_ai_project, + custom_llm_provider="vertex_ai", + ) + headers["Authorization"] = f"Bearer {access_token}" - access_token, project_id = self._ensure_access_token( - credentials=vertex_credentials, - project_id=vertex_ai_project, - custom_llm_provider="vertex_ai", - ) - - headers["Authorization"] = f"Bearer {access_token}" - else: - # Authorization already in headers, but we still need project_id - project_id = vertex_ai_project - - # Always calculate api_base if not provided, regardless of Authorization header + # Calculate api_base if not provided if api_base is None: api_base = self.get_complete_vertex_url( custom_api_base=api_base, diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/main.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/main.py index 960d3483848..78669f1e789 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/main.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/main.py @@ -194,9 +194,11 @@ class VertexAIPartnerModels(VertexBase): encoding=encoding, ) elif "claude" in model: - if headers is None: - headers = {} - headers.update({"Authorization": "Bearer {}".format(access_token)}) + # Build a new dict so we never mutate the shared deployment extra_headers object. + headers = { + **(headers or {}), + "Authorization": "Bearer {}".format(access_token), + } optional_params.update( { diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py index 6f4bb4e59c2..d64b5c8d742 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_messages_config.py @@ -1,10 +1,11 @@ -from unittest.mock import patch +from unittest.mock import MagicMock, patch import pytest from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.experimental_pass_through.transformation import ( VertexAIPartnerModelsAnthropicMessagesConfig, ) +from litellm.llms.vertex_ai.vertex_ai_partner_models.main import VertexAIPartnerModels from litellm.types.router import GenericLiteLLMParams @@ -233,40 +234,36 @@ def test_both_compact_and_context_management_headers_added(): ), f"anthropic-beta should contain 'context-management-2025-06-27', got: {updated_headers['anthropic-beta']}" -def test_validate_environment_with_authorization_header_calculates_api_base(): - """Test that api_base is calculated even when Authorization header is already present""" +def test_validate_environment_always_refreshes_token_ignoring_stale_bearer(): + """Regression: stale Authorization in shared deployment extra_headers must not + skip token refresh on /v1/messages — _ensure_access_token is always called.""" config = VertexAIPartnerModelsAnthropicMessagesConfig() - # Simulate scenario where Authorization is already in headers (e.g., from cached extra_headers) - headers = {"Authorization": "Bearer existing-token"} + headers = {"Authorization": "Bearer EXPIRED"} litellm_params = { "vertex_project": "test-project", "vertex_location": "us-central1", - "extra_headers": {"anthropic-beta": "context-1m-2025-08-07"}, } - optional_params = {} - with patch.object( - config, "get_complete_vertex_url", return_value="https://mock-vertex-url" - ) as mock_get_url: + with ( + patch.object( + config, "_ensure_access_token", return_value=("fresh-token", "test-project") + ) as mock_ensure, + patch.object( + config, "get_complete_vertex_url", return_value="https://mock-vertex-url" + ), + ): updated_headers, api_base = config.validate_anthropic_messages_environment( headers=headers, model="claude-sonnet-4", messages=[], - optional_params=optional_params, + optional_params={}, litellm_params=litellm_params, api_base=None, ) - # Verify that api_base was calculated even though Authorization was already present - assert ( - api_base == "https://mock-vertex-url" - ), f"api_base should be calculated even with Authorization header. Got: {api_base}" - assert mock_get_url.called, "get_complete_vertex_url should be called" - - # Verify Authorization header is still present - assert ( - "Authorization" in updated_headers - ), "Authorization header should be preserved" + mock_ensure.assert_called_once() + assert updated_headers["Authorization"] == "Bearer fresh-token" + assert api_base == "https://mock-vertex-url" def test_transform_anthropic_messages_request_removes_scope_from_cache_control(): @@ -371,3 +368,77 @@ def test_provider_config_manager_reuses_vertex_anthropic_messages_config_instanc assert first_config is second_config finally: ProviderConfigManager._get_provider_anthropic_messages_config_cached.cache_clear() + + +def test_validate_environment_does_not_mutate_caller_headers(): + """Regression: beta headers (e.g. web-search) must not leak into the caller's + headers dict — which may be the shared deployment extra_headers object.""" + config = VertexAIPartnerModelsAnthropicMessagesConfig() + caller_headers: dict = {} + + with ( + patch.object( + config, "_ensure_access_token", return_value=("token", "test-project") + ), + patch.object( + config, "get_complete_vertex_url", return_value="https://mock-url" + ), + ): + config.validate_anthropic_messages_environment( + headers=caller_headers, + model="claude-sonnet-4", + messages=[], + optional_params={ + "tools": [{"type": "web_search_20250305", "name": "web_search"}] + }, + litellm_params={ + "vertex_ai_project": "p", + "vertex_ai_location": "us-central1", + }, + api_base=None, + ) + + assert ( + caller_headers == {} + ), "validate_anthropic_messages_environment must not mutate the caller's headers dict" + + +def test_vertex_claude_completion_does_not_mutate_shared_extra_headers(): + """Regression: router shallow-copies litellm_params so extra_headers is a shared + reference. Verify that the chat/completions path builds a new headers dict instead + of calling .update() on the shared object.""" + handler = VertexAIPartnerModels() + shared_extra_headers = {} # simulates deployment["litellm_params"]["extra_headers"] + + mock_response = MagicMock() + + with ( + patch.object( + handler, "_ensure_access_token", return_value=("ya29.fresh", "proj") + ), + patch.object( + handler, "get_complete_vertex_url", return_value="https://mock-url" + ), + patch( + "litellm.llms.anthropic.chat.AnthropicChatCompletion.completion", + return_value=mock_response, + ), + ): + handler.completion( + model="claude-haiku-4-5@20251001", + messages=[{"role": "user", "content": "hi"}], + model_response=MagicMock(), + print_verbose=lambda *a, **k: None, + encoding=None, + logging_obj=MagicMock(), + api_base=None, + optional_params={}, + custom_prompt_dict={}, + headers=shared_extra_headers, + timeout=30, + litellm_params={}, + ) + + assert ( + shared_extra_headers == {} + ), "extra_headers must not be mutated by completion()"