From f6d5502faa59e56fb4bc38227153296fb9fb4c49 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 13 Apr 2026 16:42:25 +0530 Subject: [PATCH 001/196] feat(vertex-ai): transform batch prediction outputs to OpenAI format - Add automatic conversion of Vertex AI batch prediction JSONL to OpenAI format - Preserve custom_id via Vertex AI labels for request correlation - Fix Content-Length header mismatch in transformed responses - Add comprehensive tests for batch output transformation Made-with: Cursor --- litellm/llms/vertex_ai/files/handler.py | 29 +- .../llms/vertex_ai/files/transformation.py | 245 ++++++++++++++ .../test_vertex_ai_files_transformation.py | 300 +++++++++++++++++- 3 files changed, 571 insertions(+), 3 deletions(-) diff --git a/litellm/llms/vertex_ai/files/handler.py b/litellm/llms/vertex_ai/files/handler.py index 6636bccd6a3..81bf7852c82 100644 --- a/litellm/llms/vertex_ai/files/handler.py +++ b/litellm/llms/vertex_ai/files/handler.py @@ -1,4 +1,5 @@ import asyncio +import time import urllib.parse from typing import Any, Coroutine, Optional, Tuple, Union @@ -188,11 +189,35 @@ class VertexAIFilesHandler(GCSBucketBase): mock_response = httpx.Response( status_code=200, content=file_content, - headers={"content-type": "application/octet-stream"}, + headers={ + "content-type": "application/octet-stream", + "content-length": str(len(file_content)), + }, request=httpx.Request(method="GET", url=decoded_path), ) - return HttpxBinaryResponseContent(response=mock_response) + # Apply transformation to convert Vertex AI batch outputs to OpenAI format + from .transformation import VertexAIFilesConfig + from litellm.litellm_core_utils.litellm_logging import Logging + + config = VertexAIFilesConfig() + + # Create a logging object for transformation + logging_obj = Logging( + model="", + messages=[], + stream=False, + call_type="afile_content", + start_time=time.time(), + litellm_call_id="", + function_id="", + ) + + return config.transform_file_content_response( + raw_response=mock_response, + logging_obj=logging_obj, + litellm_params={} + ) def file_content( self, diff --git a/litellm/llms/vertex_ai/files/transformation.py b/litellm/llms/vertex_ai/files/transformation.py index 070ec508283..a967e968e4a 100644 --- a/litellm/llms/vertex_ai/files/transformation.py +++ b/litellm/llms/vertex_ai/files/transformation.py @@ -3,6 +3,7 @@ import os import time from typing import Any, Dict, List, Optional, Tuple, Union +import httpx from httpx import Headers, Response from openai.types.file_deleted import FileDeleted @@ -247,6 +248,14 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): litellm_params={}, cached_content=None, ) + + # Add custom_id as a label for correlation in batch outputs + custom_id = _openai_jsonl_content.get("custom_id") + if custom_id: + if "labels" not in vertex_request_body: + vertex_request_body["labels"] = {} + vertex_request_body["labels"]["litellm_custom_id"] = str(custom_id) + vertex_jsonl_content.append({"request": vertex_request_body}) return vertex_jsonl_content @@ -453,7 +462,235 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): logging_obj: LiteLLMLoggingObj, litellm_params: dict, ) -> HttpxBinaryResponseContent: + """ + Transform file content response, converting Vertex AI batch output to OpenAI format if applicable. + + This method automatically detects and transforms Vertex AI batch prediction outputs + (predictions.jsonl files) into OpenAI-compatible batch response format. + + If the file is not a batch output or transformation fails, the original content + is returned as-is to maintain backward compatibility. + """ + try: + # Try to transform batch output if it's a JSONL file + content = raw_response.content + if content: + transformed_content = self._try_transform_vertex_batch_output_to_openai( + content + ) + if transformed_content != content: + # Create a new response with transformed content and updated Content-Length + import httpx + + # Update headers with correct Content-Length + new_headers = dict(raw_response.headers) + new_headers["content-length"] = str(len(transformed_content)) + + mock_response = httpx.Response( + status_code=raw_response.status_code, + content=transformed_content, + headers=new_headers, + request=raw_response.request, + ) + return HttpxBinaryResponseContent(response=mock_response) + except Exception: + # If transformation fails, return as-is + pass + return HttpxBinaryResponseContent(response=raw_response) + + def _try_transform_vertex_batch_output_to_openai( + self, content: bytes + ) -> bytes: + """ + Try to transform Vertex AI batch output to OpenAI format. + If conversion fails at any point, return the original content as-is. + + Vertex AI batch output format (predictions.jsonl): + { + "request": {"contents": [...], "labels": {"litellm_custom_id": "request-1"}}, + "status": "", + "response": {"candidates": [...], "modelVersion": "gemini-2.5-flash", ...}, + "processed_time": "2026-04-13T10:18:18.102004+00:00" + } + + OpenAI batch output format: + { + "id": "batch_req_...", + "custom_id": "request-1", + "response": { + "status_code": 200, + "request_id": "chatcmpl-...", + "body": {} + }, + "error": null + } + """ + try: + # Decode content + content_str = content.decode("utf-8") + + # Check if it's JSONL (multiple lines) + lines = content_str.strip().split("\n") + if not lines: + return content + + # Try to parse the first line to see if it's Vertex AI batch output + first_line = json.loads(lines[0]) + + # Check if it has Vertex AI batch output structure + if not ("response" in first_line and "request" in first_line): + # Not a Vertex AI batch output, return as-is + return content + + # Transform all lines + transformed_lines = [] + for line in lines: + if not line.strip(): + continue + + try: + vertex_output = json.loads(line) + openai_output = self._transform_single_vertex_batch_output_to_openai( + vertex_output + ) + transformed_lines.append(json.dumps(openai_output)) + except Exception: + # If any line fails, return original content + return content + + # Return transformed content + return "\n".join(transformed_lines).encode("utf-8") + + except Exception: + # If anything fails, return original content + return content + + def _transform_single_vertex_batch_output_to_openai( + self, vertex_output: Dict[str, Any] + ) -> Dict[str, Any]: + """ + Transform a single Vertex AI batch output line to OpenAI format. + Uses the existing VertexGeminiConfig transformation for the response. + """ + from litellm.types.utils import ModelResponse + import httpx + import time + + # Extract custom_id from request labels + custom_id = "unknown" + request_data = vertex_output.get("request", {}) + labels = request_data.get("labels", {}) + if "litellm_custom_id" in labels: + custom_id = labels["litellm_custom_id"] + + # Check if there's an error + status = vertex_output.get("status", "") + has_error = bool(status) + + if has_error: + # Return error response in OpenAI format + return { + "id": f"batch_req_{uuid.uuid4()}", + "custom_id": custom_id, + "response": { + "status_code": 400, + "request_id": "", + "body": { + "error": { + "message": status, + "type": "vertex_ai_error", + "code": "vertex_ai_error" + } + } + }, + "error": { + "message": status, + "type": "vertex_ai_error", + "code": "vertex_ai_error" + } + } + + # Transform successful response using existing transformation + vertex_response = vertex_output.get("response", {}) + + # Extract model from response + model = vertex_response.get("modelVersion", "gemini-1.5-flash-001") + if "@" in model: + model = model.split("@")[0] + + # Create logging object for transformation + from litellm.litellm_core_utils.litellm_logging import Logging + + logging_obj = Logging( + model=model, + messages=[], + stream=False, + call_type="batch_transform", + start_time=time.time(), + litellm_call_id="", + function_id="", + ) + logging_obj.optional_params = {} + + # Create mock httpx response for transformation + mock_httpx_response = httpx.Response( + status_code=200, + content=json.dumps(vertex_response).encode("utf-8"), + headers={"content-type": "application/json"}, + request=httpx.Request(method="POST", url="https://example.com"), + ) + + try: + # Use existing VertexGeminiConfig transformation + vertex_gemini_config = VertexGeminiConfig() + model_response = ModelResponse() + + transformed_response = vertex_gemini_config._transform_google_generate_content_to_openai_model_response( + completion_response=vertex_response, + model_response=model_response, + model=model, + logging_obj=logging_obj, + raw_response=mock_httpx_response, + ) + + # Convert ModelResponse to dict + response_dict = transformed_response.model_dump() + + # Return in OpenAI batch format + return { + "id": f"batch_req_{uuid.uuid4()}", + "custom_id": custom_id, + "response": { + "status_code": 200, + "request_id": response_dict.get("id", ""), + "body": response_dict + }, + "error": None + } + + except Exception as e: + # If transformation fails, return error + return { + "id": f"batch_req_{uuid.uuid4()}", + "custom_id": custom_id, + "response": { + "status_code": 500, + "request_id": "", + "body": { + "error": { + "message": f"Failed to transform response: {str(e)}", + "type": "transformation_error", + "code": "transformation_error" + } + } + }, + "error": { + "message": f"Failed to transform response: {str(e)}", + "type": "transformation_error", + "code": "transformation_error" + } + } class VertexAIJsonlFilesTransformation(VertexGeminiConfig): @@ -513,6 +750,14 @@ class VertexAIJsonlFilesTransformation(VertexGeminiConfig): litellm_params={}, cached_content=None, ) + + # Add custom_id as a label for correlation in batch outputs + custom_id = _openai_jsonl_content.get("custom_id") + if custom_id: + if "labels" not in vertex_request_body: + vertex_request_body["labels"] = {} + vertex_request_body["labels"]["litellm_custom_id"] = str(custom_id) + vertex_jsonl_content.append({"request": vertex_request_body}) return vertex_jsonl_content diff --git a/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py b/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py index 598ad255aca..7549eb9eaa6 100644 --- a/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py @@ -1,5 +1,6 @@ """ Tests for VertexAIFilesConfig transformation methods (Issues 5-7). +Includes tests for Vertex AI batch output transformation to OpenAI format. """ import json @@ -9,7 +10,10 @@ import httpx import pytest from unittest.mock import MagicMock -from litellm.llms.vertex_ai.files.transformation import VertexAIFilesConfig +from litellm.llms.vertex_ai.files.transformation import ( + VertexAIFilesConfig, + VertexAIJsonlFilesTransformation, +) from litellm.types.llms.openai import OpenAIFileObject, HttpxBinaryResponseContent from openai.types.file_deleted import FileDeleted @@ -228,3 +232,297 @@ class TestTransformDeleteFile: "gs://prod-bucket/litellm-vertex-files/publishers/google/" "models/gemini-2.0-flash-001/abc-123" ) + + +class TestVertexBatchOutputTransformation: + """Test transformation of Vertex AI batch outputs to OpenAI format""" + + def test_transform_successful_vertex_batch_output(self, config): + """Test transformation of a successful Vertex AI batch output""" + # Sample Vertex AI batch output (based on actual format) + vertex_output = { + "status": "", + "processed_time": "2024-11-01T18:13:16.826+00:00", + "request": { + "contents": [{"role": "user", "parts": [{"text": "Hello world!"}]}], + "labels": {"litellm_custom_id": "request-1"} + }, + "response": { + "candidates": [{ + "content": { + "parts": [{"text": "Hello! How can I help you today?"}], + "role": "model" + }, + "finishReason": "STOP" + }], + "modelVersion": "gemini-2.0-flash-001@default", + "usageMetadata": { + "promptTokenCount": 10, + "candidatesTokenCount": 20, + "totalTokenCount": 30 + } + } + } + + content = json.dumps(vertex_output).encode("utf-8") + transformed_content = config._try_transform_vertex_batch_output_to_openai(content) + result = json.loads(transformed_content.decode("utf-8")) + + # Verify OpenAI format + assert "id" in result + assert "custom_id" in result + assert "response" in result + assert "error" in result + + # Verify custom_id was extracted from labels + assert result["custom_id"] == "request-1" + + # Verify response structure + assert result["response"]["status_code"] == 200 + assert "body" in result["response"] + + # Verify body has OpenAI format + body = result["response"]["body"] + assert "choices" in body + assert "usage" in body + assert "model" in body + + # Verify choices + assert len(body["choices"]) > 0 + choice = body["choices"][0] + assert "message" in choice + assert "content" in choice["message"] + assert "Hello! How can I help you today?" in choice["message"]["content"] + + def test_transform_error_vertex_batch_output(self, config): + """Test transformation of an error Vertex AI batch output""" + vertex_output = { + "status": "Error: Invalid request", + "processed_time": "2024-11-01T18:13:16.826+00:00", + "request": { + "contents": [{"role": "user", "parts": [{"text": "Hello world!"}]}], + "labels": {"litellm_custom_id": "request-error"} + }, + "response": {} + } + + content = json.dumps(vertex_output).encode("utf-8") + transformed_content = config._try_transform_vertex_batch_output_to_openai(content) + result = json.loads(transformed_content.decode("utf-8")) + + # Verify error format + assert result["response"]["status_code"] == 400 + assert result["error"] is not None + assert "Invalid request" in result["error"]["message"] + assert result["custom_id"] == "request-error" + + def test_transform_multiple_vertex_batch_outputs(self, config): + """Test transformation of multiple Vertex AI batch outputs (JSONL)""" + vertex_outputs = [ + { + "status": "", + "processed_time": "2024-11-01T18:13:16.826+00:00", + "request": { + "contents": [{"role": "user", "parts": [{"text": "First request"}]}], + "labels": {"litellm_custom_id": "request-1"} + }, + "response": { + "candidates": [{ + "content": {"parts": [{"text": "First response"}], "role": "model"}, + "finishReason": "STOP" + }], + "modelVersion": "gemini-2.0-flash-001@default", + "usageMetadata": { + "promptTokenCount": 5, + "candidatesTokenCount": 10, + "totalTokenCount": 15 + } + } + }, + { + "status": "", + "processed_time": "2024-11-01T18:13:17.826+00:00", + "request": { + "contents": [{"role": "user", "parts": [{"text": "Second request"}]}], + "labels": {"litellm_custom_id": "request-2"} + }, + "response": { + "candidates": [{ + "content": {"parts": [{"text": "Second response"}], "role": "model"}, + "finishReason": "STOP" + }], + "modelVersion": "gemini-2.0-flash-001@default", + "usageMetadata": { + "promptTokenCount": 6, + "candidatesTokenCount": 11, + "totalTokenCount": 17 + } + } + } + ] + + content = "\n".join(json.dumps(output) for output in vertex_outputs).encode("utf-8") + transformed_content = config._try_transform_vertex_batch_output_to_openai(content) + lines = transformed_content.decode("utf-8").strip().split("\n") + + assert len(lines) == 2 + + for i, line in enumerate(lines): + result = json.loads(line) + assert "id" in result + assert "response" in result + assert result["response"]["status_code"] == 200 + assert result["custom_id"] == f"request-{i+1}" + body = result["response"]["body"] + assert "choices" in body + assert len(body["choices"]) > 0 + + def test_non_batch_output_passthrough(self, config): + """Test that non-batch output is returned as-is""" + regular_content = b"This is just a regular file content" + transformed_content = config._try_transform_vertex_batch_output_to_openai(regular_content) + assert transformed_content == regular_content + + def test_invalid_json_passthrough(self, config): + """Test that invalid JSON is returned as-is""" + invalid_content = b'{"invalid": json content}' + transformed_content = config._try_transform_vertex_batch_output_to_openai(invalid_content) + assert transformed_content == invalid_content + + +class TestVertexBatchCustomIdLabels: + """Test custom_id handling in batch transformations""" + + def test_custom_id_added_to_labels_in_vertex_request(self): + """Test that custom_id from OpenAI format is added as a label in Vertex AI format""" + transformation = VertexAIJsonlFilesTransformation() + + openai_jsonl_content = [ + { + "custom_id": "request-1", + "method": "POST", + "url": "/v1/chat/completions", + "body": { + "model": "gemini-1.5-flash-001", + "messages": [{"role": "user", "content": "What is 2+2?"}], + "max_tokens": 10 + } + } + ] + + vertex_jsonl_content = transformation._transform_openai_jsonl_content_to_vertex_ai_jsonl_content( + openai_jsonl_content + ) + + assert len(vertex_jsonl_content) == 1 + vertex_request = vertex_jsonl_content[0] + + # Verify labels were added + assert "labels" in vertex_request["request"] + assert "litellm_custom_id" in vertex_request["request"]["labels"] + assert vertex_request["request"]["labels"]["litellm_custom_id"] == "request-1" + + def test_multiple_requests_each_get_their_own_label(self): + """Test that multiple requests each get their own custom_id label""" + transformation = VertexAIJsonlFilesTransformation() + + openai_jsonl_content = [ + { + "custom_id": f"request-{i+1}", + "method": "POST", + "url": "/v1/chat/completions", + "body": { + "model": "gemini-1.5-flash-001", + "messages": [{"role": "user", "content": f"Question {i+1}"}], + } + } + for i in range(3) + ] + + vertex_jsonl_content = transformation._transform_openai_jsonl_content_to_vertex_ai_jsonl_content( + openai_jsonl_content + ) + + assert len(vertex_jsonl_content) == 3 + + for i, vertex_request in enumerate(vertex_jsonl_content): + expected_custom_id = f"request-{i+1}" + assert vertex_request["request"]["labels"]["litellm_custom_id"] == expected_custom_id + + def test_request_without_custom_id_has_no_label(self): + """Test that requests without custom_id don't get a label""" + transformation = VertexAIJsonlFilesTransformation() + + openai_jsonl_content = [ + { + "method": "POST", + "url": "/v1/chat/completions", + "body": { + "model": "gemini-1.5-flash-001", + "messages": [{"role": "user", "content": "Question"}], + } + } + ] + + vertex_jsonl_content = transformation._transform_openai_jsonl_content_to_vertex_ai_jsonl_content( + openai_jsonl_content + ) + + # Should not have labels if no custom_id was provided + assert "labels" not in vertex_jsonl_content[0]["request"] + + def test_end_to_end_custom_id_round_trip(self): + """ + Test the full round trip: OpenAI format -> Vertex AI format -> Vertex AI output -> OpenAI output + Verify that custom_id is preserved through the entire flow. + """ + transformation = VertexAIJsonlFilesTransformation() + config = VertexAIFilesConfig() + + # Step 1: Transform OpenAI input to Vertex AI format + openai_input = [ + { + "custom_id": "my-custom-request-id", + "method": "POST", + "url": "/v1/chat/completions", + "body": { + "model": "gemini-1.5-flash-001", + "messages": [{"role": "user", "content": "Hello"}], + } + } + ] + + vertex_input = transformation._transform_openai_jsonl_content_to_vertex_ai_jsonl_content( + openai_input + ) + + # Verify label was added + assert vertex_input[0]["request"]["labels"]["litellm_custom_id"] == "my-custom-request-id" + + # Step 2: Simulate Vertex AI batch output (with the label echoed back) + vertex_output = { + "status": "", + "processed_time": "2024-11-01T18:13:16.826+00:00", + "request": vertex_input[0]["request"], + "response": { + "candidates": [{ + "content": {"parts": [{"text": "Hi there!"}], "role": "model"}, + "finishReason": "STOP" + }], + "modelVersion": "gemini-2.0-flash-001@default", + "usageMetadata": { + "promptTokenCount": 5, + "candidatesTokenCount": 10, + "totalTokenCount": 15 + } + } + } + + # Step 3: Transform Vertex AI output back to OpenAI format + content = json.dumps(vertex_output).encode("utf-8") + transformed_content = config._try_transform_vertex_batch_output_to_openai(content) + openai_output = json.loads(transformed_content.decode("utf-8")) + + # Step 4: Verify custom_id was preserved + assert openai_output["custom_id"] == "my-custom-request-id" + assert openai_output["response"]["status_code"] == 200 From 380c14e7dd020653fbc2cd9a468fee4ac5cba99b Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 13 Apr 2026 17:16:17 +0530 Subject: [PATCH 002/196] fix: address Greptile review comments - Sanitize custom_id to meet GCP label constraints (lowercase, alphanumeric, max 63 chars) - Improve batch output detection heuristic with processed_time and candidates/status checks - Move inline imports to module level - Fix Content-Length header for transformed responses - Add test for label sanitization Made-with: Cursor --- litellm/llms/vertex_ai/files/handler.py | 6 +-- .../llms/vertex_ai/files/transformation.py | 47 +++++++++++++++---- .../test_vertex_ai_files_transformation.py | 39 +++++++++++++++ 3 files changed, 79 insertions(+), 13 deletions(-) diff --git a/litellm/llms/vertex_ai/files/handler.py b/litellm/llms/vertex_ai/files/handler.py index 81bf7852c82..816fb82812c 100644 --- a/litellm/llms/vertex_ai/files/handler.py +++ b/litellm/llms/vertex_ai/files/handler.py @@ -17,9 +17,10 @@ from litellm.types.llms.openai import ( HttpxBinaryResponseContent, OpenAIFileObject, ) +from litellm.litellm_core_utils.litellm_logging import Logging from litellm.types.llms.vertex_ai import VERTEX_CREDENTIALS_TYPES -from .transformation import VertexAIJsonlFilesTransformation +from .transformation import VertexAIFilesConfig, VertexAIJsonlFilesTransformation vertex_ai_files_transformation = VertexAIJsonlFilesTransformation() @@ -197,9 +198,6 @@ class VertexAIFilesHandler(GCSBucketBase): ) # Apply transformation to convert Vertex AI batch outputs to OpenAI format - from .transformation import VertexAIFilesConfig - from litellm.litellm_core_utils.litellm_logging import Logging - config = VertexAIFilesConfig() # Create a logging object for transformation diff --git a/litellm/llms/vertex_ai/files/transformation.py b/litellm/llms/vertex_ai/files/transformation.py index a967e968e4a..5cf1166baaf 100644 --- a/litellm/llms/vertex_ai/files/transformation.py +++ b/litellm/llms/vertex_ai/files/transformation.py @@ -1,5 +1,6 @@ import json import os +import re import time from typing import Any, Dict, List, Optional, Tuple, Union @@ -9,6 +10,7 @@ from openai.types.file_deleted import FileDeleted from litellm._uuid import uuid from litellm.files.utils import FilesAPIUtils +from litellm.litellm_core_utils.litellm_logging import Logging from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.base_llm.files.transformation import ( @@ -32,12 +34,31 @@ from litellm.types.llms.openai import ( PathLike, ) from litellm.types.llms.vertex_ai import GcsBucketResponse -from litellm.types.utils import ExtractedFileData, LlmProviders +from litellm.types.utils import ExtractedFileData, LlmProviders, ModelResponse from ..common_utils import VertexAIError from ..vertex_llm_base import VertexBase +def _sanitize_gcp_label_value(value: str) -> str: + """ + Sanitize a string to meet GCP label value constraints. + + GCP label values must: + - Be lowercase + - Contain only letters, numbers, underscores, and hyphens + - Be max 63 characters + + Args: + value: The string to sanitize + + Returns: + A sanitized string that meets GCP label constraints + """ + sanitized = re.sub(r"[^a-z0-9_-]", "_", value.lower()) + return sanitized[:63] + + class VertexAIFilesConfig(VertexBase, BaseFilesConfig): """ Config for VertexAI Files @@ -254,7 +275,7 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): if custom_id: if "labels" not in vertex_request_body: vertex_request_body["labels"] = {} - vertex_request_body["labels"]["litellm_custom_id"] = str(custom_id) + vertex_request_body["labels"]["litellm_custom_id"] = _sanitize_gcp_label_value(str(custom_id)) vertex_jsonl_content.append({"request": vertex_request_body}) return vertex_jsonl_content @@ -538,8 +559,20 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): # Try to parse the first line to see if it's Vertex AI batch output first_line = json.loads(lines[0]) - # Check if it has Vertex AI batch output structure - if not ("response" in first_line and "request" in first_line): + # Check if it has Vertex AI batch output structure with discriminating fields + # Must have request, response, and processed_time + # Plus either candidates (success) or status (error) + has_base_structure = ( + "response" in first_line + and "request" in first_line + and "processed_time" in first_line + ) + has_success_or_error = ( + "candidates" in first_line.get("response", {}) + or "status" in first_line + ) + + if not (has_base_structure and has_success_or_error): # Not a Vertex AI batch output, return as-is return content @@ -573,10 +606,6 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): Transform a single Vertex AI batch output line to OpenAI format. Uses the existing VertexGeminiConfig transformation for the response. """ - from litellm.types.utils import ModelResponse - import httpx - import time - # Extract custom_id from request labels custom_id = "unknown" request_data = vertex_output.get("request", {}) @@ -756,7 +785,7 @@ class VertexAIJsonlFilesTransformation(VertexGeminiConfig): if custom_id: if "labels" not in vertex_request_body: vertex_request_body["labels"] = {} - vertex_request_body["labels"]["litellm_custom_id"] = str(custom_id) + vertex_request_body["labels"]["litellm_custom_id"] = _sanitize_gcp_label_value(str(custom_id)) vertex_jsonl_content.append({"request": vertex_request_body}) return vertex_jsonl_content diff --git a/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py b/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py index 7549eb9eaa6..78f0cde96fd 100644 --- a/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py @@ -526,3 +526,42 @@ class TestVertexBatchCustomIdLabels: # Step 4: Verify custom_id was preserved assert openai_output["custom_id"] == "my-custom-request-id" assert openai_output["response"]["status_code"] == 200 + + def test_custom_id_label_sanitization(self): + """Test that custom_id values are sanitized to meet GCP label constraints""" + from litellm.llms.vertex_ai.files.transformation import ( + VertexAIJsonlFilesTransformation, + _sanitize_gcp_label_value, + ) + + transformation = VertexAIJsonlFilesTransformation() + + # Test sanitization function + assert _sanitize_gcp_label_value("MyRequest-1") == "myrequest-1" + assert _sanitize_gcp_label_value("Request.With.Dots") == "request_with_dots" + assert _sanitize_gcp_label_value("Request With Spaces") == "request_with_spaces" + assert _sanitize_gcp_label_value("Request@#$%Special") == "request____special" + + # Test max length (63 chars) + long_id = "a" * 100 + assert len(_sanitize_gcp_label_value(long_id)) == 63 + + # Test in actual transformation + openai_input = [ + { + "custom_id": "MyRequest-1", + "method": "POST", + "url": "/v1/chat/completions", + "body": { + "model": "gemini-1.5-flash-001", + "messages": [{"role": "user", "content": "Hello"}], + } + } + ] + + vertex_input = transformation._transform_openai_jsonl_content_to_vertex_ai_jsonl_content( + openai_input + ) + + # Verify label was sanitized + assert vertex_input[0]["request"]["labels"]["litellm_custom_id"] == "myrequest-1" From 19c54b68a9e7250523bb9eaa834fb5bae3204fdb Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 14 Apr 2026 22:07:05 +0530 Subject: [PATCH 003/196] Fix code qa --- litellm/llms/vertex_ai/files/transformation.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/litellm/llms/vertex_ai/files/transformation.py b/litellm/llms/vertex_ai/files/transformation.py index 5cf1166baaf..06aa4cc0925 100644 --- a/litellm/llms/vertex_ai/files/transformation.py +++ b/litellm/llms/vertex_ai/files/transformation.py @@ -649,8 +649,6 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): model = model.split("@")[0] # Create logging object for transformation - from litellm.litellm_core_utils.litellm_logging import Logging - logging_obj = Logging( model=model, messages=[], From f6c7e4ee3abe8a75a469f1a174e834d54d8716f0 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 14 Apr 2026 22:08:13 +0530 Subject: [PATCH 004/196] Fix greptile review --- .../test_vertex_ai_files_transformation.py | 50 ++++++++++++++++--- 1 file changed, 43 insertions(+), 7 deletions(-) diff --git a/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py b/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py index 78f0cde96fd..1fecb63541f 100644 --- a/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/files/test_vertex_ai_files_transformation.py @@ -316,6 +316,38 @@ class TestVertexBatchOutputTransformation: assert "Invalid request" in result["error"]["message"] assert result["custom_id"] == "request-error" + def test_transform_vertex_batch_output_legacy_labels_only_sanitized(self, config): + """Older LiteLLM batches only stored litellm_custom_id (sanitized); read path still works.""" + vertex_output = { + "status": "", + "processed_time": "2024-11-01T18:13:16.826+00:00", + "request": { + "contents": [{"role": "user", "parts": [{"text": "Hello world!"}]}], + "labels": {"litellm_custom_id": "myrequest-1"}, + }, + "response": { + "candidates": [{ + "content": { + "parts": [{"text": "Hello!"}], + "role": "model", + }, + "finishReason": "STOP", + }], + "modelVersion": "gemini-2.0-flash-001@default", + "usageMetadata": { + "promptTokenCount": 10, + "candidatesTokenCount": 20, + "totalTokenCount": 30, + }, + }, + } + + content = json.dumps(vertex_output).encode("utf-8") + transformed_content = config._try_transform_vertex_batch_output_to_openai(content) + result = json.loads(transformed_content.decode("utf-8")) + + assert result["custom_id"] == "myrequest-1" + def test_transform_multiple_vertex_batch_outputs(self, config): """Test transformation of multiple Vertex AI batch outputs (JSONL)""" vertex_outputs = [ @@ -421,6 +453,7 @@ class TestVertexBatchCustomIdLabels: assert "labels" in vertex_request["request"] assert "litellm_custom_id" in vertex_request["request"]["labels"] assert vertex_request["request"]["labels"]["litellm_custom_id"] == "request-1" + assert vertex_request["request"]["labels"]["litellm_custom_id_raw"] == "request-1" def test_multiple_requests_each_get_their_own_label(self): """Test that multiple requests each get their own custom_id label""" @@ -448,6 +481,7 @@ class TestVertexBatchCustomIdLabels: for i, vertex_request in enumerate(vertex_jsonl_content): expected_custom_id = f"request-{i+1}" assert vertex_request["request"]["labels"]["litellm_custom_id"] == expected_custom_id + assert vertex_request["request"]["labels"]["litellm_custom_id_raw"] == expected_custom_id def test_request_without_custom_id_has_no_label(self): """Test that requests without custom_id don't get a label""" @@ -479,10 +513,10 @@ class TestVertexBatchCustomIdLabels: transformation = VertexAIJsonlFilesTransformation() config = VertexAIFilesConfig() - # Step 1: Transform OpenAI input to Vertex AI format + # Step 1: Transform OpenAI input to Vertex AI format (mixed case exercises raw label) openai_input = [ { - "custom_id": "my-custom-request-id", + "custom_id": "MyRequest-1", "method": "POST", "url": "/v1/chat/completions", "body": { @@ -496,8 +530,9 @@ class TestVertexBatchCustomIdLabels: openai_input ) - # Verify label was added - assert vertex_input[0]["request"]["labels"]["litellm_custom_id"] == "my-custom-request-id" + # Verify GCP-safe label and preserved raw for round-trip + assert vertex_input[0]["request"]["labels"]["litellm_custom_id"] == "myrequest-1" + assert vertex_input[0]["request"]["labels"]["litellm_custom_id_raw"] == "MyRequest-1" # Step 2: Simulate Vertex AI batch output (with the label echoed back) vertex_output = { @@ -523,8 +558,8 @@ class TestVertexBatchCustomIdLabels: transformed_content = config._try_transform_vertex_batch_output_to_openai(content) openai_output = json.loads(transformed_content.decode("utf-8")) - # Step 4: Verify custom_id was preserved - assert openai_output["custom_id"] == "my-custom-request-id" + # Step 4: Verify custom_id was preserved (original casing, not sanitized label) + assert openai_output["custom_id"] == "MyRequest-1" assert openai_output["response"]["status_code"] == 200 def test_custom_id_label_sanitization(self): @@ -563,5 +598,6 @@ class TestVertexBatchCustomIdLabels: openai_input ) - # Verify label was sanitized + # Verify label was sanitized and original retained for read-back assert vertex_input[0]["request"]["labels"]["litellm_custom_id"] == "myrequest-1" + assert vertex_input[0]["request"]["labels"]["litellm_custom_id_raw"] == "MyRequest-1" From e3eb1b5400fb739835cbdf3e89aca3723005f5b0 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 14 Apr 2026 22:12:47 +0530 Subject: [PATCH 005/196] Fix greptile review --- .../llms/vertex_ai/files/transformation.py | 40 ++++++++++++++----- 1 file changed, 31 insertions(+), 9 deletions(-) diff --git a/litellm/llms/vertex_ai/files/transformation.py b/litellm/llms/vertex_ai/files/transformation.py index 06aa4cc0925..3a140eb723f 100644 --- a/litellm/llms/vertex_ai/files/transformation.py +++ b/litellm/llms/vertex_ai/files/transformation.py @@ -40,6 +40,9 @@ from ..common_utils import VertexAIError from ..vertex_llm_base import VertexBase +_GCP_LABEL_VALUE_MAX_LEN = 63 + + def _sanitize_gcp_label_value(value: str) -> str: """ Sanitize a string to meet GCP label value constraints. @@ -56,7 +59,28 @@ def _sanitize_gcp_label_value(value: str) -> str: A sanitized string that meets GCP label constraints """ sanitized = re.sub(r"[^a-z0-9_-]", "_", value.lower()) - return sanitized[:63] + return sanitized[:_GCP_LABEL_VALUE_MAX_LEN] + + +def _set_litellm_batch_custom_id_labels(labels: Dict[str, str], custom_id: Any) -> None: + """ + Store OpenAI batch custom_id for Vertex batch correlation. + + ``litellm_custom_id`` is GCP-label-safe (may alter casing and characters). + ``litellm_custom_id_raw`` preserves the original string (truncated) for + round-trip correlation in batch output transforms. + """ + custom_id_str = str(custom_id) + labels["litellm_custom_id"] = _sanitize_gcp_label_value(custom_id_str) + labels["litellm_custom_id_raw"] = custom_id_str[:_GCP_LABEL_VALUE_MAX_LEN] + + +def _get_litellm_batch_custom_id_from_labels(labels: Dict[str, Any]) -> str: + """Prefer unsanitized custom_id when present (see _set_litellm_batch_custom_id_labels).""" + raw = labels.get("litellm_custom_id_raw") + if raw: + return str(raw) + return str(labels.get("litellm_custom_id", "unknown")) class VertexAIFilesConfig(VertexBase, BaseFilesConfig): @@ -275,7 +299,7 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): if custom_id: if "labels" not in vertex_request_body: vertex_request_body["labels"] = {} - vertex_request_body["labels"]["litellm_custom_id"] = _sanitize_gcp_label_value(str(custom_id)) + _set_litellm_batch_custom_id_labels(vertex_request_body["labels"], custom_id) vertex_jsonl_content.append({"request": vertex_request_body}) return vertex_jsonl_content @@ -529,7 +553,7 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): Vertex AI batch output format (predictions.jsonl): { - "request": {"contents": [...], "labels": {"litellm_custom_id": "request-1"}}, + "request": {"contents": [...], "labels": {"litellm_custom_id": "request-1", "litellm_custom_id_raw": "..."}}, "status": "", "response": {"candidates": [...], "modelVersion": "gemini-2.5-flash", ...}, "processed_time": "2026-04-13T10:18:18.102004+00:00" @@ -606,12 +630,10 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): Transform a single Vertex AI batch output line to OpenAI format. Uses the existing VertexGeminiConfig transformation for the response. """ - # Extract custom_id from request labels - custom_id = "unknown" + # Extract custom_id from request labels (prefer raw for OpenAI round-trip) request_data = vertex_output.get("request", {}) - labels = request_data.get("labels", {}) - if "litellm_custom_id" in labels: - custom_id = labels["litellm_custom_id"] + labels = request_data.get("labels", {}) or {} + custom_id = _get_litellm_batch_custom_id_from_labels(labels) # Check if there's an error status = vertex_output.get("status", "") @@ -783,7 +805,7 @@ class VertexAIJsonlFilesTransformation(VertexGeminiConfig): if custom_id: if "labels" not in vertex_request_body: vertex_request_body["labels"] = {} - vertex_request_body["labels"]["litellm_custom_id"] = _sanitize_gcp_label_value(str(custom_id)) + _set_litellm_batch_custom_id_labels(vertex_request_body["labels"], custom_id) vertex_jsonl_content.append({"request": vertex_request_body}) return vertex_jsonl_content From 7fc54ad460938ee81eae021b510db4dddb0bb2ee Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 20 Apr 2026 10:23:40 +0530 Subject: [PATCH 006/196] fix(vertex_ai): omit system_instruction/tools/toolConfig when cachedContent set Vertex generateContent returns INVALID_ARGUMENT if cachedContent is sent with system_instruction, tools, or toolConfig; those belong on CachedContent. Fixes #26014 Made-with: Cursor --- .../llms/vertex_ai/gemini/transformation.py | 9 ++-- .../test_vertex_ai_gemini_transformation.py | 47 +++++++++++++++++++ 2 files changed, 52 insertions(+), 4 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/transformation.py b/litellm/llms/vertex_ai/gemini/transformation.py index d0f0e5c1e24..ef35fc09610 100644 --- a/litellm/llms/vertex_ai/gemini/transformation.py +++ b/litellm/llms/vertex_ai/gemini/transformation.py @@ -748,13 +748,14 @@ def _transform_request_body( # noqa: PLR0915 ] data = RequestBody(contents=content) - if system_instructions is not None: + # Vertex rejects system_instruction/tools/toolConfig alongside cachedContent. + if system_instructions is not None and cached_content is None: data["system_instruction"] = system_instructions - if tools is not None: + if tools is not None and cached_content is None: data["tools"] = tools - if tool_choice is not None: + if tool_choice is not None and cached_content is None: data["toolConfig"] = tool_choice - if include_server_side_tool_invocations: + if include_server_side_tool_invocations and cached_content is None: if "toolConfig" not in data: data["toolConfig"] = {} data["toolConfig"]["includeServerSideToolInvocations"] = True diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py index 6937b4c3ba1..6952be7db3c 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py @@ -86,6 +86,53 @@ def test_check_if_part_exists_in_parts_camel_case_snake_case(): assert check_if_part_exists_in_parts(parts_mixed, part_mixed_casing) +def test_cached_content_omits_system_instruction_tools_toolconfig(): + """Regression: #26014 / #17304 — cachedContent must not ship with tools/system/toolConfig.""" + cache_name = "projects/p/locations/us-central1/cachedContents/abc123" + messages = [ + {"role": "system", "content": "You are helpful"}, + {"role": "user", "content": "hi"}, + ] + optional_params = { + "tools": [ + { + "functionDeclarations": [ + {"name": "get_weather", "description": "Get weather"}, + ] + } + ], + "tool_choice": {"functionCallingConfig": {"mode": "AUTO"}}, + } + + result = _transform_request_body( + messages=list(messages), + model="gemini-2.5-pro", + optional_params=dict(optional_params), + custom_llm_provider="vertex_ai", + litellm_params={}, + cached_content=cache_name, + ) + + assert result.get("cachedContent") == cache_name + assert "system_instruction" not in result + assert "tools" not in result + assert "toolConfig" not in result + assert "contents" in result + + # Without cache, conflicting fields are included as before + result_no_cache = _transform_request_body( + messages=list(messages), + model="gemini-2.5-pro", + optional_params=dict(optional_params), + custom_llm_provider="vertex_ai", + litellm_params={}, + cached_content=None, + ) + assert "system_instruction" in result_no_cache + assert "tools" in result_no_cache + assert "toolConfig" in result_no_cache + + # Tests for issue #14556: Labels field provider-aware filtering def test_google_genai_excludes_labels(): """Test that Google GenAI/AI Studio endpoints exclude labels when custom_llm_provider='gemini'""" From 499f5d6d6b367920b8944ce18349227b8bb48672 Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Mon, 20 Apr 2026 09:13:22 -0400 Subject: [PATCH 007/196] prevent post call guardrail called twice --- litellm/proxy/common_request_processing.py | 6 +++ .../test_deferred_guardrail_logging.py | 54 +++++++++++++++++++ 2 files changed, 60 insertions(+) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 97801baaf0c..86a7b2dfa9b 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -1507,6 +1507,12 @@ class ProxyBaseLLMRequestProcessing: # here would duplicate the guardrail API call # (e.g. double OpenAI Moderation charges). continue + if "async_post_call_streaming_iterator_hook" in type(cb).__dict__: + # Skip — the guardrail already scanned the assembled + # response via its own streaming iterator hook in the + # streaming pipeline. re running this function async_post_call_success_hook + # here would duplicate the scan and can spuriously block the guardrail that already passed / failed. + continue else: guardrail_result = await cb.async_post_call_success_hook( user_api_key_dict=captured_user_api_key_dict, diff --git a/tests/test_litellm/proxy/guardrails/test_deferred_guardrail_logging.py b/tests/test_litellm/proxy/guardrails/test_deferred_guardrail_logging.py index 160d621e60c..ee37b5e598c 100644 --- a/tests/test_litellm/proxy/guardrails/test_deferred_guardrail_logging.py +++ b/tests/test_litellm/proxy/guardrails/test_deferred_guardrail_logging.py @@ -683,6 +683,60 @@ class TestDeferredStreamingClosure: apply_guardrail_called is False ), "apply_guardrail guardrails must be SKIPPED in deferred path" + @pytest.mark.asyncio + async def test_streaming_iterator_hook_skipped_in_deferred_path(self): + """regression test: guardrails that define async_post_call_streaming_iterator_hook must be SKIPPED in _run_deferred_stream_guardrails. + The iterator hook already scanned the assembled response in the streaming + pipeline""" + success_hook_called = False + + class IteratorHookGuardrail(CustomGuardrail): + def __init__(self): + super().__init__( + guardrail_name="iterator-hook", + default_on=True, + event_hook=GuardrailEventHooks.post_call, + ) + + async def async_post_call_streaming_iterator_hook( + self, user_api_key_dict, response, request_data + ): + async for chunk in response: + yield chunk + + async def async_post_call_success_hook( + self, data: dict, user_api_key_dict: UserAPIKeyAuth, response: Any + ) -> Any: + nonlocal success_hook_called + success_hook_called = True + return response + + mock_logging_obj = MagicMock() + mock_logging_obj.model_call_details = {"metadata": {}} + + async def track_async_success(*args, **kwargs): + pass + + mock_logging_obj.async_success_handler = track_async_success + + guardrail = IteratorHookGuardrail() + + with patch("litellm.callbacks", [guardrail]): + await ProxyBaseLLMRequestProcessing._run_deferred_stream_guardrails( + captured_data={"model": "gpt-4", "metadata": {}}, + captured_user_api_key_dict=UserAPIKeyAuth(api_key="test"), + captured_logging_obj=mock_logging_obj, + assembled_response=MagicMock(), + cache_hit=False, + ) + + await asyncio.sleep(0) + + assert success_hook_called is False, ( + "Guardrails that implement async_post_call_streaming_iterator_hook " + "must be SKIPPED in deferred path — the iterator hook already ran" + ) + @pytest.mark.asyncio async def test_hooks_receive_merged_guardrail_data(self): """Hooks must receive guardrail_data (the merged dict from From 437a179612c939ee79a09b569be2ac94dce05233 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 21 Apr 2026 15:18:23 +0530 Subject: [PATCH 008/196] fix(router): constrain same-name deployment routing by access groups Filter router candidate deployments by caller-authorized model access groups when access is granted via group membership, preventing cross-group load balancing for shared public model names. Made-with: Cursor --- litellm/router.py | 77 ++++++++++++++++++++ tests/test_litellm/test_router.py | 115 ++++++++++++++++++++++++++++++ 2 files changed, 192 insertions(+) diff --git a/litellm/router.py b/litellm/router.py index f6976109bc7..2e436973e7f 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -9241,6 +9241,12 @@ class Router: healthy_deployments = self._get_all_deployments( model_name=model, team_id=request_team_id ) + healthy_deployments = self._filter_deployments_by_model_access_groups( + model=model, + healthy_deployments=healthy_deployments, + request_kwargs=request_kwargs, + request_team_id=request_team_id, + ) if len(healthy_deployments) == 0: # check if the user sent in a deployment name instead @@ -9264,6 +9270,14 @@ class Router: healthy_deployments = self._get_all_deployments( model_name=model, team_id=request_team_id ) + healthy_deployments = ( + self._filter_deployments_by_model_access_groups( + model=model, + healthy_deployments=healthy_deployments, + request_kwargs=request_kwargs, + request_team_id=request_team_id, + ) + ) # If still no deployments after checking for fallbacks, raise an error if len(healthy_deployments) == 0: @@ -9289,6 +9303,68 @@ class Router: return model, healthy_deployments + def _filter_deployments_by_model_access_groups( + self, + model: str, + healthy_deployments: List, + request_kwargs: Optional[Dict], + request_team_id: Optional[str], + ) -> List: + """ + Restrict candidate deployments to caller-authorized model access groups. + + This is only applied when: + - request metadata includes `user_api_key_auth`, and + - caller permissions for this model are access-group-only + (no explicit model, wildcard, or all-proxy grants). + """ + if not healthy_deployments or request_kwargs is None: + return healthy_deployments + + metadata = request_kwargs.get("metadata") or {} + litellm_metadata = request_kwargs.get("litellm_metadata") or {} + user_api_key_auth = metadata.get("user_api_key_auth") or litellm_metadata.get( + "user_api_key_auth" + ) + if user_api_key_auth is None: + return healthy_deployments + + object_models = set(getattr(user_api_key_auth, "models", []) or []) + object_team_models = set(getattr(user_api_key_auth, "team_models", []) or []) + allowed_models = object_models | object_team_models + if not allowed_models: + return healthy_deployments + + # If caller has direct model/wildcard/all-proxy access, do not constrain + # deployment choice by access group. + if ( + model in allowed_models + or "*" in allowed_models + or "all-proxy-models" in allowed_models + ): + return healthy_deployments + + access_groups_for_model = self.get_model_access_groups( + model_name=model, team_id=request_team_id + ) + if len(access_groups_for_model) == 0: + return healthy_deployments + + allowed_access_groups = set(access_groups_for_model.keys()) & allowed_models + if not allowed_access_groups: + return healthy_deployments + + filtered_deployments = [] + for deployment in healthy_deployments: + deployment_model_info = deployment.get("model_info") or {} + deployment_access_groups = set( + deployment_model_info.get("access_groups", []) or [] + ) + if deployment_access_groups & allowed_access_groups: + filtered_deployments.append(deployment) + + return filtered_deployments + async def async_get_healthy_deployments( self, model: str, @@ -9796,6 +9872,7 @@ class Router: messages=messages, input=input, specific_deployment=specific_deployment, + request_kwargs=request_kwargs, ) if isinstance(healthy_deployments, dict): diff --git a/tests/test_litellm/test_router.py b/tests/test_litellm/test_router.py index 2ae54f55103..af8633361ac 100644 --- a/tests/test_litellm/test_router.py +++ b/tests/test_litellm/test_router.py @@ -3204,3 +3204,118 @@ async def test_multiregion_team_failover_between_regions(): "response from us-east-1", "response from us-west-2", ] + + +def test_access_group_scoped_key_filters_deployments_with_same_public_model(): + """ + If a key can access a model only via access group membership, + router candidate deployments for that public model should be constrained + to deployments in the allowed access group. + """ + from litellm.proxy._types import UserAPIKeyAuth + + router = litellm.Router( + model_list=[ + { + "model_name": "gpt-5", + "litellm_params": { + "model": "openai/gpt-5.1", + "api_key": "key1", + "mock_response": "response-via-AG1", + }, + "model_info": {"access_groups": ["AG1"]}, + }, + { + "model_name": "gpt-5", + "litellm_params": { + "model": "openai/gpt-4o", + "api_key": "key2", + "mock_response": "response-via-AG2", + }, + "model_info": {"access_groups": ["AG2"]}, + }, + ] + ) + + scoped_key = UserAPIKeyAuth( + api_key="hashed-key", + team_id="team2", + models=["AG2"], + team_models=["AG2"], + ) + + _model, deployments = router._common_checks_available_deployment( + model="gpt-5", + request_kwargs={ + "metadata": { + "user_api_key_team_id": "team2", + "user_api_key_auth": scoped_key, + } + }, + ) + + assert len(deployments) == 1 + assert deployments[0].get("model_info", {}).get("access_groups") == ["AG2"] + + seen = set() + for _ in range(20): + response = router.completion( + model="gpt-5", + messages=[{"role": "user", "content": "hello"}], + metadata={"user_api_key_team_id": "team2", "user_api_key_auth": scoped_key}, + ) + seen.add(response.choices[0].message.content) + + assert seen == {"response-via-AG2"} + + +def test_explicit_model_access_does_not_force_access_group_filtering(): + """ + If a key has explicit model access in addition to access group entries, + do not force access-group-only filtering for deployment selection. + """ + from litellm.proxy._types import UserAPIKeyAuth + + router = litellm.Router( + model_list=[ + { + "model_name": "gpt-5", + "litellm_params": { + "model": "openai/gpt-5.1", + "api_key": "key1", + "mock_response": "response-via-AG1", + }, + "model_info": {"access_groups": ["AG1"]}, + }, + { + "model_name": "gpt-5", + "litellm_params": { + "model": "openai/gpt-4o", + "api_key": "key2", + "mock_response": "response-via-AG2", + }, + "model_info": {"access_groups": ["AG2"]}, + }, + ] + ) + + explicit_key = UserAPIKeyAuth( + api_key="hashed-key", + team_id="team2", + models=["AG2", "gpt-5"], + team_models=["AG2", "gpt-5"], + ) + + _model, deployments = router._common_checks_available_deployment( + model="gpt-5", + request_kwargs={ + "metadata": { + "user_api_key_team_id": "team2", + "user_api_key_auth": explicit_key, + } + }, + ) + + deployment_groups = [d.get("model_info", {}).get("access_groups") for d in deployments] + assert ["AG1"] in deployment_groups + assert ["AG2"] in deployment_groups From a3da4721cac04518a1acef5dddf34e83b18d4092 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 21 Apr 2026 15:21:35 +0530 Subject: [PATCH 009/196] test(router): add coverage for access-group deployment filter Add a router utils unit test that directly exercises _filter_deployments_by_model_access_groups for access-group-only key permissions. Made-with: Cursor --- .../test_router_utils_common_utils.py | 44 +++++++++++++++++++ 1 file changed, 44 insertions(+) diff --git a/tests/test_litellm/router_utils/test_router_utils_common_utils.py b/tests/test_litellm/router_utils/test_router_utils_common_utils.py index 02241d4bc92..c6fff7945d6 100644 --- a/tests/test_litellm/router_utils/test_router_utils_common_utils.py +++ b/tests/test_litellm/router_utils/test_router_utils_common_utils.py @@ -4,6 +4,7 @@ from unittest.mock import Mock import pytest from litellm import Router +from litellm.proxy._types import UserAPIKeyAuth from litellm.router_utils.common_utils import ( _deployment_supports_web_search, filter_team_based_models, @@ -362,3 +363,46 @@ def test_invalidate_model_group_info_cache(): # Invalidate and verify cache is cleared router._invalidate_model_group_info_cache() assert router._cached_get_model_group_info.cache_info().currsize == 0 + + +def test_filter_deployments_by_model_access_groups_access_group_only_key(): + """ + Access-group-only keys should only route to deployments in allowed groups, + even when multiple deployments share the same public model name. + """ + router = Router( + model_list=[ + { + "model_name": "gpt-5", + "litellm_params": {"model": "openai/gpt-5.1", "api_key": "key-1"}, + "model_info": {"access_groups": ["AG1"]}, + }, + { + "model_name": "gpt-5", + "litellm_params": {"model": "openai/gpt-4o", "api_key": "key-2"}, + "model_info": {"access_groups": ["AG2"]}, + }, + ] + ) + + scoped_key = UserAPIKeyAuth( + api_key="hashed-key", + team_id="team-2", + models=["AG2"], + team_models=["AG2"], + ) + + filtered = router._filter_deployments_by_model_access_groups( + model="gpt-5", + healthy_deployments=router._get_all_deployments(model_name="gpt-5"), + request_kwargs={ + "metadata": { + "user_api_key_team_id": "team-2", + "user_api_key_auth": scoped_key, + } + }, + request_team_id="team-2", + ) + + assert len(filtered) == 1 + assert filtered[0].get("model_info", {}).get("access_groups") == ["AG2"] From d6be59eac51a304739061559a9e5f865dc2a5e81 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 21 Apr 2026 15:28:58 +0530 Subject: [PATCH 010/196] chore(router): clarify empty access-group overlap behavior Document why empty allowed_access_groups intentionally preserves unfiltered deployments to avoid breaking non-access-group authorization paths. Made-with: Cursor --- litellm/router.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/litellm/router.py b/litellm/router.py index 2e436973e7f..9edf28df2bf 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -9352,6 +9352,8 @@ class Router: allowed_access_groups = set(access_groups_for_model.keys()) & allowed_models if not allowed_access_groups: + # No overlap means this request was not authorized via model access + # group membership for this model, so do not force group filtering. return healthy_deployments filtered_deployments = [] From 47614f967bed2966f52a27a7428c1f32c744a7a4 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Wed, 22 Apr 2026 17:47:27 +0530 Subject: [PATCH 011/196] refactor code --- .../llms/vertex_ai/gemini/transformation.py | 25 +++++++++++-------- 1 file changed, 15 insertions(+), 10 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/transformation.py b/litellm/llms/vertex_ai/gemini/transformation.py index ef35fc09610..33301a8fd8d 100644 --- a/litellm/llms/vertex_ai/gemini/transformation.py +++ b/litellm/llms/vertex_ai/gemini/transformation.py @@ -749,16 +749,21 @@ def _transform_request_body( # noqa: PLR0915 data = RequestBody(contents=content) # Vertex rejects system_instruction/tools/toolConfig alongside cachedContent. - if system_instructions is not None and cached_content is None: - data["system_instruction"] = system_instructions - if tools is not None and cached_content is None: - data["tools"] = tools - if tool_choice is not None and cached_content is None: - data["toolConfig"] = tool_choice - if include_server_side_tool_invocations and cached_content is None: - if "toolConfig" not in data: - data["toolConfig"] = {} - data["toolConfig"]["includeServerSideToolInvocations"] = True + # Treat dropping these fields as a request mutation guarded by modify_params. + can_send_cache_incompatible_fields = ( + cached_content is None or litellm.modify_params is False + ) + if can_send_cache_incompatible_fields: + if system_instructions is not None: + data["system_instruction"] = system_instructions + if tools is not None: + data["tools"] = tools + if tool_choice is not None: + data["toolConfig"] = tool_choice + if include_server_side_tool_invocations: + if "toolConfig" not in data: + data["toolConfig"] = {} + data["toolConfig"]["includeServerSideToolInvocations"] = True if safety_settings is not None: data["safetySettings"] = safety_settings if generation_config is not None and len(generation_config) > 0: From f92594f2c68a67d77d744c1344ace26e2a575efa Mon Sep 17 00:00:00 2001 From: Ryan Crabbe Date: Wed, 22 Apr 2026 14:28:58 -0700 Subject: [PATCH 012/196] fix: honor key access_group_ids when team restricts models Two model-access gates run per request in `common_checks` and they're asymmetric: `can_key_call_model` falls back to the key's `access_group_ids`, but `can_team_access_model` only looks at `team.models` + `team.access_group_ids`. A key granted a model via its own access group on a model-restricted team is silently denied at the team gate. Wrap `can_team_access_model` in try/except in `common_checks`: on `team_model_access_denied`, consult a new `_key_access_group_grants_model` helper that expands `valid_token.access_group_ids` via the existing `_get_models_from_access_groups` and checks via `_can_object_call_model`. Re-raise if the key's access groups don't grant the model. Any other exception propagates unchanged. Effect: request allowed if `team allows X` OR `key's access group grants X`, making the two gates symmetric. Test: add three unit tests for `_key_access_group_grants_model` covering: group covers model, key has no groups, group resolves but does not cover model. --- litellm/proxy/auth/auth_checks.py | 66 ++++++++++++++---- tests/proxy_unit_tests/test_auth_checks.py | 81 ++++++++++++++++++++++ 2 files changed, 133 insertions(+), 14 deletions(-) diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index 2c8299e77a9..e959867091e 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -494,23 +494,27 @@ async def common_checks( # noqa: PLR0915 f"Team={team_object.team_id} is blocked. Update via `/team/unblock` if you're an admin." ) - # 2. If team can call model + # 2. If team can call model (or key's access_group_ids grant it) if _model and team_object: with tracer.trace("litellm.proxy.auth.common_checks.can_team_access_model"): - if not await can_team_access_model( - model=_model, - team_object=team_object, - llm_router=llm_router, - team_model_aliases=( - valid_token.team_model_aliases if valid_token else None - ), - ): - raise ProxyException( - message=f"Team not allowed to access model. Team={team_object.team_id}, Model={_model}. Allowed team models = {team_object.models}", - type=ProxyErrorTypes.team_model_access_denied, - param="model", - code=status.HTTP_401_UNAUTHORIZED, + try: + await can_team_access_model( + model=_model, + team_object=team_object, + llm_router=llm_router, + team_model_aliases=( + valid_token.team_model_aliases if valid_token else None + ), ) + except ProxyException as team_denial: + if team_denial.type != ProxyErrorTypes.team_model_access_denied: + raise + if not await _key_access_group_grants_model( + model=_model, + valid_token=valid_token, + llm_router=llm_router, + ): + raise # 2.2. If team member has per-member model scope, enforce it if _model and team_object and valid_token and valid_token.user_id: @@ -2863,6 +2867,40 @@ async def can_team_access_model( raise +async def _key_access_group_grants_model( + model: Union[str, List[str]], + valid_token: Optional[UserAPIKeyAuth], + llm_router: Optional[Router], +) -> bool: + """ + Returns True if the key's `access_group_ids` expand to models that grant + access to `model`. Used to let a key's access group override a team's + model restriction in `common_checks`. + """ + if valid_token is None: + return False + key_access_group_ids = valid_token.access_group_ids or [] + if not key_access_group_ids: + return False + models_from_groups = await _get_models_from_access_groups( + access_group_ids=key_access_group_ids, + ) + if not models_from_groups: + return False + try: + _can_object_call_model( + model=model, + llm_router=llm_router, + models=models_from_groups, + team_model_aliases=valid_token.team_model_aliases, + team_id=valid_token.team_id, + object_type="key", + ) + return True + except ProxyException: + return False + + def can_project_access_model( model: Union[str, List[str]], project_object: LiteLLM_ProjectTableCachedObj, diff --git a/tests/proxy_unit_tests/test_auth_checks.py b/tests/proxy_unit_tests/test_auth_checks.py index 86cd5c0c413..6a404712c11 100644 --- a/tests/proxy_unit_tests/test_auth_checks.py +++ b/tests/proxy_unit_tests/test_auth_checks.py @@ -1146,3 +1146,84 @@ async def test_can_key_call_model_via_access_group_ids(): valid_token=user_api_key_object, llm_router=router, ) + + +# --------------------------------------------------------------------------- +# _key_access_group_grants_model (key access group overriding team restriction) +# --------------------------------------------------------------------------- + + +@pytest.mark.asyncio +async def test_key_access_group_grants_model_when_group_covers_model(): + """Key's access_group_ids expand to a set that includes the requested model.""" + from unittest.mock import AsyncMock, patch + + from litellm.proxy.auth.auth_checks import _key_access_group_grants_model + + valid_token = UserAPIKeyAuth( + token="test-token", + models=[], + access_group_ids=["ryan-access-group"], + ) + + with patch( + "litellm.proxy.auth.auth_checks._get_models_from_access_groups", + new_callable=AsyncMock, + return_value=["claude-haiku-4-5"], + ): + assert ( + await _key_access_group_grants_model( + model="claude-haiku-4-5", + valid_token=valid_token, + llm_router=None, + ) + is True + ) + + +@pytest.mark.asyncio +async def test_key_access_group_grants_model_when_key_has_no_groups(): + """Key with no access_group_ids cannot override team denial.""" + from litellm.proxy.auth.auth_checks import _key_access_group_grants_model + + valid_token = UserAPIKeyAuth( + token="test-token", + models=[], + access_group_ids=[], + ) + assert ( + await _key_access_group_grants_model( + model="claude-haiku-4-5", + valid_token=valid_token, + llm_router=None, + ) + is False + ) + + +@pytest.mark.asyncio +async def test_key_access_group_grants_model_when_group_does_not_cover_model(): + """Key's access_group_ids expand to models that do not include the request.""" + from unittest.mock import AsyncMock, patch + + from litellm.proxy.auth.auth_checks import _key_access_group_grants_model + + valid_token = UserAPIKeyAuth( + token="test-token", + models=[], + access_group_ids=["other-group"], + ) + + with patch( + "litellm.proxy.auth.auth_checks._get_models_from_access_groups", + new_callable=AsyncMock, + return_value=["gpt-4o-mini"], + ): + assert ( + await _key_access_group_grants_model( + model="claude-haiku-4-5", + valid_token=valid_token, + llm_router=None, + ) + is False + ) From e03bb3437feb03d3f91a9cdbaa45b942e1351a6b Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 24 Apr 2026 20:10:30 +0530 Subject: [PATCH 013/196] Fix test --- .../test_vertex_ai_gemini_transformation.py | 77 ++++++++++++------- 1 file changed, 50 insertions(+), 27 deletions(-) diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py index 6952be7db3c..d09c1afcdb5 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_ai_gemini_transformation.py @@ -86,8 +86,10 @@ def test_check_if_part_exists_in_parts_camel_case_snake_case(): assert check_if_part_exists_in_parts(parts_mixed, part_mixed_casing) -def test_cached_content_omits_system_instruction_tools_toolconfig(): - """Regression: #26014 / #17304 — cachedContent must not ship with tools/system/toolConfig.""" +def test_cached_content_respects_modify_params_for_cache_incompatible_fields(): + """Regression: cachedContent drops system/tools/toolConfig only when modify_params=True.""" + import litellm + cache_name = "projects/p/locations/us-central1/cachedContents/abc123" messages = [ {"role": "system", "content": "You are helpful"}, @@ -104,33 +106,54 @@ def test_cached_content_omits_system_instruction_tools_toolconfig(): "tool_choice": {"functionCallingConfig": {"mode": "AUTO"}}, } - result = _transform_request_body( - messages=list(messages), - model="gemini-2.5-pro", - optional_params=dict(optional_params), - custom_llm_provider="vertex_ai", - litellm_params={}, - cached_content=cache_name, - ) + original_modify_params = litellm.modify_params + try: + # With modify_params=False (default), keep fields even with cachedContent. + litellm.modify_params = False + result = _transform_request_body( + messages=list(messages), + model="gemini-2.5-pro", + optional_params=dict(optional_params), + custom_llm_provider="vertex_ai", + litellm_params={}, + cached_content=cache_name, + ) + assert result.get("cachedContent") == cache_name + assert "system_instruction" in result + assert "tools" in result + assert "toolConfig" in result + assert "contents" in result - assert result.get("cachedContent") == cache_name - assert "system_instruction" not in result - assert "tools" not in result - assert "toolConfig" not in result - assert "contents" in result + # With modify_params=True, drop cache-incompatible fields. + litellm.modify_params = True + result_modify_true = _transform_request_body( + messages=list(messages), + model="gemini-2.5-pro", + optional_params=dict(optional_params), + custom_llm_provider="vertex_ai", + litellm_params={}, + cached_content=cache_name, + ) + assert result_modify_true.get("cachedContent") == cache_name + assert "system_instruction" not in result_modify_true + assert "tools" not in result_modify_true + assert "toolConfig" not in result_modify_true + assert "contents" in result_modify_true - # Without cache, conflicting fields are included as before - result_no_cache = _transform_request_body( - messages=list(messages), - model="gemini-2.5-pro", - optional_params=dict(optional_params), - custom_llm_provider="vertex_ai", - litellm_params={}, - cached_content=None, - ) - assert "system_instruction" in result_no_cache - assert "tools" in result_no_cache - assert "toolConfig" in result_no_cache + # Without cache, fields are always included. + result_no_cache = _transform_request_body( + messages=list(messages), + model="gemini-2.5-pro", + optional_params=dict(optional_params), + custom_llm_provider="vertex_ai", + litellm_params={}, + cached_content=None, + ) + assert "system_instruction" in result_no_cache + assert "tools" in result_no_cache + assert "toolConfig" in result_no_cache + finally: + litellm.modify_params = original_modify_params # Tests for issue #14556: Labels field provider-aware filtering From b9e46cbdb75675decb93e5ddd340f979145d7388 Mon Sep 17 00:00:00 2001 From: Darien Kindlund Date: Fri, 24 Apr 2026 11:48:39 -0400 Subject: [PATCH 014/196] fix(adapters,vertex): pass output_config through to backends that accept it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolves the silent strip of Anthropic Structured Outputs across the Vertex AI Claude transformation paths and the Anthropic-adapter re-merge. Consolidates and supersedes four stalled community PRs addressing overlapping aspects of the same root bug: - #23475 (Vertex AI Claude blanket-strip removal) - #23396 (Vertex AI Claude conditional passthrough) - #23706 (Anthropic adapter exclude output_config from non-Anthropic backends) - #22727 (Anthropic adapter strip output_config for non-Anthropic backends) Closes / addresses: #23380 (Vertex AI Claude output_config drop), related: #26423, #25079, #24549, #25971, #25957, #26163, #24856. What was broken --------------- * Vertex AI Claude paths called ``data.pop("output_config")`` and ``data.pop("output_format")`` unconditionally even when Vertex accepted those fields. Callers asking for Structured Outputs got a 200 with prose and never knew the schema constraints had been silently dropped (often masked for months by permissive fallback parsers). * The ``/v1/messages`` -> ``/chat/completions`` adapter (``LiteLLMMessagesToCompletionTransformationHandler``) re-merged the raw Anthropic-shaped ``output_config`` into ``completion_kwargs`` AFTER the translator already mapped its meaningful parts to ``response_format`` / ``reasoning_effort``. Non-Anthropic backends (Azure OpenAI, Fireworks, Bedrock Nova, etc.) then 400'd with "Extra inputs are not permitted". Approach -------- Vertex AI Claude (chat-completion + experimental_pass_through paths): Replace the unconditional pop with a sanitizer ``_sanitize_vertex_anthropic_output_params`` that strips only the Vertex-unsupported keys (today: ``effort``) from ``output_config`` while forwarding ``format`` and the legacy top-level ``output_format``. Defensive: non-dict ``output_config`` values are dropped to avoid sending malformed payloads downstream. Greptile P1 from PR #23396 addressed: when ``output_config`` carries both ``format`` and ``effort``, the prior conditional pass-through forwarded ``effort`` and reproduced the 400. The new helper filters per-key. Anthropic ``/v1/messages`` adapter: Add ``output_config`` to a named module-level constant ``ANTHROPIC_ONLY_REQUEST_KEYS`` and wire it into ``excluded_keys`` so the post-translation re-merge skips re-adding the raw key. This fixes the 400 on non-Anthropic backends and avoids the conflicting duplicate (``response_format`` + raw ``output_config``) on Anthropic-family backends. Greptile P2 from PR #23706 addressed: the constant gives reviewers one grep target instead of an inline literal that silently grows. Greptile P2 from PR #22727 addressed: ``extra_kwargs or {}`` is replaced with explicit ``is None`` checks so empty-dict callers no longer skip the fallback path. Tests ----- * tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/ test_vertex_ai_partner_models_anthropic_transformation.py: - 5 new/updated cases plus a direct unit test for ``_sanitize_vertex_anthropic_output_params``. - Updated ``test_vertex_ai_claude_sonnet_4_5_structured_output_fix`` so its mock-injected ``output_format`` is asserted to FLOW THROUGH (the original test asserted the now-buggy strip behavior). * tests/test_litellm/llms/anthropic/experimental_pass_through/ adapters/test_handler_output_config_passthrough.py (new): - Constant export sanity, output_config strip with ``effort`` only, output_config strip with ``format`` only, regression guard that unrelated extras still flow, explicit-empty-dict path, and the ``extra_kwargs=None`` no-crash path. Test-quality fixes incorporated from Greptile review on the superseded PRs: * No ``inspect.getsource`` source-text assertions (PR #24114 / #23475). * ``sys.path`` insertion is anchored to ``__file__`` (PR #23706). * Assertion messages are positional, not tuple (PR #24114-class bug). * No ``or {}`` masking explicit empty dicts in helper signatures (PR #22727). Verified locally: 26/26 pass with this commit. The new tests fail (or fail to import) on ``main`` without it. Out of scope ------------ * The ``max_tokens`` capping logic from PR #22727 — independent concern, deserves its own PR with a focused test plan. * Architectural rework of the ``excluded_keys`` mechanism (Greptile P2 on PR #23706 noted point-fix growth). The named constant gives maintainers a clear place to extend; a registry-based approach would be a follow-up. Co-Authored-By: netbrah Co-Authored-By: s-zx Co-Authored-By: invoicepulse Co-Authored-By: cfdude Co-Authored-By: Claude Opus 4.7 (1M context) --- .../adapters/handler.py | 36 ++- .../transformation.py | 13 +- .../anthropic/transformation.py | 50 +++- .../test_handler_output_config_passthrough.py | 167 ++++++++++++ ...partner_models_anthropic_transformation.py | 243 +++++++++++++----- 5 files changed, 431 insertions(+), 78 deletions(-) create mode 100644 tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index d16f5afb45c..10455825e41 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -27,6 +27,16 @@ from litellm.utils import get_model_info if TYPE_CHECKING: pass + +# Anthropic-only fields that the translator above already maps into the +# OpenAI-format completion_kwargs (output_config → reasoning_effort / +# response_format, etc.). They must be filtered out of the raw +# extra_kwargs re-merge below or non-Anthropic backends reject the call +# with 400 "Extra inputs are not permitted". Add new entries here when +# extending AnthropicMessagesRequestOptionalParams with another Anthropic- +# specific key. +ANTHROPIC_ONLY_REQUEST_KEYS: frozenset[str] = frozenset({"output_config"}) + ######################################################## # init adapter ANTHROPIC_ADAPTER = AnthropicAdapter() @@ -202,8 +212,12 @@ class LiteLLMMessagesToCompletionTransformationHandler: request_data["output_format"] = output_format # Extract output_config from extra_kwargs so the translator can use it - # (e.g. output_config.effort for adaptive thinking → reasoning_effort) - extra_kwargs = extra_kwargs or {} + # (e.g. output_config.effort for adaptive thinking → reasoning_effort, + # output_config.format → response_format for structured outputs). + # Use explicit None check rather than `or {}` so an explicit empty dict + # caller-passed argument is preserved (matters for tests that drive + # the fallback inference path). + extra_kwargs = extra_kwargs if extra_kwargs is not None else {} if "output_config" in extra_kwargs: request_data["output_config"] = extra_kwargs["output_config"] @@ -225,8 +239,22 @@ class LiteLLMMessagesToCompletionTransformationHandler: "include_usage": True, } - excluded_keys = {"anthropic_messages"} - extra_kwargs = extra_kwargs or {} + # Keys that must NOT be forwarded as raw extras into the OpenAI-format + # ``completion_kwargs`` after translation. The translator above has + # already consumed the meaningful parts of these inputs (e.g. + # ``output_config.format`` → ``response_format``, ``output_config.effort`` + # → ``reasoning_effort`` for non-Claude targets). Re-adding the raw + # Anthropic-shaped key here causes 400 "Extra inputs are not permitted" + # on non-Anthropic backends (Azure OpenAI, Fireworks, Bedrock Nova, + # etc.) and is silently lossy on Anthropic-family targets, which would + # see the translated key ``response_format`` AND a duplicate, conflicting + # ``output_config``. + # + # Maintainability: when adding a new Anthropic-only request param to + # ``AnthropicMessagesRequestOptionalParams``, also extend + # ``ANTHROPIC_ONLY_REQUEST_KEYS`` here so it doesn't silently leak. + excluded_keys = ANTHROPIC_ONLY_REQUEST_KEYS | {"anthropic_messages"} + extra_kwargs = extra_kwargs if extra_kwargs is not None else {} for key, value in extra_kwargs.items(): if ( key == "litellm_logging_obj" diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py index 5c3bbf61ee2..9080ac02330 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py @@ -13,6 +13,7 @@ from litellm.types.llms.vertex_ai import VertexPartnerProvider from litellm.types.router import GenericLiteLLMParams from ....vertex_llm_base import VertexBase +from ..transformation import _sanitize_vertex_anthropic_output_params class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, VertexBase): @@ -158,12 +159,10 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert "model", None ) # do not pass model in request body to vertex ai - anthropic_messages_request.pop( - "output_format", None - ) # do not pass output_format in request body to vertex ai - vertex ai does not support output_format as yet - - anthropic_messages_request.pop( - "output_config", None - ) # do not pass output_config in request body to vertex ai - vertex ai does not support output_config + # Vertex AI Claude accepts ``output_config.format`` (structured outputs) + # and ``output_format``, but rejects ``output_config.effort`` with 400 + # "Extra inputs are not permitted". Sanitize in place so the supported + # bits flow through. + _sanitize_vertex_anthropic_output_params(anthropic_messages_request) return anthropic_messages_request diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index 504914c4796..a9dea6646ff 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -11,6 +11,45 @@ from litellm.types.utils import ModelResponse from ....anthropic.chat.transformation import AnthropicConfig +# Keys inside ``output_config`` that Vertex AI Claude does not accept. +# Today only ``effort`` triggers "Extra inputs are not permitted"; add new +# entries here as Vertex parity drifts. Keep this list narrow — anything +# Vertex DOES accept (e.g. ``format`` for structured outputs) must be +# preserved so callers can rely on Anthropic-native features. +_VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS = frozenset({"effort"}) + + +def _sanitize_vertex_anthropic_output_params(data: dict) -> None: + """ + Strip Vertex-unsupported keys from ``output_config`` / ``output_format`` + in-place; forward whatever remains. + + Behavior: + * ``output_config`` containing only unsupported keys (e.g. ``effort`` + alone) is removed entirely so the request body has no empty dict. + * ``output_config`` containing a mix of supported + unsupported keys has + the unsupported subset filtered out and the rest forwarded. + * ``output_config`` that is supported in full passes through unchanged. + * ``output_format`` is forwarded as-is (Vertex AI Claude accepts it). + * Non-dict values for ``output_config`` are dropped to avoid sending + malformed payloads downstream. + """ + output_config = data.get("output_config") + if output_config is None: + return + if not isinstance(output_config, dict): + data.pop("output_config", None) + return + sanitized = { + k: v + for k, v in output_config.items() + if k not in _VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS + } + if sanitized: + data["output_config"] = sanitized + else: + data.pop("output_config", None) + class VertexAIError(Exception): def __init__(self, status_code, message): @@ -105,11 +144,12 @@ class VertexAIAnthropicConfig(AnthropicConfig): data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter - # VertexAI doesn't support output_format parameter, remove it if present - data.pop("output_format", None) - - # VertexAI doesn't support output_config parameter, remove it if present - data.pop("output_config", None) + # Vertex AI Claude accepts ``output_config.format`` (structured outputs / + # JSON Schema) but NOT ``output_config.effort`` — sending ``effort`` to + # Vertex returns 400 "Extra inputs are not permitted". Sanitize in place: + # forward the structured-output bits, drop the unsupported keys. + # Same treatment for the legacy top-level ``output_format`` field. + _sanitize_vertex_anthropic_output_params(data) tools = optional_params.get("tools") tool_search_used = self.is_tool_search_used(tools) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py new file mode 100644 index 00000000000..50e5fd884e9 --- /dev/null +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py @@ -0,0 +1,167 @@ +""" +Regression tests for output_config passthrough through the Anthropic +``/v1/messages`` → ``/chat/completions`` adapter. + +Background — what was broken: +* When a client sent ``output_config`` to ``/v1/messages`` and the request + was routed to a non-Anthropic backend (Azure OpenAI, Fireworks, Bedrock + Nova, etc.), the adapter forwarded the raw Anthropic-shaped ``output_config`` + field as-is into the OpenAI-format ``completion_kwargs``. The non-Anthropic + backend then rejected the request with 400 "Extra inputs are not permitted". +* The translator above the re-merge already extracts the meaningful parts of + ``output_config`` (``format`` → ``response_format``, ``effort`` → + ``reasoning_effort`` for non-Claude targets), so re-adding the raw key was + always either redundant (Anthropic-family) or harmful (non-Anthropic). + +Tests cover (consolidating PRs #23706 and #22727): +1. ``output_config`` is excluded from the post-translation re-merge. +2. ``ANTHROPIC_ONLY_REQUEST_KEYS`` constant is exported and contains + ``output_config`` so future maintainers know where to extend it. +3. The translator-extracted fields (``response_format`` / ``reasoning_effort``) + are still present after the strip — the strip removes only the raw + Anthropic-shaped duplicate. +4. Helper-level coverage for empty ``extra_kwargs`` (PR #22727 Greptile P2 — + the original ``or {}`` pattern silently substituted a default and prevented + the fallback inference path from being exercised). +""" + +import os +import sys +from unittest.mock import MagicMock, patch + +import pytest + +# Anchor sys.path to this file's location — not the working-directory-relative +# pattern Greptile flagged on PR #23706. Resolves correctly regardless of +# where pytest is invoked from. +sys.path.insert( + 0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../..")) +) + +from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( + ANTHROPIC_ONLY_REQUEST_KEYS, + LiteLLMMessagesToCompletionTransformationHandler, +) + +MESSAGES = [{"role": "user", "content": "hello"}] + + +def _call_prepare(extra_kwargs, model="gpt-4o", **overrides): + """ + Drive ``_prepare_completion_kwargs`` with the minimum scaffolding needed. + + Uses an explicit-None check on ``extra_kwargs`` so callers can test the + falsy-empty-dict path. The fallback ``or {}`` pattern PR #22727 used here + masked the no-extra-kwargs case from ever exercising the test's intent. + """ + return LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs( + max_tokens=overrides.get("max_tokens", 1024), + messages=overrides.get("messages", MESSAGES), + model=model, + metadata=None, + stop_sequences=None, + stream=False, + system=None, + temperature=None, + thinking=None, + tool_choice=None, + tools=None, + top_k=None, + top_p=None, + output_format=None, + extra_kwargs=extra_kwargs, + ) + + +class TestAnthropicOnlyRequestKeysExport: + """The exclusion list must be a public, named constant for maintainability — + Greptile P2 on PR #23706: ``excluded_keys`` was silently growing as a + point-fix pattern. A named module-level constant gives reviewers a single + grep target when extending Anthropic-only fields.""" + + def test_constant_exposed(self): + assert isinstance(ANTHROPIC_ONLY_REQUEST_KEYS, frozenset) + + def test_contains_output_config(self): + assert "output_config" in ANTHROPIC_ONLY_REQUEST_KEYS + + +class TestOutputConfigStrippedFromCompletionKwargs: + """``output_config`` must not survive the post-translation re-merge into + ``completion_kwargs`` regardless of the target provider — the translator + has already consumed its meaningful parts.""" + + def test_output_config_with_effort_is_stripped(self): + extra_kwargs = { + "custom_llm_provider": "azure", + "output_config": {"effort": "high"}, + } + + result = _call_prepare(extra_kwargs=extra_kwargs) + + # Returns (completion_kwargs, original_messages, ...) — first element + # is the dict we care about. + completion_kwargs = result[0] if isinstance(result, tuple) else result + assert "output_config" not in completion_kwargs, ( + "Raw output_config must not be forwarded — non-Anthropic backends " + "reject it with 400 'Extra inputs are not permitted'" + ) + + def test_output_config_with_format_is_stripped_format_already_translated(self): + """Even when ``output_config`` carries useful structured-output info, + the raw key must be excluded — the translator above has already mapped + ``output_config.format`` to ``response_format`` (the OpenAI-shaped key + the downstream backend understands).""" + extra_kwargs = { + "custom_llm_provider": "azure", + "output_config": { + "format": {"type": "json_schema", "schema": {"type": "object"}} + }, + } + + result = _call_prepare(extra_kwargs=extra_kwargs) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + assert "output_config" not in completion_kwargs + + def test_other_extra_kwargs_still_passed_through(self): + """Regression guard: the strip must be narrow. Unrelated fields like + ``api_key`` / ``timeout`` continue to flow through.""" + extra_kwargs = { + "custom_llm_provider": "azure", + "output_config": {"effort": "high"}, + "timeout": 30, + "user": "end-user-123", + } + + result = _call_prepare(extra_kwargs=extra_kwargs) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + assert "output_config" not in completion_kwargs + assert completion_kwargs.get("timeout") == 30 + assert completion_kwargs.get("user") == "end-user-123" + + +class TestEmptyExtraKwargsPath: + """Greptile P2 on PR #22727: ``extra_kwargs or {default}`` substitutes a + default for an explicitly-passed empty dict, hiding the no-extra-kwargs + path. The new explicit-None pattern lets ``extra_kwargs={}`` reach the + code under test as written.""" + + def test_explicit_empty_dict_does_not_substitute_default(self): + # Explicit empty dict must be honored — not silently replaced with a + # default that adds back a custom_llm_provider this test wants absent. + result = _call_prepare(extra_kwargs={}) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + # No output_config because nothing supplied it. + assert "output_config" not in completion_kwargs + + def test_none_extra_kwargs_handled_safely(self): + """The signature documents ``extra_kwargs: Optional[Dict] = None``; + passing None must not crash with KeyError or AttributeError.""" + result = _call_prepare(extra_kwargs=None) + # Just exercising the path; assert no exception and we get back a + # dict-like result. + completion_kwargs = result[0] if isinstance(result, tuple) else result + assert isinstance(completion_kwargs, dict) diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 79fc66a74b8..376c48d9e95 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -203,12 +203,15 @@ def test_vertex_ai_anthropic_structured_output_header_not_added(): def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): """ Test fix for issue #18625: Claude Sonnet 4.5 on VertexAI should use tool-based - structured outputs instead of output_format parameter. + structured outputs when ``response_format`` is supplied via the OpenAI-compat + interface (``map_openai_params``). This test verifies that: - 1. Claude Sonnet 4.5 uses tool-based structured outputs on VertexAI - 2. output_format parameter is removed from the final request - 3. The fix prevents "Extra inputs are not permitted" error + 1. Claude Sonnet 4.5 uses tool-based structured outputs when ``response_format`` + is given to the OpenAI-compat path (the path that triggered #18625). + 2. ``output_format`` is forwarded to Vertex AI when present — Vertex now + accepts the field; the prior blanket-strip behavior was the silent drop + of Anthropic Structured Outputs that this PR fixes. """ config = VertexAIAnthropicConfig() @@ -294,11 +297,15 @@ def test_vertex_ai_claude_sonnet_4_5_structured_output_fix(): headers={}, ) - # Verify that output_format was removed (fixes the "Extra inputs are not permitted" error) + # output_format is now forwarded to Vertex (Vertex parity has shifted — + # it accepts the field and uses it to enforce the JSON schema). The + # prior behavior silently stripped it, hiding Structured Outputs from + # callers who explicitly requested them. + assert "output_format" in final_data + assert final_data["output_format"]["type"] == "json_schema" assert ( - "output_format" not in final_data - ), "output_format should be removed for VertexAI" - assert "model" not in final_data, "model should be removed for VertexAI" + "model" not in final_data + ), "model is still stripped (Vertex routes by URL)" assert "tools" in final_data, "tools should still be present" assert "tool_choice" in final_data, "tool_choice should still be present" @@ -491,28 +498,22 @@ def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_hea ), "Header should be removed if no supported values remain" -def test_vertex_ai_anthropic_output_config_dropped(): +def test_vertex_ai_anthropic_output_config_effort_only_dropped(): """ - Test that output_config parameter is dropped from Vertex AI Anthropic requests. - - Vertex AI does not support the output_config parameter (used for effort settings - in Anthropic API). This test ensures it's properly removed to prevent - "Extra inputs are not permitted" errors. + ``output_config`` containing only ``effort`` (an Anthropic-only key Vertex + rejects with "Extra inputs are not permitted") is dropped entirely so the + request body has no empty dict. """ config = VertexAIAnthropicConfig() messages = [{"role": "user", "content": "What is 2+2?"}] - headers = {} + headers: dict = {} - # Simulate optional_params with output_config that would be passed in optional_params = { "max_tokens": 1024, - "output_config": { - "effort": "high" # This is Anthropic-specific and not supported by Vertex AI - }, + "output_config": {"effort": "high"}, } - # Call transform_request which should drop output_config result = config.transform_request( model="claude-3-5-sonnet-20241022", messages=messages, @@ -521,54 +522,144 @@ def test_vertex_ai_anthropic_output_config_dropped(): headers=headers, ) - # Verify output_config was removed assert ( "output_config" not in result - ), "output_config should be dropped from Vertex AI Anthropic requests" - - # Verify other parameters are preserved - assert result["max_tokens"] == 1024, "max_tokens should be preserved" - assert "messages" in result, "messages should be present" + ), "output_config containing only effort must be dropped" + assert result["max_tokens"] == 1024 + assert "messages" in result -def test_vertex_ai_anthropic_output_format_and_output_config_both_dropped(): +def test_vertex_ai_anthropic_output_config_format_passes_through(): """ - Test that both output_format and output_config are dropped from Vertex AI requests. - - This ensures that even if both parameters somehow make it to the transform_request, - they are properly cleaned up before sending to Vertex AI. + ``output_config`` containing structured-output ``format`` is FORWARDED to + Vertex AI Claude — Vertex now accepts it and uses it for JSON Schema + enforcement. Previously the entire field was being silently stripped, so + Anthropic Structured Outputs never engaged on Vertex even when callers + requested it. """ config = VertexAIAnthropicConfig() + messages = [{"role": "user", "content": "Return a person object."}] + output_config = { + "format": { + "type": "json_schema", + "schema": { + "type": "object", + "additionalProperties": False, + "properties": { + "name": {"type": "string"}, + "age": {"type": "integer"}, + }, + }, + } + } + optional_params = {"max_tokens": 1024, "output_config": output_config} + + result = config.transform_request( + model="claude-3-5-sonnet-20241022", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert result["output_config"] == output_config + + +def test_vertex_ai_anthropic_output_config_format_plus_effort_strips_only_effort(): + """ + Greptile P1 on PR #23396: when ``output_config`` contains BOTH ``format`` + and ``effort``, the prior conditional-passthrough logic forwarded the + full dict including the unsupported ``effort`` key, reproducing the + 400 error the fix was meant to resolve. Only ``effort`` (and any future + Vertex-unsupported keys) should be filtered; ``format`` must survive. + """ + config = VertexAIAnthropicConfig() + messages = [{"role": "user", "content": "Return a person object."}] + + output_config = { + "format": { + "type": "json_schema", + "schema": { + "type": "object", + "additionalProperties": False, + "properties": {"name": {"type": "string"}}, + }, + }, + "effort": "high", + } + optional_params = {"max_tokens": 1024, "output_config": output_config} + + result = config.transform_request( + model="claude-3-5-sonnet-20241022", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "output_config" in result + assert ( + "effort" not in result["output_config"] + ), "effort must be stripped — Vertex returns 400 on unknown keys" + assert result["output_config"]["format"] == output_config["format"] + + +def test_vertex_ai_anthropic_output_config_non_dict_dropped(): + """Defensive: if ``output_config`` is somehow not a dict, drop it rather + than forwarding malformed data downstream.""" + config = VertexAIAnthropicConfig() + messages = [{"role": "user", "content": "hi"}] + optional_params = {"max_tokens": 64, "output_config": "not-a-dict"} + + result = config.transform_request( + model="claude-3-5-sonnet-20241022", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "output_config" not in result + + +def test_vertex_ai_anthropic_output_format_preserved_output_config_effort_dropped(): + """ + When the request carries both ``output_format`` (top-level structured + outputs) AND an ``output_config`` whose only useful key for Vertex is + ``effort``: ``output_format`` must be forwarded (Vertex accepts it), + while ``output_config`` is dropped because Vertex returns 400 on + ``effort``. This replaces the old "drop both" behavior, which was the + silent strip the bug report flagged. + """ + config = VertexAIAnthropicConfig() messages = [{"role": "user", "content": "Extract structured data"}] - headers = {} + + output_format = { + "type": "json_schema", + "json_schema": { + "name": "data", + "schema": { + "type": "object", + "properties": {"result": {"type": "string"}}, + }, + }, + } optional_params = { "max_tokens": 2048, - "output_format": { - "type": "json_schema", - "json_schema": { - "name": "data", - "schema": { - "type": "object", - "properties": {"result": {"type": "string"}}, - }, - }, - }, + "output_format": output_format, "output_config": {"effort": "high"}, } - # Simulate parent class creating test_data with both parameters - # (as if the parent transform_request added them) test_data = { "model": "claude-3-5-sonnet-20241022", "messages": messages, "max_tokens": 2048, - "output_format": optional_params["output_format"], - "output_config": optional_params["output_config"], + "output_format": output_format, + "output_config": {"effort": "high"}, } - # Mock the parent transform_request to return data with both parameters original_transform = config.__class__.__bases__[0].transform_request def mock_transform_request( @@ -584,22 +675,50 @@ def test_vertex_ai_anthropic_output_format_and_output_config_both_dropped(): messages=messages, optional_params=optional_params, litellm_params={}, - headers=headers, + headers={}, ) - # Verify both were removed - assert ( - "output_format" not in result - ), "output_format should be dropped from Vertex AI requests" - assert ( - "output_config" not in result - ), "output_config should be dropped from Vertex AI requests" - - # Verify essential params are preserved - assert result["max_tokens"] == 2048, "max_tokens should be preserved" - assert "messages" in result, "messages should be present" - assert "model" not in result, "model should also be dropped for Vertex AI" - + # output_format flows through unchanged — Vertex AI Claude accepts it. + assert result["output_format"] == output_format + # output_config containing only ``effort`` is dropped to avoid the + # 400 "Extra inputs are not permitted" the silent strip used to mask. + assert "output_config" not in result + assert result["max_tokens"] == 2048 + assert "model" not in result, "model is still stripped (Vertex routes by URL)" finally: - # Restore original method config.__class__.__bases__[0].transform_request = original_transform + + +def test_sanitize_vertex_anthropic_output_params_unit(): + """Direct unit coverage for the helper itself (used by both Vertex + Anthropic transformation paths). Mirrors the integration assertions + above without going through the full ``transform_request`` stack.""" + from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import ( + _sanitize_vertex_anthropic_output_params, + ) + + # No-op when output_config absent. + data: dict = {"max_tokens": 8} + _sanitize_vertex_anthropic_output_params(data) + assert data == {"max_tokens": 8} + + # Effort-only → dropped entirely. + data = {"output_config": {"effort": "high"}} + _sanitize_vertex_anthropic_output_params(data) + assert "output_config" not in data + + # Format-only → preserved unchanged. + fmt = {"format": {"type": "json_schema", "schema": {"type": "object"}}} + data = {"output_config": dict(fmt)} + _sanitize_vertex_anthropic_output_params(data) + assert data["output_config"] == fmt + + # Mixed → effort filtered, format kept. + data = {"output_config": {"format": fmt["format"], "effort": "high"}} + _sanitize_vertex_anthropic_output_params(data) + assert data["output_config"] == fmt + + # Non-dict → dropped defensively. + data = {"output_config": "garbage"} + _sanitize_vertex_anthropic_output_params(data) + assert "output_config" not in data From 79517bc6282c43d0d844b6d3f7b2597e3c2735bd Mon Sep 17 00:00:00 2001 From: Darien Kindlund Date: Fri, 24 Apr 2026 12:03:55 -0400 Subject: [PATCH 015/196] fix: address Greptile review feedback on PR #26439 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three concerns raised by bot reviewers, all addressed: 1. CodeQL cyclic-import warning ``experimental_pass_through/transformation.py`` imported from the parent ``..transformation`` module, which CodeQL flagged as a potential cycle. Extracted the helper into a new leaf module ``vertex_ai_partner_models/anthropic/output_params_utils.py`` that has no heavy imports of its own. Both transformation files now import from it cleanly. Renamed the helper from the underscore- prefixed ``_sanitize_vertex_anthropic_output_params`` to the public ``sanitize_vertex_anthropic_output_params`` since it is now shared across modules. 2. Greptile P2: redundant ``None`` guard on ``extra_kwargs`` ``handler.py`` had two ``extra_kwargs = extra_kwargs if ... else {}`` coercions; the second was a no-op because line 220 already coerced. Removed the second one and added a NOTE comment so future readers understand ``extra_kwargs`` is guaranteed non-None at the point of use. 3. Greptile P2: misleading "already translated" docstring The docstring claimed the translator above mapped ``output_config.format`` to ``response_format``, but Greptile correctly traced the code and found that only the legacy top-level ``output_format`` was being translated — ``output_config.format`` was being silently dropped on the adapter path. Two-part fix: a. Code: extended ``_translate_output_format_to_openai`` to accept both shapes (top-level ``output_format`` AND ``output_config.format`` sub-key). Top-level still takes precedence when both are supplied. This means callers using the newer Anthropic Structured Outputs API now have their schema properly forwarded to non-Anthropic backends as ``response_format``. b. Tests: rewrote the misleading docstring to describe what actually happens, plus added two new tests: * ``test_output_format_top_level_still_translates`` — regression guard for the legacy path * ``test_output_format_takes_precedence_over_output_config_format`` — documents the precedence rule explicitly Tests: 28/28 pass (was 26/26 before; +2 for the new translation behavior + precedence). All run in ~0.5s, no real network calls. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../adapters/handler.py | 3 +- .../adapters/transformation.py | 23 +++-- .../transformation.py | 4 +- .../anthropic/output_params_utils.py | 50 +++++++++++ .../anthropic/transformation.py | 42 +-------- .../test_handler_output_config_passthrough.py | 85 ++++++++++++++++--- ...partner_models_anthropic_transformation.py | 14 +-- 7 files changed, 156 insertions(+), 65 deletions(-) create mode 100644 litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index 10455825e41..ac55aac8062 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -254,7 +254,8 @@ class LiteLLMMessagesToCompletionTransformationHandler: # ``AnthropicMessagesRequestOptionalParams``, also extend # ``ANTHROPIC_ONLY_REQUEST_KEYS`` here so it doesn't silently leak. excluded_keys = ANTHROPIC_ONLY_REQUEST_KEYS | {"anthropic_messages"} - extra_kwargs = extra_kwargs if extra_kwargs is not None else {} + # NOTE: extra_kwargs was already coerced from None to {} at the top of + # this method (line ~220). It is guaranteed to be a dict here. for key, value in extra_kwargs.items(): if ( key == "litellm_logging_obj" diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index 20fa4f125de..f7bc67ccd3a 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -664,7 +664,7 @@ class LiteLLMAnthropicMessagesAdapter: @staticmethod def translate_anthropic_thinking_to_reasoning_effort( - thinking: Dict[str, Any] + thinking: Dict[str, Any], ) -> Optional[str]: """ Translate Anthropic's thinking parameter to OpenAI's reasoning_effort. @@ -1081,10 +1081,23 @@ class LiteLLMAnthropicMessagesAdapter: anthropic_message_request: AnthropicMessagesRequest, new_kwargs: ChatCompletionRequest, ) -> None: - """Translate output_format to response_format when applicable.""" - if "output_format" not in anthropic_message_request: - return - output_format = anthropic_message_request["output_format"] + """Translate Anthropic structured-output config to OpenAI ``response_format``. + + Accepts either the legacy top-level ``output_format`` field OR the + newer ``output_config.format`` (sub-key on ``output_config``) so that + both shapes flow through to non-Anthropic backends as + ``response_format``. Without the ``output_config.format`` branch, + callers using the new Anthropic Structured Outputs API would have + their schema silently dropped on the adapter path — only the legacy + top-level ``output_format`` was being mapped. + + ``output_format`` takes precedence when both are provided. + """ + output_format: Any = anthropic_message_request.get("output_format") + if not output_format: + output_config = anthropic_message_request.get("output_config") + if isinstance(output_config, dict): + output_format = output_config.get("format") if not output_format: return response_format = self.translate_anthropic_output_format_to_openai( diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py index 9080ac02330..d450f7a4635 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py @@ -13,7 +13,7 @@ from litellm.types.llms.vertex_ai import VertexPartnerProvider from litellm.types.router import GenericLiteLLMParams from ....vertex_llm_base import VertexBase -from ..transformation import _sanitize_vertex_anthropic_output_params +from ..output_params_utils import sanitize_vertex_anthropic_output_params class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, VertexBase): @@ -163,6 +163,6 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert # and ``output_format``, but rejects ``output_config.effort`` with 400 # "Extra inputs are not permitted". Sanitize in place so the supported # bits flow through. - _sanitize_vertex_anthropic_output_params(anthropic_messages_request) + sanitize_vertex_anthropic_output_params(anthropic_messages_request) return anthropic_messages_request diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py new file mode 100644 index 00000000000..982d8edbf20 --- /dev/null +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py @@ -0,0 +1,50 @@ +""" +Shared sanitization for ``output_config`` / ``output_format`` on Vertex AI +Claude. Lives in its own module so both the chat-completion transformation +(``transformation.py``) and the Messages pass-through transformation +(``experimental_pass_through/transformation.py``) can import it without +forming a cycle through the parent module's heavier imports. + +CodeQL flagged the ``..transformation`` import path as a potential cyclic +import; extracting the helper into a leaf module resolves the warning and +keeps the parent module's import surface narrow. +""" + +# Keys inside ``output_config`` that Vertex AI Claude does not accept. +# Today only ``effort`` triggers "Extra inputs are not permitted"; add new +# entries here as Vertex parity drifts. Keep this list narrow — anything +# Vertex DOES accept (e.g. ``format`` for structured outputs) must be +# preserved so callers can rely on Anthropic-native features. +VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS: frozenset = frozenset({"effort"}) + + +def sanitize_vertex_anthropic_output_params(data: dict) -> None: + """ + Strip Vertex-unsupported keys from ``output_config`` / + ``output_format`` in-place; forward whatever remains. + + Behavior: + * ``output_config`` containing only unsupported keys (e.g. ``effort`` + alone) is removed entirely so the request body has no empty dict. + * ``output_config`` containing a mix of supported + unsupported keys + has the unsupported subset filtered out and the rest forwarded. + * ``output_config`` that is supported in full passes through unchanged. + * ``output_format`` is forwarded as-is (Vertex AI Claude accepts it). + * Non-dict values for ``output_config`` are dropped to avoid sending + malformed payloads downstream. + """ + output_config = data.get("output_config") + if output_config is None: + return + if not isinstance(output_config, dict): + data.pop("output_config", None) + return + sanitized = { + k: v + for k, v in output_config.items() + if k not in VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS + } + if sanitized: + data["output_config"] = sanitized + else: + data.pop("output_config", None) diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index a9dea6646ff..914c7e92e5e 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -10,45 +10,7 @@ from litellm.types.llms.openai import AllMessageValues from litellm.types.utils import ModelResponse from ....anthropic.chat.transformation import AnthropicConfig - -# Keys inside ``output_config`` that Vertex AI Claude does not accept. -# Today only ``effort`` triggers "Extra inputs are not permitted"; add new -# entries here as Vertex parity drifts. Keep this list narrow — anything -# Vertex DOES accept (e.g. ``format`` for structured outputs) must be -# preserved so callers can rely on Anthropic-native features. -_VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS = frozenset({"effort"}) - - -def _sanitize_vertex_anthropic_output_params(data: dict) -> None: - """ - Strip Vertex-unsupported keys from ``output_config`` / ``output_format`` - in-place; forward whatever remains. - - Behavior: - * ``output_config`` containing only unsupported keys (e.g. ``effort`` - alone) is removed entirely so the request body has no empty dict. - * ``output_config`` containing a mix of supported + unsupported keys has - the unsupported subset filtered out and the rest forwarded. - * ``output_config`` that is supported in full passes through unchanged. - * ``output_format`` is forwarded as-is (Vertex AI Claude accepts it). - * Non-dict values for ``output_config`` are dropped to avoid sending - malformed payloads downstream. - """ - output_config = data.get("output_config") - if output_config is None: - return - if not isinstance(output_config, dict): - data.pop("output_config", None) - return - sanitized = { - k: v - for k, v in output_config.items() - if k not in _VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS - } - if sanitized: - data["output_config"] = sanitized - else: - data.pop("output_config", None) +from .output_params_utils import sanitize_vertex_anthropic_output_params class VertexAIError(Exception): @@ -149,7 +111,7 @@ class VertexAIAnthropicConfig(AnthropicConfig): # Vertex returns 400 "Extra inputs are not permitted". Sanitize in place: # forward the structured-output bits, drop the unsupported keys. # Same treatment for the legacy top-level ``output_format`` field. - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) tools = optional_params.get("tools") tool_search_used = self.is_tool_search_used(tools) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py index 50e5fd884e9..615dc5cfebc 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_handler_output_config_passthrough.py @@ -46,10 +46,13 @@ from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( MESSAGES = [{"role": "user", "content": "hello"}] -def _call_prepare(extra_kwargs, model="gpt-4o", **overrides): +def _call_prepare(extra_kwargs, model="gpt-4o", output_format=None, **overrides): """ Drive ``_prepare_completion_kwargs`` with the minimum scaffolding needed. + ``output_format`` is a top-level parameter on the function, so callers + pass it explicitly here rather than tucking it into ``extra_kwargs``. + Uses an explicit-None check on ``extra_kwargs`` so callers can test the falsy-empty-dict path. The fallback ``or {}`` pattern PR #22727 used here masked the no-extra-kwargs case from ever exercising the test's intent. @@ -68,7 +71,7 @@ def _call_prepare(extra_kwargs, model="gpt-4o", **overrides): tools=None, top_k=None, top_p=None, - output_format=None, + output_format=output_format, extra_kwargs=extra_kwargs, ) @@ -107,22 +110,84 @@ class TestOutputConfigStrippedFromCompletionKwargs: "reject it with 400 'Extra inputs are not permitted'" ) - def test_output_config_with_format_is_stripped_format_already_translated(self): - """Even when ``output_config`` carries useful structured-output info, - the raw key must be excluded — the translator above has already mapped - ``output_config.format`` to ``response_format`` (the OpenAI-shaped key - the downstream backend understands).""" + def test_output_config_format_translated_to_response_format(self): + """When ``output_config`` carries structured-output ``format``, the + translator now maps it to OpenAI's ``response_format`` so non-Anthropic + backends see the schema in their native shape. The raw + ``output_config`` key is still stripped from ``completion_kwargs`` — + only the translated ``response_format`` survives. + + Before this PR, only the legacy top-level ``output_format`` was + translated; ``output_config.format`` was silently dropped on the + adapter path even when the schema was correctly supplied (issue + flagged by Greptile review of the initial fix). + """ + schema = { + "type": "object", + "additionalProperties": False, + "properties": {"name": {"type": "string"}}, + } extra_kwargs = { "custom_llm_provider": "azure", - "output_config": { - "format": {"type": "json_schema", "schema": {"type": "object"}} - }, + "output_config": {"format": {"type": "json_schema", "schema": schema}}, } result = _call_prepare(extra_kwargs=extra_kwargs) completion_kwargs = result[0] if isinstance(result, tuple) else result + # Raw Anthropic-shaped key is gone (would 400 on non-Anthropic backends). assert "output_config" not in completion_kwargs + # Translated OpenAI-shaped key is present so the schema actually + # reaches the downstream backend. + assert "response_format" in completion_kwargs, ( + "output_config.format must be translated to response_format — " + "without this, structured-output schemas are silently dropped on " + "the adapter path" + ) + + def test_output_format_top_level_still_translates(self): + """Regression guard: the legacy top-level ``output_format`` field must + continue to translate to ``response_format``. The new + ``output_config.format`` path must not break this existing behavior.""" + schema = {"type": "object", "properties": {"name": {"type": "string"}}} + result = _call_prepare( + extra_kwargs={"custom_llm_provider": "azure"}, + output_format={"type": "json_schema", "schema": schema}, + ) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + assert "response_format" in completion_kwargs + + def test_output_format_takes_precedence_over_output_config_format(self): + """When both top-level ``output_format`` and ``output_config.format`` + are present, the legacy top-level ``output_format`` wins. Documents + which one the translator picks rather than leaving it implementation- + defined.""" + winning_schema = { + "type": "object", + "properties": {"top_level": {"type": "string"}}, + } + losing_schema = { + "type": "object", + "properties": {"nested": {"type": "string"}}, + } + result = _call_prepare( + extra_kwargs={ + "custom_llm_provider": "azure", + "output_config": { + "format": {"type": "json_schema", "schema": losing_schema} + }, + }, + output_format={"type": "json_schema", "schema": winning_schema}, + ) + completion_kwargs = result[0] if isinstance(result, tuple) else result + + assert "response_format" in completion_kwargs + # Verify the winning_schema (top-level output_format) was used, + # not the losing one nested under output_config. + rendered = str(completion_kwargs["response_format"]) + assert "top_level" in rendered + assert "nested" not in rendered def test_other_extra_kwargs_still_passed_through(self): """Regression guard: the strip must be narrow. Unrelated fields like diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 376c48d9e95..d0be476d72e 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -693,32 +693,32 @@ def test_sanitize_vertex_anthropic_output_params_unit(): """Direct unit coverage for the helper itself (used by both Vertex Anthropic transformation paths). Mirrors the integration assertions above without going through the full ``transform_request`` stack.""" - from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import ( - _sanitize_vertex_anthropic_output_params, + from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.output_params_utils import ( + sanitize_vertex_anthropic_output_params, ) # No-op when output_config absent. data: dict = {"max_tokens": 8} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert data == {"max_tokens": 8} # Effort-only → dropped entirely. data = {"output_config": {"effort": "high"}} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert "output_config" not in data # Format-only → preserved unchanged. fmt = {"format": {"type": "json_schema", "schema": {"type": "object"}}} data = {"output_config": dict(fmt)} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert data["output_config"] == fmt # Mixed → effort filtered, format kept. data = {"output_config": {"format": fmt["format"], "effort": "high"}} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert data["output_config"] == fmt # Non-dict → dropped defensively. data = {"output_config": "garbage"} - _sanitize_vertex_anthropic_output_params(data) + sanitize_vertex_anthropic_output_params(data) assert "output_config" not in data From 34c93645e9aaed8546032ef52ff71b54b264e454 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 24 Apr 2026 14:30:06 -0700 Subject: [PATCH 016/196] fix(openai): gpt-5.5 does not support reasoning_effort=minimal Verified against OpenAI's live Chat Completions API on 2026-04-24: POST /v1/chat/completions {"model": "gpt-5.5", "reasoning_effort": "minimal", ...} -> 400 Unsupported value: 'reasoning_effort' does not support 'minimal' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'. POST /v1/chat/completions {"model": "gpt-5.5-pro", "reasoning_effort": "minimal", ...} -> 400 Unsupported value: 'minimal' is not supported with the 'gpt-5.5-pro' model. Supported values are: 'medium', 'high', and 'xhigh'. Set supports_minimal_reasoning_effort=false on all four entries (gpt-5.5, gpt-5.5-2026-04-23, gpt-5.5-pro, gpt-5.5-pro-2026-04-23) so OpenAIGPT5Config._is_reasoning_effort_level_explicitly_disabled fires and LiteLLM either drops the param (drop_params=True) or raises a local UnsupportedParamsError, instead of round-tripping to OpenAI for a 400. Adds a parametrized test_gpt55_reasoning_effort_flags_match_live_openai_api test that pins supports_{none,minimal,xhigh}_reasoning_effort on each entry to OpenAI's actual API contract. Note: gpt-5.5-pro additionally rejects 'none' and 'low'. 'none' is already handled (supports_none_reasoning_effort=false). 'low' is not representable in the current JSON schema (no supports_low flag); filing separately. --- ...odel_prices_and_context_window_backup.json | 8 ++-- model_prices_and_context_window.json | 8 ++-- .../llm_cost_calc/test_llm_cost_calc_utils.py | 40 +++++++++++++++++++ 3 files changed, 48 insertions(+), 8 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 49ce5022c56..ffe5b47bac7 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -19319,7 +19319,7 @@ "supports_web_search": true, "supports_none_reasoning_effort": true, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": true + "supports_minimal_reasoning_effort": false }, "gpt-5.5-2026-04-23": { "cache_read_input_token_cost": 5e-07, @@ -19367,7 +19367,7 @@ "supports_web_search": true, "supports_none_reasoning_effort": true, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": true + "supports_minimal_reasoning_effort": false }, "gpt-5.5-pro": { "cache_read_input_token_cost": 6e-06, @@ -19410,7 +19410,7 @@ "supports_web_search": true, "supports_none_reasoning_effort": false, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": true + "supports_minimal_reasoning_effort": false }, "gpt-5.5-pro-2026-04-23": { "cache_read_input_token_cost": 6e-06, @@ -19453,7 +19453,7 @@ "supports_web_search": true, "supports_none_reasoning_effort": false, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": true + "supports_minimal_reasoning_effort": false }, "gpt-5.4": { "cache_read_input_token_cost": 2.5e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 3733f07a30d..39f85650d1f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -19333,7 +19333,7 @@ "supports_web_search": true, "supports_none_reasoning_effort": true, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": true + "supports_minimal_reasoning_effort": false }, "gpt-5.5-2026-04-23": { "cache_read_input_token_cost": 5e-07, @@ -19381,7 +19381,7 @@ "supports_web_search": true, "supports_none_reasoning_effort": true, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": true + "supports_minimal_reasoning_effort": false }, "gpt-5.5-pro": { "cache_read_input_token_cost": 6e-06, @@ -19424,7 +19424,7 @@ "supports_web_search": true, "supports_none_reasoning_effort": false, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": true + "supports_minimal_reasoning_effort": false }, "gpt-5.5-pro-2026-04-23": { "cache_read_input_token_cost": 6e-06, @@ -19467,7 +19467,7 @@ "supports_web_search": true, "supports_none_reasoning_effort": false, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": true + "supports_minimal_reasoning_effort": false }, "gpt-5.4": { "cache_read_input_token_cost": 2.5e-07, diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index 5e37e2a3424..8b1b39848ea 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -411,6 +411,46 @@ def test_generic_cost_per_token_gpt55_pro(): ) +@pytest.mark.parametrize( + "model,expected_none,expected_xhigh,expected_minimal", + [ + # Verified against OpenAI's live API on 2026-04-24: + # gpt-5.5 -> supports: none, low, medium, high, xhigh + # gpt-5.5-pro -> supports: medium, high, xhigh + # Neither supports "minimal"; gpt-5.5-pro additionally does not support "none". + # The JSON must reflect this so LiteLLM rejects unsupported values locally + # (or drops them with drop_params=True) instead of round-tripping to OpenAI + # for a 400. + ("gpt-5.5", True, True, False), + ("gpt-5.5-2026-04-23", True, True, False), + ("gpt-5.5-pro", False, True, False), + ("gpt-5.5-pro-2026-04-23", False, True, False), + ], +) +def test_gpt55_reasoning_effort_flags_match_live_openai_api( + model, expected_none, expected_xhigh, expected_minimal +): + """Pin reasoning_effort capability flags to OpenAI's actual API contract. + + Observed via `POST /v1/chat/completions` with reasoning_effort=minimal: + ``Unsupported value: 'reasoning_effort' does not support 'minimal' with + this model``. gpt-5.5-pro additionally rejects 'none' and 'low'. + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + m = litellm.model_cost[model] + assert ( + m.get("supports_none_reasoning_effort") is expected_none + ), f"{model}: supports_none_reasoning_effort expected {expected_none}" + assert ( + m.get("supports_xhigh_reasoning_effort") is expected_xhigh + ), f"{model}: supports_xhigh_reasoning_effort expected {expected_xhigh}" + assert ( + m.get("supports_minimal_reasoning_effort") is expected_minimal + ), f"{model}: supports_minimal_reasoning_effort expected {expected_minimal}" + + @pytest.mark.parametrize( "base_model,dated_model", [ From 94f8f12a00d63253a78187699e585c74c49c08c7 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 24 Apr 2026 15:05:43 -0700 Subject: [PATCH 017/196] feat(openai): add supports_low_reasoning_effort flag; reject low on gpt-5.5-pro MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit gpt-5.5-pro only accepts reasoning_effort in {medium, high, xhigh} (verified live against OpenAI's API on 2026-04-24). LiteLLM previously had no way to express this constraint — the existing JSON schema covered none/minimal/xhigh but not low. Result: drop_params=true users saw an avoidable 400 from OpenAI. Add supports_low_reasoning_effort following the existing opt-out pattern (default-allow, explicit false to block). Mirror the minimal branch in OpenAIGPT5Config.map_openai_params so 'low' goes through the same _is_reasoning_effort_level_explicitly_disabled gate. Set the flag to false on gpt-5.5-pro and gpt-5.5-pro-2026-04-23 in both model_prices JSON files (kept in sync). Other models leave the key absent so behavior is unchanged. Tests cover: rejection on pro variants (no drop_params), drop on pro with drop_params=True, passthrough on gpt-5.5 chat, passthrough on unknown models, and the helper-level _is_reasoning_effort_level_explicitly_disabled contract. --- .../llms/openai/chat/gpt_5_transformation.py | 8 +- ...odel_prices_and_context_window_backup.json | 6 +- litellm/types/utils.py | 1 + litellm/utils.py | 3 + model_prices_and_context_window.json | 6 +- .../llms/openai/test_gpt5_transformation.py | 75 +++++++++++++++++++ 6 files changed, 92 insertions(+), 7 deletions(-) diff --git a/litellm/llms/openai/chat/gpt_5_transformation.py b/litellm/llms/openai/chat/gpt_5_transformation.py index 34941a545eb..4e34d10b187 100644 --- a/litellm/llms/openai/chat/gpt_5_transformation.py +++ b/litellm/llms/openai/chat/gpt_5_transformation.py @@ -244,9 +244,11 @@ class OpenAIGPT5Config(OpenAIGPTConfig): ), status_code=400, ) - elif effective_effort == "minimal": - # minimal is opt-out: unknown models pass through; only block when - # the model map explicitly sets supports_minimal_reasoning_effort=false. + elif effective_effort in ("minimal", "low"): + # minimal/low are opt-out: unknown models pass through; only block when + # the model map explicitly sets supports_{level}_reasoning_effort=false. + # Example: gpt-5.5-pro only accepts {medium, high, xhigh}, so it sets + # supports_low_reasoning_effort=false (and supports_minimal=false). if self._is_reasoning_effort_level_explicitly_disabled( model, effective_effort ): diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ffe5b47bac7..82aa4a6a0d7 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -19410,7 +19410,8 @@ "supports_web_search": true, "supports_none_reasoning_effort": false, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": false + "supports_minimal_reasoning_effort": false, + "supports_low_reasoning_effort": false }, "gpt-5.5-pro-2026-04-23": { "cache_read_input_token_cost": 6e-06, @@ -19453,7 +19454,8 @@ "supports_web_search": true, "supports_none_reasoning_effort": false, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": false + "supports_minimal_reasoning_effort": false, + "supports_low_reasoning_effort": false }, "gpt-5.4": { "cache_read_input_token_cost": 2.5e-07, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index c347956cba7..c81bbca19e8 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -140,6 +140,7 @@ class ProviderSpecificModelInfo(TypedDict, total=False): supports_url_context: Optional[bool] supports_none_reasoning_effort: Optional[bool] supports_minimal_reasoning_effort: Optional[bool] + supports_low_reasoning_effort: Optional[bool] supports_xhigh_reasoning_effort: Optional[bool] supports_max_reasoning_effort: Optional[bool] diff --git a/litellm/utils.py b/litellm/utils.py index e1ad1db63ef..b60b5aca546 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -5896,6 +5896,9 @@ def _get_model_info_helper( # noqa: PLR0915 supports_minimal_reasoning_effort=_model_info.get( "supports_minimal_reasoning_effort", None ), + supports_low_reasoning_effort=_model_info.get( + "supports_low_reasoning_effort", None + ), supports_xhigh_reasoning_effort=_model_info.get( "supports_xhigh_reasoning_effort", None ), diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 39f85650d1f..830988b7a2e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -19424,7 +19424,8 @@ "supports_web_search": true, "supports_none_reasoning_effort": false, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": false + "supports_minimal_reasoning_effort": false, + "supports_low_reasoning_effort": false }, "gpt-5.5-pro-2026-04-23": { "cache_read_input_token_cost": 6e-06, @@ -19467,7 +19468,8 @@ "supports_web_search": true, "supports_none_reasoning_effort": false, "supports_xhigh_reasoning_effort": true, - "supports_minimal_reasoning_effort": false + "supports_minimal_reasoning_effort": false, + "supports_low_reasoning_effort": false }, "gpt-5.4": { "cache_read_input_token_cost": 2.5e-07, diff --git a/tests/test_litellm/llms/openai/test_gpt5_transformation.py b/tests/test_litellm/llms/openai/test_gpt5_transformation.py index aebab33e808..a0584f89855 100644 --- a/tests/test_litellm/llms/openai/test_gpt5_transformation.py +++ b/tests/test_litellm/llms/openai/test_gpt5_transformation.py @@ -536,6 +536,81 @@ def test_gpt5_unknown_model_passes_through_minimal(config: OpenAIConfig): assert params["reasoning_effort"] == "minimal" +def test_gpt5_5_pro_rejects_reasoning_effort_low(config: OpenAIConfig): + """gpt-5.5-pro only accepts {medium, high, xhigh} — 'low' must raise. + + Verified against OpenAI's live API: /v1/chat/completions with + reasoning_effort='low' on gpt-5.5-pro returns HTTP 400. + """ + with pytest.raises(litellm.utils.UnsupportedParamsError): + config.map_openai_params( + non_default_params={"reasoning_effort": "low"}, + optional_params={}, + model="gpt-5.5-pro", + drop_params=False, + ) + + +def test_gpt5_5_pro_dated_rejects_reasoning_effort_low(config: OpenAIConfig): + """Dated snapshot must inherit the base alias's low-rejection behavior.""" + with pytest.raises(litellm.utils.UnsupportedParamsError): + config.map_openai_params( + non_default_params={"reasoning_effort": "low"}, + optional_params={}, + model="gpt-5.5-pro-2026-04-23", + drop_params=False, + ) + + +def test_gpt5_5_pro_drops_reasoning_effort_low_when_requested(config: OpenAIConfig): + """drop_params=True silently strips 'low' instead of round-tripping a 400.""" + params = config.map_openai_params( + non_default_params={"reasoning_effort": "low"}, + optional_params={}, + model="gpt-5.5-pro", + drop_params=True, + ) + assert "reasoning_effort" not in params + + +def test_gpt5_5_chat_allows_reasoning_effort_low(config: OpenAIConfig): + """gpt-5.5 (chat) supports 'low'; flag absent → opt-out check passes.""" + params = config.map_openai_params( + non_default_params={"reasoning_effort": "low"}, + optional_params={}, + model="gpt-5.5", + drop_params=False, + ) + assert params["reasoning_effort"] == "low" + + +def test_gpt5_unknown_model_passes_through_low(config: OpenAIConfig): + """Unknown gpt-5 models pass 'low' through (opt-out, not opt-in).""" + params = config.map_openai_params( + non_default_params={"reasoning_effort": "low"}, + optional_params={}, + model="gpt-5.4-turbo-preview", + drop_params=False, + ) + assert params["reasoning_effort"] == "low" + + +def test_gpt5_low_explicitly_disabled_check(gpt5_config: OpenAIGPT5Config): + """supports_low_reasoning_effort=false → disabled; missing/true → not disabled.""" + assert gpt5_config._is_reasoning_effort_level_explicitly_disabled( + "gpt-5.5-pro", "low" + ) + assert gpt5_config._is_reasoning_effort_level_explicitly_disabled( + "gpt-5.5-pro-2026-04-23", "low" + ) + assert not gpt5_config._is_reasoning_effort_level_explicitly_disabled( + "gpt-5.5", "low" + ) + assert not gpt5_config._is_reasoning_effort_level_explicitly_disabled( + "gpt-5.4", "low" + ) + + def test_gpt5_normalizes_reasoning_effort_dict_with_summary(config: OpenAIConfig): """Dict with summary/generate_summary is normalized for chat completions.""" params = config.map_openai_params( From c3338384c93c467ebf447a52b0d5bba740a19395 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 24 Apr 2026 15:33:30 -0700 Subject: [PATCH 018/196] test: register supports_low_reasoning_effort in cost-map JSON schema The strict 'additionalProperties: false' schema in test_aaamodel_prices_and_context_window_json_is_valid rejected the new flag added in this PR's earlier commit. Register it alongside the other supports_*_reasoning_effort entries so the schema validation passes. --- tests/test_litellm/test_utils.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 67b62696196..c6b0f49ff61 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -769,6 +769,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "uses_embed_content": {"type": "boolean"}, "supports_reasoning": {"type": "boolean"}, "supports_minimal_reasoning_effort": {"type": "boolean"}, + "supports_low_reasoning_effort": {"type": "boolean"}, "supports_none_reasoning_effort": {"type": "boolean"}, "supports_xhigh_reasoning_effort": {"type": "boolean"}, "supports_max_reasoning_effort": {"type": "boolean"}, From e2d0fd9eacdeadf46c0e18057e5f51ea1f28eb49 Mon Sep 17 00:00:00 2001 From: RoomWithOutRoof <166608075+Jah-yee@users.noreply.github.com> Date: Sun, 26 Apr 2026 01:45:01 +0800 Subject: [PATCH 019/196] fix: remove duplicate MAX_SIZE + add Cloudflare response_text support (#26385) - Remove duplicate MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB definition (kept the one with default 1024, removed the one with default 512) - Add fallback from 'response' to 'response_text' key in Cloudflare Workers AI transformation for newer Nemotron models Co-authored-by: yuneng-jiang Co-authored-by: Jah-yee <110645028+Jah-yee@users.noreply.github.com> --- litellm/constants.py | 3 --- litellm/llms/cloudflare/chat/transformation.py | 6 +++--- 2 files changed, 3 insertions(+), 6 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index 012599ab6ab..385e3723bee 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -409,9 +409,6 @@ CACHED_STREAMING_CHUNK_DELAY = float(os.getenv("CACHED_STREAMING_CHUNK_DELAY", 0 AUDIO_SPEECH_CHUNK_SIZE = int( os.getenv("AUDIO_SPEECH_CHUNK_SIZE", 8192) ) # chunk_size for audio speech streaming. Balance between latency and memory usage -MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = int( - os.getenv("MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB", 512) -) DEFAULT_MAX_TOKENS_FOR_TRITON = int(os.getenv("DEFAULT_MAX_TOKENS_FOR_TRITON", 2000)) #### Networking settings #### # Sentinel used when `REQUEST_TIMEOUT` is unset: `litellm.request_timeout` keeps this diff --git a/litellm/llms/cloudflare/chat/transformation.py b/litellm/llms/cloudflare/chat/transformation.py index 9e59782bf73..d0c2e86f708 100644 --- a/litellm/llms/cloudflare/chat/transformation.py +++ b/litellm/llms/cloudflare/chat/transformation.py @@ -147,9 +147,9 @@ class CloudflareChatConfig(BaseConfig): ) -> ModelResponse: completion_response = raw_response.json() - model_response.choices[0].message.content = completion_response["result"][ # type: ignore - "response" - ] + # Support both "response" and "response_text" keys (newer models like Nemotron use "response_text") + result = completion_response["result"] + model_response.choices[0].message.content = result.get("response") if result.get("response") is not None else result.get("response_text", "") # type: ignore prompt_tokens = litellm.utils.get_token_count(messages=messages, model=model) completion_tokens = len( From 334aedf2d49cbc1992b153610950a12b6603a97d Mon Sep 17 00:00:00 2001 From: Blossom Date: Sun, 26 Apr 2026 01:52:17 +0800 Subject: [PATCH 020/196] fix(ui): add missing 'zai' (Z.AI / Zhipu AI) provider to Add-Model dropdown (#25482) (#26419) The Z.AI (Zhipu AI) provider was missing from the Add-Model dropdown in the admin UI, even though the rest of the stack already supports it: - /public/providers returns 'zai' in the provider list - provider_endpoints_support.json includes a full 'zai' entry with endpoints and a docs URL (https://docs.litellm.ai/docs/providers/zai) - Backend routing works for zai/* models (e.g. zai/glm-4.5, zai/glm-5) - There are many zai/* entries in model_prices_and_context_window.json The dropdown is driven by the hard-coded Providers enum and provider_map in provider_info_helpers.tsx, which did not include 'zai', so users could not select Z.AI when adding a model through the UI. This PR: - Adds Providers.ZAI ('Z.AI (Zhipu AI)') to the enum. - Maps it to 'zai' in provider_map so the UI round-trips the existing backend provider key. - Wires a reasonable placeholder 'zai/glm-4.5' in getPlaceholder, since glm-4.5 is an established zai/* model in the pricing catalog. - Adds two regression tests in provider_info_helpers.test.tsx: 1. getProviderLogoAndName('zai') resolves to Providers.ZAI. 2. getPlaceholder(Providers.ZAI) returns 'zai/glm-4.5'. No logo asset is added in this PR; getProviderLogoAndName already gracefully returns an empty logo string for providers missing from providerLogoMap, matching the existing pattern for several other providers. A follow-up can add a dedicated logo. Fixes #25482 Co-authored-by: yuneng-jiang --- .../src/components/provider_info_helpers.test.tsx | 14 ++++++++++++++ .../src/components/provider_info_helpers.tsx | 4 ++++ 2 files changed, 18 insertions(+) diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx index a8021f94d84..fa014de4e62 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.test.tsx @@ -68,6 +68,16 @@ describe("provider_info_helpers", () => { expect(result.logo).toBe(providerLogoMap[Providers.OpenAI]); }); + it("should resolve the zai (Z.AI) provider value to the Z.AI display name", () => { + // Regression test for https://github.com/BerriAI/litellm/issues/25482 — + // the backend already returns `zai` from /public/providers and the docs + // have a dedicated page, but the UI dropdown was missing an entry, so + // `getProviderLogoAndName("zai")` previously returned the raw value as + // the display name (no mapping). + const result = getProviderLogoAndName("zai"); + expect(result.displayName).toBe(Providers.ZAI); + }); + it("should return provider value as display name when no mapping exists", () => { const unknownProvider = "unknown_provider"; const result = getProviderLogoAndName(unknownProvider); @@ -156,6 +166,10 @@ describe("provider_info_helpers", () => { expect(getPlaceholder(Providers.WATSONX)).toBe("watsonx/ibm/granite-3-3-8b-instruct"); }); + it("should return zai/glm-4.5 placeholder for Z.AI provider", () => { + expect(getPlaceholder(Providers.ZAI)).toBe("zai/glm-4.5"); + }); + it("should return default gpt-3.5-turbo placeholder for unknown provider", () => { expect(getPlaceholder("UnknownProvider" as any)).toBe("gpt-3.5-turbo"); }); diff --git a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx index e833d0eb4fb..62c0633d117 100644 --- a/ui/litellm-dashboard/src/components/provider_info_helpers.tsx +++ b/ui/litellm-dashboard/src/components/provider_info_helpers.tsx @@ -103,6 +103,7 @@ export enum Providers { WATSONX_TEXT = "Watsonx Text", xAI = "xAI", XINFERENCE = "Xinference", + ZAI = "Z.AI (Zhipu AI)", } export const provider_map: Record = { @@ -210,6 +211,7 @@ export const provider_map: Record = { WATSONX_TEXT: "watsonx_text", xAI: "xai", XINFERENCE: "xinference", + ZAI: "zai", }; const asset_logos_folder = "../ui/assets/logos/"; @@ -366,6 +368,8 @@ export const getPlaceholder = (selectedProvider: string): string => { return "watsonx/ibm/granite-3-3-8b-instruct"; } else if (selectedProvider === Providers.Cursor) { return "cursor/claude-4-sonnet"; + } else if (selectedProvider === Providers.ZAI) { + return "zai/glm-4.5"; } else { return "gpt-3.5-turbo"; } From f63a6f1b263130e8529b699acb19b7481489518e Mon Sep 17 00:00:00 2001 From: Yufeng He <40085740+he-yufeng@users.noreply.github.com> Date: Sun, 26 Apr 2026 03:13:13 +0800 Subject: [PATCH 021/196] fix(proxy): set verbose_logger level when LITELLM_LOG=INFO (#26401) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fixes #26396. The `LITELLM_LOG=INFO` branch in proxy_server only set `verbose_router_logger` and `verbose_proxy_logger`. The third logger `verbose_logger` (used by e.g. `token_based_routing.py`) inherited the Python root default (WARNING) and its INFO-level messages were silently filtered — inconsistent with the neighbouring DEBUG branch which configures all three and with the `debug=True` / `detailed_debug` paths above. Include `verbose_logger` in the INFO branch as well so all three loggers behave the same. Co-authored-by: yuneng-jiang Co-authored-by: Yufeng He <40085740+universeplayer@users.noreply.github.com> --- litellm/proxy/proxy_server.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 4ca6895c2ce..4afac7173cd 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5845,10 +5845,15 @@ async def initialize( # noqa: PLR0915 if litellm_log_setting.upper() == "INFO": import logging - from litellm._logging import verbose_proxy_logger, verbose_router_logger + from litellm._logging import ( + verbose_logger, + verbose_proxy_logger, + verbose_router_logger, + ) # this must ALWAYS remain logging.INFO, DO NOT MODIFY THIS + verbose_logger.setLevel(level=logging.INFO) # set package log to info verbose_router_logger.setLevel( level=logging.INFO ) # set router logs to info From 98a9005c765cf6ceee0eec498e3517166c0e0b7e Mon Sep 17 00:00:00 2001 From: Alvin Tang Date: Sun, 26 Apr 2026 05:11:51 +0800 Subject: [PATCH 022/196] fix(arize): _set_usage_outputs handles raw OpenAI Pydantic CompletionUsage (#26506) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * [Feat] Day-0 support for GPT-5.5 and GPT-5.5 Pro (#26449) * feat(openai): day-0 support for GPT-5.5 and GPT-5.5 Pro Add pricing + capability entries for the new GPT-5.5 family launched by OpenAI on 2026-04-24: - gpt-5.5 / gpt-5.5-2026-04-23 (chat): $5/$30/$0.50 per 1M input/output/cached input - gpt-5.5-pro / gpt-5.5-pro-2026-04-23 (responses-only): $60/$360/$6 per 1M input/output/cached input Other fees (long-context >272k, flex, batches, priority, cache discounts) follow the same ratios as GPT-5.4, with context window retained at 1.05M input / 128K output. No transformation / classifier code changes are required: OpenAIGPT5Config.is_model_gpt_5_4_plus_model() already matches 5.5+ via numeric version parsing, and model registration is driven from the JSON. The existing responses-API bridge for tools + reasoning_effort (litellm/main.py:970) already covers gpt-5.5-pro. Tests: - GPT5_MODELS regression list now covers gpt-5.5-pro and dated variants - New test_generic_cost_per_token_gpt55_pro cost-calc test - Updated test_generic_cost_per_token_gpt55 for long-context fields * fix(openai): mirror reasoning_effort flags onto gpt-5.5 dated variants gpt-5.5-2026-04-23 and gpt-5.5-pro-2026-04-23 were missing the supports_none_reasoning_effort, supports_xhigh_reasoning_effort, and supports_minimal_reasoning_effort flags that their non-dated counterparts define. Reasoning-effort routing in OpenAIGPT5Config is fully capability-driven from these JSON flags — since an absent flag is treated as False for opt-in levels (xhigh), users pinning to a dated snapshot would silently lose xhigh support and diverge from the base alias on logprobs + flexible temperature handling. Copy the flags onto both dated variants so every dated snapshot inherits the base model's reasoning-effort capability profile. Adds a parametrized regression test that asserts supports_{none,minimal,xhigh}_reasoning_effort parity between each dated variant and its non-dated counterpart, preventing future drift when new snapshots are added. * [Feat] Add azure/gpt-5.5 + azure/gpt-5.5-pro entries (+ dated variants) (#26361) * feat(azure): add azure/gpt-5.5 + azure/gpt-5.5-pro entries (+ dated variants) Azure variants of OpenAI's GPT-5.5 family. Microsoft has not yet shipped GPT-5.5 on Azure OpenAI (latest GA on the Foundry models page is GPT-5.4 as of 2026-04-24), but adding the entries day-0 mirrors the established precedent for azure/gpt-5.4* (which were in the cost map before the Azure rollout) so cost tracking and capability flags work the moment customers deploy. Schema follows the existing azure/gpt-5.4* shape: - Same base/long-context pricing as openai/gpt-5.5*: $5/$30 chat, $60/$360 pro per 1M, with priority tier 2x base - Azure variants drop the flex/batches keys (Azure has no flex tier) but keep priority pricing, matching gpt-5.4* precedent - mode=chat for the thinking model, mode=responses for pro reasoning_effort capability flags mirror the OpenAI variants exactly since Azure proxies the same API contract: minimal rejection on both chat and pro, low/none rejection on pro. Once #26456 (which sets supports_low_reasoning_effort + minimal=false on openai/gpt-5.5*) lands, OpenAI and Azure flag profiles align. Tests pin entry presence + pricing for all four Azure variants and verify the live-API-derived reasoning_effort flags. * test: register supports_low_reasoning_effort in cost-map JSON schema azure/gpt-5.5-pro and azure/gpt-5.5-pro-2026-04-23 added in this branch carry supports_low_reasoning_effort=false. The strict 'additionalProperties: false' schema in test_aaamodel_prices_and_context_window_json_is_valid rejected the new key. Register it alongside the other supports_*_reasoning_effort entries. Note: the runtime side of this flag (code that reads it) lands in #26456. Until that PR merges the flag is inert for both Azure and OpenAI pro entries, but having the schema accept it lets cost-map tests pass on either merge order. * fix(arize/langfuse_otel): handle Pydantic usage objects without `.get` `_set_usage_outputs` called `usage.get(...)` and `usage.get('output_tokens_details', {}).get('reasoning_tokens')`. These crash with `AttributeError: 'CompletionUsage' object has no attribute 'get'` when `usage` (or the nested token-details object) is a raw OpenAI Pydantic model rather than a dict / litellm `Usage` wrapper. Reproduces on the langfuse_otel + arize Responses API logging paths. Fixes #13672. Changes: - Add `_safe_get(obj, key, default)` that prefers dict-style `.get` when available and otherwise falls back to `getattr`. Works uniformly for dicts, litellm's `Usage`, and plain Pydantic models like `openai.types.completion_usage.CompletionUsage` / `CompletionTokensDetails` / `OutputTokensDetails`. - Use `_safe_get` for total / completion / prompt / output tokens. - Look for reasoning tokens in `completion_tokens_details` (Chat Completions API) before falling back to `output_tokens_details` (Responses API). Previously reasoning tokens from the Chat Completions API were silently dropped. Tests: - `test_set_usage_outputs_pydantic_completion_usage` — covers the chat completions path with raw `CompletionUsage` + `CompletionTokensDetails`. - `test_set_usage_outputs_pydantic_response_api_usage` — covers the Responses API path with a Pydantic usage object lacking `.get`. Both tests fail on main before this commit and pass after. --------- Co-authored-by: yuneng-jiang Co-authored-by: Mateo Wang <277851410+mateo-berri@users.noreply.github.com> Co-authored-by: alvinttang Co-authored-by: Krrish Dholakia --- litellm/integrations/arize/_utils.py | 42 ++++- ...odel_prices_and_context_window_backup.json | 163 ++++++++++++++++++ model_prices_and_context_window.json | 163 ++++++++++++++++++ .../integrations/arize/test_arize_utils.py | 85 +++++++++ tests/test_litellm/test_utils.py | 1 + 5 files changed, 450 insertions(+), 4 deletions(-) diff --git a/litellm/integrations/arize/_utils.py b/litellm/integrations/arize/_utils.py index 8dfaa8b1425..a1bf65141c9 100644 --- a/litellm/integrations/arize/_utils.py +++ b/litellm/integrations/arize/_utils.py @@ -220,23 +220,57 @@ def _set_structured_outputs(span: "Span", response_obj, msg_attrs, span_attrs): safe_set_attribute(span, f"{prefix}.{msg_attrs.MESSAGE_ROLE}", message_role) +def _safe_get(obj, key, default=None): + """Read ``key`` from a dict-like or Pydantic-model-like object. + + The arize/langfuse_otel logger receives ``usage`` objects from many sources: + plain dicts, litellm ``Usage`` (which exposes ``.get``), and raw OpenAI + Pydantic models (e.g. ``openai.types.completion_usage.CompletionUsage`` and + nested ``CompletionTokensDetails`` / ``OutputTokensDetails``) which do NOT + expose ``.get``. Calling ``.get`` on the latter raised ``AttributeError`` — + see https://github.com/BerriAI/litellm/issues/13672. + """ + if obj is None: + return default + getter = getattr(obj, "get", None) + if callable(getter): + try: + return getter(key, default) + except TypeError: + # Some objects expose `.get` with a different signature + pass + return getattr(obj, key, default) + + def _set_usage_outputs(span: "Span", response_obj, span_attrs): usage = response_obj and response_obj.get("usage") if not usage: return safe_set_attribute( - span, span_attrs.LLM_TOKEN_COUNT_TOTAL, usage.get("total_tokens") + span, span_attrs.LLM_TOKEN_COUNT_TOTAL, _safe_get(usage, "total_tokens") + ) + completion_tokens = _safe_get(usage, "completion_tokens") or _safe_get( + usage, "output_tokens" ) - completion_tokens = usage.get("completion_tokens") or usage.get("output_tokens") if completion_tokens: safe_set_attribute( span, span_attrs.LLM_TOKEN_COUNT_COMPLETION, completion_tokens ) - prompt_tokens = usage.get("prompt_tokens") or usage.get("input_tokens") + prompt_tokens = _safe_get(usage, "prompt_tokens") or _safe_get( + usage, "input_tokens" + ) if prompt_tokens: safe_set_attribute(span, span_attrs.LLM_TOKEN_COUNT_PROMPT, prompt_tokens) - reasoning_tokens = usage.get("output_tokens_details", {}).get("reasoning_tokens") + + # Reasoning tokens live in `completion_tokens_details` for Chat Completions + # API (Usage) and in `output_tokens_details` for Responses API + # (ResponseAPIUsage). Both nested objects may be plain Pydantic models + # without `.get`. + token_details = _safe_get(usage, "completion_tokens_details") or _safe_get( + usage, "output_tokens_details" + ) + reasoning_tokens = _safe_get(token_details, "reasoning_tokens") if reasoning_tokens: safe_set_attribute( span, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index f6de40717d1..5cccd5f00af 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -4645,6 +4645,169 @@ "supports_vision": true, "supports_web_search": true }, + "azure/gpt-5.5": { + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_priority": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 2e-06, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "input_cost_per_token_priority": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 2e-05, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, + "output_cost_per_token_priority": 6e-05, + "output_cost_per_token_above_272k_tokens_priority": 9e-05, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_service_tier": true, + "supports_vision": true, + "supports_web_search": true, + "supports_none_reasoning_effort": true, + "supports_xhigh_reasoning_effort": true, + "supports_minimal_reasoning_effort": false + }, + "azure/gpt-5.5-2026-04-23": { + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_priority": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 2e-06, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "input_cost_per_token_priority": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 2e-05, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, + "output_cost_per_token_priority": 6e-05, + "output_cost_per_token_above_272k_tokens_priority": 9e-05, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_service_tier": true, + "supports_vision": true, + "supports_web_search": true + }, + "azure/gpt-5.5-pro": { + "cache_read_input_token_cost": 6e-06, + "cache_read_input_token_cost_above_272k_tokens": 1.2e-05, + "input_cost_per_token": 6e-05, + "input_cost_per_token_above_272k_tokens": 0.00012, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 0.00036, + "output_cost_per_token_above_272k_tokens": 0.00054, + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true, + "supports_none_reasoning_effort": false, + "supports_xhigh_reasoning_effort": true, + "supports_minimal_reasoning_effort": false, + "supports_low_reasoning_effort": false + }, + "azure/gpt-5.5-pro-2026-04-23": { + "cache_read_input_token_cost": 6e-06, + "cache_read_input_token_cost_above_272k_tokens": 1.2e-05, + "input_cost_per_token": 6e-05, + "input_cost_per_token_above_272k_tokens": 0.00012, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 0.00036, + "output_cost_per_token_above_272k_tokens": 0.00054, + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, "azure/gpt-5.4-mini": { "cache_read_input_token_cost": 7.5e-08, "input_cost_per_token": 7.5e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 2d13c5cd00f..12a0d8fe0a7 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -4659,6 +4659,169 @@ "supports_vision": true, "supports_web_search": true }, + "azure/gpt-5.5": { + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_priority": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 2e-06, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "input_cost_per_token_priority": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 2e-05, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, + "output_cost_per_token_priority": 6e-05, + "output_cost_per_token_above_272k_tokens_priority": 9e-05, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_service_tier": true, + "supports_vision": true, + "supports_web_search": true, + "supports_none_reasoning_effort": true, + "supports_xhigh_reasoning_effort": true, + "supports_minimal_reasoning_effort": false + }, + "azure/gpt-5.5-2026-04-23": { + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_priority": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 2e-06, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "input_cost_per_token_priority": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 2e-05, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, + "output_cost_per_token_priority": 6e-05, + "output_cost_per_token_above_272k_tokens_priority": 9e-05, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_service_tier": true, + "supports_vision": true, + "supports_web_search": true + }, + "azure/gpt-5.5-pro": { + "cache_read_input_token_cost": 6e-06, + "cache_read_input_token_cost_above_272k_tokens": 1.2e-05, + "input_cost_per_token": 6e-05, + "input_cost_per_token_above_272k_tokens": 0.00012, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 0.00036, + "output_cost_per_token_above_272k_tokens": 0.00054, + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true, + "supports_none_reasoning_effort": false, + "supports_xhigh_reasoning_effort": true, + "supports_minimal_reasoning_effort": false, + "supports_low_reasoning_effort": false + }, + "azure/gpt-5.5-pro-2026-04-23": { + "cache_read_input_token_cost": 6e-06, + "cache_read_input_token_cost_above_272k_tokens": 1.2e-05, + "input_cost_per_token": 6e-05, + "input_cost_per_token_above_272k_tokens": 0.00012, + "litellm_provider": "azure", + "max_input_tokens": 1050000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 0.00036, + "output_cost_per_token_above_272k_tokens": 0.00054, + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, "azure/gpt-5.4-mini": { "cache_read_input_token_cost": 7.5e-08, "input_cost_per_token": 7.5e-07, diff --git a/tests/test_litellm/integrations/arize/test_arize_utils.py b/tests/test_litellm/integrations/arize/test_arize_utils.py index a87a4167899..86c5448d468 100644 --- a/tests/test_litellm/integrations/arize/test_arize_utils.py +++ b/tests/test_litellm/integrations/arize/test_arize_utils.py @@ -273,6 +273,91 @@ def test_arize_set_attributes_responses_api(): ) +def test_set_usage_outputs_pydantic_completion_usage(): + """ + Regression test for https://github.com/BerriAI/litellm/issues/13672 + + `_set_usage_outputs` previously called `usage.get(...)` which crashes when + `usage` is a plain Pydantic model (e.g. openai.types.completion_usage.CompletionUsage) + that does not implement dict-style `.get()`. Same crash for nested + `output_tokens_details` / `completion_tokens_details`. + + The function must: + 1. Read total/prompt/completion tokens from a Pydantic usage without `.get`. + 2. Read reasoning_tokens from `completion_tokens_details` (chat completions API) + OR `output_tokens_details` (responses API), even when those nested objects + are Pydantic models without `.get`. + 3. Not raise AttributeError; not call span.record_exception. + """ + from unittest.mock import MagicMock + + from openai.types.completion_usage import ( + CompletionTokensDetails, + CompletionUsage, + ) + + from litellm.integrations.arize._utils import _set_usage_outputs + + span = MagicMock() + + # Plain OpenAI Pydantic model — has no `.get()` + usage = CompletionUsage( + completion_tokens=60, + prompt_tokens=40, + total_tokens=100, + completion_tokens_details=CompletionTokensDetails(reasoning_tokens=25), + ) + assert not hasattr(usage, "get"), "precondition: CompletionUsage must lack .get" + + response_obj = {"usage": usage} + + # Must not raise + _set_usage_outputs(span, response_obj, SpanAttributes) + + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 100) + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 40) + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 60) + # reasoning_tokens for chat completions live in completion_tokens_details + span.set_attribute.assert_any_call( + SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25 + ) + + +def test_set_usage_outputs_pydantic_response_api_usage(): + """ + Same crash also affects Responses API with `output_tokens_details` as a + Pydantic model that lacks `.get()`. Verifies the responses-API path. + """ + from unittest.mock import MagicMock + + from litellm.integrations.arize._utils import _set_usage_outputs + from litellm.types.llms.openai import OutputTokensDetails + + # Build an object that mimics openai ResponsesAPI usage but lacks `.get` + # (uses a plain class — not BaseLiteLLMOpenAIResponseObject) + class PlainResponsesUsage: + def __init__(self): + self.total_tokens = 370 + self.input_tokens = 120 + self.output_tokens = 250 + self.output_tokens_details = OutputTokensDetails(reasoning_tokens=180) + + usage = PlainResponsesUsage() + assert not hasattr(usage, "get") + + span = MagicMock() + response_obj = {"usage": usage} + + _set_usage_outputs(span, response_obj, SpanAttributes) + + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370) + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120) + span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250) + span.set_attribute.assert_any_call( + SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180 + ) + + class TestArizeLogger(CustomLogger): """ Custom logger implementation to capture standard_callback_dynamic_params. diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index dc344a433bc..93c61e003d9 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -769,6 +769,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "uses_embed_content": {"type": "boolean"}, "supports_reasoning": {"type": "boolean"}, "supports_minimal_reasoning_effort": {"type": "boolean"}, + "supports_low_reasoning_effort": {"type": "boolean"}, "supports_none_reasoning_effort": {"type": "boolean"}, "supports_xhigh_reasoning_effort": {"type": "boolean"}, "supports_max_reasoning_effort": {"type": "boolean"}, From 3c9a8690d1e61495916f1a0cef5e07020921e848 Mon Sep 17 00:00:00 2001 From: user <70670632+stuxf@users.noreply.github.com> Date: Wed, 29 Apr 2026 22:18:18 +0000 Subject: [PATCH 023/196] fix(auth): gate oauth2-proxy header trust on premium + privileged-field denylist MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ``handle_oauth2_proxy_request`` reads HTTP request headers per the admin-set ``oauth2_config_mappings`` and constructs a ``UserAPIKeyAuth`` from the values. Two failure modes: 1. **Premium parity.** Sibling auth paths (``enable_oauth2_auth``, ``enable_jwt_auth``) require ``premium_user``; this path did not, so any open-source deployment could turn the feature on without realising it requires a hardened reverse-proxy topology. Added the ``premium_user`` gate. 2. **Privileged-field denylist.** Without a denylist, an admin who maps the wrong header to ``user_role`` (or whose reverse proxy leaks the header from upstream user input) lets any caller send ``X-User-Role: proxy_admin`` and gain full admin access — Pydantic coerces the string into the ``LitellmUserRoles.PROXY_ADMIN`` enum. Mapping any field in ``PRIVILEGED_OAUTH2_PROXY_FIELDS`` (``user_role``, ``api_key``, ``token``, ``permissions``, ``allowed_routes``, budget/limit fields, ``metadata``) raises at request time so the misconfiguration surfaces loudly rather than as a silent privesc. Operators who genuinely need a trusted upstream to assert one of these privileged fields should switch to JWT auth (signature-validated) rather than header-trust. Tests: - ``test_returns_auth_for_simple_user_id_mapping``: legitimate identity-only mapping still works. - ``test_rejects_when_not_premium``: open-source deployments get a clear enterprise-feature error. - ``test_refuses_to_map_privileged_fields``: parametrized over every entry in the denylist — each is rejected at request time. - ``test_user_role_header_forgery_attack_is_blocked``: end-to-end shape of the GHSA-5c3m-qffq-4r9m attack; rejected before auth object construction. - ``test_safe_fields_still_pass_through``: documented usage (``user_id``, ``user_email``, ``team_id``, ``models``) is unaffected. Co-Authored-By: Claude Opus 4.7 (1M context) --- litellm/proxy/auth/oauth2_proxy_hook.py | 106 ++++++++-- .../proxy/auth/test_oauth2_proxy_hook.py | 189 ++++++++++++++++++ 2 files changed, 277 insertions(+), 18 deletions(-) create mode 100644 tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py diff --git a/litellm/proxy/auth/oauth2_proxy_hook.py b/litellm/proxy/auth/oauth2_proxy_hook.py index 0dc696bc455..341d2b477b0 100644 --- a/litellm/proxy/auth/oauth2_proxy_hook.py +++ b/litellm/proxy/auth/oauth2_proxy_hook.py @@ -1,19 +1,78 @@ -from typing import Any, Dict +from typing import Any, Dict, FrozenSet from fastapi import Request from litellm._logging import verbose_proxy_logger -from litellm.proxy._types import UserAPIKeyAuth +from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth + +# Fields on ``UserAPIKeyAuth`` that grant privileges directly (``user_role`` +# is the canonical privesc — coerced from the string ``"proxy_admin"`` into +# ``LitellmUserRoles.PROXY_ADMIN`` by Pydantic) or break trust assumptions +# (``api_key`` / ``token`` short-circuit the validated-key contract; +# ``permissions`` / ``allowed_routes`` directly grant route access; budget +# and limit fields can be set to wild values to bypass enforcement; +# ``metadata`` is too broad to safely admit from caller-controlled headers). +# +# Operators who legitimately need any of these to flow from a trusted +# upstream proxy should switch to JWT authentication, which validates a +# signature on the assertion rather than blindly trusting headers. +PRIVILEGED_OAUTH2_PROXY_FIELDS: FrozenSet[str] = frozenset( + { + "user_role", + "api_key", + "token", + "key_alias", + "key_name", + "permissions", + "allowed_routes", + "max_budget", + "spend", + "model_max_budget", + "model_spend", + "tpm_limit", + "rpm_limit", + "team_max_budget", + "team_spend", + "blocked", + "metadata", + } +) async def handle_oauth2_proxy_request(request: Request) -> UserAPIKeyAuth: """ - Handle request from oauth2 proxy. + Resolve a ``UserAPIKeyAuth`` from request headers per the admin-set + ``oauth2_config_mappings``. + + The auth model assumes the proxy is deployed behind a trusted OAuth2 + reverse proxy that injects authenticated identity headers (e.g. + oauth2-proxy, Authelia). Two safeguards above and beyond that + deployment assumption: + + 1. **Premium gate.** The sibling auth paths (``enable_oauth2_auth`` + and ``enable_jwt_auth``) require ``premium_user``; this path + previously did not, which let any open-source deployment turn + the feature on without realising it requires a hardened + deployment topology. + 2. **Privileged-field denylist.** ``oauth2_config_mappings`` maps + header names to ``UserAPIKeyAuth`` fields. Without a denylist, + an admin who maps the wrong header to ``user_role`` (or who + hasn't fully locked down their reverse proxy) lets any caller + set the ``user_role`` header to ``"proxy_admin"`` and gain full + admin privileges — Pydantic coerces the string into the enum. + Mapping any privileged field is rejected at startup-style auth + time so the misconfiguration surfaces loudly rather than as a + silent privesc. """ - from litellm.proxy.proxy_server import general_settings + from litellm.proxy.proxy_server import general_settings, premium_user + + if premium_user is not True: + raise ValueError( + "Oauth2 proxy auth is an enterprise-only feature. " + + CommonProxyErrors.not_premium_user.value + ) verbose_proxy_logger.debug("Handling oauth2 proxy request") - # Define the OAuth2 config mappings oauth2_config_mappings: Dict[str, str] = ( general_settings.get("oauth2_config_mappings") or {} ) @@ -21,21 +80,33 @@ async def handle_oauth2_proxy_request(request: Request) -> UserAPIKeyAuth: if not oauth2_config_mappings: raise ValueError("Oauth2 config mappings not found in general_settings") - # Initialize a dictionary to store the mapped values - auth_data: Dict[str, Any] = {} - # Extract values from headers based on the mappings + privileged_mapped = sorted( + set(oauth2_config_mappings.keys()) & PRIVILEGED_OAUTH2_PROXY_FIELDS + ) + if privileged_mapped: + raise ValueError( + "Oauth2 proxy auth refuses to map privileged UserAPIKeyAuth " + f"fields from request headers: {privileged_mapped}. These " + "fields would grant privileges (e.g. proxy_admin), bypass " + "budget enforcement, or short-circuit key validation if a " + "caller can spoof the corresponding header. If you need a " + "trusted upstream to assert one of these, use JWT auth " + "(signature-validated) instead of header-trust." + ) + + auth_data: Dict[str, Any] = {} for key, header in oauth2_config_mappings.items(): value = request.headers.get(header) - if value: - # Convert max_budget to float if present - if key == "max_budget": - auth_data[key] = float(value) - # Convert models to list if present - elif key == "models": - auth_data[key] = [model.strip() for model in value.split(",")] - else: - auth_data[key] = value + if not value: + continue + if key == "max_budget": + auth_data[key] = float(value) + elif key == "models": + auth_data[key] = [model.strip() for model in value.split(",")] + else: + auth_data[key] = value + verbose_proxy_logger.debug( "Auth data before creating UserAPIKeyAuth object: keys=%s", list(auth_data.keys()), @@ -45,5 +116,4 @@ async def handle_oauth2_proxy_request(request: Request) -> UserAPIKeyAuth: "UserAPIKeyAuth object created with keys: %s", list(user_api_key_auth.__fields_set__), ) - # Create and return UserAPIKeyAuth object return user_api_key_auth diff --git a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py new file mode 100644 index 00000000000..87f8c45087e --- /dev/null +++ b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py @@ -0,0 +1,189 @@ +""" +Regression tests for the OAuth2-proxy header-forgery fix +(GHSA-5c3m-qffq-4r9m). + +The hook reads HTTP request headers per ``oauth2_config_mappings`` and +constructs a ``UserAPIKeyAuth`` from them. Two separate failure modes +the fix closes: + +1. The path was not gated on ``premium_user`` (the sibling + ``enable_oauth2_auth`` and ``enable_jwt_auth`` paths are). Open-source + deployments could enable the feature without realising it requires + a hardened deployment topology. +2. Any ``UserAPIKeyAuth`` field could be mapped from a header — including + ``user_role``, which Pydantic coerces from the string ``"proxy_admin"`` + into ``LitellmUserRoles.PROXY_ADMIN``. An attacker who reaches the + proxy directly (or via a misconfigured reverse proxy) sets the mapped + header and gains full admin privileges. +""" + +import os +import sys +from unittest.mock import patch + +import pytest +from fastapi import Request +from starlette.datastructures import Headers + +sys.path.insert(0, os.path.abspath("../../../..")) + +from litellm.proxy._types import LitellmUserRoles +from litellm.proxy.auth.oauth2_proxy_hook import ( + PRIVILEGED_OAUTH2_PROXY_FIELDS, + handle_oauth2_proxy_request, +) + + +def _request_with_headers(headers: dict) -> Request: + scope = { + "type": "http", + "headers": [(k.lower().encode(), v.encode()) for k, v in headers.items()], + } + request = Request(scope=scope) + request._headers = Headers(headers) + return request + + +@pytest.fixture +def premium_proxy_settings(monkeypatch): + """ + Patch the proxy_server module attributes the hook reads so each test + starts from "premium=True, mapping={user_id: x-user-id}". + """ + import litellm.proxy.proxy_server as proxy_server + + monkeypatch.setattr(proxy_server, "premium_user", True, raising=False) + monkeypatch.setattr( + proxy_server, + "general_settings", + {"oauth2_config_mappings": {"user_id": "x-user-id"}}, + raising=False, + ) + + +@pytest.mark.asyncio +async def test_returns_auth_for_simple_user_id_mapping(premium_proxy_settings): + request = _request_with_headers({"x-user-id": "alice"}) + + auth = await handle_oauth2_proxy_request(request) + + assert auth.user_id == "alice" + assert auth.user_role is None + + +@pytest.mark.asyncio +async def test_rejects_when_not_premium(monkeypatch): + import litellm.proxy.proxy_server as proxy_server + + monkeypatch.setattr(proxy_server, "premium_user", False, raising=False) + monkeypatch.setattr( + proxy_server, + "general_settings", + {"oauth2_config_mappings": {"user_id": "x-user-id"}}, + raising=False, + ) + request = _request_with_headers({"x-user-id": "alice"}) + + with pytest.raises(ValueError, match="enterprise"): + await handle_oauth2_proxy_request(request) + + +@pytest.mark.parametrize( + "privileged_field", + sorted(PRIVILEGED_OAUTH2_PROXY_FIELDS), +) +@pytest.mark.asyncio +async def test_refuses_to_map_privileged_fields(monkeypatch, privileged_field): + """ + The exact privesc shape from GHSA-5c3m-qffq-4r9m: an admin maps + ``user_role`` (or any other privileged field) to a header and a + caller forges ``X-User-Role: proxy_admin``. The hook must reject + this configuration outright at request time. + """ + import litellm.proxy.proxy_server as proxy_server + + monkeypatch.setattr(proxy_server, "premium_user", True, raising=False) + monkeypatch.setattr( + proxy_server, + "general_settings", + {"oauth2_config_mappings": {privileged_field: f"x-{privileged_field}"}}, + raising=False, + ) + request = _request_with_headers({f"x-{privileged_field}": "proxy_admin"}) + + with pytest.raises(ValueError) as exc: + await handle_oauth2_proxy_request(request) + assert privileged_field in str(exc.value) + + +@pytest.mark.asyncio +async def test_user_role_header_forgery_attack_is_blocked(monkeypatch): + """ + End-to-end shape from the GHSA: with ``user_role`` mapped, a forged + ``X-User-Role: proxy_admin`` header would have produced a + ``UserAPIKeyAuth`` with PROXY_ADMIN role. Now the request raises + before any auth object is constructed. + """ + import litellm.proxy.proxy_server as proxy_server + + monkeypatch.setattr(proxy_server, "premium_user", True, raising=False) + monkeypatch.setattr( + proxy_server, + "general_settings", + { + "oauth2_config_mappings": { + "user_id": "x-user-id", + "user_role": "x-user-role", + } + }, + raising=False, + ) + request = _request_with_headers( + { + "x-user-id": "attacker", + "x-user-role": LitellmUserRoles.PROXY_ADMIN.value, + } + ) + + with pytest.raises(ValueError, match="user_role"): + await handle_oauth2_proxy_request(request) + + +@pytest.mark.asyncio +async def test_safe_fields_still_pass_through(monkeypatch): + """ + Sanity check that non-privileged fields (the documented use case + for OAuth2 proxy auth — asserting identity from a trusted upstream) + still work after the fix. + """ + import litellm.proxy.proxy_server as proxy_server + + monkeypatch.setattr(proxy_server, "premium_user", True, raising=False) + monkeypatch.setattr( + proxy_server, + "general_settings", + { + "oauth2_config_mappings": { + "user_id": "x-user-id", + "user_email": "x-user-email", + "team_id": "x-team-id", + "models": "x-models", + } + }, + raising=False, + ) + request = _request_with_headers( + { + "x-user-id": "alice", + "x-user-email": "alice@example.com", + "x-team-id": "team-corp", + "x-models": "gpt-4, gpt-3.5-turbo", + } + ) + + auth = await handle_oauth2_proxy_request(request) + + assert auth.user_id == "alice" + assert auth.user_email == "alice@example.com" + assert auth.team_id == "team-corp" + assert auth.models == ["gpt-4", "gpt-3.5-turbo"] From e6867c143ae831ed9c6927034f4233048711820a Mon Sep 17 00:00:00 2001 From: user <70670632+stuxf@users.noreply.github.com> Date: Wed, 29 Apr 2026 22:23:29 +0000 Subject: [PATCH 024/196] =?UTF-8?q?chore(oauth2-proxy):=20/simplify=20pass?= =?UTF-8?q?=20=E2=80=94=20drop=20dead=20max=5Fbudget=20branch=20+=20DRY=20?= =?UTF-8?q?tests?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two cleanups from the /simplify review pass: * The header-mapping loop had a special-case ``if key == "max_budget": auth_data[key] = float(value)`` branch. Since ``max_budget`` is now in ``PRIVILEGED_OAUTH2_PROXY_FIELDS``, the denylist check rejects the configuration before the loop runs — the float-conversion branch is unreachable. Removed. * Four tests independently called ``monkeypatch.setattr(proxy_server, "premium_user", ...)`` and ``monkeypatch.setattr(proxy_server, "general_settings", ...)`` with almost-identical bodies. Replaced with a ``configure_proxy`` fixture that yields a single callable — ``configure_proxy(premium=False)`` / ``configure_proxy(mappings={...})`` — so each test's setup is one line. The previously-unused ``premium_proxy_settings`` fixture is removed. Co-Authored-By: Claude Opus 4.7 (1M context) --- litellm/proxy/auth/oauth2_proxy_hook.py | 4 +- .../proxy/auth/test_oauth2_proxy_hook.py | 119 +++++++----------- 2 files changed, 43 insertions(+), 80 deletions(-) diff --git a/litellm/proxy/auth/oauth2_proxy_hook.py b/litellm/proxy/auth/oauth2_proxy_hook.py index 341d2b477b0..0f7cfa42229 100644 --- a/litellm/proxy/auth/oauth2_proxy_hook.py +++ b/litellm/proxy/auth/oauth2_proxy_hook.py @@ -100,9 +100,7 @@ async def handle_oauth2_proxy_request(request: Request) -> UserAPIKeyAuth: value = request.headers.get(header) if not value: continue - if key == "max_budget": - auth_data[key] = float(value) - elif key == "models": + if key == "models": auth_data[key] = [model.strip() for model in value.split(",")] else: auth_data[key] = value diff --git a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py index 87f8c45087e..e51882a3b16 100644 --- a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py +++ b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py @@ -45,24 +45,32 @@ def _request_with_headers(headers: dict) -> Request: @pytest.fixture -def premium_proxy_settings(monkeypatch): +def configure_proxy(monkeypatch): """ - Patch the proxy_server module attributes the hook reads so each test - starts from "premium=True, mapping={user_id: x-user-id}". + Yields a callable that sets ``premium_user`` and + ``oauth2_config_mappings`` on the proxy_server module for the + duration of one test. Default is premium=True with a single + ``user_id -> x-user-id`` mapping. """ import litellm.proxy.proxy_server as proxy_server - monkeypatch.setattr(proxy_server, "premium_user", True, raising=False) - monkeypatch.setattr( - proxy_server, - "general_settings", - {"oauth2_config_mappings": {"user_id": "x-user-id"}}, - raising=False, - ) + def _configure(*, premium=True, mappings=None): + if mappings is None: + mappings = {"user_id": "x-user-id"} + monkeypatch.setattr(proxy_server, "premium_user", premium, raising=False) + monkeypatch.setattr( + proxy_server, + "general_settings", + {"oauth2_config_mappings": mappings}, + raising=False, + ) + + return _configure @pytest.mark.asyncio -async def test_returns_auth_for_simple_user_id_mapping(premium_proxy_settings): +async def test_returns_auth_for_simple_user_id_mapping(configure_proxy): + configure_proxy() request = _request_with_headers({"x-user-id": "alice"}) auth = await handle_oauth2_proxy_request(request) @@ -72,16 +80,8 @@ async def test_returns_auth_for_simple_user_id_mapping(premium_proxy_settings): @pytest.mark.asyncio -async def test_rejects_when_not_premium(monkeypatch): - import litellm.proxy.proxy_server as proxy_server - - monkeypatch.setattr(proxy_server, "premium_user", False, raising=False) - monkeypatch.setattr( - proxy_server, - "general_settings", - {"oauth2_config_mappings": {"user_id": "x-user-id"}}, - raising=False, - ) +async def test_rejects_when_not_premium(configure_proxy): + configure_proxy(premium=False) request = _request_with_headers({"x-user-id": "alice"}) with pytest.raises(ValueError, match="enterprise"): @@ -93,22 +93,11 @@ async def test_rejects_when_not_premium(monkeypatch): sorted(PRIVILEGED_OAUTH2_PROXY_FIELDS), ) @pytest.mark.asyncio -async def test_refuses_to_map_privileged_fields(monkeypatch, privileged_field): - """ - The exact privesc shape from GHSA-5c3m-qffq-4r9m: an admin maps - ``user_role`` (or any other privileged field) to a header and a - caller forges ``X-User-Role: proxy_admin``. The hook must reject - this configuration outright at request time. - """ - import litellm.proxy.proxy_server as proxy_server - - monkeypatch.setattr(proxy_server, "premium_user", True, raising=False) - monkeypatch.setattr( - proxy_server, - "general_settings", - {"oauth2_config_mappings": {privileged_field: f"x-{privileged_field}"}}, - raising=False, - ) +async def test_refuses_to_map_privileged_fields(configure_proxy, privileged_field): + # GHSA-5c3m-qffq-4r9m attack shape: admin maps a privileged field + # to a header and a caller forges the value. The hook must reject + # the misconfiguration outright at request time. + configure_proxy(mappings={privileged_field: f"x-{privileged_field}"}) request = _request_with_headers({f"x-{privileged_field}": "proxy_admin"}) with pytest.raises(ValueError) as exc: @@ -117,26 +106,13 @@ async def test_refuses_to_map_privileged_fields(monkeypatch, privileged_field): @pytest.mark.asyncio -async def test_user_role_header_forgery_attack_is_blocked(monkeypatch): - """ - End-to-end shape from the GHSA: with ``user_role`` mapped, a forged - ``X-User-Role: proxy_admin`` header would have produced a - ``UserAPIKeyAuth`` with PROXY_ADMIN role. Now the request raises - before any auth object is constructed. - """ - import litellm.proxy.proxy_server as proxy_server - - monkeypatch.setattr(proxy_server, "premium_user", True, raising=False) - monkeypatch.setattr( - proxy_server, - "general_settings", - { - "oauth2_config_mappings": { - "user_id": "x-user-id", - "user_role": "x-user-role", - } - }, - raising=False, +async def test_user_role_header_forgery_attack_is_blocked(configure_proxy): + # End-to-end form of the privesc: with ``user_role`` mapped, the + # forged ``X-User-Role: proxy_admin`` header would have produced + # a ``UserAPIKeyAuth(user_role=PROXY_ADMIN)``. Now rejected before + # any auth object is constructed. + configure_proxy( + mappings={"user_id": "x-user-id", "user_role": "x-user-role"}, ) request = _request_with_headers( { @@ -150,27 +126,16 @@ async def test_user_role_header_forgery_attack_is_blocked(monkeypatch): @pytest.mark.asyncio -async def test_safe_fields_still_pass_through(monkeypatch): - """ - Sanity check that non-privileged fields (the documented use case - for OAuth2 proxy auth — asserting identity from a trusted upstream) - still work after the fix. - """ - import litellm.proxy.proxy_server as proxy_server - - monkeypatch.setattr(proxy_server, "premium_user", True, raising=False) - monkeypatch.setattr( - proxy_server, - "general_settings", - { - "oauth2_config_mappings": { - "user_id": "x-user-id", - "user_email": "x-user-email", - "team_id": "x-team-id", - "models": "x-models", - } +async def test_safe_fields_still_pass_through(configure_proxy): + # The documented use case for OAuth2 proxy auth: identity assertion + # from a trusted upstream. Must remain unaffected by the denylist. + configure_proxy( + mappings={ + "user_id": "x-user-id", + "user_email": "x-user-email", + "team_id": "x-team-id", + "models": "x-models", }, - raising=False, ) request = _request_with_headers( { From b35287a062dbdd99eac053223f099200b12db9c0 Mon Sep 17 00:00:00 2001 From: user <70670632+stuxf@users.noreply.github.com> Date: Wed, 29 Apr 2026 22:28:57 +0000 Subject: [PATCH 025/196] fix(oauth2-proxy): switch privileged-field denylist to identity-only allowlist MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Greptile flagged that the denylist was incomplete: ``user_max_budget``, ``user_tpm_limit``, ``user_rpm_limit``, and ``user_spend`` were not on it. Inspection of the auth model showed dozens more privileged fields across the ``LiteLLM_VerificationTokenView`` hierarchy (team / org / end-user / region budget / spend / limit fields, plus ``allowed_model_region``, ``rpm_limit_per_model``, etc.) — a denylist of "privileged fields" is unmaintainable here. Inverted the model. ``ALLOWED_OAUTH2_PROXY_FIELDS`` is now an identity-only allowlist: ``user_id``, ``user_email``, ``team_id``, ``team_alias``, ``org_id``, ``models``. Any mapping to a non-identity field is rejected at request time. Default-secure: a future field added to ``UserAPIKeyAuth`` is automatically blocked from header-trust. Use case for OAuth2-proxy auth is identity assertion from a trusted upstream. Anything beyond that (privileges, budgets, rate limits) is policy and should be authenticated with a signature, not a header — operators who need this should switch to JWT auth. Tests: - ``test_refuses_to_map_non_identity_fields`` parametrized over 22 fields including all four ``user_*`` Greptile flagged, plus team/org/end-user budget/limit fields, plus a fabricated field name to confirm "anything not on the allowlist" is the rule. - ``test_allowlist_is_identity_only`` locks in the allowlist's intent so future additions of budget / role / permission entries are caught in review. Co-Authored-By: Claude Opus 4.7 (1M context) --- litellm/proxy/auth/oauth2_proxy_hook.py | 87 +++++++++---------- .../proxy/auth/test_oauth2_proxy_hook.py | 59 +++++++++++-- 2 files changed, 96 insertions(+), 50 deletions(-) diff --git a/litellm/proxy/auth/oauth2_proxy_hook.py b/litellm/proxy/auth/oauth2_proxy_hook.py index 0f7cfa42229..1ba1b100a86 100644 --- a/litellm/proxy/auth/oauth2_proxy_hook.py +++ b/litellm/proxy/auth/oauth2_proxy_hook.py @@ -5,36 +5,32 @@ from fastapi import Request from litellm._logging import verbose_proxy_logger from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth -# Fields on ``UserAPIKeyAuth`` that grant privileges directly (``user_role`` -# is the canonical privesc — coerced from the string ``"proxy_admin"`` into -# ``LitellmUserRoles.PROXY_ADMIN`` by Pydantic) or break trust assumptions -# (``api_key`` / ``token`` short-circuit the validated-key contract; -# ``permissions`` / ``allowed_routes`` directly grant route access; budget -# and limit fields can be set to wild values to bypass enforcement; -# ``metadata`` is too broad to safely admit from caller-controlled headers). +# OAuth2-proxy header trust is for **identity assertion** from a trusted +# upstream auth proxy (oauth2-proxy, Authelia, etc.). The allowlist below +# is the only safe surface — anything else (``user_role``, ``api_key``, +# ``permissions``, ``max_budget``, ``user_max_budget``, +# ``team_tpm_limit``, ``end_user_max_budget``, ``allowed_model_region``, +# and dozens of similar policy fields scattered across the +# ``LiteLLM_VerificationTokenView`` hierarchy) is a privilege grant that +# would let a caller forge their own enforcement parameters by sending +# the matching header. # -# Operators who legitimately need any of these to flow from a trusted -# upstream proxy should switch to JWT authentication, which validates a +# A denylist of "privileged fields" is unmaintainable in this codebase: +# the auth model has ~50 budget/spend/limit/permission fields and gains +# more with each release. An allowlist scoped to identity assertion is +# default-secure — new fields are blocked automatically. +# +# Operators who need a trusted upstream to assert anything beyond +# identity should switch to JWT authentication, which validates a # signature on the assertion rather than blindly trusting headers. -PRIVILEGED_OAUTH2_PROXY_FIELDS: FrozenSet[str] = frozenset( +ALLOWED_OAUTH2_PROXY_FIELDS: FrozenSet[str] = frozenset( { - "user_role", - "api_key", - "token", - "key_alias", - "key_name", - "permissions", - "allowed_routes", - "max_budget", - "spend", - "model_max_budget", - "model_spend", - "tpm_limit", - "rpm_limit", - "team_max_budget", - "team_spend", - "blocked", - "metadata", + "user_id", + "user_email", + "team_id", + "team_alias", + "org_id", + "models", } ) @@ -54,15 +50,15 @@ async def handle_oauth2_proxy_request(request: Request) -> UserAPIKeyAuth: previously did not, which let any open-source deployment turn the feature on without realising it requires a hardened deployment topology. - 2. **Privileged-field denylist.** ``oauth2_config_mappings`` maps - header names to ``UserAPIKeyAuth`` fields. Without a denylist, - an admin who maps the wrong header to ``user_role`` (or who - hasn't fully locked down their reverse proxy) lets any caller - set the ``user_role`` header to ``"proxy_admin"`` and gain full - admin privileges — Pydantic coerces the string into the enum. - Mapping any privileged field is rejected at startup-style auth - time so the misconfiguration surfaces loudly rather than as a - silent privesc. + 2. **Identity-only allowlist.** ``oauth2_config_mappings`` maps + header names to ``UserAPIKeyAuth`` fields. Without an allowlist, + an admin who maps the wrong header to ``user_role`` lets any + caller send ``X-User-Role: proxy_admin`` and gain full admin + privileges (Pydantic coerces the string into the enum). Only + fields in ``ALLOWED_OAUTH2_PROXY_FIELDS`` (identity assertion + only — see the constant's comment) may be mapped; any other + mapping is rejected at request time so the misconfiguration + surfaces loudly rather than as a silent privesc. """ from litellm.proxy.proxy_server import general_settings, premium_user @@ -81,17 +77,18 @@ async def handle_oauth2_proxy_request(request: Request) -> UserAPIKeyAuth: if not oauth2_config_mappings: raise ValueError("Oauth2 config mappings not found in general_settings") - privileged_mapped = sorted( - set(oauth2_config_mappings.keys()) & PRIVILEGED_OAUTH2_PROXY_FIELDS + disallowed = sorted( + set(oauth2_config_mappings.keys()) - ALLOWED_OAUTH2_PROXY_FIELDS ) - if privileged_mapped: + if disallowed: raise ValueError( - "Oauth2 proxy auth refuses to map privileged UserAPIKeyAuth " - f"fields from request headers: {privileged_mapped}. These " - "fields would grant privileges (e.g. proxy_admin), bypass " - "budget enforcement, or short-circuit key validation if a " - "caller can spoof the corresponding header. If you need a " - "trusted upstream to assert one of these, use JWT auth " + "Oauth2 proxy auth refuses to map non-identity UserAPIKeyAuth " + f"fields from request headers: {disallowed}. Only identity " + f"fields are accepted ({sorted(ALLOWED_OAUTH2_PROXY_FIELDS)}); " + "anything else (privileges, budgets, rate limits, metadata) " + "would let a caller forge enforcement parameters by spoofing " + "the matching header. If you need a trusted upstream to " + "assert anything beyond identity, use JWT auth " "(signature-validated) instead of header-trust." ) diff --git a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py index e51882a3b16..e73ac571d59 100644 --- a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py +++ b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py @@ -29,7 +29,7 @@ sys.path.insert(0, os.path.abspath("../../../..")) from litellm.proxy._types import LitellmUserRoles from litellm.proxy.auth.oauth2_proxy_hook import ( - PRIVILEGED_OAUTH2_PROXY_FIELDS, + ALLOWED_OAUTH2_PROXY_FIELDS, handle_oauth2_proxy_request, ) @@ -90,13 +90,46 @@ async def test_rejects_when_not_premium(configure_proxy): @pytest.mark.parametrize( "privileged_field", - sorted(PRIVILEGED_OAUTH2_PROXY_FIELDS), + [ + # The GHSA-5c3m-qffq-4r9m primary privesc field. + "user_role", + # Key-level enforcement bypass shapes. + "api_key", + "token", + "permissions", + "allowed_routes", + "max_budget", + "spend", + "tpm_limit", + "rpm_limit", + "model_max_budget", + "metadata", + # User-level enforcement bypass — flagged by Greptile as a denylist gap. + "user_max_budget", + "user_tpm_limit", + "user_rpm_limit", + "user_spend", + # Team / org / end-user / region — same class, all denied by the + # identity-only allowlist. + "team_max_budget", + "team_spend", + "team_member_tpm_limit", + "organization_max_budget", + "organization_tpm_limit", + "end_user_max_budget", + "allowed_model_region", + # Anything not on ALLOWED_OAUTH2_PROXY_FIELDS is blocked, even + # fabricated field names admins might try. + "definitely_not_a_real_field", + ], ) @pytest.mark.asyncio -async def test_refuses_to_map_privileged_fields(configure_proxy, privileged_field): +async def test_refuses_to_map_non_identity_fields(configure_proxy, privileged_field): # GHSA-5c3m-qffq-4r9m attack shape: admin maps a privileged field - # to a header and a caller forges the value. The hook must reject - # the misconfiguration outright at request time. + # to a header and a caller forges the value. The allowlist rejects + # any non-identity mapping at request time, regardless of whether + # the field ever appeared on a denylist — which is the whole reason + # we use an allowlist instead. configure_proxy(mappings={privileged_field: f"x-{privileged_field}"}) request = _request_with_headers({f"x-{privileged_field}": "proxy_admin"}) @@ -105,6 +138,22 @@ async def test_refuses_to_map_privileged_fields(configure_proxy, privileged_fiel assert privileged_field in str(exc.value) +@pytest.mark.parametrize("identity_field", sorted(ALLOWED_OAUTH2_PROXY_FIELDS)) +def test_allowlist_is_identity_only(identity_field): + # Lock in the allowlist's intent: only identity-assertion fields are + # safe to populate from a header. If anyone proposes adding budget / + # spend / role / permission to ``ALLOWED_OAUTH2_PROXY_FIELDS``, this + # assertion forces them to update the test deliberately. + assert identity_field in { + "user_id", + "user_email", + "team_id", + "team_alias", + "org_id", + "models", + } + + @pytest.mark.asyncio async def test_user_role_header_forgery_attack_is_blocked(configure_proxy): # End-to-end form of the privesc: with ``user_role`` mapped, the From fbcfd59b1a23edc17d6a93880726f26e308efedd Mon Sep 17 00:00:00 2001 From: user <70670632+stuxf@users.noreply.github.com> Date: Wed, 29 Apr 2026 23:04:05 +0000 Subject: [PATCH 026/196] fix(oauth2-proxy): drop premium gate; identity-only allowlist is the security fix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Greptile flagged the ``premium_user is not True`` check as a hard backwards-incompatible break for OSS users currently running ``enable_oauth2_proxy_auth=True``. They were right: unlike the api_base case (where the docs already required admin opt-in), this path was documented as available to OSS users. Adding the gate would have closed a documented feature, not fixed a vuln. Reframed the change: * The **identity-only allowlist** (``ALLOWED_OAUTH2_PROXY_FIELDS`` = ``{user_id, user_email, team_id, team_alias, org_id, models}``) is the actual security fix — it closes the privesc by rejecting any mapping to a non-identity field at request time. This is unchanged. * The **premium gate** was parity-with-siblings (a product decision, not a security one). Removed. BerriAI can re-add it on their own schedule with a proper deprecation cycle if they want enterprise- only gating. Tests: removed ``test_rejects_when_not_premium``; everything else (allowlist enforcement, identity passthrough, attack-shape regression) still passes — 14 tests. Co-Authored-By: Claude Opus 4.7 (1M context) --- litellm/proxy/auth/oauth2_proxy_hook.py | 36 +++++++------------ .../proxy/auth/test_oauth2_proxy_hook.py | 19 +++------- 2 files changed, 16 insertions(+), 39 deletions(-) diff --git a/litellm/proxy/auth/oauth2_proxy_hook.py b/litellm/proxy/auth/oauth2_proxy_hook.py index 1ba1b100a86..389a5b2b9e2 100644 --- a/litellm/proxy/auth/oauth2_proxy_hook.py +++ b/litellm/proxy/auth/oauth2_proxy_hook.py @@ -3,7 +3,7 @@ from typing import Any, Dict, FrozenSet from fastapi import Request from litellm._logging import verbose_proxy_logger -from litellm.proxy._types import CommonProxyErrors, UserAPIKeyAuth +from litellm.proxy._types import UserAPIKeyAuth # OAuth2-proxy header trust is for **identity assertion** from a trusted # upstream auth proxy (oauth2-proxy, Authelia, etc.). The allowlist below @@ -42,31 +42,19 @@ async def handle_oauth2_proxy_request(request: Request) -> UserAPIKeyAuth: The auth model assumes the proxy is deployed behind a trusted OAuth2 reverse proxy that injects authenticated identity headers (e.g. - oauth2-proxy, Authelia). Two safeguards above and beyond that - deployment assumption: + oauth2-proxy, Authelia). - 1. **Premium gate.** The sibling auth paths (``enable_oauth2_auth`` - and ``enable_jwt_auth``) require ``premium_user``; this path - previously did not, which let any open-source deployment turn - the feature on without realising it requires a hardened - deployment topology. - 2. **Identity-only allowlist.** ``oauth2_config_mappings`` maps - header names to ``UserAPIKeyAuth`` fields. Without an allowlist, - an admin who maps the wrong header to ``user_role`` lets any - caller send ``X-User-Role: proxy_admin`` and gain full admin - privileges (Pydantic coerces the string into the enum). Only - fields in ``ALLOWED_OAUTH2_PROXY_FIELDS`` (identity assertion - only — see the constant's comment) may be mapped; any other - mapping is rejected at request time so the misconfiguration - surfaces loudly rather than as a silent privesc. + **Identity-only allowlist.** ``oauth2_config_mappings`` maps header + names to ``UserAPIKeyAuth`` fields. Without an allowlist, an admin + who maps the wrong header to ``user_role`` lets any caller send + ``X-User-Role: proxy_admin`` and gain full admin privileges + (Pydantic coerces the string into the enum). Only fields in + ``ALLOWED_OAUTH2_PROXY_FIELDS`` (identity assertion only — see the + constant's comment) may be mapped; any other mapping is rejected at + request time so the misconfiguration surfaces loudly rather than as + a silent privesc. """ - from litellm.proxy.proxy_server import general_settings, premium_user - - if premium_user is not True: - raise ValueError( - "Oauth2 proxy auth is an enterprise-only feature. " - + CommonProxyErrors.not_premium_user.value - ) + from litellm.proxy.proxy_server import general_settings verbose_proxy_logger.debug("Handling oauth2 proxy request") oauth2_config_mappings: Dict[str, str] = ( diff --git a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py index e73ac571d59..42af9e6f030 100644 --- a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py +++ b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py @@ -47,17 +47,15 @@ def _request_with_headers(headers: dict) -> Request: @pytest.fixture def configure_proxy(monkeypatch): """ - Yields a callable that sets ``premium_user`` and - ``oauth2_config_mappings`` on the proxy_server module for the - duration of one test. Default is premium=True with a single - ``user_id -> x-user-id`` mapping. + Yields a callable that sets ``oauth2_config_mappings`` on the + proxy_server module for the duration of one test. Default mapping + is a single ``user_id -> x-user-id`` (identity-only). """ import litellm.proxy.proxy_server as proxy_server - def _configure(*, premium=True, mappings=None): + def _configure(*, mappings=None): if mappings is None: mappings = {"user_id": "x-user-id"} - monkeypatch.setattr(proxy_server, "premium_user", premium, raising=False) monkeypatch.setattr( proxy_server, "general_settings", @@ -79,15 +77,6 @@ async def test_returns_auth_for_simple_user_id_mapping(configure_proxy): assert auth.user_role is None -@pytest.mark.asyncio -async def test_rejects_when_not_premium(configure_proxy): - configure_proxy(premium=False) - request = _request_with_headers({"x-user-id": "alice"}) - - with pytest.raises(ValueError, match="enterprise"): - await handle_oauth2_proxy_request(request) - - @pytest.mark.parametrize( "privileged_field", [ From 722bc63e37d5a3773f95bb6c22b11f74bb3cb65e Mon Sep 17 00:00:00 2001 From: user <70670632+stuxf@users.noreply.github.com> Date: Wed, 29 Apr 2026 23:47:29 +0000 Subject: [PATCH 027/196] chore(oauth2-proxy): drop unused patch import + tighten docstring Greptile flagged the unused ``from unittest.mock import patch`` left over from before the ``configure_proxy`` fixture refactor (the fixture uses ``monkeypatch``, no ``patch`` calls remain). Also pruned the now-stale "premium gate" paragraph from the module docstring since that gate was removed in fbcfd59b1a. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../proxy/auth/test_oauth2_proxy_hook.py | 19 ++++++------------- 1 file changed, 6 insertions(+), 13 deletions(-) diff --git a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py index 42af9e6f030..9d0bdcf3513 100644 --- a/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py +++ b/tests/test_litellm/proxy/auth/test_oauth2_proxy_hook.py @@ -3,23 +3,16 @@ Regression tests for the OAuth2-proxy header-forgery fix (GHSA-5c3m-qffq-4r9m). The hook reads HTTP request headers per ``oauth2_config_mappings`` and -constructs a ``UserAPIKeyAuth`` from them. Two separate failure modes -the fix closes: - -1. The path was not gated on ``premium_user`` (the sibling - ``enable_oauth2_auth`` and ``enable_jwt_auth`` paths are). Open-source - deployments could enable the feature without realising it requires - a hardened deployment topology. -2. Any ``UserAPIKeyAuth`` field could be mapped from a header — including - ``user_role``, which Pydantic coerces from the string ``"proxy_admin"`` - into ``LitellmUserRoles.PROXY_ADMIN``. An attacker who reaches the - proxy directly (or via a misconfigured reverse proxy) sets the mapped - header and gains full admin privileges. +constructs a ``UserAPIKeyAuth`` from them. Without the +identity-only allowlist any field could be mapped — including +``user_role``, which Pydantic coerces from the string +``"proxy_admin"`` into ``LitellmUserRoles.PROXY_ADMIN``. An attacker +who reaches the proxy directly (or via a misconfigured reverse +proxy) sets the mapped header and gains full admin privileges. """ import os import sys -from unittest.mock import patch import pytest from fastapi import Request From b1ee9c4fc5375a62403ae46a3a2a38493891913b Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Wed, 29 Apr 2026 19:21:41 -0700 Subject: [PATCH 028/196] fix(rbac): restore admin-viewer read parity for Logs page + settings reads Admin Viewer (proxy_admin_viewer) was being blocked from endpoints it should be able to read. Most visibly the UI Logs page rendered empty because every filter and detail call (/spend/logs/ui, /spend/logs/ui/{id}, /spend/logs/session/ui, /customer/list) was rejected at the route_checks layer even though the underlying handlers permit admin-viewer. Backend: - Extend admin_viewer_routes to include spend_tracking_routes, /customer/{list,info}, /spend/logs/* detail routes, callback / config / budget / alerting reads, and model cost map status/source. - Replace bare `user_role != PROXY_ADMIN` checks in read-only handlers (/budget/list, /budget/settings, /alerting/settings, /invitation/info, /config/field/info, /config/list, /schedule/model_cost_map_reload/status, /model/cost_map/source) with `_user_has_admin_view()`. UI: - Add `rolesAllowedToViewWriteScopedPages` (rolesWithWriteAccess + Admin Viewer) and use it for the "Models + Endpoints" and "Agents" sidebar items so admin viewers see them read-only. Playground stays gated by rolesWithWriteAccess (cost-incurring). - Hide Add / Edit / Delete buttons in the LLM Credentials panel for non-proxy-admin viewers. Tests: - 31 parametrized route_checks cases for the Logs + settings endpoints, with internal-user negative coverage to ensure the gate isn't widened. - 9 handler-level integration tests (FastAPI TestClient) verifying admin viewer is no longer blocked at the handler layer. - New leftnav cases asserting Playground hidden / Models + Agents / Logs visible to Admin Viewer. - New roles + credentials test cases for the UI write-gate. --- litellm/proxy/_types.py | 64 +++++-- .../budget_management_endpoints.py | 5 +- litellm/proxy/proxy_server.py | 16 +- .../auth/test_admin_viewer_handler_access.py | 168 ++++++++++++++++++ .../proxy/auth/test_route_checks.py | 164 +++++++++++++++++ .../src/components/leftnav.test.tsx | 45 ++++- .../src/components/leftnav.tsx | 15 +- .../components/model_add/credentials.test.tsx | 67 ++++++- .../src/components/model_add/credentials.tsx | 45 +++-- ui/litellm-dashboard/src/utils/roles.test.ts | 32 +++- ui/litellm-dashboard/src/utils/roles.ts | 9 + 11 files changed, 579 insertions(+), 51 deletions(-) create mode 100644 tests/test_litellm/proxy/auth/test_admin_viewer_handler_access.py diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 92c920ca594..522bb47936d 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -717,20 +717,56 @@ class LiteLLMRoutes(enum.Enum): ] # Routes accessible by Admin Viewer (read-only admin access) - admin_viewer_routes = [ - "/user/list", - "/user/available_users", - "/user/available_roles", - "/user/daily/activity", - "/team/daily/activity", - "/tag/daily/activity", - "/tag/list", - "/audit", - "/audit/{id}", - "/global/activity", - "/global/activity/model", - "/global/activity/cache_hits", - ] + info_routes + # + # Admin Viewer follows a read-parity-with-Proxy-Admin rule: anything Proxy + # Admin can read/list/get, Admin Viewer can too (no writes, no cost-incurring + # actions). When extending this list, the only valid exclusions are write + # endpoints and cost-incurring endpoints (e.g. /chat/completions, the + # Playground). Pure GET/list/info endpoints belong here. + admin_viewer_routes = ( + [ + "/user/list", + "/user/available_users", + "/user/available_roles", + "/user/daily/activity", + "/team/daily/activity", + "/tag/daily/activity", + "/tag/list", + "/audit", + "/audit/{id}", + "/global/activity", + "/global/activity/model", + "/global/activity/cache_hits", + # Customer / end-user listing (handlers already gate on + # PROXY_ADMIN_VIEW_ONLY — the route gate must match). + "/customer/list", + "/customer/info", + # UI Logs page detail drawer (single + session). The list endpoint + # `/spend/logs/ui` is covered via spend_tracking_routes below. + "/spend/logs/ui/{logId}", + "/spend/logs/session/ui", + # Settings / observability read endpoints exposed in admin-only + # sidebar groups (Logging & Alerts, Admin Settings, Budgets, + # Invitations). + "/callbacks/list", + "/callbacks/configs", + "/get/config/callbacks", + "/alerting/settings", + "/config/list", + "/config/field/info", + "/budget/list", + "/budget/settings", + # Model cost map maintenance views (read-only status / source). + "/schedule/model_cost_map_reload/status", + "/model/cost_map/source", + ] + # Spend tracking reads (/spend/logs, /spend/logs/ui, /spend/keys, + # /spend/users, /spend/tags, /spend/calculate, /cost/estimate). Admin + # Viewer can already read /global/spend/* via global_spend_tracking_routes; + # the per-tenant /spend/* views were the missing peer. + + spend_tracking_routes + + info_routes + ) # All routes accesible by an Org Admin org_admin_allowed_routes = ( diff --git a/litellm/proxy/management_endpoints/budget_management_endpoints.py b/litellm/proxy/management_endpoints/budget_management_endpoints.py index 90c0d02d1e0..81b133e6c81 100644 --- a/litellm/proxy/management_endpoints/budget_management_endpoints.py +++ b/litellm/proxy/management_endpoints/budget_management_endpoints.py @@ -17,6 +17,7 @@ from fastapi import APIRouter, Depends, HTTPException from litellm.proxy.common_utils.timezone_utils import get_budget_reset_time from litellm.proxy._types import * from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.proxy.management_endpoints.common_utils import _user_has_admin_view from litellm.proxy.utils import jsonify_object router = APIRouter() @@ -238,7 +239,7 @@ async def budget_settings( detail={"error": CommonProxyErrors.db_not_connected_error.value}, ) - if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN: + if not _user_has_admin_view(user_api_key_dict): raise HTTPException( status_code=400, detail={ @@ -305,7 +306,7 @@ async def list_budget( detail={"error": CommonProxyErrors.db_not_connected_error.value}, ) - if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN: + if not _user_has_admin_view(user_api_key_dict): raise HTTPException( status_code=400, detail={ diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 870ea78f17a..6450a16df53 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -358,6 +358,7 @@ from litellm.proxy.management_endpoints.callback_management_endpoints import ( ) from litellm.proxy.management_endpoints.common_utils import ( _user_has_admin_privileges, + _user_has_admin_view, admin_can_invite_user, ) from litellm.proxy.management_endpoints.compliance_endpoints import ( @@ -11499,7 +11500,7 @@ async def alerting_settings( detail={"error": CommonProxyErrors.db_not_connected_error.value}, ) - if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN: + if not _user_has_admin_view(user_api_key_dict): raise HTTPException( status_code=400, detail={ @@ -12485,7 +12486,7 @@ async def invitation_info( detail={"error": CommonProxyErrors.db_not_connected_error.value}, ) - if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN: + if not _user_has_admin_view(user_api_key_dict): raise HTTPException( status_code=400, detail={ @@ -12917,7 +12918,7 @@ async def get_config_general_settings( detail={"error": CommonProxyErrors.db_not_connected_error.value}, ) - if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN: + if not _user_has_admin_view(user_api_key_dict): raise HTTPException( status_code=400, detail={"error": CommonProxyErrors.not_allowed_access.value}, @@ -12981,7 +12982,7 @@ async def get_config_list( detail={"error": CommonProxyErrors.db_not_connected_error.value}, ) - if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN: + if not _user_has_admin_view(user_api_key_dict): raise HTTPException( status_code=400, detail={ @@ -13693,8 +13694,8 @@ async def get_model_cost_map_reload_status( Get the status of the scheduled model cost map reload job. """ - # Check if user is admin - if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN: + # Read-only status check — admin viewers can read. + if not _user_has_admin_view(user_api_key_dict): raise HTTPException( status_code=403, detail=f"Access denied. Admin role required. Current role: {user_api_key_dict.user_role}", @@ -13796,7 +13797,8 @@ async def get_model_cost_map_source( - fallback_reason: human-readable reason why remote failed (null on success) - model_count: number of models in the currently loaded cost map """ - if user_api_key_dict.user_role != LitellmUserRoles.PROXY_ADMIN: + # Read-only source info — admin viewers can read. + if not _user_has_admin_view(user_api_key_dict): raise HTTPException( status_code=403, detail=f"Access denied. Admin role required. Current role: {user_api_key_dict.user_role}", diff --git a/tests/test_litellm/proxy/auth/test_admin_viewer_handler_access.py b/tests/test_litellm/proxy/auth/test_admin_viewer_handler_access.py new file mode 100644 index 00000000000..b0d2595e48c --- /dev/null +++ b/tests/test_litellm/proxy/auth/test_admin_viewer_handler_access.py @@ -0,0 +1,168 @@ +""" +Handler-level admin viewer parity tests. + +These tests assert that PROXY_ADMIN_VIEW_ONLY callers are NOT blocked at the +handler level for read-only admin endpoints. The route_checks layer is tested +separately in `test_route_checks.py`; here we verify each individual endpoint +function has been updated to use `_user_has_admin_view()` rather than a bare +`user_role != PROXY_ADMIN` check. + +The principle (see Admin Viewer role doc): anything Proxy Admin can read, +Admin Viewer can read. No writes, no cost-incurring actions. +""" + +import os +import sys +import types +from unittest.mock import AsyncMock, MagicMock + +import pytest +from fastapi.testclient import TestClient + +sys.path.insert(0, os.path.abspath("../../../")) + +import litellm.proxy.proxy_server as ps +from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth +from litellm.proxy.proxy_server import app + + +def _make_admin_viewer_auth() -> UserAPIKeyAuth: + return UserAPIKeyAuth( + user_id="viewer_user", + user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, + ) + + +def _override_auth(role: LitellmUserRoles) -> None: + fake_user = UserAPIKeyAuth(user_id="viewer_user", user_role=role) + app.dependency_overrides[ps.user_api_key_auth] = lambda: fake_user + + +def _clear_overrides() -> None: + app.dependency_overrides.clear() + + +@pytest.fixture +def admin_viewer_client(monkeypatch): + """TestClient where auth always returns PROXY_ADMIN_VIEW_ONLY + a mocked Prisma.""" + mock_prisma = MagicMock() + + # Common DB tables touched by the read endpoints under test. + mock_budget_table = MagicMock() + mock_budget_table.find_many = AsyncMock(return_value=[]) + mock_budget_table.find_first = AsyncMock(return_value=None) + + mock_invitation_table = MagicMock() + mock_invitation_table.find_unique = AsyncMock(return_value=None) + + mock_config_table = MagicMock() + mock_config_table.find_first = AsyncMock(return_value=None) + + mock_prisma.db = types.SimpleNamespace( + litellm_budgettable=mock_budget_table, + litellm_invitationlink=mock_invitation_table, + litellm_config=mock_config_table, + ) + + monkeypatch.setattr(ps, "prisma_client", mock_prisma) + _override_auth(LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY) + + yield TestClient(app) + + _clear_overrides() + + +def _assert_not_role_blocked(response) -> None: + """The endpoint must not return a role-block error. + + Detects both the 400 ``not_allowed_access`` pattern (used by most + management endpoints) and the 403 ``Admin role required`` pattern + (used by model cost map endpoints). + """ + if response.status_code in (400, 401, 403): + body = response.json() + detail = body.get("detail", body) + if isinstance(detail, dict): + err = detail.get("error", "") + else: + err = str(detail) + err_lower = err.lower() + role_block_signals = ( + "your role=", + "not allowed to access", + "admin role required", + "admin-only endpoint", + ) + for signal in role_block_signals: + assert ( + signal not in err_lower + ), f"endpoint blocked PROXY_ADMIN_VIEW_ONLY at handler level: {err}" + + +def test_budget_list_allows_admin_viewer(admin_viewer_client): + """`/budget/list` is read-only and must be accessible to Admin Viewer.""" + resp = admin_viewer_client.get("/budget/list") + _assert_not_role_blocked(resp) + assert resp.status_code == 200, resp.text + + +def test_budget_settings_allows_admin_viewer(admin_viewer_client): + """`/budget/settings` describes a budget's fields; read-only.""" + resp = admin_viewer_client.get("/budget/settings", params={"budget_id": "b1"}) + _assert_not_role_blocked(resp) + assert resp.status_code == 200, resp.text + + +def test_alerting_settings_allows_admin_viewer(admin_viewer_client): + """`/alerting/settings` describes alerting params; read-only.""" + resp = admin_viewer_client.get("/alerting/settings") + _assert_not_role_blocked(resp) + # Endpoint may 400 for *config* reasons (no proxy config loaded), but it + # must not 400 because of role. + assert resp.status_code != 403, resp.text + + +def test_get_config_field_info_allows_admin_viewer(admin_viewer_client): + """`/config/field/info` describes a single general-settings field; read-only.""" + resp = admin_viewer_client.get( + "/config/field/info", params={"field_name": "alerting"} + ) + _assert_not_role_blocked(resp) + + +def test_get_config_list_allows_admin_viewer(admin_viewer_client): + """`/config/list` lists configurable params for a config_type; read-only.""" + resp = admin_viewer_client.get( + "/config/list", params={"config_type": "general_settings"} + ) + _assert_not_role_blocked(resp) + + +def test_get_config_callbacks_allows_admin_viewer(admin_viewer_client): + """`/get/config/callbacks` lists current callbacks; read-only.""" + resp = admin_viewer_client.get("/get/config/callbacks") + _assert_not_role_blocked(resp) + + +def test_invitation_info_allows_admin_viewer(admin_viewer_client): + """`/invitation/info` reads a single invitation; read-only. + + The invitation lookup will return 400 because no invitation exists in our + mock DB — that's fine. We only assert it doesn't hit the role-block path. + """ + resp = admin_viewer_client.get( + "/invitation/info", params={"invitation_id": "nonexistent"} + ) + _assert_not_role_blocked(resp) + + +def test_model_cost_map_reload_status_allows_admin_viewer(admin_viewer_client): + """`/schedule/model_cost_map_reload/status` is read-only operations status.""" + resp = admin_viewer_client.get("/schedule/model_cost_map_reload/status") + _assert_not_role_blocked(resp) + + +def test_model_cost_map_source_allows_admin_viewer(admin_viewer_client): + """`/model/cost_map/source` reads the configured cost map source URL.""" + resp = admin_viewer_client.get("/model/cost_map/source") + _assert_not_role_blocked(resp) diff --git a/tests/test_litellm/proxy/auth/test_route_checks.py b/tests/test_litellm/proxy/auth/test_route_checks.py index a5d405cfc2d..4c8832dee60 100644 --- a/tests/test_litellm/proxy/auth/test_route_checks.py +++ b/tests/test_litellm/proxy/auth/test_route_checks.py @@ -1198,6 +1198,170 @@ def test_proxy_admin_viewer_can_access_audit_logs(route): ) +# ── Admin Viewer parity: Logs page endpoints ────────────────────────────────── +# +# The Admin Viewer (PROXY_ADMIN_VIEW_ONLY) role is documented as +# "view all keys, view all spend" and follows a read-parity-with-Proxy-Admin +# rule. The UI Logs page is the most user-visible failure mode: filtering and +# log details break entirely when these routes are blocked at the route_checks +# layer, even though the underlying handlers already gate on PROXY_ADMIN_VIEW_ONLY. +# +# Each route below corresponds to a network call made by the Logs page +# (ui/litellm-dashboard/src/components/view_logs/) — see the comment on each. +ADMIN_VIEWER_LOGS_PAGE_ROUTES = [ + # Main paginated log list — uiSpendLogsCall in log_filter_logic.tsx & index.tsx + "/spend/logs/ui", + # Single-log detail drawer — fetched on row click in LogDetailsDrawer + "/spend/logs/ui/abc-request-id", + # Multi-call session drawer — sessionSpendLogsCall in LogDetailsDrawer + "/spend/logs/session/ui", + # End User filter dropdown — allEndUsersCall in index.tsx + "/customer/list", + "/customer/info", + # Cost estimation — used by some log views + "/cost/estimate", + # Public spend logs / spend tracking routes that admin viewer should read + "/spend/logs", + "/spend/keys", + "/spend/users", + "/spend/tags", + "/spend/calculate", +] + + +@pytest.mark.parametrize("route", ADMIN_VIEWER_LOGS_PAGE_ROUTES) +def test_proxy_admin_viewer_can_access_logs_page_endpoints(route): + """ + PROXY_ADMIN_VIEW_ONLY must pass route_checks for every endpoint the UI + Logs page depends on. Without these, the page renders empty / errors. + """ + user_obj = LiteLLM_UserTable( + user_id="viewer_user", + user_email="viewer@example.com", + user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value, + ) + valid_token = UserAPIKeyAuth( + user_id="viewer_user", + user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value, + ) + request = MagicMock(spec=Request) + request.query_params = {} + + try: + RouteChecks.non_proxy_admin_allowed_routes_check( + user_obj=user_obj, + _user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value, + route=route, + request=request, + valid_token=valid_token, + request_data={}, + ) + except Exception as e: + pytest.fail( + f"proxy_admin_viewer should be able to access {route}. Got error: {str(e)}" + ) + + +@pytest.mark.parametrize("route", ADMIN_VIEWER_LOGS_PAGE_ROUTES) +def test_internal_user_blocked_from_admin_viewer_logs_routes(route): + """ + The Logs-page route opening above must NOT also widen access for + INTERNAL_USER. Plain internal users still see only their own logs and + must be blocked from proxy-wide spend tracking + customer routes. + """ + user_obj = LiteLLM_UserTable( + user_id="internal_user", + user_email="user@example.com", + user_role=LitellmUserRoles.INTERNAL_USER.value, + ) + valid_token = UserAPIKeyAuth( + user_id="internal_user", + user_role=LitellmUserRoles.INTERNAL_USER.value, + ) + request = MagicMock(spec=Request) + request.query_params = {} + + # Routes already in `spend_tracking_routes` (which is part of + # `internal_user_routes`) are intentionally accessible to internal users + # for their own scoped spend — those handlers enforce per-user filtering. + # /cost/estimate is similarly per-user. The /customer/* routes are + # admin-only. + INTERNAL_USER_BLOCKED_SUBSET = { + "/customer/list", + "/customer/info", + } + if route not in INTERNAL_USER_BLOCKED_SUBSET: + return + + with pytest.raises(Exception) as exc_info: + RouteChecks.non_proxy_admin_allowed_routes_check( + user_obj=user_obj, + _user_role=LitellmUserRoles.INTERNAL_USER.value, + route=route, + request=request, + valid_token=valid_token, + request_data={}, + ) + assert "Only proxy admin" in str(exc_info.value) + + +# ── Admin Viewer parity: Settings/observability read endpoints ──────────────── +# +# These are GET endpoints accessible to PROXY_ADMIN that the UI exposes to +# admin viewers via sidebar items gated by `all_admin_roles` (which includes +# proxy_admin_viewer). Without these, the Logging & Alerts, Caching, Budgets, +# and Admin Settings pages break for admin viewers. +ADMIN_VIEWER_SETTINGS_ROUTES = [ + # Logging & Alerts page + "/callbacks/list", + "/callbacks/configs", + "/get/config/callbacks", + "/alerting/settings", + # Admin Settings / Router Settings pages + "/config/list", + "/config/field/info", + # Budgets page + "/budget/list", + "/budget/settings", + # Model cost map (read-only status / source) + "/schedule/model_cost_map_reload/status", + "/model/cost_map/source", +] + + +@pytest.mark.parametrize("route", ADMIN_VIEWER_SETTINGS_ROUTES) +def test_proxy_admin_viewer_can_access_settings_read_endpoints(route): + """ + PROXY_ADMIN_VIEW_ONLY must pass route_checks for the read-only + settings/observability endpoints exposed in admin-only sidebar groups. + """ + user_obj = LiteLLM_UserTable( + user_id="viewer_user", + user_email="viewer@example.com", + user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value, + ) + valid_token = UserAPIKeyAuth( + user_id="viewer_user", + user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value, + ) + request = MagicMock(spec=Request) + request.query_params = {} + + try: + RouteChecks.non_proxy_admin_allowed_routes_check( + user_obj=user_obj, + _user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY.value, + route=route, + request=request, + valid_token=valid_token, + request_data={}, + ) + except Exception as e: + pytest.fail( + f"proxy_admin_viewer should be able to access {route}. Got error: {str(e)}" + ) + + class TestModelsRouteExemptFromDisableLLMEndpoints: """ Test that /models and /v1/models are exempt from DISABLE_LLM_API_ENDPOINTS. diff --git a/ui/litellm-dashboard/src/components/leftnav.test.tsx b/ui/litellm-dashboard/src/components/leftnav.test.tsx index c168ffa21e0..b8f30084127 100644 --- a/ui/litellm-dashboard/src/components/leftnav.test.tsx +++ b/ui/litellm-dashboard/src/components/leftnav.test.tsx @@ -5,10 +5,11 @@ import Sidebar from "./leftnav"; vi.mock("../utils/roles", () => { return { - all_admin_roles: ["admin"], + all_admin_roles: ["admin", "admin_viewer"], internalUserRoles: ["internal"], rolesWithWriteAccess: ["admin", "internal"], - isAdminRole: (role: string) => role === "admin", + rolesAllowedToViewWriteScopedPages: ["admin", "internal", "admin_viewer"], + isAdminRole: (role: string) => role === "admin" || role === "admin_viewer", isUserTeamAdminForAnyTeam: () => false, }; }); @@ -135,6 +136,46 @@ describe("Sidebar (leftnav)", () => { expect(duplicates).toHaveLength(0); }); + describe("Admin Viewer parity", () => { + // Admin Viewer follows a "read parity with Proxy Admin, no writes, no + // cost-incurring actions" rule. Playground stays hidden (incurs LLM + // cost); Models + Endpoints and Agents must be visible read-only. + const adminViewerAuth = { + userId: "admin-viewer-user-id", + accessToken: "test-access-token", + userRole: "admin_viewer", + token: "test-token", + userEmail: "viewer@example.com", + premiumUser: false, + disabledPersonalKeyCreation: false, + showSSOBanner: false, + }; + + it("hides Playground from Admin Viewer (cost-incurring action)", () => { + mockUseAuthorized.mockReturnValueOnce(adminViewerAuth); + renderWithProviders(); + expect(screen.queryByText("Playground")).not.toBeInTheDocument(); + }); + + it("shows Models + Endpoints to Admin Viewer (read-only)", () => { + mockUseAuthorized.mockReturnValueOnce(adminViewerAuth); + renderWithProviders(); + expect(screen.getByText("Models + Endpoints")).toBeInTheDocument(); + }); + + it("shows Agents to Admin Viewer (read-only)", () => { + mockUseAuthorized.mockReturnValueOnce(adminViewerAuth); + renderWithProviders(); + expect(screen.getByText("Agents")).toBeInTheDocument(); + }); + + it("shows Logs to Admin Viewer", () => { + mockUseAuthorized.mockReturnValueOnce(adminViewerAuth); + renderWithProviders(); + expect(screen.getByText("Logs")).toBeInTheDocument(); + }); + }); + it("should show Organizations tab for organization admins", () => { mockUseAuthorized.mockReturnValueOnce({ userId: "org-admin-user-id", diff --git a/ui/litellm-dashboard/src/components/leftnav.tsx b/ui/litellm-dashboard/src/components/leftnav.tsx index c340b65496c..029bdc49b5d 100644 --- a/ui/litellm-dashboard/src/components/leftnav.tsx +++ b/ui/litellm-dashboard/src/components/leftnav.tsx @@ -31,7 +31,14 @@ import { import type { MenuProps } from "antd"; import { ConfigProvider, Layout, Menu } from "antd"; import { useMemo } from "react"; -import { all_admin_roles, internalUserRoles, isAdminRole, isUserTeamAdminForAnyTeam, rolesWithWriteAccess } from "../utils/roles"; +import { + all_admin_roles, + internalUserRoles, + isAdminRole, + isUserTeamAdminForAnyTeam, + rolesAllowedToViewWriteScopedPages, + rolesWithWriteAccess, +} from "../utils/roles"; import NewBadge from "./common_components/NewBadge"; import type { Organization } from "./networking"; import UsageIndicator from "./UsageIndicator"; @@ -117,14 +124,16 @@ const menuGroups: MenuGroup[] = [ page: "models", label: "Models + Endpoints", icon: , - roles: rolesWithWriteAccess, + // Admin Viewer can view models read-only (write actions are + // hidden inside the page); Playground above stays write-only. + roles: rolesAllowedToViewWriteScopedPages, }, { key: "agents", page: "agents", label: "Agents", icon: , - roles: rolesWithWriteAccess, + roles: rolesAllowedToViewWriteScopedPages, }, { key: "mcp-servers", diff --git a/ui/litellm-dashboard/src/components/model_add/credentials.test.tsx b/ui/litellm-dashboard/src/components/model_add/credentials.test.tsx index 2504b2a9789..6d555621ca8 100644 --- a/ui/litellm-dashboard/src/components/model_add/credentials.test.tsx +++ b/ui/litellm-dashboard/src/components/model_add/credentials.test.tsx @@ -30,7 +30,7 @@ const createQueryClient = () => describe("CredentialsPanel", () => { it("should render", () => { - mockUseAuthorized.mockReturnValue({ accessToken: "test-token" }); + mockUseAuthorized.mockReturnValue({ accessToken: "test-token", userRole: "Admin" }); mockUseCredentials.mockReturnValue({ data: { credentials: [] }, refetch: vi.fn(), @@ -54,7 +54,7 @@ describe("CredentialsPanel", () => { }, ]; - mockUseAuthorized.mockReturnValue({ accessToken: "test-token" }); + mockUseAuthorized.mockReturnValue({ accessToken: "test-token", userRole: "Admin" }); mockUseCredentials.mockReturnValue({ data: { credentials }, refetch: vi.fn(), @@ -70,7 +70,7 @@ describe("CredentialsPanel", () => { }); it("should display empty state when no credentials are provided", () => { - mockUseAuthorized.mockReturnValue({ accessToken: "test-token" }); + mockUseAuthorized.mockReturnValue({ accessToken: "test-token", userRole: "Admin" }); mockUseCredentials.mockReturnValue({ data: { credentials: [] }, refetch: vi.fn(), @@ -86,7 +86,7 @@ describe("CredentialsPanel", () => { }); it("should open add modal when add button is clicked", async () => { - mockUseAuthorized.mockReturnValue({ accessToken: "test-token" }); + mockUseAuthorized.mockReturnValue({ accessToken: "test-token", userRole: "Admin" }); mockUseCredentials.mockReturnValue({ data: { credentials: [] }, refetch: vi.fn(), @@ -108,4 +108,63 @@ describe("CredentialsPanel", () => { expect(screen.getByText("Add New Credential")).toBeInTheDocument(); }); }); + + describe("Admin Viewer write-action gating", () => { + // Admin Viewer can VIEW credentials but must not be able to add / edit / + // delete them. The page shows the credential list read-only. + const credentials: CredentialItem[] = [ + { + credential_name: "openai-key", + credential_values: {}, + credential_info: { custom_llm_provider: "openai" }, + }, + ]; + + it("hides the Add Credential button for Admin Viewer", () => { + mockUseAuthorized.mockReturnValue({ + accessToken: "test-token", + userRole: "Admin Viewer", + }); + mockUseCredentials.mockReturnValue({ + data: { credentials }, + refetch: vi.fn(), + }); + + render( + + + , + ); + + // Credential row still renders (read parity). + expect(screen.getByText("openai-key")).toBeInTheDocument(); + // But no Add Credential button (write blocked). + expect( + screen.queryByRole("button", { name: /add credential/i }), + ).not.toBeInTheDocument(); + }); + + it("hides Edit / Delete buttons on existing credentials for Admin Viewer", () => { + mockUseAuthorized.mockReturnValue({ + accessToken: "test-token", + userRole: "Admin Viewer", + }); + mockUseCredentials.mockReturnValue({ + data: { credentials }, + refetch: vi.fn(), + }); + + const { container } = render( + + + , + ); + + // The Actions cell should be empty (no edit/delete buttons rendered). + // We rely on the row being visible but containing no ` + {canModifyCredentials && ( + + )}
Configured credentials for different AI providers. Add and manage your API credentials.
@@ -166,22 +171,26 @@ const CredentialsPanel: React.FC = ({ uploadProps }) => { {renderProviderBadge((credential.credential_info?.custom_llm_provider as string) || "-")} - @@ -470,7 +474,7 @@ const ModelHubTable: React.FC = ({ accessToken, publicPage, {/* Header with Make Public Button */} - {publicPage == false && isAdminRole(userRole || "") && ( + {publicPage == false && canModify && (
@@ -496,7 +500,7 @@ const ModelHubTable: React.FC = ({ accessToken, publicPage, {/* Header with Make Public Button */} - {publicPage == false && isAdminRole(userRole || "") && ( + {publicPage == false && canModify && (
@@ -520,7 +524,7 @@ const ModelHubTable: React.FC = ({ accessToken, publicPage, {/* Skill Hub Tab */} - {publicPage == false && isAdminRole(userRole || "") && ( + {publicPage == false && canModify && (
+ {canModify && ( + + )} diff --git a/ui/litellm-dashboard/src/components/Settings/RouterSettings/Fallbacks/Fallbacks.tsx b/ui/litellm-dashboard/src/components/Settings/RouterSettings/Fallbacks/Fallbacks.tsx index 9c5933aba3a..b4ab7ddddba 100644 --- a/ui/litellm-dashboard/src/components/Settings/RouterSettings/Fallbacks/Fallbacks.tsx +++ b/ui/litellm-dashboard/src/components/Settings/RouterSettings/Fallbacks/Fallbacks.tsx @@ -8,6 +8,7 @@ import DeleteResourceModal from "../../../common_components/DeleteResourceModal" import { ProviderLogo } from "../../../molecules/models/ProviderLogo"; import NotificationsManager from "../../../molecules/notifications_manager"; import { getCallbacksCall, setCallbacksCall } from "../../../networking"; +import { isProxyAdminRole } from "@/utils/roles"; import AddFallbacks from "./AddFallbacks"; type FallbackEntry = { [modelName: string]: string[] }; @@ -243,15 +244,19 @@ const Fallbacks: React.FC = ({ accessToken, userRole, userID, mo }; const hasFallbacks = Array.isArray(routerSettings.fallbacks) && routerSettings.fallbacks.length > 0; + // Admin Viewer follows the read-parity rule: see fallbacks, no writes. + const canModify = isProxyAdminRole(userRole ?? ""); return ( <> - data.model_name) : []} - accessToken={accessToken || ""} - value={routerSettings.fallbacks || []} - onChange={handleFallbacksChange} - /> + {canModify && ( + data.model_name) : []} + accessToken={accessToken || ""} + value={routerSettings.fallbacks || []} + onChange={handleFallbacksChange} + /> + )} {!hasFallbacks ? (
@@ -280,30 +285,34 @@ const Fallbacks: React.FC = ({ accessToken, userRole, userID, mo {renderFallbacksChain(key, Array.isArray(value) ? value : [], getProviderFromModel)} - - testFallbackModelResponse(Object.keys(item)[0], accessToken || "")} - className="cursor-pointer hover:text-blue-600" - /> - - - handleDeleteClick(item)} - onKeyDown={(e) => e.key === "Enter" && handleDeleteClick(item)} - className="cursor-pointer inline-flex" - > - - - + {canModify && ( + <> + + testFallbackModelResponse(Object.keys(item)[0], accessToken || "")} + className="cursor-pointer hover:text-blue-600" + /> + + + handleDeleteClick(item)} + onKeyDown={(e) => e.key === "Enter" && handleDeleteClick(item)} + className="cursor-pointer inline-flex" + > + + + + + )} )), diff --git a/ui/litellm-dashboard/src/components/budgets/budget_panel.tsx b/ui/litellm-dashboard/src/components/budgets/budget_panel.tsx index e42d0569652..d90737b130e 100644 --- a/ui/litellm-dashboard/src/components/budgets/budget_panel.tsx +++ b/ui/litellm-dashboard/src/components/budgets/budget_panel.tsx @@ -28,6 +28,8 @@ import { useBudgets, useDeleteBudget } from "@/app/(dashboard)/hooks/budgets/use import BudgetModal from "./budget_modal"; import EditBudgetModal from "./edit_budget_modal"; import { CREATE_END_USER_CURL_COMMAND, CHAT_COMPLETIONS_CURL_COMMAND, OPENAI_SDK_PYTHON_CODE } from "./constants"; +import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized"; +import { isProxyAdminRole } from "@/utils/roles"; interface BudgetSettingsPageProps { accessToken: string | null; @@ -47,6 +49,10 @@ const BudgetPanel: React.FC = ({ accessToken }) => { const [selectedBudget, setSelectedBudget] = useState(null); const [isDeleteModalVisible, setIsDeleteModalVisible] = useState(false); + const { userRole } = useAuthorized(); + // Admin Viewer follows the read-parity rule: see budgets, no writes. + const canModify = isProxyAdminRole(userRole ?? ""); + const { data: budgetList = [] } = useBudgets(); const deleteBudget = useDeleteBudget(); @@ -89,9 +95,11 @@ const BudgetPanel: React.FC = ({ accessToken }) => { return (
- + {canModify && ( + + )} Budgets @@ -133,18 +141,22 @@ const BudgetPanel: React.FC = ({ accessToken }) => { {value.max_budget ? value.max_budget : "n/a"} {value.tpm_limit ? value.tpm_limit : "n/a"} {value.rpm_limit ? value.rpm_limit : "n/a"} - handleEditCall(value)} - dataTestId="edit-budget-button" - /> - handleDeleteClick(value)} - dataTestId="delete-budget-button" - /> + {canModify && ( + <> + handleEditCall(value)} + dataTestId="edit-budget-button" + /> + handleDeleteClick(value)} + dataTestId="delete-budget-button" + /> + + )} ))} diff --git a/ui/litellm-dashboard/src/components/prompts.tsx b/ui/litellm-dashboard/src/components/prompts.tsx index 8e2e9c8b112..1e0155a7738 100644 --- a/ui/litellm-dashboard/src/components/prompts.tsx +++ b/ui/litellm-dashboard/src/components/prompts.tsx @@ -8,7 +8,7 @@ import PromptInfoView from "./prompts/prompt_info"; import AddPromptForm from "./prompts/add_prompt_form"; import PromptEditorView from "./prompts/prompt_editor_view"; import NotificationsManager from "./molecules/notifications_manager"; -import { isAdminRole } from "@/utils/roles"; +import { isAdminRole, isProxyAdminRole } from "@/utils/roles"; interface PromptsProps { accessToken: string | null; @@ -27,6 +27,8 @@ const PromptsPanel: React.FC = ({ accessToken, userRole }) => { const [promptToDelete, setPromptToDelete] = useState<{ id: string; name: string } | null>(null); const isAdmin = userRole ? isAdminRole(userRole) : false; + // Admin Viewer follows the read-parity rule: see prompts, no writes. + const canModify = userRole ? isProxyAdminRole(userRole) : false; const fetchPrompts = async () => { if (!accessToken) { @@ -128,7 +130,7 @@ const PromptsPanel: React.FC = ({ accessToken, userRole }) => { promptId={selectedPromptId} onClose={() => setSelectedPromptId(null)} accessToken={accessToken} - isAdmin={isAdmin} + isAdmin={canModify} onDelete={fetchPrompts} onEdit={handleEditPrompt} /> @@ -136,12 +138,16 @@ const PromptsPanel: React.FC = ({ accessToken, userRole }) => { <>
- - + {canModify && ( + <> + + + + )}