From 10bee3ef974c7ea8d0864f414c4e760fdc598314 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:59:06 -0700 Subject: [PATCH 01/96] feat(cost-map): add vertex ai llama 3.3 70b, veo 2/3, virtual try-on and 2.5 tts rows (#42837) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 100 ++++++++++++++++++ model_prices_and_context_window.json | 100 ++++++++++++++++++ 2 files changed, 200 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ebea1a6044d..844e613977f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -74121,5 +74121,105 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true + }, + "vertex_ai/meta/llama-3.3-70b-instruct-maas": { + "input_cost_per_token": 7.2e-07, + "input_cost_per_token_batches": 3.6e-07, + "litellm_provider": "vertex_ai-llama_models", + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 7.2e-07, + "output_cost_per_token_batches": 3.6e-07, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "text", + "code" + ], + "supports_function_calling": true, + "supports_tool_choice": true + }, + "vertex_ai/veo-3.0-generate-001": { + "deprecation_date": "2026-06-30", + "litellm_provider": "vertex_ai-video-models", + "max_input_tokens": 1024, + "max_tokens": 1024, + "mode": "video_generation", + "output_cost_per_second": 0.4, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#veo", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "vertex_ai/veo-3.0-fast-generate-001": { + "deprecation_date": "2026-06-30", + "litellm_provider": "vertex_ai-video-models", + "max_input_tokens": 1024, + "max_tokens": 1024, + "mode": "video_generation", + "output_cost_per_second": 0.1, + "output_cost_per_second_1080p": 0.12, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#veo", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "vertex_ai/veo-2.0-generate-001": { + "litellm_provider": "vertex_ai-video-models", + "max_input_tokens": 1024, + "max_tokens": 1024, + "mode": "video_generation", + "output_cost_per_second": 0.5, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#veo", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "vertex_ai/virtual-try-on-001": { + "deprecation_date": "2027-03-15", + "litellm_provider": "vertex_ai", + "mode": "image_generation", + "output_cost_per_image": 0.06, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", + "supported_modalities": [ + "image" + ], + "supported_output_modalities": [ + "image" + ] + }, + "vertex_ai/gemini-2.5-flash-tts": { + "input_cost_per_token": 5e-07, + "input_cost_per_token_batches": 2.5e-07, + "litellm_provider": "vertex_ai", + "mode": "audio_speech", + "output_cost_per_audio_token": 1e-05, + "output_cost_per_token": 1e-05, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" + }, + "vertex_ai/gemini-2.5-pro-tts": { + "input_cost_per_token": 1e-06, + "input_cost_per_token_batches": 5e-07, + "litellm_provider": "vertex_ai", + "mode": "audio_speech", + "output_cost_per_audio_token": 2e-05, + "output_cost_per_token": 2e-05, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" } } diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ebea1a6044d..844e613977f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -74121,5 +74121,105 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true + }, + "vertex_ai/meta/llama-3.3-70b-instruct-maas": { + "input_cost_per_token": 7.2e-07, + "input_cost_per_token_batches": 3.6e-07, + "litellm_provider": "vertex_ai-llama_models", + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 7.2e-07, + "output_cost_per_token_batches": 3.6e-07, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "text", + "code" + ], + "supports_function_calling": true, + "supports_tool_choice": true + }, + "vertex_ai/veo-3.0-generate-001": { + "deprecation_date": "2026-06-30", + "litellm_provider": "vertex_ai-video-models", + "max_input_tokens": 1024, + "max_tokens": 1024, + "mode": "video_generation", + "output_cost_per_second": 0.4, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#veo", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "vertex_ai/veo-3.0-fast-generate-001": { + "deprecation_date": "2026-06-30", + "litellm_provider": "vertex_ai-video-models", + "max_input_tokens": 1024, + "max_tokens": 1024, + "mode": "video_generation", + "output_cost_per_second": 0.1, + "output_cost_per_second_1080p": 0.12, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#veo", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "vertex_ai/veo-2.0-generate-001": { + "litellm_provider": "vertex_ai-video-models", + "max_input_tokens": 1024, + "max_tokens": 1024, + "mode": "video_generation", + "output_cost_per_second": 0.5, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#veo", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "video" + ] + }, + "vertex_ai/virtual-try-on-001": { + "deprecation_date": "2027-03-15", + "litellm_provider": "vertex_ai", + "mode": "image_generation", + "output_cost_per_image": 0.06, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", + "supported_modalities": [ + "image" + ], + "supported_output_modalities": [ + "image" + ] + }, + "vertex_ai/gemini-2.5-flash-tts": { + "input_cost_per_token": 5e-07, + "input_cost_per_token_batches": 2.5e-07, + "litellm_provider": "vertex_ai", + "mode": "audio_speech", + "output_cost_per_audio_token": 1e-05, + "output_cost_per_token": 1e-05, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" + }, + "vertex_ai/gemini-2.5-pro-tts": { + "input_cost_per_token": 1e-06, + "input_cost_per_token_batches": 5e-07, + "litellm_provider": "vertex_ai", + "mode": "audio_speech", + "output_cost_per_audio_token": 2e-05, + "output_cost_per_token": 2e-05, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" } } From cad49ee1718cabc7a133c1f3e195368007944265 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:03:02 -0700 Subject: [PATCH 02/96] fix(proxy): gate disable_global_guardrails on keys and teams to proxy admins (#42699) * fix(proxy): gate disable_global_guardrails on keys and teams to proxy admins Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test: cover metadata smuggle with explicit false and UI toggle gating Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(proxy): satisfy PT017 in resend-stored guardrail flag test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(proxy): keep regenerate_key_fn under the C901 ceiling via a guardrail opt-out helper Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(ui): regenerate schema.d.ts for guardrail opt-out docstrings Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): gate disable_global_guardrails on caller-sent metadata, not server defaults Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): audit cells for disable_global_guardrails admin gate Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): restore contracts.json formatting Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): share guardrail opt-out helpers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(ui): hide the team disable_global_guardrails switch from non proxy admins Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): drop covers markers and bound the slow sink check to the sink delay Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../management_endpoints/common_utils.py | 28 ++ .../key_management_endpoints.py | 45 ++- .../management_endpoints/team_endpoints.py | 18 +- .../authorization/_guardrail_opt_out.py | 71 ++++ .../test_key_guardrail_opt_out.py | 371 ++++++++++++++++++ .../test_key_guardrail_opt_out_chaos.py | 319 +++++++++++++++ .../test_key_guardrail_opt_out_runtime.py | 230 +++++++++++ .../management_endpoints/test_common_utils.py | 129 ++++++ .../test_key_management_endpoints.py | 193 +++++++++ .../test_team_endpoints.py | 77 ++++ .../src/components/Teams.test.tsx | 47 +++ ui/litellm-dashboard/src/components/Teams.tsx | 48 +-- .../create_key_button.integration.test.tsx | 15 + .../organisms/create_key_button.tsx | 70 ++-- .../src/components/team/TeamInfo.test.tsx | 47 +++ .../src/components/team/TeamInfo.tsx | 40 +- .../key_edit_view.integration.test.tsx | 28 ++ .../components/templates/key_edit_view.tsx | 26 +- ui/litellm-dashboard/src/lib/http/schema.d.ts | 8 +- 19 files changed, 1714 insertions(+), 96 deletions(-) create mode 100644 tests/integration/authorization/_guardrail_opt_out.py create mode 100644 tests/integration/authorization/test_key_guardrail_opt_out.py create mode 100644 tests/integration/authorization/test_key_guardrail_opt_out_chaos.py create mode 100644 tests/integration/authorization/test_key_guardrail_opt_out_runtime.py diff --git a/litellm/proxy/management_endpoints/common_utils.py b/litellm/proxy/management_endpoints/common_utils.py index 78e3ac7bd66..0bd4eb5a5d8 100644 --- a/litellm/proxy/management_endpoints/common_utils.py +++ b/litellm/proxy/management_endpoints/common_utils.py @@ -173,6 +173,34 @@ def _check_passthrough_routes_caller_permission( ) +def _check_disable_global_guardrails_caller_permission( + disable_global_guardrails: bool | None, + metadata: Mapping[str, object] | None, + user_api_key_dict: UserAPIKeyAuth, + *, + entity: str = "key", + existing_metadata: Mapping[str, object] | None = None, +) -> None: + """ + Only proxy admins may opt a key or team out of default-on guardrails, whether the + flag is top-level or under `metadata`. Re-sending a flag that is already stored is + not an opt-out, so non-admin edits of an already exempted object still go through. + """ + if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: + return + requested: Final = bool(disable_global_guardrails) or ( + metadata is not None and bool(metadata.get("disable_global_guardrails")) + ) + if not requested: + return + if existing_metadata is not None and existing_metadata.get("disable_global_guardrails") is True: + return + raise HTTPException( + status_code=403, + detail={"error": f"Only proxy admins can set `disable_global_guardrails` on a {entity}."}, + ) + + def _is_user_team_admin(user_api_key_dict: UserAPIKeyAuth, team_obj: LiteLLM_TeamTable) -> bool: for member in team_obj.members_with_roles: if (member.user_id is not None and member.user_id == user_api_key_dict.user_id) and member.role == "admin": diff --git a/litellm/proxy/management_endpoints/key_management_endpoints.py b/litellm/proxy/management_endpoints/key_management_endpoints.py index 854b80ba2c3..306ea90d7f1 100644 --- a/litellm/proxy/management_endpoints/key_management_endpoints.py +++ b/litellm/proxy/management_endpoints/key_management_endpoints.py @@ -85,6 +85,7 @@ from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache from litellm.proxy.hooks.key_management_event_hooks import KeyManagementEventHooks from litellm.proxy.hooks.model_max_budget_limiter import build_model_max_budget_usage from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, _check_passthrough_routes_caller_permission, _is_user_org_admin_for_team, _is_user_team_admin, @@ -1224,6 +1225,7 @@ async def _common_key_generation_helper( # default_key_generate_params injected. _requested_max_budget: Final = data.max_budget _requested_team_id: Final = data.team_id + _requested_metadata: Final = data.metadata # pyright: ignore[reportUnknownMemberType] # request models declare `metadata` as bare dict # check if user set default key/generate params on config.yaml if litellm.default_key_generate_params is not None: @@ -1311,6 +1313,11 @@ async def _common_key_generation_helper( data=data, user_api_key_dict=user_api_key_dict, ) + _check_disable_global_guardrails_caller_permission( + data.disable_global_guardrails, + _requested_metadata, + user_api_key_dict, + ) # APPLY ENTERPRISE KEY MANAGEMENT PARAMS try: @@ -1966,7 +1973,7 @@ async def generate_key_fn( - metadata: Optional[dict] - Metadata for key, store information for key. Example metadata = {"team": "core-infra", "app": "app2", "email": "ishaan@berri.ai" } - guardrails: Optional[List[str]] - List of active guardrails for the key - policies: Optional[List[str]] - List of policy names to apply to the key. Policies define guardrails, conditions, and inheritance rules. - - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. + - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. Proxy admin only. - throttle_on_budget_exceeded: Optional[bool] - When the key exceeds its max_budget, throttle its tpm/rpm to the global budget_exceeded_throttle_percentage instead of blocking the key entirely. - enable_prompt_caching: Optional[bool] - Auto-inject prompt caching breakpoints (Anthropic cache_control markers) on requests made with this key. Supported Claude models on Anthropic, Bedrock, Vertex AI, and Azure AI only. - permissions: Optional[dict] - key-specific permissions. Currently just used for turning off pii masking (if connected). Example - {"pii": false} @@ -2729,6 +2736,13 @@ async def _process_single_key_update( prisma_client=prisma_client, ) + _check_disable_global_guardrails_caller_permission( + update_key_request.disable_global_guardrails, + update_key_request.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # request models declare `metadata` as bare dict + user_api_key_dict, + existing_metadata=existing_key_row.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # LiteLLM_VerificationToken.metadata is a bare dict + ) + enforce_batch_enqueued_token_limit_is_admin_only( data=update_key_request, existing_metadata=existing_key_row.metadata, @@ -3025,6 +3039,12 @@ async def _validate_update_key_data( data=data, user_api_key_dict=user_api_key_dict, ) + _check_disable_global_guardrails_caller_permission( + data.disable_global_guardrails, + data.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # request models declare `metadata` as bare dict + user_api_key_dict, + existing_metadata=existing_key_row.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # LiteLLM_VerificationToken.metadata is a bare dict + ) _validate_caller_can_change_key_ownership( data=data, @@ -3328,7 +3348,7 @@ async def update_key_fn( - send_invite_email: Optional[bool] - Send invite email to user_id - guardrails: Optional[List[str]] - List of active guardrails for the key - policies: Optional[List[str]] - List of policy names to apply to the key. Policies define guardrails, conditions, and inheritance rules. - - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. + - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. Proxy admin only. - throttle_on_budget_exceeded: Optional[bool] - When the key exceeds its max_budget, throttle its tpm/rpm to the global budget_exceeded_throttle_percentage instead of blocking the key entirely. - enable_prompt_caching: Optional[bool] - Auto-inject prompt caching breakpoints (Anthropic cache_control markers) on requests made with this key. Supported Claude models on Anthropic, Bedrock, Vertex AI, and Azure AI only. - prompts: Optional[List[str]] - List of prompts that the key is allowed to use. @@ -5622,6 +5642,21 @@ async def _execute_virtual_key_regeneration( return response +def _check_regenerate_guardrail_opt_out( + data: RegenerateKeyRequest | None, + existing_metadata: Mapping[str, object] | None, + user_api_key_dict: UserAPIKeyAuth, +) -> None: + if data is None: + return + _check_disable_global_guardrails_caller_permission( + data.disable_global_guardrails, + data.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # request models declare `metadata` as bare dict + user_api_key_dict, + existing_metadata=existing_metadata, + ) + + @router.post( "/key/{key:path}/regenerate", tags=["key management"], @@ -5805,6 +5840,12 @@ async def regenerate_key_fn( detail={"error": f"Key {key} not found."}, ) + _check_regenerate_guardrail_opt_out( + data, + _key_in_db.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # LiteLLM_VerificationToken.metadata is a bare dict + user_api_key_dict, + ) + # check if user has permission to regenerate key await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint( user_api_key_dict=user_api_key_dict, diff --git a/litellm/proxy/management_endpoints/team_endpoints.py b/litellm/proxy/management_endpoints/team_endpoints.py index 09c6c05e22b..6cec3e714ec 100644 --- a/litellm/proxy/management_endpoints/team_endpoints.py +++ b/litellm/proxy/management_endpoints/team_endpoints.py @@ -127,6 +127,7 @@ from litellm.proxy.management_endpoints.common_daily_activity import ( get_daily_activity_aggregated, ) from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, _check_passthrough_routes_caller_permission, _is_user_org_admin_for_team, _is_user_team_admin, @@ -1416,7 +1417,7 @@ async def new_team( - model_max_budget: Optional[dict] - Per-model max budget every key on the team inherits unless the key sets its own for that model. Example: {"gpt-4o": {"max_budget": 10, "budget_duration": "1d"}} - guardrails: Optional[List[str]] - Guardrails for the team. [Docs](https://docs.litellm.ai/docs/proxy/guardrails) - policies: Optional[List[str]] - Policies for the team. [Docs](https://docs.litellm.ai/docs/proxy/guardrails/guardrail_policies) - - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. + - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the team. Proxy admin only. - object_permission: Optional[LiteLLM_ObjectPermissionBase] - team-specific object permission. Example - {"vector_stores": ["vector_store_1", "vector_store_2"], "agents": ["agent_1", "agent_2"], "agent_access_groups": ["dev_group"]}. IF null or {} then no object permission. - team_member_budget: Optional[float] - The maximum budget allocated to an individual team member. - team_member_budget_duration: Optional[str] - The duration of the budget for the team member. Doc [here](https://docs.litellm.ai/docs/proxy/team_budgets) @@ -1638,6 +1639,12 @@ async def new_team( data.members_with_roles.append(Member(role="admin", user_id=user_api_key_dict.user_id)) _check_passthrough_routes_caller_permission(data, user_api_key_dict, entity="team") + _check_disable_global_guardrails_caller_permission( + data.disable_global_guardrails, + data.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # request models declare `metadata` as bare dict + user_api_key_dict, + entity="team", + ) if isinstance(data.metadata, dict): TeamMemberBudgetHandler.strip_system_managed_metadata_keys(data.metadata) @@ -2172,7 +2179,7 @@ async def update_team( - model_max_budget: Optional[dict] - Per-model max budget every key on the team inherits unless the key sets its own for that model. Example: {"gpt-4o": {"max_budget": 10, "budget_duration": "1d"}} - guardrails: Optional[List[str]] - Guardrails for the team. [Docs](https://docs.litellm.ai/docs/proxy/guardrails) - policies: Optional[List[str]] - Policies for the team. [Docs](https://docs.litellm.ai/docs/proxy/guardrails/guardrail_policies) - - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. + - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the team. Proxy admin only. - object_permission: Optional[LiteLLM_ObjectPermissionBase] - team-specific object permission. Example - {"vector_stores": ["vector_store_1", "vector_store_2"], "agents": ["agent_1", "agent_2"], "agent_access_groups": ["dev_group"]}. IF null or {} then no object permission. - team_member_budget: Optional[float] - The maximum budget allocated to an individual team member. - team_member_budget_duration: Optional[str] - The duration of the budget for the team member. Doc [here](https://docs.litellm.ai/docs/proxy/team_budgets) @@ -2313,6 +2320,13 @@ async def update_team( ) _check_passthrough_routes_caller_permission(data, user_api_key_dict, entity="team") + _check_disable_global_guardrails_caller_permission( + data.disable_global_guardrails, + data.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # request models declare `metadata` as bare dict + user_api_key_dict, + entity="team", + existing_metadata=_existing_team_metadata if isinstance(_existing_team_metadata, dict) else None, # pyright: ignore[reportUnknownArgumentType] # existing_team_row.metadata is a bare dict + ) if data.soft_budget is not None: max_budget_to_check = data.max_budget if data.max_budget is not None else existing_team_row.max_budget diff --git a/tests/integration/authorization/_guardrail_opt_out.py b/tests/integration/authorization/_guardrail_opt_out.py new file mode 100644 index 00000000000..e813993edf4 --- /dev/null +++ b/tests/integration/authorization/_guardrail_opt_out.py @@ -0,0 +1,71 @@ +import json +import uuid +from hashlib import sha256 +from pathlib import Path +from typing import Final + +import httpx +import yaml +from pydantic import JsonValue + +from integration._support.client import Gateway, Scenario, object_value +from integration._support.database import read_rows +from integration._support.wire import Reply, Request + +MANAGEMENT_ROUTES: Final = ["/key/*", "/team/new", "/team/update", "/v1/chat/completions"] + + +def denying_guardrail(request: Request) -> Reply: + assert request.target == "/beta/litellm_basic_guardrail_api" + return Reply(body=json.dumps({"action": "BLOCKED", "blocked_reason": "synthetic policy denial"}).encode()) + + +def guardrail_config(policy_url: str, path: Path) -> Path: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["guardrails"] = [ + { + "guardrail_name": "guardrail" + uuid.uuid4().hex, + "litellm_params": { + "guardrail": "generic_guardrail_api", + "mode": "pre_call", + "default_on": True, + "api_base": policy_url, + "api_key": "synthetic-guardrail-key", + }, + } + ] + path.write_text(yaml.safe_dump(config)) + return path + + +def stored_metadata(token: str) -> dict[str, object]: + rows: Final = read_rows( + 'SELECT metadata FROM "LiteLLM_VerificationToken" WHERE token = %s', (sha256(token.encode()).hexdigest(),) + ) + assert len(rows) == 1, rows + return rows[0]["metadata"] + + +def non_admin_caller(scenario: Scenario, member: str, team: str, model: str) -> str: + return scenario.key(user_id=member, team_id=team, models=[model], allowed_routes=MANAGEMENT_ROUTES) + + +def chat(candidate: Gateway, model: str, key: str, marker: str, *, stream: bool = False) -> httpx.Response: + return candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": marker}], "stream": stream}, + key=key, + ) + + +def upstream_observations(gateway: Gateway) -> tuple[dict[str, JsonValue], ...]: + with httpx.Client(timeout=5, trust_env=False) as client: + drained: Final = object_value(client.get(f"{gateway.upstream_url}/__observations").json()) + requests: Final = drained["requests"] + assert isinstance(requests, list) + return tuple(object_value(entry) for entry in requests) + + +def upstream_hits(gateway: Gateway, marker: str) -> int: + return sum(1 for entry in upstream_observations(gateway) if marker in json.dumps(entry.get("body"))) diff --git a/tests/integration/authorization/test_key_guardrail_opt_out.py b/tests/integration/authorization/test_key_guardrail_opt_out.py new file mode 100644 index 00000000000..6591b0b3993 --- /dev/null +++ b/tests/integration/authorization/test_key_guardrail_opt_out.py @@ -0,0 +1,371 @@ +import uuid +from pathlib import Path +from typing import Final + +import httpx +import yaml + +from integration._support.client import Gateway, Scenario, object_value, string_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy +from integration._support.wire import wire_server +from integration.authorization._guardrail_opt_out import ( + denying_guardrail, + guardrail_config, + non_admin_caller, + stored_metadata, +) + +_KEY_ROUTES: Final = ["/key/generate", "/key/update", "/key/regenerate", "/v1/chat/completions"] + + +def test_non_admin_cannot_opt_key_out_of_default_on_guardrail(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy: + config: Final = guardrail_config(policy.url, tmp_path / "default_on.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = scenario.key(user_id=member, models=[model], allowed_routes=_KEY_ROUTES) + own: Final = scenario.key(team_id=team, models=[model]) + + plain: Final = candidate.request("POST", "/key/generate", {"team_id": team, "models": [model]}, key=caller) + assert plain.status_code == 200, plain.text + scenario.cleanups.callback(scenario.delete_key, string_value(plain.json()["key"])) + + generated: Final = candidate.request( + "POST", + "/key/generate", + {"team_id": team, "models": [model], "disable_global_guardrails": True}, + key=caller, + ) + if generated.status_code == 200: + scenario.cleanups.callback(scenario.delete_key, string_value(generated.json()["key"])) + assert generated.status_code == 403, generated.text + assert "disable_global_guardrails" in generated.text + + smuggled: Final = candidate.request( + "POST", + "/key/generate", + {"team_id": team, "models": [model], "metadata": {"disable_global_guardrails": True}}, + key=caller, + ) + if smuggled.status_code == 200: + scenario.cleanups.callback(scenario.delete_key, string_value(smuggled.json()["key"])) + assert smuggled.status_code == 403, smuggled.text + + updated: Final = candidate.request( + "POST", "/key/update", {"key": own, "disable_global_guardrails": True}, key=caller + ) + assert updated.status_code == 403, updated.text + regenerated: Final = candidate.request( + "POST", "/key/regenerate", {"key": own, "disable_global_guardrails": True}, key=caller + ) + assert regenerated.status_code == 403, regenerated.text + assert "disable_global_guardrails" not in stored_metadata(own) + + blocked: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": "synthetic denied marker"}]}, + key=own, + ) + assert blocked.status_code == 400 and "synthetic policy denial" in blocked.text, blocked.text + + exempt: Final = scenario.key(team_id=team, models=[model], disable_global_guardrails=True) + assert stored_metadata(exempt)["disable_global_guardrails"] is True + resaved: Final = candidate.request( + "POST", + "/key/update", + { + "key": exempt, + "key_alias": "renamed" + uuid.uuid4().hex, + "metadata": {"disable_global_guardrails": True}, + }, + key=caller, + ) + assert resaved.status_code == 200, resaved.text + assert stored_metadata(exempt)["disable_global_guardrails"] is True + served: Final = candidate.chat(model, key=exempt, text="synthetic denied marker") + assert object_value(served["usage"])["total_tokens"] == 40 + assert len(policy.drain()) == 1 + + +def _team_metadata(team_id: str) -> dict[str, object]: + rows: Final = read_rows('SELECT metadata FROM "LiteLLM_TeamTable" WHERE team_id = %s', (team_id,)) + assert len(rows) == 1, rows + return rows[0]["metadata"] + + +def _drop_created_key(scenario: Scenario, response: httpx.Response) -> None: + if response.status_code == 200: + scenario.cleanups.callback(scenario.delete_key, string_value(response.json()["key"])) + + +def test_non_admin_flag_denied_on_every_key_write_route(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy: + config: Final = guardrail_config(policy.url, tmp_path / "denied-routes.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + own: Final = scenario.key(team_id=team, models=[model]) + + attempts: Final = ( + ("POST", "/key/generate", {"team_id": team, "models": [model], "disable_global_guardrails": True}), + ( + "POST", + "/key/generate", + {"team_id": team, "models": [model], "metadata": {"disable_global_guardrails": True}}, + ), + ( + "POST", + "/key/generate", + { + "team_id": team, + "models": [model], + "disable_global_guardrails": False, + "metadata": {"disable_global_guardrails": True}, + }, + ), + ("POST", "/key/update", {"key": own, "disable_global_guardrails": True}), + ("POST", "/key/update", {"key": own, "metadata": {"disable_global_guardrails": True}}), + ("POST", "/key/regenerate", {"key": own, "disable_global_guardrails": True}), + ("POST", f"/key/{own}/regenerate", {"disable_global_guardrails": True}), + ( + "POST", + "/key/service-account/generate", + {"team_id": team, "disable_global_guardrails": True}, + ), + ) + for method, path, body in attempts: + response: Final = candidate.request(method, path, body, key=caller) + _drop_created_key(scenario, response) + assert response.status_code == 403, f"{method} {path}: {response.text}" + assert "disable_global_guardrails" in response.text, response.text + assert "disable_global_guardrails" not in stored_metadata(own) + + service_alias: Final = "audit-sa-" + uuid.uuid4().hex + service_denied: Final = candidate.request( + "POST", + "/key/service-account/generate", + {"team_id": team, "key_alias": service_alias, "disable_global_guardrails": True}, + key=caller, + ) + _drop_created_key(scenario, service_denied) + assert ( + read_rows('SELECT token FROM "LiteLLM_VerificationToken" WHERE key_alias = %s', (service_alias,)) == [] + ), service_denied.text + + +def test_non_admin_flag_denied_on_team_new(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy: + config: Final = guardrail_config(policy.url, tmp_path / "denied-team.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + + alias: Final = "audit-team-" + uuid.uuid4().hex + denied: Final = candidate.request( + "POST", + "/team/new", + {"team_alias": alias, "models": [model], "disable_global_guardrails": True}, + key=caller, + ) + created: Final = read_rows('SELECT team_id FROM "LiteLLM_TeamTable" WHERE team_alias = %s', (alias,)) + for row in created: + scenario.cleanups.callback(scenario.delete_team, str(row["team_id"])) + assert denied.status_code == 403, denied.text + assert "disable_global_guardrails" in denied.text, denied.text + + +def test_admin_flag_writes_succeed_on_all_routes(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy: + config: Final = guardrail_config(policy.url, tmp_path / "admin-routes.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + team: Final = scenario.team(models=[model]) + + generated: Final = candidate.post( + "/key/generate", {"team_id": team, "models": [model], "disable_global_guardrails": True} + ) + generated_key: Final = string_value(generated["key"]) + scenario.cleanups.callback(scenario.delete_key, generated_key) + assert stored_metadata(generated_key)["disable_global_guardrails"] is True + + plain: Final = scenario.key(team_id=team, models=[model]) + candidate.post("/key/update", {"key": plain, "disable_global_guardrails": True}) + assert stored_metadata(plain)["disable_global_guardrails"] is True + + regen_source: Final = string_value( + candidate.post("/key/generate", {"team_id": team, "models": [model]})["key"] + ) + regenerated: Final = candidate.post( + "/key/regenerate", {"key": regen_source, "disable_global_guardrails": True} + ) + regenerated_key: Final = string_value(regenerated["key"]) + scenario.cleanups.callback(scenario.delete_key, regenerated_key) + assert stored_metadata(regenerated_key)["disable_global_guardrails"] is True + + new_team: Final = candidate.post( + "/team/new", {"team_alias": "audit-admin-" + uuid.uuid4().hex, "disable_global_guardrails": True} + ) + new_team_id: Final = string_value(new_team["team_id"]) + scenario.cleanups.callback(scenario.delete_team, new_team_id) + assert _team_metadata(new_team_id)["disable_global_guardrails"] is True + + candidate.post("/team/update", {"team_id": team, "disable_global_guardrails": True}) + assert _team_metadata(team)["disable_global_guardrails"] is True + + +def test_non_admin_resave_omit_and_revoke_sequences(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy: + config: Final = guardrail_config(policy.url, tmp_path / "resave.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + exempt: Final = scenario.key(team_id=team, models=[model], disable_global_guardrails=True) + assert stored_metadata(exempt)["disable_global_guardrails"] is True + + resaved: Final = candidate.request( + "POST", + "/key/update", + { + "key": exempt, + "key_alias": "audit-resave-" + uuid.uuid4().hex, + "metadata": {"disable_global_guardrails": True}, + }, + key=caller, + ) + assert resaved.status_code == 200, resaved.text + assert stored_metadata(exempt)["disable_global_guardrails"] is True + + omitted: Final = candidate.request( + "POST", + "/key/update", + {"key": exempt, "key_alias": "audit-omit-" + uuid.uuid4().hex}, + key=caller, + ) + assert omitted.status_code == 200, omitted.text + + candidate.post("/key/update", {"key": exempt, "disable_global_guardrails": False}) + assert stored_metadata(exempt)["disable_global_guardrails"] is False + + rejected: Final = candidate.request( + "POST", "/key/update", {"key": exempt, "disable_global_guardrails": True}, key=caller + ) + assert rejected.status_code == 403, rejected.text + assert "disable_global_guardrails" in rejected.text, rejected.text + assert stored_metadata(exempt)["disable_global_guardrails"] is False + + +def test_generate_ignores_server_default_metadata_flag(gateway: Gateway, tmp_path: Path) -> None: + raw: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + raw.setdefault("litellm_settings", {})["default_key_generate_params"] = { + "metadata": {"disable_global_guardrails": True} + } + path: Final = tmp_path / "server-defaults.yaml" + path.write_text(yaml.safe_dump(raw)) + with owned_proxy(gateway, tmp_path, {}, config=path) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + + generated: Final = candidate.post("/key/generate", {"team_id": team, "models": [model]}, key=caller) + generated_key: Final = string_value(generated["key"]) + scenario.cleanups.callback(scenario.delete_key, generated_key) + assert stored_metadata(generated_key)["disable_global_guardrails"] is True + + explicit: Final = candidate.request( + "POST", + "/key/generate", + {"team_id": team, "models": [model], "metadata": {"disable_global_guardrails": True}}, + key=caller, + ) + _drop_created_key(scenario, explicit) + assert explicit.status_code == 403, explicit.text + assert "disable_global_guardrails" in explicit.text, explicit.text + + +def test_sad_flag_inputs_on_key_generate(gateway: Gateway, tmp_path: Path) -> None: + with owned_proxy(gateway, tmp_path, {}) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + + denied_bodies: Final = ( + {"team_id": team, "models": [model], "disable_global_guardrails": "true"}, + {"team_id": team, "models": [model], "disable_global_guardrails": 1}, + {"team_id": team, "models": [model], "metadata": {"disable_global_guardrails": "true"}}, + {"team_id": team, "models": [model], "metadata": {"disable_global_guardrails": 1}}, + {"team_id": team, "models": [model], "metadata": {"disable_global_guardrails": "x" * 5120}}, + ) + for body in denied_bodies: + response: Final = candidate.request("POST", "/key/generate", body, key=caller) + _drop_created_key(scenario, response) + assert response.status_code == 403, response.text + assert "disable_global_guardrails" in response.text, response.text + + invalid_bodies: Final = ( + {"team_id": team, "models": [model], "disable_global_guardrails": []}, + {"team_id": team, "models": [model], "disable_global_guardrails": {}}, + ) + for body in invalid_bodies: + rejected: Final = candidate.request("POST", "/key/generate", body, key=caller) + assert rejected.status_code == 422, rejected.text + + unauthenticated: Final = candidate.request( + "POST", + "/key/generate", + {"team_id": team, "models": [model], "disable_global_guardrails": True}, + key="sk-not-a-real-key-" + uuid.uuid4().hex, + ) + assert unauthenticated.status_code == 401, unauthenticated.text + + repeat_alias: Final = "audit-repeat-" + uuid.uuid4().hex + for _ in range(2): + repeated: Final = candidate.request( + "POST", + "/key/generate", + {"team_id": team, "models": [model], "key_alias": repeat_alias, "disable_global_guardrails": True}, + key=caller, + ) + _drop_created_key(scenario, repeated) + assert repeated.status_code == 403, repeated.text + assert read_rows('SELECT token FROM "LiteLLM_VerificationToken" WHERE key_alias = %s', (repeat_alias,)) == [] + + +def test_falsy_metadata_flag_shapes_stay_stored_and_guarded(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy: + config: Final = guardrail_config(policy.url, tmp_path / "falsy.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + + for shape in ([], {}): + created: Final = candidate.request( + "POST", + "/key/generate", + {"team_id": team, "models": [model], "metadata": {"disable_global_guardrails": shape}}, + key=caller, + ) + _drop_created_key(scenario, created) + assert created.status_code == 200, created.text + token: Final = string_value(created.json()["key"]) + assert stored_metadata(token)["disable_global_guardrails"] == shape + blocked: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": "synthetic denied marker"}]}, + key=token, + ) + assert blocked.status_code == 400 and "synthetic policy denial" in blocked.text, blocked.text diff --git a/tests/integration/authorization/test_key_guardrail_opt_out_chaos.py b/tests/integration/authorization/test_key_guardrail_opt_out_chaos.py new file mode 100644 index 00000000000..f205b3804f6 --- /dev/null +++ b/tests/integration/authorization/test_key_guardrail_opt_out_chaos.py @@ -0,0 +1,319 @@ +import json +import os +import signal +import socket +import threading +import time +import uuid +from concurrent.futures import ThreadPoolExecutor +from hashlib import sha256 +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from queue import SimpleQueue +from typing import Final + +import httpx +import psutil + +from integration._support.client import Gateway, eventually, string_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy, owned_proxy_process +from integration.authorization._guardrail_opt_out import ( + chat, + guardrail_config, + non_admin_caller, + stored_metadata, + upstream_hits, + upstream_observations, +) + + +class _GuardrailSink: + """Test-owned guardrail endpoint that can be stopped and restarted on the same port.""" + + def __init__(self, *, delay_seconds: float = 0.0, action: str = "BLOCKED") -> None: + self.received: SimpleQueue[bytes] = SimpleQueue() + self._delay: Final = delay_seconds + self._action: Final = action + self._server: ThreadingHTTPServer | None = None + self._thread: threading.Thread | None = None + self._port: Final = self._claim_port() + self.start() + + def _claim_port(self) -> int: + with socket.socket() as probe: + probe.bind(("127.0.0.1", 0)) + return probe.getsockname()[1] + + @property + def url(self) -> str: + return f"http://127.0.0.1:{self._port}" + + def start(self) -> None: + received = self.received + delay = self._delay + action = self._action + + class Handler(BaseHTTPRequestHandler): + def do_POST(self) -> None: + body: Final = self.rfile.read(int(self.headers.get("content-length", "0"))) + received.put(body) + if delay: + time.sleep(delay) + payload: Final = json.dumps({"action": action, "blocked_reason": "synthetic policy denial"}).encode() + self.send_response(200) + self.send_header("content-type", "application/json") + self.send_header("content-length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, format: str, *args: object) -> None: + pass + + class Server(ThreadingHTTPServer): + daemon_threads = True + allow_reuse_address = True + + self._server = Server(("127.0.0.1", self._port), Handler) + self._thread = threading.Thread(target=self._server.serve_forever, kwargs={"poll_interval": 0.05}) + self._thread.start() + + def stop(self) -> None: + assert self._server is not None and self._thread is not None + self._server.shutdown() + self._server.server_close() + self._thread.join(timeout=6) + assert not self._thread.is_alive() + self._server = None + + def drain(self) -> tuple[bytes, ...]: + return tuple(self.received.get_nowait() for _ in range(self.received.qsize())) + + def __enter__(self) -> "_GuardrailSink": + return self + + def __exit__(self, *exc_info: object) -> None: + if self._server is not None: + self.stop() + + +def test_concurrent_flag_writes_split_expected_outcomes(gateway: Gateway, tmp_path: Path) -> None: + with owned_proxy(gateway, tmp_path, {}) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + alias: Final = "audit-concurrent-" + uuid.uuid4().hex + + bodies: Final = [ + {"team_id": team, "models": [model], "key_alias": f"{alias}-{index}", "disable_global_guardrails": flag} + for index in range(20) + for flag in (True, False) + ] + with ThreadPoolExecutor(max_workers=20) as pool: + responses: Final = tuple( + pool.map(lambda body: candidate.request("POST", "/key/generate", body, key=caller), bodies) + ) + created_aliases: Final = [ + row["key_alias"] + for row in read_rows( + 'SELECT key_alias FROM "LiteLLM_VerificationToken" WHERE key_alias LIKE %s', (f"{alias}-%",) + ) + ] + for response in responses: + if response.status_code == 200: + scenario.cleanups.callback(scenario.delete_key, string_value(response.json()["key"])) + flagged: Final = tuple( + response for response, body in zip(responses, bodies) if body["disable_global_guardrails"] is True + ) + flagless: Final = tuple( + response for response, body in zip(responses, bodies) if body["disable_global_guardrails"] is False + ) + assert sorted(response.status_code for response in flagged) == [403] * 20, [ + response.text for response in flagged + ] + assert sorted(response.status_code for response in flagless) == [200] * 20, [ + response.text for response in flagless + ] + assert len(created_aliases) == 20, created_aliases + for entry in created_aliases: + stored: Final = read_rows('SELECT metadata FROM "LiteLLM_VerificationToken" WHERE key_alias = %s', (entry,)) + assert stored[0]["metadata"].get("disable_global_guardrails") is not True, entry + + +def test_revoked_exemption_denies_later_non_admin_resave(gateway: Gateway, tmp_path: Path) -> None: + with owned_proxy(gateway, tmp_path, {}) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + exempt: Final = scenario.key(team_id=team, models=[model], disable_global_guardrails=True) + assert stored_metadata(exempt)["disable_global_guardrails"] is True + + candidate.post("/key/update", {"key": exempt, "disable_global_guardrails": False}) + assert stored_metadata(exempt)["disable_global_guardrails"] is False + + resave: Final = candidate.request( + "POST", + "/key/update", + { + "key": exempt, + "key_alias": "audit-revoked-" + uuid.uuid4().hex, + "metadata": {"disable_global_guardrails": True}, + }, + key=caller, + ) + assert resave.status_code == 403, resave.text + assert "disable_global_guardrails" in resave.text, resave.text + assert stored_metadata(exempt)["disable_global_guardrails"] is False + + +def test_revoked_exemption_blocks_chats_on_both_workers(gateway: Gateway, tmp_path: Path) -> None: + with _GuardrailSink() as sink: + config: Final = guardrail_config(sink.url, tmp_path / "revoke-workers.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as first: + with owned_proxy(gateway, tmp_path, {}, config=config) as second: + with first.scenario() as scenario: + model: Final = scenario.model() + exempt: Final = scenario.key(models=[model], disable_global_guardrails=True) + for worker in (first, second): + served: Final = chat(worker, model, exempt, "audit-both-" + uuid.uuid4().hex) + assert served.status_code == 200, served.text + first.post("/key/update", {"key": exempt, "disable_global_guardrails": False}) + for worker in (first, second): + denied: Final = eventually( + lambda w=worker: chat(w, model, exempt, "audit-both-" + uuid.uuid4().hex), + lambda response: response.status_code == 400 and "synthetic policy denial" in response.text, + seconds=70, + ) + assert denied.status_code == 400, denied.text + + +def test_exempt_burst_survives_guardrail_sink_outage(gateway: Gateway, tmp_path: Path) -> None: + with _GuardrailSink() as sink: + config: Final = guardrail_config(sink.url, tmp_path / "sink-outage.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + exempt: Final = scenario.key(models=[model], disable_global_guardrails=True) + plain: Final = scenario.key(models=[model]) + + warm: Final = chat(candidate, model, plain, "warm-" + uuid.uuid4().hex) + assert warm.status_code == 400 and "synthetic policy denial" in warm.text, warm.text + assert sink.drain() != () + + def burst(keys: tuple[str, ...], tag: str) -> tuple[httpx.Response, ...]: + with ThreadPoolExecutor(max_workers=15) as pool: + return tuple( + pool.map( + lambda pair: chat(candidate, model, pair[1], f"{tag}-{pair[0]}-{uuid.uuid4().hex}"), + enumerate(keys * 10), + ) + ) + + outage_keys: Final = (exempt, plain) + with ThreadPoolExecutor(max_workers=2) as pool: + bursts: Final = pool.submit(burst, outage_keys, "outage") + eventually( + lambda: sink.received.qsize(), + lambda count: count >= 2, + seconds=30, + ) + sink.stop() + outage_responses: Final = bursts.result(timeout=90) + exempt_outage: Final = [response for index, response in enumerate(outage_responses) if index % 2 == 0] + non_exempt_outage: Final = [response for index, response in enumerate(outage_responses) if index % 2 == 1] + assert all(response.status_code == 200 for response in exempt_outage), [ + response.status_code for response in exempt_outage + ] + outage_statuses: Final = {response.status_code for response in non_exempt_outage} + assert outage_statuses <= {400, 500}, outage_statuses + assert all( + "synthetic policy denial" in response.text or response.status_code == 500 + for response in non_exempt_outage + ), [response.text for response in non_exempt_outage if response.status_code not in {400, 500}] + assert all(upstream_hits(gateway, f"outage-{index}-") == 0 for index in range(1, 20, 2)), ( + upstream_observations(gateway) + ) + + sink.start() + recovered: Final = chat(candidate, model, plain, "recovered-" + uuid.uuid4().hex) + assert recovered.status_code == 400 and "synthetic policy denial" in recovered.text, recovered.text + + +def test_flag_denial_survives_worker_kill(gateway: Gateway, tmp_path: Path) -> None: + with owned_proxy_process(gateway, tmp_path, {}, workers=2) as owned: + candidate: Final = owned.gateway + with candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + team: Final = scenario.team(models=[model], members_with_roles=[{"role": "admin", "user_id": member}]) + caller: Final = non_admin_caller(scenario, member, team, model) + alias: Final = "audit-kill-" + uuid.uuid4().hex + + workers: Final = eventually( + lambda: psutil.Process(owned.process.pid).children(recursive=True), + lambda children: len(children) >= 2, + seconds=30, + ) + victim: Final = workers[0] + os.kill(victim.pid, signal.SIGKILL) + + probe: Final = eventually( + lambda: candidate.request( + "POST", + "/key/generate", + {"team_id": team, "models": [model], "key_alias": f"{alias}-probe"}, + key=caller, + ), + lambda response: response.status_code in (200, 403), + seconds=30, + ) + if probe.status_code == 200: + scenario.cleanups.callback(scenario.delete_key, string_value(probe.json()["key"])) + for index in range(10): + denied: Final = candidate.request( + "POST", + "/key/generate", + { + "team_id": team, + "models": [model], + "key_alias": f"{alias}-{index}", + "disable_global_guardrails": True, + }, + key=caller, + ) + if denied.status_code == 200: + scenario.cleanups.callback(scenario.delete_key, string_value(denied.json()["key"])) + assert denied.status_code == 403, denied.text + assert "disable_global_guardrails" in denied.text, denied.text + assert ( + read_rows( + 'SELECT token FROM "LiteLLM_VerificationToken" WHERE key_alias LIKE %s AND metadata::text LIKE %s', + (f"{alias}-%", '%"disable_global_guardrails": true%'), + ) + == [] + ) + + +def test_exempt_chats_do_not_wait_on_slow_guardrail_sink(gateway: Gateway, tmp_path: Path) -> None: + sink_delay: Final = 10.0 + with _GuardrailSink(delay_seconds=sink_delay) as sink: + config: Final = guardrail_config(sink.url, tmp_path / "slow-sink.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + exempt: Final = scenario.key(models=[model], disable_global_guardrails=True) + + started: Final = time.monotonic() + with ThreadPoolExecutor(max_workers=10) as pool: + responses: Final = tuple( + pool.map( + lambda index: chat(candidate, model, exempt, f"slow-sink-{index}-{uuid.uuid4().hex}"), + range(10), + ) + ) + elapsed: Final = time.monotonic() - started + assert all(response.status_code == 200 for response in responses), [ + (response.status_code, response.text) for response in responses + ] + assert elapsed < sink_delay, f"exempt chats waited on the guardrail sink: {elapsed}s" + assert sink.drain() == () diff --git a/tests/integration/authorization/test_key_guardrail_opt_out_runtime.py b/tests/integration/authorization/test_key_guardrail_opt_out_runtime.py new file mode 100644 index 00000000000..dc549d3be8d --- /dev/null +++ b/tests/integration/authorization/test_key_guardrail_opt_out_runtime.py @@ -0,0 +1,230 @@ +import asyncio +import json +import uuid +from pathlib import Path +from typing import Final + +import httpx +from anthropic import Anthropic +from openai import AsyncOpenAI, OpenAI + +from integration._support.client import Gateway, eventually, string_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy +from integration._support.wire import Reply, Request, Wire, wire_server +from integration.authorization._guardrail_opt_out import ( + chat, + denying_guardrail, + guardrail_config, + stored_metadata, + upstream_hits, +) + + +def _wire_hits(wire: Wire, marker: str) -> int: + return sum(1 for request in wire.drain() if marker.encode() in request.body) + + +def _sink_hits(policy: Wire, marker: str) -> int: + return sum(1 for request in policy.drain() if marker.encode() in request.body) + + +def _anthropic_provider(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages", request.target + body: Final = json.loads(request.body) + if body.get("stream") is True: + identity: Final = "msg_" + uuid.uuid4().hex + frames: Final = ( + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": body["model"], + "content": [], + "stop_reason": None, + "usage": {"input_tokens": 10, "output_tokens": 1}, + }, + }, + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "synthetic"}}, + {"type": "content_block_stop", "index": 0}, + {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {"output_tokens": 4}}, + {"type": "message_stop"}, + ) + return Reply( + content_type="text/event-stream", + chunks=tuple(f"event: {frame['type']}\ndata: {json.dumps(frame)}\n\n".encode() for frame in frames), + ) + return Reply( + body=json.dumps( + { + "id": "msg_" + uuid.uuid4().hex, + "type": "message", + "role": "assistant", + "model": body["model"], + "content": [{"type": "text", "text": "synthetic"}], + "stop_reason": "end_turn", + "usage": {"input_tokens": 10, "output_tokens": 4}, + } + ).encode() + ) + + +def _messages(candidate: Gateway, model: str, key: str, marker: str, *, stream: bool) -> httpx.Response: + return candidate.request( + "POST", + "/v1/messages", + { + "model": model, + "messages": [{"role": "user", "content": marker}], + "max_tokens": 16, + "stream": stream, + }, + key=key, + ) + + +def _responses(candidate: Gateway, model: str, key: str, marker: str, *, stream: bool) -> httpx.Response: + return candidate.request( + "POST", + "/v1/responses", + {"model": model, "input": marker, "stream": stream}, + key=key, + ) + + +def test_guardrail_denies_non_exempt_key_on_all_surfaces(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy, wire_server(_anthropic_provider) as anthropic_wire: + config: Final = guardrail_config(policy.url, tmp_path / "denied.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + openai_model: Final = scenario.model() + claude_model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", + api_base=anthropic_wire.url, + api_key="synthetic-anthropic-key", + ) + deepseek_model: Final = scenario.model(model="deepseek/gpt-4o-mini", api_base=gateway.upstream_url + "/v1") + key: Final = scenario.key(models=[openai_model, claude_model, deepseek_model]) + surfaces: Final = ( + ("chat", openai_model, chat), + ("messages", claude_model, _messages), + ("responses", deepseek_model, _responses), + ) + for surface, model, call in surfaces: + for stream in (False, True): + marker: Final = f"denied-{surface}-{stream}-{uuid.uuid4().hex}" + response: Final = call(candidate, model, key, marker, stream=stream) + response.read() + assert response.status_code == 400, f"{surface} stream={stream}: {response.text}" + assert "synthetic policy denial" in response.text, response.text + assert _sink_hits(policy, marker) == 1 + assert upstream_hits(gateway, marker) == 0 + assert _wire_hits(anthropic_wire, marker) == 0 + + +def test_guardrail_skipped_for_admin_exempt_key_on_all_surfaces_and_clients(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy, wire_server(_anthropic_provider) as anthropic_wire: + config: Final = guardrail_config(policy.url, tmp_path / "exempt.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + openai_model: Final = scenario.model() + claude_model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", + api_base=anthropic_wire.url, + api_key="synthetic-anthropic-key", + ) + deepseek_model: Final = scenario.model(model="deepseek/gpt-4o-mini", api_base=gateway.upstream_url + "/v1") + exempt: Final = scenario.key( + models=[openai_model, claude_model, deepseek_model], disable_global_guardrails=True + ) + assert stored_metadata(exempt)["disable_global_guardrails"] is True + surfaces: Final = ( + ("chat", openai_model, chat), + ("messages", claude_model, _messages), + ("responses", deepseek_model, _responses), + ) + for surface, model, call in surfaces: + for stream in (False, True): + marker: Final = f"exempt-{surface}-{stream}-{uuid.uuid4().hex}" + response: Final = call(candidate, model, exempt, marker, stream=stream) + response.read() + assert response.status_code == 200, f"{surface} stream={stream}: {response.text}" + assert "synthetic policy denial" not in response.text + provider_hits: Final = ( + _wire_hits(anthropic_wire, marker) if surface == "messages" else upstream_hits(gateway, marker) + ) + assert provider_hits == 1, f"{surface} stream={stream} marker={marker}" + assert _sink_hits(policy, marker) == 0 + + base_url: Final = str(candidate.client.base_url).rstrip("/") + "/v1" + sync_marker: Final = "exempt-sdk-sync-" + uuid.uuid4().hex + OpenAI(api_key=exempt, base_url=base_url, max_retries=0).chat.completions.create( + model=openai_model, messages=[{"role": "user", "content": sync_marker}] + ) + assert upstream_hits(gateway, sync_marker) == 1 + + async_marker: Final = "exempt-sdk-async-" + uuid.uuid4().hex + + async def _asyncchat() -> None: + async with AsyncOpenAI(api_key=exempt, base_url=base_url, max_retries=0) as client: + await client.chat.completions.create( + model=openai_model, messages=[{"role": "user", "content": async_marker}] + ) + + asyncio.run(_asyncchat()) + assert upstream_hits(gateway, async_marker) == 1 + + anthropic_marker: Final = "exempt-anthropic-" + uuid.uuid4().hex + Anthropic( + api_key=exempt, base_url=str(candidate.client.base_url).rstrip("/"), max_retries=0 + ).messages.create( + model=claude_model, max_tokens=16, messages=[{"role": "user", "content": anthropic_marker}] + ) + assert _wire_hits(anthropic_wire, anthropic_marker) == 1 + + +def test_team_flag_resaved_key_and_spend_log(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(denying_guardrail) as policy: + config: Final = guardrail_config(policy.url, tmp_path / "team-exempt.yaml") + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model() + member: Final = scenario.user(user_role="internal_user") + + exempt_team: Final = scenario.team(models=[model], disable_global_guardrails=True) + team_key: Final = scenario.key(team_id=exempt_team, models=[model]) + team_marker: Final = "team-exempt-" + uuid.uuid4().hex + team_response: Final = chat(candidate, model, team_key, team_marker, stream=False) + assert team_response.status_code == 200, team_response.text + assert upstream_hits(gateway, team_marker) == 1 + assert _sink_hits(policy, team_marker) == 0 + + caller_team: Final = scenario.team( + models=[model], members_with_roles=[{"role": "admin", "user_id": member}] + ) + admin_exempt: Final = scenario.key(team_id=caller_team, models=[model], disable_global_guardrails=True) + resave_caller: Final = scenario.key( + user_id=member, team_id=caller_team, models=[model], allowed_routes=["/key/*", "/v1/chat/completions"] + ) + resaved: Final = candidate.request( + "POST", + "/key/update", + { + "key": admin_exempt, + "key_alias": "audit-runtime-resave-" + uuid.uuid4().hex, + "metadata": {"disable_global_guardrails": True}, + }, + key=resave_caller, + ) + assert resaved.status_code == 200, resaved.text + resave_marker: Final = "resaved-exempt-" + uuid.uuid4().hex + resave_response: Final = chat(candidate, model, admin_exempt, resave_marker, stream=False) + assert resave_response.status_code == 200, resave_response.text + response_id: Final = string_value(resave_response.json()["id"]) + assert upstream_hits(gateway, resave_marker) == 1 + assert _sink_hits(policy, resave_marker) == 0 + eventually( + lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id = %s', (response_id,)), + lambda rows: len(rows) == 1, + seconds=70, + ) diff --git a/tests/test_litellm/proxy/management_endpoints/test_common_utils.py b/tests/test_litellm/proxy/management_endpoints/test_common_utils.py index 2b614632346..69013408962 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_common_utils.py +++ b/tests/test_litellm/proxy/management_endpoints/test_common_utils.py @@ -774,6 +774,135 @@ class TestCheckPassthroughRoutesCallerPermission: ) +class TestCheckDisableGlobalGuardrailsCallerPermission: + """Only proxy admins may set disable_global_guardrails (top-level or under + metadata); non-admins get a 403 naming the entity.""" + + def _non_admin(self): + return UserAPIKeyAuth( + user_id="u1", api_key="sk-x", user_role=LitellmUserRoles.INTERNAL_USER + ) + + def _admin(self): + return UserAPIKeyAuth( + user_id="u2", api_key="sk-y", user_role=LitellmUserRoles.PROXY_ADMIN + ) + + def test_top_level_flag_rejected_with_default_entity(self): + from fastapi import HTTPException + + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + with pytest.raises(HTTPException) as exc_info: + _check_disable_global_guardrails_caller_permission(True, None, self._non_admin()) + + assert exc_info.value.status_code == 403 + assert exc_info.value.detail == {"error": "Only proxy admins can set `disable_global_guardrails` on a key."} + + def test_metadata_flag_rejected_with_default_entity(self): + from fastapi import HTTPException + + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + with pytest.raises(HTTPException) as exc_info: + _check_disable_global_guardrails_caller_permission( + None, {"disable_global_guardrails": True}, self._non_admin() + ) + + assert exc_info.value.status_code == 403 + assert exc_info.value.detail == {"error": "Only proxy admins can set `disable_global_guardrails` on a key."} + + def test_explicit_false_with_metadata_true_is_rejected(self): + from fastapi import HTTPException + + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + with pytest.raises(HTTPException) as exc_info: + _check_disable_global_guardrails_caller_permission( + False, {"disable_global_guardrails": True}, self._non_admin() + ) + + assert exc_info.value.status_code == 403 + assert exc_info.value.detail == {"error": "Only proxy admins can set `disable_global_guardrails` on a key."} + + def test_rejection_names_the_team_entity(self): + from fastapi import HTTPException + + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + with pytest.raises(HTTPException) as exc_info: + _check_disable_global_guardrails_caller_permission(True, None, self._non_admin(), entity="team") + + assert exc_info.value.detail == {"error": "Only proxy admins can set `disable_global_guardrails` on a team."} + + def test_false_and_absent_flag_do_not_raise(self): + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + non_admin = self._non_admin() + assert _check_disable_global_guardrails_caller_permission(False, None, non_admin) is None + assert _check_disable_global_guardrails_caller_permission(None, None, non_admin) is None + assert _check_disable_global_guardrails_caller_permission(None, {}, non_admin) is None + assert ( + _check_disable_global_guardrails_caller_permission(None, {"disable_global_guardrails": False}, non_admin) + is None + ) + + def test_unchanged_stored_flag_does_not_raise(self): + """Re-sending a flag that is already stored is not an opt-out.""" + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + non_admin = self._non_admin() + assert ( + _check_disable_global_guardrails_caller_permission( + True, + {"disable_global_guardrails": True}, + non_admin, + existing_metadata={"disable_global_guardrails": True}, + ) + is None + ) + + def test_stored_false_does_not_exempt(self): + from fastapi import HTTPException + + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + with pytest.raises(HTTPException) as exc_info: + _check_disable_global_guardrails_caller_permission( + True, + None, + self._non_admin(), + existing_metadata={"disable_global_guardrails": False}, + ) + + assert exc_info.value.status_code == 403 + assert exc_info.value.detail == {"error": "Only proxy admins can set `disable_global_guardrails` on a key."} + + def test_proxy_admin_may_set_the_flag(self): + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + assert ( + _check_disable_global_guardrails_caller_permission(True, {"disable_global_guardrails": True}, self._admin()) + is None + ) + + class TestIsUserOrgAdminForTeam: """The caller must be looked up with its exact identity; a nulled or omitted lookup argument would silently mis-resolve org-admin status.""" diff --git a/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py index 69f66ca3939..3b86f1f6d20 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py @@ -18020,6 +18020,199 @@ async def test_regenerate_key_non_admin_permissions_rejected_before_enterprise_g assert "Enterprise" not in str(exc.value.message) +@pytest.mark.asyncio +async def test_generate_key_non_admin_disable_global_guardrails_rejected(monkeypatch): + """`_common_key_generation_helper` rejects a non-admin setting + `disable_global_guardrails` on the request body.""" + monkeypatch.setattr( + "litellm.proxy.management_endpoints.key_management_endpoints.litellm.default_key_generate_params", + None, + raising=False, + ) + caller = UserAPIKeyAuth( + user_role=LitellmUserRoles.INTERNAL_USER, + user_id="user-1", + max_budget=100.0, + ) + request = GenerateKeyRequest(disable_global_guardrails=True) + with pytest.raises(HTTPException) as exc_info: + await _common_key_generation_helper( + data=request, + user_api_key_dict=caller, + litellm_changed_by=None, + team_table=None, + ) + assert exc_info.value.status_code == 403 + assert "disable_global_guardrails" in str(exc_info.value.detail) + + +@pytest.mark.asyncio +async def test_generate_key_non_admin_metadata_disable_global_guardrails_rejected(monkeypatch): + """`_common_key_generation_helper` rejects a non-admin smuggling + `disable_global_guardrails` under `metadata`.""" + monkeypatch.setattr( + "litellm.proxy.management_endpoints.key_management_endpoints.litellm.default_key_generate_params", + None, + raising=False, + ) + caller = UserAPIKeyAuth( + user_role=LitellmUserRoles.INTERNAL_USER, + user_id="user-1", + max_budget=100.0, + ) + request = GenerateKeyRequest(metadata={"disable_global_guardrails": True}) + with pytest.raises(HTTPException) as exc_info: + await _common_key_generation_helper( + data=request, + user_api_key_dict=caller, + litellm_changed_by=None, + team_table=None, + ) + assert exc_info.value.status_code == 403 + assert "disable_global_guardrails" in str(exc_info.value.detail) + + +@pytest.mark.asyncio +async def test_generate_key_non_admin_server_default_guardrail_flag_not_treated_as_requested(monkeypatch): + """An admin-configured `default_key_generate_params.metadata` containing + `disable_global_guardrails: true` must not 403 a non-admin who sent no flag; + only caller-sent metadata counts as requesting the opt-out.""" + monkeypatch.setattr( + "litellm.proxy.management_endpoints.key_management_endpoints.litellm.default_key_generate_params", + {"metadata": {"disable_global_guardrails": True}}, + raising=False, + ) + caller = UserAPIKeyAuth( + user_role=LitellmUserRoles.INTERNAL_USER, + user_id="user-1", + max_budget=100.0, + ) + + raised: Exception | None = None + try: + await _common_key_generation_helper( + data=GenerateKeyRequest(team_id="team-1", models=["gpt-4o"]), + user_api_key_dict=caller, + litellm_changed_by=None, + team_table=None, + ) + except Exception as exc: + raised = exc + assert not (isinstance(raised, HTTPException) and "disable_global_guardrails" in str(raised.detail)), raised + + with pytest.raises(HTTPException) as exc_info: + await _common_key_generation_helper( + data=GenerateKeyRequest( + team_id="team-1", + models=["gpt-4o"], + metadata={"disable_global_guardrails": True}, + ), + user_api_key_dict=caller, + litellm_changed_by=None, + team_table=None, + ) + assert exc_info.value.status_code == 403 + assert "disable_global_guardrails" in str(exc_info.value.detail) + + +@pytest.mark.asyncio +async def test_update_key_non_admin_disable_global_guardrails_rejected(monkeypatch): + """`_validate_update_key_data` rejects a non-admin when + `disable_global_guardrails` is true in the request body.""" + mock_prisma_client = AsyncMock() + mock_prisma_client.jsonify_object = lambda data: data + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client) + + data = UpdateKeyRequest( + key="sk-alice-personal", + disable_global_guardrails=True, + ) + + with pytest.raises(HTTPException) as exc: + await _validate_update_key_data( + data=data, + existing_key_row=_make_personal_key_row_for_alice(), + user_api_key_dict=_make_alice_internal_user(), + llm_router=None, + premium_user=True, + prisma_client=mock_prisma_client, + user_api_key_cache=MagicMock(), + ) + assert exc.value.status_code == 403 + assert "disable_global_guardrails" in str(exc.value.detail) + + +@pytest.mark.asyncio +async def test_update_key_non_admin_resending_stored_disable_global_guardrails_allowed(monkeypatch): + """`_validate_update_key_data` must not 403 when a non-admin edit form + re-sends `metadata.disable_global_guardrails` that is already stored on + the key (the Admin UI edit form round-trips the whole metadata JSON).""" + mock_prisma_client = AsyncMock() + mock_prisma_client.jsonify_object = lambda data: data + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client) + + existing_key_row = _make_personal_key_row_for_alice() + existing_key_row.metadata = {"disable_global_guardrails": True} + data = UpdateKeyRequest( + key="sk-alice-personal", + metadata={"disable_global_guardrails": True, "x": 1}, + ) + + raised: HTTPException | None = None + try: + await _validate_update_key_data( + data=data, + existing_key_row=existing_key_row, + user_api_key_dict=_make_alice_internal_user(), + llm_router=None, + premium_user=True, + prisma_client=mock_prisma_client, + user_api_key_cache=MagicMock(), + ) + except HTTPException as exc: + raised = exc + assert raised is None or "disable_global_guardrails" not in str(raised.detail) + + +@pytest.mark.asyncio +async def test_regenerate_key_non_admin_disable_global_guardrails_rejected(monkeypatch): + """`regenerate_key_fn` rejects a non-admin setting + `disable_global_guardrails` once the stored key row is loaded (the + already-stored exemption check needs the row's metadata).""" + from litellm.proxy._types import RegenerateKeyRequest + from litellm.proxy.management_endpoints.key_management_endpoints import ( + regenerate_key_fn, + ) + + existing_key = _make_regenerate_existing_key() + mock_prisma_client = AsyncMock() + mock_repo = MagicMock() + mock_repo.table.find_unique = AsyncMock(return_value=existing_key) + + data = RegenerateKeyRequest( + key="sk-alice-personal", + disable_global_guardrails=True, + ) + + with ( + patch("litellm.proxy.proxy_server.premium_user", True), + patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client), + patch( + "litellm.proxy.management_endpoints.key_management_endpoints.VerificationTokenRepository", + return_value=mock_repo, + ), + pytest.raises(ProxyException) as exc, + ): + await regenerate_key_fn( + key=None, + data=data, + user_api_key_dict=_make_alice_internal_user(), + litellm_changed_by=None, + ) + assert int(exc.value.code) == 403 + assert "disable_global_guardrails" in str(exc.value.message) + + def test_generate_key_helper_fn_accepts_per_tag_rate_limits(): """ Regression: new_user / SSO sign-in forward NewUserRequest fields to diff --git a/tests/test_litellm/proxy/management_endpoints/test_team_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_team_endpoints.py index 7a9b6b66946..b066b3b80e6 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_team_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_team_endpoints.py @@ -11592,6 +11592,83 @@ async def test_update_team_blocks_non_admin_passthrough_routes(mock_db_client): assert "allowed_passthrough_routes" in str(exc.value.message) +def test_check_disable_global_guardrails_caller_permission_team(): + from litellm.proxy._types import NewTeamRequest + from litellm.proxy.management_endpoints.common_utils import ( + _check_disable_global_guardrails_caller_permission, + ) + + admin = UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN) + non_admin = _non_admin_auth() + + _check_disable_global_guardrails_caller_permission(True, {"disable_global_guardrails": True}, admin, entity="team") + _check_disable_global_guardrails_caller_permission(None, None, non_admin, entity="team") + _check_disable_global_guardrails_caller_permission(False, None, non_admin, entity="team") + + with pytest.raises(HTTPException) as exc: + _check_disable_global_guardrails_caller_permission(True, None, non_admin, entity="team") + assert exc.value.status_code == 403 + assert "disable_global_guardrails" in str(exc.value.detail) + assert "team" in str(exc.value.detail) + + with pytest.raises(HTTPException) as exc: + _check_disable_global_guardrails_caller_permission( + None, {"disable_global_guardrails": True}, non_admin, entity="team" + ) + assert exc.value.status_code == 403 + assert "disable_global_guardrails" in str(exc.value.detail) + + +@pytest.mark.asyncio +async def test_new_team_blocks_non_admin_disable_global_guardrails(mock_db_client): + """A non-proxy-admin cannot opt a team out of global guardrails via /team/new.""" + mock_db_client.db.litellm_teamtable.count = AsyncMock(return_value=0) + from fastapi import Request + + from litellm.proxy._types import NewTeamRequest, ProxyException + from litellm.proxy.management_endpoints.team_endpoints import new_team + + with patch( + "litellm.proxy.management_endpoints.team_endpoints._check_user_team_limits", + AsyncMock(return_value=None), + ): + with pytest.raises(ProxyException) as exc: + await new_team( + data=NewTeamRequest(team_alias="t", disable_global_guardrails=True), + http_request=MagicMock(spec=Request), + user_api_key_dict=_non_admin_auth(), + ) + assert str(exc.value.code) == "403" + assert "disable_global_guardrails" in str(exc.value.message) + + +@pytest.mark.asyncio +async def test_update_team_blocks_non_admin_disable_global_guardrails(mock_db_client): + """Even a team manager (non-proxy-admin) cannot set + disable_global_guardrails via /team/update.""" + from fastapi import Request + + from litellm.proxy._types import ProxyException, UpdateTeamRequest + from litellm.proxy.management_endpoints.team_endpoints import update_team + + existing = MagicMock() + existing.model_dump.return_value = {"team_id": "t1"} + mock_db_client.db.litellm_teamtable.find_unique = AsyncMock(return_value=existing) + + with patch( + "litellm.proxy.management_endpoints.team_endpoints._resolve_team_access", + AsyncMock(return_value="org_admin"), + ): + with pytest.raises(ProxyException) as exc: + await update_team( + data=UpdateTeamRequest(team_id="t1", disable_global_guardrails=True), + http_request=MagicMock(spec=Request), + user_api_key_dict=_non_admin_auth(), + ) + assert str(exc.value.code) == "403" + assert "disable_global_guardrails" in str(exc.value.message) + + def test_set_budget_reset_at_clears_when_budget_duration_null(): """ When budget_duration is explicitly set to null, _set_budget_reset_at diff --git a/ui/litellm-dashboard/src/components/Teams.test.tsx b/ui/litellm-dashboard/src/components/Teams.test.tsx index b6c6bbbcdd9..5ff95d2af0c 100644 --- a/ui/litellm-dashboard/src/components/Teams.test.tsx +++ b/ui/litellm-dashboard/src/components/Teams.test.tsx @@ -1779,3 +1779,50 @@ describe("Teams - the create form keeps the organization and models picks while expect(modelsField()).toHaveValue(""); }); }); + +describe("Teams - disable_global_guardrails switch gating", () => { + const openCreateModal = async () => { + act(() => { + fireEvent.click(screen.getAllByRole("button", { name: /create team/i })[0]); + }); + await screen.findByLabelText(/team name/i); + }; + + beforeEach(() => { + vi.clearAllMocks(); + mockTeamInfoView.mockClear(); + vi.mocked(fetchAvailableModelsForTeamOrKey).mockResolvedValue(["gpt-4"]); + vi.mocked(fetchMCPAccessGroups).mockResolvedValue([]); + vi.mocked(getGuardrailsList).mockResolvedValue({ guardrails: [] }); + vi.mocked(getDefaultTeamSettings).mockResolvedValue({ values: {} }); + mockUseOrganizations.mockReturnValue({ data: null }); + }); + + it("hides the Disable Global Guardrails switch from a non-admin", async () => { + mockUseOrganizations.mockReturnValue({ + data: [ + { + organization_id: "org-1", + organization_alias: "Org 1", + models: [], + members: [{ user_id: "user-123", user_role: "org_admin" }], + }, + ], + }); + renderWithQueryClient(); + await openCreateModal(); + + fireEvent.click(screen.getByText("Additional Settings")); + + expect(screen.queryByRole("switch", { name: /Disable Global Guardrails/i })).not.toBeInTheDocument(); + }); + + it("shows the Disable Global Guardrails switch to a proxy admin", async () => { + renderWithQueryClient(); + await openCreateModal(); + + fireEvent.click(screen.getByText("Additional Settings")); + + expect(await screen.findByRole("switch", { name: /Disable Global Guardrails/i })).toBeInTheDocument(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/Teams.tsx b/ui/litellm-dashboard/src/components/Teams.tsx index 7214d16f665..c2a23cef83a 100644 --- a/ui/litellm-dashboard/src/components/Teams.tsx +++ b/ui/litellm-dashboard/src/components/Teams.tsx @@ -983,29 +983,31 @@ const Teams: React.FC = ({ accessToken, userID, userRole, premiumUser /> )} - - {({ id, value, onChange }) => ( - - )} - + {isProxyAdminRole(userRole || "") && ( + + {({ id, value, onChange }) => ( + + )} + + )} {canViewPolicies && ( { expect((await createdPayload()).disable_global_guardrails).toBe(true); }); + it("hides the disable_global_guardrails switch from a non-admin", async () => { + state.authorized = { ...state.authorized, userRole: "Internal User" }; + await openModal(); + await openSection(/Optional Settings/i); + + expect(screen.queryByRole("switch", { name: /Disable Global Guardrails/i })).not.toBeInTheDocument(); + }); + + it("shows the disable_global_guardrails switch to a proxy admin", async () => { + await openModal(); + await openSection(/Optional Settings/i); + + expect(await screen.findByRole("switch", { name: /Disable Global Guardrails/i })).toBeInTheDocument(); + }); + it("folds a metadata JSON string back through JSON.stringify", async () => { await openModal(); await nameTheKey(); diff --git a/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx b/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx index 25b986e4c9e..45245ff95b3 100644 --- a/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx +++ b/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx @@ -1293,40 +1293,42 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp /> )} - - Disable Global Guardrails{" "} - - e.stopPropagation()} // Prevent accordion from collapsing when clicking link - > - - - - - } - name="disable_global_guardrails" - className="mt-4" - help={ - canEditGuardrails - ? "Bypass global guardrails for this key" - : "Premium feature - Upgrade to disable global guardrails by key" - } - > - {(control) => ( - - )} - + {userRole != null && isProxyAdminRole(userRole) && ( + + Disable Global Guardrails{" "} + + e.stopPropagation()} // Prevent accordion from collapsing when clicking link + > + + + + + } + name="disable_global_guardrails" + className="mt-4" + help={ + canEditGuardrails + ? "Bypass global guardrails for this key" + : "Premium feature - Upgrade to disable global guardrails by key" + } + > + {(control) => ( + + )} + + )} {canViewPolicies && ( { errorToast.mockRestore(); }); }); + +describe("TeamInfoView - disable_global_guardrails switch gating", () => { + beforeEach(() => { + seedDefaultMocks(); + vi.mocked(networking.teamInfoCall).mockResolvedValue(createMockTeamData()); + }); + + afterEach(() => { + vi.clearAllMocks(); + authState.userRole = "Admin"; + }); + + const props = { + teamId: "123", + onUpdate: vi.fn(), + onClose: vi.fn(), + accessToken: "test-token", + is_team_admin: true, + is_proxy_admin: true, + userModels: ["gpt-4"], + editTeam: false, + premiumUser: false, + }; + + const openEditForm = async () => { + const user = userEvent.setup({ delay: null }); + await waitFor(() => expect(screen.queryAllByText("Test Team").length).toBeGreaterThan(0)); + await user.click(screen.getByRole("tab", { name: "Settings" })); + await user.click(await screen.findByRole("button", { name: /edit settings/i })); + await screen.findByLabelText("Team Name"); + }; + + it("hides the Disable all global guardrails switch from a non-admin", async () => { + authState.userRole = "Internal User"; + renderWithProviders(); + await openEditForm(); + + expect(screen.queryByRole("switch", { name: /Disable all global guardrails/i })).not.toBeInTheDocument(); + }); + + it("shows the Disable all global guardrails switch to a proxy admin", async () => { + renderWithProviders(); + await openEditForm(); + + expect(await screen.findByRole("switch", { name: /Disable all global guardrails/i })).toBeInTheDocument(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/team/TeamInfo.tsx b/ui/litellm-dashboard/src/components/team/TeamInfo.tsx index 22cc99b32c8..d9e308e9d6f 100644 --- a/ui/litellm-dashboard/src/components/team/TeamInfo.tsx +++ b/ui/litellm-dashboard/src/components/team/TeamInfo.tsx @@ -1785,25 +1785,27 @@ const TeamInfoView: React.FC = ({ )} - - {({ id, value, onChange }) => ( - { - onChange(checked); - applyKillSwitchToGuardrails(checked); - }} - /> - )} - + {is_proxy_admin && ( + + {({ id, value, onChange }) => ( + { + onChange(checked); + applyKillSwitchToGuardrails(checked); + }} + /> + )} + + )} {canViewPolicies && ( { }, ); }); + + describe("disable_global_guardrails toggle gating", () => { + const renderAs = (userRole: string) => + renderWithProviders( + {}} + onSubmit={async () => {}} + accessToken="test-token" + userID="test-user" + userRole={userRole} + premiumUser={true} + />, + ); + + it("hides the switch from a non-admin", async () => { + renderAs("Internal User"); + await screen.findByRole("button", { name: /save changes/i }); + + expect(screen.queryByRole("switch", { name: /disable global guardrails/i })).not.toBeInTheDocument(); + }); + + it("shows the switch to a proxy admin", async () => { + renderAs("Admin"); + + expect(await screen.findByRole("switch", { name: /disable global guardrails/i })).toBeInTheDocument(); + }); + }); }); diff --git a/ui/litellm-dashboard/src/components/templates/key_edit_view.tsx b/ui/litellm-dashboard/src/components/templates/key_edit_view.tsx index c668958be74..9cd97f4ef98 100644 --- a/ui/litellm-dashboard/src/components/templates/key_edit_view.tsx +++ b/ui/litellm-dashboard/src/components/templates/key_edit_view.tsx @@ -618,18 +618,20 @@ export function KeyEditView({ } - - {({ value, onChange, ref: _ref, ...field }) => ( - - )} - + {userRole != null && isProxyAdminRole(userRole) && ( + + {({ value, onChange, ref: _ref, ...field }) => ( + + )} + + )} {canViewPolicies && ( Date: Wed, 23 Sep 2026 18:03:09 -0700 Subject: [PATCH 03/96] fix(models): sync openrouter prices from the models API (#42832) * fix(models): sync openrouter prices from the models API Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(models): allow above_32k_tokens cost fields in price map schema test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 77 +++++++++++++------ model_prices_and_context_window.json | 77 +++++++++++++------ model_prices_and_context_window.schema.json | 20 +++++ tests/test_litellm/test_utils.py | 4 + 4 files changed, 128 insertions(+), 50 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 844e613977f..93bcfc9b120 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -40246,30 +40246,30 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 8.8044e-07, + "input_cost_per_token": 9.396e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.76088e-06, + "output_cost_per_token": 1.8792e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.337e-08, + "cache_read_input_token_cost": 7.83e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, "supports_web_search": false }, "openrouter/deepseek/deepseek-v4.1-flash": { - "input_cost_per_token": 3e-07, - "output_cost_per_token": 1.2e-06, - "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 1.4e-07, + "output_cost_per_token": 4.2e-07, + "cache_read_input_token_cost": 4.2e-09, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 943718, @@ -40288,21 +40288,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro-0813": { - "input_cost_per_token": 1.32e-06, + "input_cost_per_token": 4.62e-07, "input_cost_per_token_cache_hit": 1.9272e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 3.96e-06, + "output_cost_per_token": 1.386e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 4.4e-08, + "cache_read_input_token_cost": 1.54e-08, "off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":0.00000132,"output_cost_per_token":0.00000396,"cache_read_input_token_cost":4.4e-8}, "supports_audio_input": false, "supports_pdf_input": false, @@ -40630,13 +40630,13 @@ "max_output_tokens": 8000 }, "openrouter/minimax/minimax-m2": { - "input_cost_per_token": 2.55e-07, + "input_cost_per_token": 3e-07, "litellm_provider": "openrouter", "max_input_tokens": 204800, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 1.02e-06, + "output_cost_per_token": 1.2e-06, "source": "https://openrouter.ai/api/v1/models", "supports_audio_input": false, "supports_function_calling": true, @@ -40843,7 +40843,7 @@ }, "openrouter/nvidia/nemotron-3.5-lightning": { "cache_read_input_token_cost": 4e-08, - "input_cost_per_token": 7e-08, + "input_cost_per_token": 8e-08, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 235929, @@ -41484,6 +41484,12 @@ }, "openrouter/qwen/qwen3-coder-plus": { "cache_creation_input_token_cost": 8.125e-07, + "cache_creation_input_token_cost_above_128k_tokens": 2.4375e-06, + "cache_read_input_token_cost_above_128k_tokens": 3.9e-07, + "input_cost_per_token_above_32k_tokens": 1.17e-06, + "cache_creation_input_token_cost_above_32k_tokens": 1.4625e-06, + "cache_read_input_token_cost_above_32k_tokens": 2.34e-07, + "output_cost_per_token_above_32k_tokens": 5.85e-06, "cache_read_input_token_cost": 1.3e-07, "input_cost_per_token": 6.5e-07, "input_cost_per_token_above_128k_tokens": 1.95e-06, @@ -41546,6 +41552,9 @@ }, "openrouter/qwen/qwen3.6-plus": { "cache_creation_input_token_cost": 4.0625e-07, + "input_cost_per_token_above_256k_tokens": 1.3e-06, + "cache_creation_input_token_cost_above_256k_tokens": 1.625e-06, + "output_cost_per_token_above_256k_tokens": 3.9e-06, "input_cost_per_token": 3.25e-07, "litellm_provider": "openrouter", "max_input_tokens": 1000000, @@ -41643,14 +41652,14 @@ }, "openrouter/qwen/qwen3.5-plus-02-15": { "input_cost_per_token": 2.6e-07, - "input_cost_per_token_above_256k_tokens": 5e-07, + "input_cost_per_token_above_256k_tokens": 3.25e-07, "litellm_provider": "openrouter", "max_input_tokens": 1000000, "max_output_tokens": 65536, "max_tokens": 65536, "mode": "chat", "output_cost_per_token": 1.56e-06, - "output_cost_per_token_above_256k_tokens": 3e-06, + "output_cost_per_token_above_256k_tokens": 1.95e-06, "source": "https://openrouter.ai/api/v1/models", "supports_audio_input": false, "supports_function_calling": true, @@ -64209,6 +64218,10 @@ }, "openrouter/qwen/qwen3.7-plus": { "input_cost_per_token": 3.2e-07, + "input_cost_per_token_above_256k_tokens": 9.6e-07, + "cache_creation_input_token_cost_above_256k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_256k_tokens": 1.92e-07, + "output_cost_per_token_above_256k_tokens": 3.84e-06, "output_cost_per_token": 1.28e-06, "litellm_provider": "openrouter", "max_input_tokens": 1000000, @@ -64340,9 +64353,9 @@ "supports_web_search": false }, "openrouter/z-ai/glm-5.3": { - "input_cost_per_token": 6.538e-07, - "output_cost_per_token": 2.0548e-06, - "cache_read_input_token_cost": 1.2142e-07, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, "litellm_provider": "openrouter", "max_input_tokens": 1310720, "max_output_tokens": 131072, @@ -64499,6 +64512,10 @@ }, "openrouter/qwen/qwen3.7-flash": { "input_cost_per_token": 3e-08, + "input_cost_per_token_above_32k_tokens": 1e-07, + "cache_creation_input_token_cost_above_32k_tokens": 1.25e-07, + "cache_read_input_token_cost_above_32k_tokens": 2e-08, + "output_cost_per_token_above_32k_tokens": 4e-07, "output_cost_per_token": 1.3e-07, "cache_read_input_token_cost": 6e-09, "cache_creation_input_token_cost": 3.8e-08, @@ -65047,9 +65064,9 @@ "supports_web_search": true }, "openrouter/deepseek/deepseek-v4-flash": { - "input_cost_per_token": 4.9e-08, - "output_cost_per_token": 9.8e-08, - "cache_read_input_token_cost": 9.8e-09, + "input_cost_per_token": 8.8606e-08, + "output_cost_per_token": 1.77212e-07, + "cache_read_input_token_cost": 1.77212e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, @@ -65389,6 +65406,8 @@ }, "openrouter/qwen/qwen3-max-thinking": { "input_cost_per_token": 7.8e-07, + "input_cost_per_token_above_32k_tokens": 1.56e-06, + "output_cost_per_token_above_32k_tokens": 7.8e-06, "output_cost_per_token": 3.9e-06, "input_cost_per_token_above_128k_tokens": 1.95e-06, "output_cost_per_token_above_128k_tokens": 9.75e-06, @@ -65835,6 +65854,10 @@ }, "openrouter/qwen/qwen3-max": { "input_cost_per_token": 7.8e-07, + "input_cost_per_token_above_32k_tokens": 1.56e-06, + "cache_creation_input_token_cost_above_32k_tokens": 1.95e-06, + "cache_read_input_token_cost_above_32k_tokens": 3.12e-07, + "output_cost_per_token_above_32k_tokens": 7.8e-06, "output_cost_per_token": 3.9e-06, "cache_read_input_token_cost": 1.56e-07, "cache_creation_input_token_cost": 9.75e-07, @@ -65881,6 +65904,10 @@ }, "openrouter/qwen/qwen3-coder-flash": { "input_cost_per_token": 1.95e-07, + "input_cost_per_token_above_32k_tokens": 3.25e-07, + "cache_creation_input_token_cost_above_32k_tokens": 4.0625e-07, + "cache_read_input_token_cost_above_32k_tokens": 6.5e-08, + "output_cost_per_token_above_32k_tokens": 1.625e-06, "output_cost_per_token": 9.75e-07, "cache_read_input_token_cost": 3.9e-08, "cache_creation_input_token_cost": 2.4375e-07, @@ -65924,7 +65951,7 @@ "supports_web_search": false }, "openrouter/qwen/qwen3-next-80b-a3b-instruct": { - "input_cost_per_token": 9e-08, + "input_cost_per_token": 1e-07, "output_cost_per_token": 1.1e-06, "cache_read_input_token_cost": 7e-08, "litellm_provider": "openrouter", @@ -72817,14 +72844,14 @@ "supports_web_search": false }, "openrouter/z-ai/glm-5.3:batch": { - "cache_read_input_token_cost": 1.2e-07, - "input_cost_per_token": 7.2e-07, + "cache_read_input_token_cost": 1e-07, + "input_cost_per_token": 4.5e-07, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 2.4e-06, + "output_cost_per_token": 2e-06, "source": "https://openrouter.ai/api/v1/models", "supports_audio_input": false, "supports_function_calling": true, @@ -72876,7 +72903,7 @@ "supports_web_search": false }, "openrouter/z-ai/glm-5.3-flashx": { - "cache_read_input_token_cost": 7.5e-08, + "cache_read_input_token_cost": 9e-08, "deprecation_date": "2098-12-31", "input_cost_per_token": 3.7e-07, "litellm_provider": "openrouter", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 844e613977f..93bcfc9b120 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -40246,30 +40246,30 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 8.8044e-07, + "input_cost_per_token": 9.396e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.76088e-06, + "output_cost_per_token": 1.8792e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.337e-08, + "cache_read_input_token_cost": 7.83e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, "supports_web_search": false }, "openrouter/deepseek/deepseek-v4.1-flash": { - "input_cost_per_token": 3e-07, - "output_cost_per_token": 1.2e-06, - "cache_read_input_token_cost": 6e-09, + "input_cost_per_token": 1.4e-07, + "output_cost_per_token": 4.2e-07, + "cache_read_input_token_cost": 4.2e-09, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 943718, @@ -40288,21 +40288,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro-0813": { - "input_cost_per_token": 1.32e-06, + "input_cost_per_token": 4.62e-07, "input_cost_per_token_cache_hit": 1.9272e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 3.96e-06, + "output_cost_per_token": 1.386e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 4.4e-08, + "cache_read_input_token_cost": 1.54e-08, "off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":0.00000132,"output_cost_per_token":0.00000396,"cache_read_input_token_cost":4.4e-8}, "supports_audio_input": false, "supports_pdf_input": false, @@ -40630,13 +40630,13 @@ "max_output_tokens": 8000 }, "openrouter/minimax/minimax-m2": { - "input_cost_per_token": 2.55e-07, + "input_cost_per_token": 3e-07, "litellm_provider": "openrouter", "max_input_tokens": 204800, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 1.02e-06, + "output_cost_per_token": 1.2e-06, "source": "https://openrouter.ai/api/v1/models", "supports_audio_input": false, "supports_function_calling": true, @@ -40843,7 +40843,7 @@ }, "openrouter/nvidia/nemotron-3.5-lightning": { "cache_read_input_token_cost": 4e-08, - "input_cost_per_token": 7e-08, + "input_cost_per_token": 8e-08, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 235929, @@ -41484,6 +41484,12 @@ }, "openrouter/qwen/qwen3-coder-plus": { "cache_creation_input_token_cost": 8.125e-07, + "cache_creation_input_token_cost_above_128k_tokens": 2.4375e-06, + "cache_read_input_token_cost_above_128k_tokens": 3.9e-07, + "input_cost_per_token_above_32k_tokens": 1.17e-06, + "cache_creation_input_token_cost_above_32k_tokens": 1.4625e-06, + "cache_read_input_token_cost_above_32k_tokens": 2.34e-07, + "output_cost_per_token_above_32k_tokens": 5.85e-06, "cache_read_input_token_cost": 1.3e-07, "input_cost_per_token": 6.5e-07, "input_cost_per_token_above_128k_tokens": 1.95e-06, @@ -41546,6 +41552,9 @@ }, "openrouter/qwen/qwen3.6-plus": { "cache_creation_input_token_cost": 4.0625e-07, + "input_cost_per_token_above_256k_tokens": 1.3e-06, + "cache_creation_input_token_cost_above_256k_tokens": 1.625e-06, + "output_cost_per_token_above_256k_tokens": 3.9e-06, "input_cost_per_token": 3.25e-07, "litellm_provider": "openrouter", "max_input_tokens": 1000000, @@ -41643,14 +41652,14 @@ }, "openrouter/qwen/qwen3.5-plus-02-15": { "input_cost_per_token": 2.6e-07, - "input_cost_per_token_above_256k_tokens": 5e-07, + "input_cost_per_token_above_256k_tokens": 3.25e-07, "litellm_provider": "openrouter", "max_input_tokens": 1000000, "max_output_tokens": 65536, "max_tokens": 65536, "mode": "chat", "output_cost_per_token": 1.56e-06, - "output_cost_per_token_above_256k_tokens": 3e-06, + "output_cost_per_token_above_256k_tokens": 1.95e-06, "source": "https://openrouter.ai/api/v1/models", "supports_audio_input": false, "supports_function_calling": true, @@ -64209,6 +64218,10 @@ }, "openrouter/qwen/qwen3.7-plus": { "input_cost_per_token": 3.2e-07, + "input_cost_per_token_above_256k_tokens": 9.6e-07, + "cache_creation_input_token_cost_above_256k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_256k_tokens": 1.92e-07, + "output_cost_per_token_above_256k_tokens": 3.84e-06, "output_cost_per_token": 1.28e-06, "litellm_provider": "openrouter", "max_input_tokens": 1000000, @@ -64340,9 +64353,9 @@ "supports_web_search": false }, "openrouter/z-ai/glm-5.3": { - "input_cost_per_token": 6.538e-07, - "output_cost_per_token": 2.0548e-06, - "cache_read_input_token_cost": 1.2142e-07, + "input_cost_per_token": 8.4e-07, + "output_cost_per_token": 2.64e-06, + "cache_read_input_token_cost": 1.56e-07, "litellm_provider": "openrouter", "max_input_tokens": 1310720, "max_output_tokens": 131072, @@ -64499,6 +64512,10 @@ }, "openrouter/qwen/qwen3.7-flash": { "input_cost_per_token": 3e-08, + "input_cost_per_token_above_32k_tokens": 1e-07, + "cache_creation_input_token_cost_above_32k_tokens": 1.25e-07, + "cache_read_input_token_cost_above_32k_tokens": 2e-08, + "output_cost_per_token_above_32k_tokens": 4e-07, "output_cost_per_token": 1.3e-07, "cache_read_input_token_cost": 6e-09, "cache_creation_input_token_cost": 3.8e-08, @@ -65047,9 +65064,9 @@ "supports_web_search": true }, "openrouter/deepseek/deepseek-v4-flash": { - "input_cost_per_token": 4.9e-08, - "output_cost_per_token": 9.8e-08, - "cache_read_input_token_cost": 9.8e-09, + "input_cost_per_token": 8.8606e-08, + "output_cost_per_token": 1.77212e-07, + "cache_read_input_token_cost": 1.77212e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, @@ -65389,6 +65406,8 @@ }, "openrouter/qwen/qwen3-max-thinking": { "input_cost_per_token": 7.8e-07, + "input_cost_per_token_above_32k_tokens": 1.56e-06, + "output_cost_per_token_above_32k_tokens": 7.8e-06, "output_cost_per_token": 3.9e-06, "input_cost_per_token_above_128k_tokens": 1.95e-06, "output_cost_per_token_above_128k_tokens": 9.75e-06, @@ -65835,6 +65854,10 @@ }, "openrouter/qwen/qwen3-max": { "input_cost_per_token": 7.8e-07, + "input_cost_per_token_above_32k_tokens": 1.56e-06, + "cache_creation_input_token_cost_above_32k_tokens": 1.95e-06, + "cache_read_input_token_cost_above_32k_tokens": 3.12e-07, + "output_cost_per_token_above_32k_tokens": 7.8e-06, "output_cost_per_token": 3.9e-06, "cache_read_input_token_cost": 1.56e-07, "cache_creation_input_token_cost": 9.75e-07, @@ -65881,6 +65904,10 @@ }, "openrouter/qwen/qwen3-coder-flash": { "input_cost_per_token": 1.95e-07, + "input_cost_per_token_above_32k_tokens": 3.25e-07, + "cache_creation_input_token_cost_above_32k_tokens": 4.0625e-07, + "cache_read_input_token_cost_above_32k_tokens": 6.5e-08, + "output_cost_per_token_above_32k_tokens": 1.625e-06, "output_cost_per_token": 9.75e-07, "cache_read_input_token_cost": 3.9e-08, "cache_creation_input_token_cost": 2.4375e-07, @@ -65924,7 +65951,7 @@ "supports_web_search": false }, "openrouter/qwen/qwen3-next-80b-a3b-instruct": { - "input_cost_per_token": 9e-08, + "input_cost_per_token": 1e-07, "output_cost_per_token": 1.1e-06, "cache_read_input_token_cost": 7e-08, "litellm_provider": "openrouter", @@ -72817,14 +72844,14 @@ "supports_web_search": false }, "openrouter/z-ai/glm-5.3:batch": { - "cache_read_input_token_cost": 1.2e-07, - "input_cost_per_token": 7.2e-07, + "cache_read_input_token_cost": 1e-07, + "input_cost_per_token": 4.5e-07, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 2.4e-06, + "output_cost_per_token": 2e-06, "source": "https://openrouter.ai/api/v1/models", "supports_audio_input": false, "supports_function_calling": true, @@ -72876,7 +72903,7 @@ "supports_web_search": false }, "openrouter/z-ai/glm-5.3-flashx": { - "cache_read_input_token_cost": 7.5e-08, + "cache_read_input_token_cost": 9e-08, "deprecation_date": "2098-12-31", "input_cost_per_token": 3.7e-07, "litellm_provider": "openrouter", diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 737a9b7fa60..395b2db1137 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -128,6 +128,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "cache_creation_input_token_cost_above_32k_tokens": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "cache_creation_input_token_cost_batches": { "type": "number", "minimum": 0 @@ -195,6 +200,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "cache_read_input_token_cost_above_32k_tokens": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "cache_read_input_token_cost_above_512k_tokens": { "type": "number", "minimum": 0, @@ -371,6 +381,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "input_cost_per_token_above_32k_tokens": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "input_cost_per_token_above_512k_tokens": { "type": "number", "minimum": 0, @@ -707,6 +722,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "output_cost_per_token_above_32k_tokens": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "output_cost_per_token_above_512k_tokens": { "type": "number", "minimum": 0, diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 2a8b31c12cc..1c2d86841f7 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -755,6 +755,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_audio_token_cost": {"type": "number"}, "cache_creation_input_token_cost": {"type": "number"}, "cache_creation_input_token_cost_above_1hr": {"type": "number"}, + "cache_creation_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_256k_tokens": {"type": "number"}, @@ -766,6 +767,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_token_cost_flex": {"type": "number"}, "cache_creation_input_token_cost_priority": {"type": "number"}, "cache_read_input_token_cost": {"type": "number"}, + "cache_read_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_256k_tokens": {"type": "number"}, @@ -789,6 +791,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "input_cost_per_image": {"type": "number"}, "input_cost_per_image_above_128k_tokens": {"type": "number"}, "input_cost_per_video_token": {"type": "number"}, + "input_cost_per_token_above_32k_tokens": {"type": "number"}, "input_cost_per_token_above_200k_tokens": {"type": "number"}, "input_cost_per_token_above_256k_tokens": {"type": "number"}, "input_cost_per_token_above_272k_tokens": {"type": "number"}, @@ -883,6 +886,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "output_cost_per_second_1080p": {"type": "number"}, "output_cost_per_second_4k": {"type": "number"}, "output_cost_per_token": {"type": "number"}, + "output_cost_per_token_above_32k_tokens": {"type": "number"}, "output_cost_per_token_above_128k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens": {"type": "number"}, "output_cost_per_token_above_256k_tokens": {"type": "number"}, From d8f032cda5483ebd83821f97aadf14e3c27addd2 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:05:06 -0700 Subject: [PATCH 04/96] feat(models): add gemini preview aliases and deep research 04-2026 rows (#42833) * feat(models): add gemini preview aliases and deep research 04-2026 rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(models): add tpm and rpm to gemini deep-research 04-2026 rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 247 ++++++++++++++++++ model_prices_and_context_window.json | 247 ++++++++++++++++++ 2 files changed, 494 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 93bcfc9b120..24dfef170c3 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -27474,6 +27474,253 @@ "supports_vision": true, "tpm": 10000000 }, + "gemini/gemini-3-pro-image-preview": { + "input_cost_per_image": 0.0011, + "input_cost_per_token": 2e-06, + "input_cost_per_token_batches": 1e-06, + "input_cost_per_token_flex": 1e-06, + "input_cost_per_token_priority": 3.6e-06, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "image_generation", + "output_cost_per_image": 0.134, + "output_cost_per_image_token": 0.00012, + "output_cost_per_token": 1.2e-05, + "rpm": 1000, + "tpm": 4000000, + "output_cost_per_token_batches": 6e-06, + "output_cost_per_token_flex": 6e-06, + "output_cost_per_token_priority": 2.16e-05, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text", + "image" + ], + "supports_function_calling": false, + "supports_prompt_caching": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_vision": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.014, + "search_context_size_medium": 0.014, + "search_context_size_high": 0.014 + }, + "web_search_billing_unit": "per_query", + "supports_reasoning": false + }, + "gemini/gemini-3.1-flash-image-preview": { + "input_cost_per_token": 5e-07, + "input_cost_per_token_batches": 2.5e-07, + "litellm_provider": "gemini", + "max_input_tokens": 65536, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "image_generation", + "output_cost_per_image": 0.045, + "output_cost_per_image_token": 6e-05, + "output_cost_per_token": 3e-06, + "output_cost_per_token_batches": 1.5e-06, + "rpm": 1000, + "tpm": 4000000, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text", + "image" + ], + "supports_function_calling": false, + "supports_prompt_caching": true, + "supports_reasoning": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_vision": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.014, + "search_context_size_medium": 0.014, + "search_context_size_high": 0.014 + }, + "web_search_billing_unit": "per_query" + }, + "gemini/gemini-3.1-flash-lite-preview": { + "cache_read_input_audio_token_cost": 5e-08, + "cache_read_input_token_cost": 2.5e-08, + "cache_read_input_token_cost_batches": 1.25e-08, + "cache_read_input_token_cost_flex": 1.25e-08, + "cache_read_input_token_cost_priority": 4.5e-08, + "input_cost_per_audio_token": 5e-07, + "input_cost_per_token": 2.5e-07, + "input_cost_per_token_batches": 1.25e-07, + "input_cost_per_token_flex": 1.25e-07, + "input_cost_per_token_priority": 4.5e-07, + "litellm_provider": "gemini", + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_reasoning_token": 1.5e-06, + "output_cost_per_token": 1.5e-06, + "output_cost_per_token_batches": 7.5e-07, + "output_cost_per_token_flex": 7.5e-07, + "output_cost_per_token_priority": 2.7e-06, + "rpm": 15, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_input": true, + "supports_audio_output": false, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_url_context": true, + "supports_video_input": true, + "supports_vision": true, + "supports_web_search": true, + "supports_native_streaming": true, + "tpm": 250000, + "search_context_cost_per_query": { + "search_context_size_low": 0.014, + "search_context_size_medium": 0.014, + "search_context_size_high": 0.014 + }, + "web_search_billing_unit": "per_query", + "google_maps_grounding_cost_per_query": 0.014, + "input_cost_per_audio_token_batches": 2.5e-07 + }, + "gemini/gemini-embedding-2-preview": { + "input_cost_per_audio_token": 6.5e-06, + "input_cost_per_audio_token_batches": 3.25e-06, + "input_cost_per_image_token": 4.5e-07, + "input_cost_per_image_token_batches": 2.25e-07, + "input_cost_per_token": 2e-07, + "input_cost_per_token_batches": 1e-07, + "input_cost_per_video_token": 1.2e-05, + "input_cost_per_video_token_batches": 6e-06, + "litellm_provider": "gemini", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "embedding", + "output_cost_per_token": 0, + "output_vector_size": 3072, + "rpm": 10000, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supports_audio_input": true, + "supports_multimodal": true, + "supports_vision": true, + "tpm": 10000000 + }, + "gemini/deep-research-preview-04-2026": { + "input_cost_per_token": 2e-06, + "input_cost_per_token_batches": 1e-06, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 1.2e-05, + "rpm": 1000, + "tpm": 4000000, + "output_cost_per_token_batches": 6e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_vision": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.035, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.035 + } + }, + "gemini/deep-research-max-preview-04-2026": { + "input_cost_per_token": 2e-06, + "input_cost_per_token_batches": 1e-06, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 1.2e-05, + "rpm": 1000, + "tpm": 4000000, + "output_cost_per_token_batches": 6e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_vision": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.035, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.035 + } + }, "gemini/gemini-2.5-flash": { "cache_read_input_audio_token_cost": 1e-07, "cache_read_input_token_cost": 3e-08, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 93bcfc9b120..24dfef170c3 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -27474,6 +27474,253 @@ "supports_vision": true, "tpm": 10000000 }, + "gemini/gemini-3-pro-image-preview": { + "input_cost_per_image": 0.0011, + "input_cost_per_token": 2e-06, + "input_cost_per_token_batches": 1e-06, + "input_cost_per_token_flex": 1e-06, + "input_cost_per_token_priority": 3.6e-06, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "image_generation", + "output_cost_per_image": 0.134, + "output_cost_per_image_token": 0.00012, + "output_cost_per_token": 1.2e-05, + "rpm": 1000, + "tpm": 4000000, + "output_cost_per_token_batches": 6e-06, + "output_cost_per_token_flex": 6e-06, + "output_cost_per_token_priority": 2.16e-05, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text", + "image" + ], + "supports_function_calling": false, + "supports_prompt_caching": true, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_vision": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.014, + "search_context_size_medium": 0.014, + "search_context_size_high": 0.014 + }, + "web_search_billing_unit": "per_query", + "supports_reasoning": false + }, + "gemini/gemini-3.1-flash-image-preview": { + "input_cost_per_token": 5e-07, + "input_cost_per_token_batches": 2.5e-07, + "litellm_provider": "gemini", + "max_input_tokens": 65536, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "image_generation", + "output_cost_per_image": 0.045, + "output_cost_per_image_token": 6e-05, + "output_cost_per_token": 3e-06, + "output_cost_per_token_batches": 1.5e-06, + "rpm": 1000, + "tpm": 4000000, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text", + "image" + ], + "supports_function_calling": false, + "supports_prompt_caching": true, + "supports_reasoning": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_vision": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.014, + "search_context_size_medium": 0.014, + "search_context_size_high": 0.014 + }, + "web_search_billing_unit": "per_query" + }, + "gemini/gemini-3.1-flash-lite-preview": { + "cache_read_input_audio_token_cost": 5e-08, + "cache_read_input_token_cost": 2.5e-08, + "cache_read_input_token_cost_batches": 1.25e-08, + "cache_read_input_token_cost_flex": 1.25e-08, + "cache_read_input_token_cost_priority": 4.5e-08, + "input_cost_per_audio_token": 5e-07, + "input_cost_per_token": 2.5e-07, + "input_cost_per_token_batches": 1.25e-07, + "input_cost_per_token_flex": 1.25e-07, + "input_cost_per_token_priority": 4.5e-07, + "litellm_provider": "gemini", + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_reasoning_token": 1.5e-06, + "output_cost_per_token": 1.5e-06, + "output_cost_per_token_batches": 7.5e-07, + "output_cost_per_token_flex": 7.5e-07, + "output_cost_per_token_priority": 2.7e-06, + "rpm": 15, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_input": true, + "supports_audio_output": false, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_url_context": true, + "supports_video_input": true, + "supports_vision": true, + "supports_web_search": true, + "supports_native_streaming": true, + "tpm": 250000, + "search_context_cost_per_query": { + "search_context_size_low": 0.014, + "search_context_size_medium": 0.014, + "search_context_size_high": 0.014 + }, + "web_search_billing_unit": "per_query", + "google_maps_grounding_cost_per_query": 0.014, + "input_cost_per_audio_token_batches": 2.5e-07 + }, + "gemini/gemini-embedding-2-preview": { + "input_cost_per_audio_token": 6.5e-06, + "input_cost_per_audio_token_batches": 3.25e-06, + "input_cost_per_image_token": 4.5e-07, + "input_cost_per_image_token_batches": 2.25e-07, + "input_cost_per_token": 2e-07, + "input_cost_per_token_batches": 1e-07, + "input_cost_per_video_token": 1.2e-05, + "input_cost_per_video_token_batches": 6e-06, + "litellm_provider": "gemini", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "embedding", + "output_cost_per_token": 0, + "output_vector_size": 3072, + "rpm": 10000, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supports_audio_input": true, + "supports_multimodal": true, + "supports_vision": true, + "tpm": 10000000 + }, + "gemini/deep-research-preview-04-2026": { + "input_cost_per_token": 2e-06, + "input_cost_per_token_batches": 1e-06, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 1.2e-05, + "rpm": 1000, + "tpm": 4000000, + "output_cost_per_token_batches": 6e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_vision": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.035, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.035 + } + }, + "gemini/deep-research-max-preview-04-2026": { + "input_cost_per_token": 2e-06, + "input_cost_per_token_batches": 1e-06, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 1.2e-05, + "rpm": 1000, + "tpm": 4000000, + "output_cost_per_token_batches": 6e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions", + "/v1/batch" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_vision": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.035, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.035 + } + }, "gemini/gemini-2.5-flash": { "cache_read_input_audio_token_cost": 1e-07, "cache_read_input_token_cost": 3e-08, From 2340dcc30c9914e3eee825ae7b22f2d923fc46ac Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:15:20 -0700 Subject: [PATCH 05/96] docs(pr-template): drop empty sections from the PR body and tighten the User Flow (#42794) --- .github/pull_request_template.md | 13 ++++++++----- AGENTS.md | 4 ++-- 2 files changed, 10 insertions(+), 7 deletions(-) diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index 7a9883df356..4b3878bed11 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -1,6 +1,8 @@ + the TLDR, User Flow, and Caveats sections + Drop every section you have nothing to put in, heading included: a bare "## Relevant issues" or + "## Affected release" with nothing under it must not appear in the final description --> ## TLDR @@ -21,6 +23,7 @@ How it solves it: + ## Affected release - + ## Linear ticket - + ## Pre-Submission checklist @@ -134,7 +137,7 @@ If you're seeing a delay in your PR being merged, ping the LiteLLM Team on [Slac human reader If you assumed something instead of testing it, e.g. "only reproduces with X on" or "no user-observable behavior difference", list it here too with what breaks if it is wrong - Leave this section empty if there are none --> + Drop this section if there are none --> ## QA runbook diff --git a/AGENTS.md b/AGENTS.md index cade08bdd02..820ea64d4f9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -33,11 +33,11 @@ End-to-end tests belong in `tests/e2e/` and must follow the harness conventions When creating PRs, target the repository's current default branch for both internal and external / OSS contributions. Check it with `python3 scripts/default_branch.py --branch` instead of assuming a branch name or relying on cached `origin/HEAD` -When writing a PR body, treat the comments and imperative instructions inside .github/pull_request_template.md as rules to follow, not just layout. Agent harnesses may strip HTML comments from copies of that file injected into context, so read .github/pull_request_template.md from disk before writing a PR body to make sure you see every comment rule +When writing a PR body, treat the comments and imperative instructions inside .github/pull_request_template.md as rules to follow, not just layout. Agent harnesses may strip HTML comments from copies of that file injected into context, so read .github/pull_request_template.md from disk before writing a PR body to make sure you see every comment rule. A section you have nothing to put in (Relevant issues, Affected release, Linear ticket, Caveats, QA runbook, and so on) is removed entirely, heading included, never left as an empty title Same applies for filing bug reports and feature requests, with .github/ISSUE_TEMPLATE/bug_report.yml and .github/ISSUE_TEMPLATE/feature_request.yml, respectively -If you're resolving a linear ticket, in the "## Linear ticket" section of the PR, say "Resolves LIT-1234", replacing "LIT-1234" with the actual ticket id that you're resolving. If you don't have the ticket id, don't make one up or search for it. Just leave the section blank +If you're resolving a linear ticket, in the "## Linear ticket" section of the PR, say "Resolves LIT-1234", replacing "LIT-1234" with the actual ticket id that you're resolving. If you don't have the ticket id, don't make one up or search for it. Just drop the section Never use `pytest` commands or the like as "Screenshots / Proof of Fix". We prefer curl'ing a live proxy instance running on localhost:4000 (I like to run it with `python litellm/proxy/proxy_cli.py --config litellm/proxy/dev_config.yaml --detailed_debug --reload --use_v2_migration_resolver 2>&1 | tee litellm.log`; the Admin UI dev server is `npm run dev` in `ui/litellm-dashboard`, served on port 3000) and showing both the command run and the output. Also, it should hit real LLM provider APIs, not mocks, and cost real $$$ because that is the most realistic test. The proof of fix should be exactly what the end user / customer would see / do. The run logs in PR #27703 is a prime example of how to do it (not a huge fan of using a python test script that future me and the team will have no visibility into; I prefer just curl commands or a short list of bash commands (e.g., using `for`)). If it's a UI thing, or the main use case runs through a headful agentic coding tool like Claude Code or Codex, drive that surface yourself and embed your own before and after screenshots of it in the PR (the Admin UI page, or what the coding tool shows), next to an ordered list of the URLs to go to (e.g., http://localhost:4000/ui/?page=logs), where to click, and what fields to fill out so a reviewer can reproduce it From eef80535ec80a282aa210939dbdec5f7dada92c3 Mon Sep 17 00:00:00 2001 From: tin-berri Date: Wed, 23 Sep 2026 18:25:50 -0700 Subject: [PATCH 06/96] feat(ui): configure prompt caching request rows per page (#42842) * feat(ui): configure prompt caching request rows per page * style(ui): match prompt caching pagination arrows --- ...tCachingRequestsTable.integration.test.tsx | 109 +++++++++++++++--- .../PromptCachingRequestsTable.tsx | 70 ++++++++--- 2 files changed, 143 insertions(+), 36 deletions(-) diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/PromptCachingRequestsTable.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/PromptCachingRequestsTable.integration.test.tsx index e7594a18105..d1e8520015a 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/PromptCachingRequestsTable.integration.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/PromptCachingRequestsTable.integration.test.tsx @@ -1,5 +1,15 @@ import { Profiler } from "react"; -import { act, fireEvent, renderWithProviders, screen, testQueryClient, waitFor, within } from "@/../tests/test-utils"; +import userEvent from "@testing-library/user-event"; +import { + act, + chooseSelectOption, + fireEvent, + renderWithProviders, + screen, + testQueryClient, + waitFor, + within, +} from "@/../tests/test-utils"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; import type { components } from "@/lib/http/schema"; @@ -23,8 +33,13 @@ const request = (overrides: Partial = {}): CacheRequest => ({ net_savings: -0.0075, ...overrides, }); -const response = (requests: CacheRequest[], nextCursor: RequestsResponse["next_cursor"] = null) => { - const body: RequestsResponse = { requests, has_more: nextCursor !== null, next_cursor: nextCursor, page_size: 10 }; +const response = (requests: CacheRequest[], nextCursor: RequestsResponse["next_cursor"] = null, pageSize = 10) => { + const body: RequestsResponse = { + requests, + has_more: nextCursor !== null, + next_cursor: nextCursor, + page_size: pageSize, + }; return Response.json(body); }; const lastQuery = () => new URL(String(fetchMock.mock.calls.at(-1)?.[0]), "http://localhost").searchParams; @@ -100,18 +115,76 @@ describe("PromptCachingRequestsTable", () => { const table = await screen.findByRole("table", { name: "Prompt caching requests" }); expect(within(table).getAllByRole("link")).toHaveLength(10); expect(within(table).queryByRole("link", { name: "request-11" })).not.toBeInTheDocument(); - fireEvent.click(screen.getByRole("button", { name: "Next" })); + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); await screen.findByRole("link", { name: "request-11" }); expect(within(screen.getByRole("table", { name: "Prompt caching requests" })).getAllByRole("link")).toHaveLength(1); - expect(screen.getByRole("button", { name: "Next" })).toBeDisabled(); - fireEvent.click(screen.getByRole("button", { name: "Previous" })); + expect(screen.getByRole("button", { name: "Go to next page" })).toBeDisabled(); + fireEvent.click(screen.getByRole("button", { name: "Go to previous page" })); await screen.findByRole("link", { name: "request-1" }); expect(within(screen.getByRole("table", { name: "Prompt caching requests" })).getAllByRole("link")).toHaveLength( 10, ); - expect(screen.getByRole("button", { name: "Previous" })).toBeDisabled(); + expect(screen.getByRole("button", { name: "Go to previous page" })).toBeDisabled(); }); + it.each([25, 50, 100])( + "restarts at page one with %i rows and retains the size across navigation and filters", + async (pageSize) => { + const user = userEvent.setup(); + const rows = Array.from({ length: 101 }, (_, index) => request({ request_id: `request-${index + 1}` })); + fetchMock.mockImplementation(async (input) => { + const query = new URL(String(input), "http://localhost").searchParams; + const start = rows.findIndex((row) => row.request_id === query.get("cursor_request_id")) + 1; + const size = Number(query.get("page_size")); + const end = start + size; + const page = rows.slice(start, end); + const last = page.at(-1); + return response( + page, + end < rows.length && last ? { start_time: last.start_time, request_id: last.request_id } : null, + size, + ); + }); + renderWithProviders(); + await screen.findByRole("link", { name: "request-1" }); + expect(screen.getByRole("combobox", { name: "Rows per page" })).toHaveTextContent("10"); + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); + await screen.findByRole("link", { name: "request-11" }); + + await chooseSelectOption(user, screen.getByRole("combobox", { name: "Rows per page" }), String(pageSize)); + await screen.findByRole("link", { name: "request-1" }); + expect(within(screen.getByRole("table", { name: "Prompt caching requests" })).getAllByRole("link")).toHaveLength( + pageSize, + ); + expect(lastQuery().get("page_size")).toBe(String(pageSize)); + expect(lastQuery().has("cursor_request_id")).toBe(false); + expect(lastQuery().has("cursor_start_time")).toBe(false); + expect(screen.getByText("Page 1")).toBeInTheDocument(); + expect(screen.getByRole("button", { name: "Go to previous page" })).toBeDisabled(); + + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); + await screen.findByRole("link", { name: `request-${pageSize + 1}` }); + expect(lastQuery().get("page_size")).toBe(String(pageSize)); + expect(lastQuery().get("cursor_request_id")).toBe(`request-${pageSize}`); + expect(screen.getByText("Page 2")).toBeInTheDocument(); + fireEvent.click(screen.getByRole("button", { name: "Go to previous page" })); + await screen.findByRole("link", { name: "request-1" }); + expect(within(screen.getByRole("table", { name: "Prompt caching requests" })).getAllByRole("link")).toHaveLength( + pageSize, + ); + + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); + await screen.findByRole("link", { name: `request-${pageSize + 1}` }); + fireEvent.click(screen.getByRole("tab", { name: "Cache hits" })); + await screen.findByRole("link", { name: "request-1" }); + expect(lastQuery().get("filter")).toBe("hits"); + expect(lastQuery().get("page_size")).toBe(String(pageSize)); + expect(lastQuery().has("cursor_request_id")).toBe(false); + expect(screen.getByText("Page 1")).toBeInTheDocument(); + expect(screen.getByRole("combobox", { name: "Rows per page" })).toHaveTextContent(String(pageSize)); + }, + ); + it("forwards complete server cursors, goes back to prior cursors, and clears them for each caching filter", async () => { fetchMock.mockImplementation(async (input) => { const query = new URL(String(input), "http://localhost").searchParams; @@ -130,33 +203,33 @@ describe("PromptCachingRequestsTable", () => { }); renderWithProviders(); await screen.findByRole("link", { name: "all-1" }); - expect(screen.getByRole("button", { name: "Previous" })).toBeDisabled(); + expect(screen.getByRole("button", { name: "Go to previous page" })).toBeDisabled(); expect(lastQuery().has("page")).toBe(false); expect(lastQuery().has("cursor_request_id")).toBe(false); - fireEvent.click(screen.getByRole("button", { name: "Next" })); + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); await screen.findByRole("link", { name: "all-2" }); expect(screen.getByText("Page 2")).toBeInTheDocument(); expect(lastQuery().get("cursor_start_time")).toBe(firstCursor.start_time); expect(lastQuery().get("cursor_request_id")).toBe(firstCursor.request_id); - fireEvent.click(screen.getByRole("button", { name: "Next" })); + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); await screen.findByRole("link", { name: "all-3" }); expect(screen.getByText("Page 3")).toBeInTheDocument(); expect(lastQuery().get("cursor_start_time")).toBe(secondCursor.start_time); expect(lastQuery().get("cursor_request_id")).toBe(secondCursor.request_id); - expect(screen.getByRole("button", { name: "Next" })).toBeDisabled(); + expect(screen.getByRole("button", { name: "Go to next page" })).toBeDisabled(); await testQueryClient.invalidateQueries({ refetchType: "none" }); - fireEvent.click(screen.getByRole("button", { name: "Previous" })); + fireEvent.click(screen.getByRole("button", { name: "Go to previous page" })); await screen.findByRole("link", { name: "all-2" }); await waitFor(() => expect(lastQuery().get("cursor_request_id")).toBe(firstCursor.request_id)); expect(lastQuery().get("cursor_start_time")).toBe(firstCursor.start_time); expect(screen.getByText("Page 2")).toBeInTheDocument(); - fireEvent.click(screen.getByRole("button", { name: "Previous" })); + fireEvent.click(screen.getByRole("button", { name: "Go to previous page" })); await screen.findByRole("link", { name: "all-1" }); await waitFor(() => expect(lastQuery().has("cursor_request_id")).toBe(false)); expect(lastQuery().has("cursor_start_time")).toBe(false); - fireEvent.click(screen.getByRole("button", { name: "Next" })); + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); await screen.findByRole("link", { name: "all-2" }); fireEvent.click(screen.getByRole("tab", { name: "LiteLLM injected" })); @@ -166,7 +239,7 @@ describe("PromptCachingRequestsTable", () => { expect(lastQuery().has("cursor_request_id")).toBe(false); expect(lastQuery().has("cursor_start_time")).toBe(false); - fireEvent.click(screen.getByRole("button", { name: "Next" })); + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); await screen.findByRole("link", { name: "injected-2" }); fireEvent.click(screen.getByRole("tab", { name: "Cache hits" })); await screen.findByRole("link", { name: "hits-1" }); @@ -203,7 +276,7 @@ describe("PromptCachingRequestsTable", () => { ); const { rerender } = renderWithProviders(tree("token-a", dates)); await screen.findByRole("link", { name: "old-first" }); - fireEvent.click(screen.getByRole("button", { name: "Next" })); + fireEvent.click(screen.getByRole("button", { name: "Go to next page" })); await screen.findByRole("link", { name: "old-second" }); const pending = Promise.withResolvers(); @@ -253,7 +326,7 @@ describe("PromptCachingRequestsTable", () => { expect(screen.getByRole("link", { name: "current-hit" })).toBeInTheDocument(); expect(screen.queryByRole("link", { name: "stale-all" })).not.toBeInTheDocument(); - expect(screen.getByRole("button", { name: "Next" })).toBeDisabled(); + expect(screen.getByRole("button", { name: "Go to next page" })).toBeDisabled(); }); it("offers retry after a failed read and shows the empty state after it succeeds", async () => { @@ -265,7 +338,7 @@ describe("PromptCachingRequestsTable", () => { fireEvent.click(screen.getByRole("button", { name: "Retry" })); expect(await screen.findByText("No matching prompt caching requests in this range")).toBeInTheDocument(); expect(screen.queryByRole("alert")).not.toBeInTheDocument(); - expect(screen.getByRole("button", { name: "Next" })).toBeDisabled(); + expect(screen.getByRole("button", { name: "Go to next page" })).toBeDisabled(); expect(fetchMock).toHaveBeenCalledTimes(2); }); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/PromptCachingRequestsTable.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/PromptCachingRequestsTable.tsx index 140c11d2318..ba9cfd8ca22 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/PromptCachingRequestsTable.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/PromptCachingRequestsTable.tsx @@ -1,12 +1,14 @@ "use client"; import { useQuery, type UseQueryOptions } from "@tanstack/react-query"; +import { ChevronLeft, ChevronRight } from "lucide-react"; import Link from "next/link"; import { useState } from "react"; import { apiClient } from "@/components/networking"; import { Button } from "@/components/ui/button"; import { Card, CardContent, CardHeader, CardTitle } from "@/components/ui/card"; +import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select"; import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table"; import { Tabs, TabsList, TabsTrigger } from "@/components/ui/tabs"; import { LOG_ID_QUERY_PARAM } from "@/components/view_logs/logDetailRouting"; @@ -18,6 +20,7 @@ import { benchmarksWindow as activityWindow } from "./useAutoRouterBenchmarks"; import type { DateRange } from "./useDailyActivityRange"; const REQUESTS_PATH = "/cost_optimization/prompt_caching/requests"; +const PAGE_SIZE_OPTIONS = [10, 25, 50, 100]; type RequestsEndpoint = paths[typeof REQUESTS_PATH]["get"]; type RequestsResponse = RequestsEndpoint["responses"][200]["content"]["application/json"]; type RequestsQuery = NonNullable; @@ -31,10 +34,11 @@ interface PromptCachingRequestsTableProps { export default function PromptCachingRequestsTable({ accessToken, dateValue }: PromptCachingRequestsTableProps) { const [filter, setFilter] = useState("all"); + const [pageSize, setPageSize] = useState(10); const window = activityWindow(dateValue, new Date()); const startDate = window.start_date ? `${window.start_date}T00:00:00.000Z` : ""; const endDate = window.end_date ? `${window.end_date}T23:59:59.999Z` : ""; - const scope = JSON.stringify([accessToken, startDate, endDate, filter]); + const scope = JSON.stringify([accessToken, startDate, endDate, filter, pageSize]); const [pagination, setPagination] = useState<{ scope: string; cursors: readonly RequestCursor[] }>({ scope, cursors: [null], @@ -52,7 +56,7 @@ export default function PromptCachingRequestsTable({ accessToken, dateValue }: P start_date: startDate, end_date: endDate, filter, - page_size: 10, + page_size: pageSize, cursor_start_time: cursor?.start_time, cursor_request_id: cursor?.request_id, }; @@ -161,22 +165,52 @@ export default function PromptCachingRequestsTable({ accessToken, dateValue }: P )} -
- - Page {page} - +
+
+ Rows per page + +
+
+ Page {page} +
+ + +
+
)} From ca8689863307152d126eab375720d5d09923a67c Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:26:02 -0700 Subject: [PATCH 07/96] fix(ollama): read the JSON thinking field on non-streaming completions (#42838) * fix(ollama): read the JSON thinking field on non-streaming completions Ollama's /api/generate returns reasoning in a top-level `thinking` field, but the completion transport only looked for inline tags. reasoning_content was therefore always null, and a model that spent its whole turn reasoning returned an empty assistant message with tokens billed. Port the precedence the ollama_chat transport already uses: the field wins and inline tags stay the fallback. Applied to both non-streaming paths, including the JSON-mode text fallback. The two fields are read through a small validated model rather than off the untyped JSON, so absent and explicitly null `response` stay distinct exactly as before. * fix(ollama): keep the thinking field on JSON-mode completions The first pass read `thinking` for plain replies and for JSON-mode text that failed to parse, but the three JSON-mode branches that succeed still dropped it: an empty `response`, a valid JSON object, and a function-call shaped one. A model that spent its whole turn reasoning under `format: json` therefore still came back blank with the tokens billed. Carry the field on all three, type the new test helper's parameters, and cover the null and malformed `response` fallbacks. --------- Co-authored-by: Pawan-Shahane --- .../llms/ollama/completion/transformation.py | 55 ++++-- .../test_ollama_completion_transformation.py | 157 ++++++++++++++++++ 2 files changed, 195 insertions(+), 17 deletions(-) diff --git a/litellm/llms/ollama/completion/transformation.py b/litellm/llms/ollama/completion/transformation.py index 0fc1cd926b8..b1f69220de7 100644 --- a/litellm/llms/ollama/completion/transformation.py +++ b/litellm/llms/ollama/completion/transformation.py @@ -4,6 +4,7 @@ from collections.abc import AsyncIterator, Iterator from typing import TYPE_CHECKING, Any, Final from httpx._models import Headers, Response +from pydantic import BaseModel, ConfigDict, ValidationError import litellm from litellm._logging import verbose_proxy_logger @@ -43,6 +44,37 @@ else: LiteLLMLoggingObj = Any +class _OllamaGenerateReasoning(BaseModel): + """The two `/api/generate` fields a reply's reasoning can arrive in.""" + + model_config = ConfigDict(extra="ignore") + + # Absent and explicitly null are distinct here: Ollama omits `response` where it sends + # no text, and sends null where the reply carries none, which stay "" and None downstream. + response: str | None = "" + thinking: str | None = None + + @classmethod + def from_response(cls, response_json: object) -> "_OllamaGenerateReasoning": + try: + return cls.model_validate(response_json) + except ValidationError: + return cls() + + def split(self) -> tuple[str | None, str | None]: + """Reasoning reaches `/api/generate` either in the top-level `thinking` field or + inline in `` tags, never both. The field wins, matching `ollama_chat`.""" + from litellm.litellm_core_utils.prompt_templates.common_utils import ( + _parse_content_for_reasoning, + ) + + if self.thinking: + return self.thinking, self.response + if self.response is None: + return None, None + return _parse_content_for_reasoning(self.response) + + class OllamaConfig(BaseConfig): """ Reference: https://github.com/ollama/ollama/blob/main/docs/api.md#parameters @@ -255,20 +287,17 @@ class OllamaConfig(BaseConfig): api_key: str | None = None, json_mode: bool | None = None, ) -> ModelResponse: - from litellm.litellm_core_utils.prompt_templates.common_utils import ( - _parse_content_for_reasoning, - ) - response_json: Final = raw_response.json() ## RESPONSE OBJECT model_response.choices[0].finish_reason = "stop" if request_data.get("format", "") == "json": # Check if response field exists and is not empty before parsing JSON response_text = response_json.get("response", "") + thinking: Final = _OllamaGenerateReasoning.from_response(response_json).thinking or None if not response_text or not response_text.strip(): # Handle empty response gracefully - set empty content - message = litellm.Message(content="") + message = litellm.Message(content="", reasoning_content=thinking) model_response.choices[0].message = message model_response.choices[0].finish_reason = "stop" else: @@ -285,6 +314,7 @@ class OllamaConfig(BaseConfig): function_call: Final = response_content message = litellm.Message( content=None, + reasoning_content=thinking, tool_calls=[ { "id": f"call_{uuid.uuid4()}", @@ -302,27 +332,18 @@ class OllamaConfig(BaseConfig): # Handle as regular JSON (new behavior) message = litellm.Message( content=json.dumps(response_content), + reasoning_content=thinking, ) model_response.choices[0].message = message model_response.choices[0].finish_reason = "stop" except json.JSONDecodeError: # If JSON parsing fails, treat as regular text response - ## output parse reasoning content from response_text - reasoning_content: str | None = None - content: str | None = None - if response_text is not None: - reasoning_content, content = _parse_content_for_reasoning(response_text) + reasoning_content, content = _OllamaGenerateReasoning.from_response(response_json).split() message = litellm.Message(content=content, reasoning_content=reasoning_content) model_response.choices[0].message = message model_response.choices[0].finish_reason = "stop" else: - response_text = response_json.get("response", "") - content = None - reasoning_content = None - if response_text is not None and isinstance(response_text, str): - reasoning_content, content = _parse_content_for_reasoning(response_text) - else: - content = response_text + reasoning_content, content = _OllamaGenerateReasoning.from_response(response_json).split() model_response.choices[0].message.content = content model_response.choices[0].message.reasoning_content = reasoning_content model_response.created = int(time.time()) diff --git a/tests/test_litellm/llms/ollama/test_ollama_completion_transformation.py b/tests/test_litellm/llms/ollama/test_ollama_completion_transformation.py index 28e86e40944..d6215a742f0 100644 --- a/tests/test_litellm/llms/ollama/test_ollama_completion_transformation.py +++ b/tests/test_litellm/llms/ollama/test_ollama_completion_transformation.py @@ -414,6 +414,163 @@ class TestOllamaConfig: ) assert result.choices[0]["finish_reason"] == "stop" + def _transform( + self, response_json: dict[str, object], request_data: dict[str, object] | None = None + ) -> ModelResponse: + config = OllamaConfig() + + raw_response = MagicMock() + raw_response.json.return_value = response_json + + mock_encoding = MagicMock() + mock_encoding.encode.return_value = [1, 2, 3] + + return config.transform_response( + model="gpt-oss:120b", + raw_response=raw_response, + model_response=ModelResponse( + id="test_id", + choices=[{"message": Message(content="")}], + ), + logging_obj=MagicMock(), + request_data=request_data or {}, + messages=[], + optional_params={}, + litellm_params={}, + encoding=mock_encoding, + ) + + def test_transform_response_with_thinking_field(self): + """`/api/generate` returns reasoning in a top-level `thinking` field, which must + reach `reasoning_content` instead of being dropped.""" + result = self._transform( + { + "response": "OK", + "thinking": 'We need to reply with exactly "OK".', + "prompt_eval_count": 15, + "eval_count": 8, + } + ) + + assert result.choices[0]["message"].reasoning_content == 'We need to reply with exactly "OK".' + assert result.choices[0]["message"].content == "OK" + assert result.choices[0]["finish_reason"] == "stop" + + def test_transform_response_with_thinking_field_and_empty_response(self): + """A model that spends its whole turn reasoning leaves `response` empty; the + reasoning still has to be surfaced rather than billed and discarded.""" + result = self._transform( + { + "response": "", + "thinking": "Entire turn went into reasoning.", + "eval_count": 96, + } + ) + + assert result.choices[0]["message"].reasoning_content == "Entire turn went into reasoning." + assert result.choices[0]["message"].content == "" + + def test_transform_response_thinking_field_wins_over_inline_tags(self): + """When both shapes are present the field wins, matching the `ollama_chat` transport.""" + result = self._transform( + { + "response": "inlineAnswer", + "thinking": "from field", + } + ) + + assert result.choices[0]["message"].reasoning_content == "from field" + assert result.choices[0]["message"].content == "inlineAnswer" + + def test_transform_response_json_mode_non_json_text_with_thinking_field(self): + """JSON mode falls back to text handling when the payload is not JSON, so the + `thinking` field has to be picked up on that path too.""" + result = self._transform( + { + "response": "not valid json", + "thinking": "reasoning in json mode", + }, + request_data={"format": "json"}, + ) + + assert result.choices[0]["message"].reasoning_content == "reasoning in json mode" + assert result.choices[0]["message"].content == "not valid json" + + def test_transform_response_empty_thinking_field_falls_back_to_tags(self): + """An empty `thinking` field must not mask inline `` tags.""" + result = self._transform( + { + "response": "inline reasoningAnswer", + "thinking": "", + } + ) + + assert result.choices[0]["message"].reasoning_content == "inline reasoning" + assert result.choices[0]["message"].content == "Answer" + + def test_transform_response_json_mode_valid_json_keeps_thinking_field(self): + """A valid JSON `response` is returned as content, and the reasoning that came + with it must not be dropped.""" + result = self._transform( + { + "response": '{"answer": 42}', + "thinking": "reasoned before answering in json", + }, + request_data={"format": "json"}, + ) + + assert result.choices[0]["message"].content == '{"answer": 42}' + assert result.choices[0]["message"].reasoning_content == "reasoned before answering in json" + assert result.choices[0]["finish_reason"] == "stop" + + def test_transform_response_json_mode_function_call_keeps_thinking_field(self): + """A JSON `response` shaped like a function call becomes a tool call, and the + reasoning behind the call must survive alongside it.""" + result = self._transform( + { + "response": '{"name": "get_weather", "arguments": {"city": "Paris"}}', + "thinking": "the user wants weather, so call the tool", + }, + request_data={"format": "json"}, + ) + + message = result.choices[0]["message"] + assert message.tool_calls is not None + assert message.tool_calls[0].function.name == "get_weather" + assert message.reasoning_content == "the user wants weather, so call the tool" + assert result.choices[0]["finish_reason"] == "tool_calls" + + def test_transform_response_json_mode_empty_response_keeps_thinking_field(self): + """In JSON mode a model that spends its whole turn reasoning leaves `response` + empty; the reasoning must still come back instead of a blank message.""" + result = self._transform( + { + "response": "", + "thinking": "all of the tokens went into reasoning", + }, + request_data={"format": "json"}, + ) + + assert result.choices[0]["message"].content == "" + assert result.choices[0]["message"].reasoning_content == "all of the tokens went into reasoning" + + def test_transform_response_null_response_keeps_content_null(self): + """Ollama sends `response: null` when the reply carries no text; that stays null + rather than becoming an empty string, while `thinking` is still surfaced.""" + result = self._transform({"response": None, "thinking": "reasoning only"}) + + assert result.choices[0]["message"].content is None + assert result.choices[0]["message"].reasoning_content == "reasoning only" + + def test_transform_response_malformed_reasoning_fields_do_not_crash(self): + """A reply whose `response` and `thinking` are not strings must still produce a + response instead of raising, with no reasoning invented.""" + result = self._transform({"response": 5, "thinking": ["not", "a", "string"]}) + + assert result.choices[0]["message"].reasoning_content is None + assert result.choices[0]["message"].content == "" + assert result.choices[0]["finish_reason"] == "stop" + class TestOllamaTextCompletionResponseIterator: def test_chunk_parser_with_thinking_field(self): From 61996e1837a9d8aa0fd8bb832e21f2a298723a2d Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:27:01 -0700 Subject: [PATCH 08/96] feat(models): add openai chat-latest, codex and deep-research rows from the model docs (#42834) * feat(models): add openai chat-latest, codex and deep-research rows from the model docs Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(models): mark new openai vision models as supporting pdf input Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 340 ++++++++++++++++++ model_prices_and_context_window.json | 340 ++++++++++++++++++ 2 files changed, 680 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 24dfef170c3..50bcc6f71bf 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -32091,6 +32091,37 @@ "supports_xhigh_reasoning_effort": false, "supports_minimal_reasoning_effort": true }, + "gpt-5-codex": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 272000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5-codex", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, "gpt-5.1": { "cache_read_input_token_cost": 1.25e-07, "cache_read_input_token_cost_batches": 6.25e-08, @@ -32143,6 +32174,129 @@ "supports_xhigh_reasoning_effort": false, "supports_minimal_reasoning_effort": false }, + "gpt-5.1-codex-mini": { + "cache_read_input_token_cost": 2.5e-08, + "input_cost_per_token": 2.5e-07, + "litellm_provider": "openai", + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 2e-06, + "source": "https://developers.openai.com/api/docs/models/gpt-5.1-codex-mini", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "gpt-5.1-codex-max": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.1-codex-max", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, + "gpt-5.1-codex": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.1-codex", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, + "gpt-5.1-chat-latest": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.1-chat-latest", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "gpt-5.1-2025-11-13": { "cache_read_input_token_cost": 1.25e-07, "cache_read_input_token_cost_batches": 6.25e-08, @@ -32248,6 +32402,68 @@ "supports_xhigh_reasoning_effort": true, "supports_minimal_reasoning_effort": false }, + "gpt-5.2-codex": { + "cache_read_input_token_cost": 1.75e-07, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "openai", + "max_input_tokens": 272000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 1.4e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.2-codex", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, + "gpt-5.2-chat-latest": { + "cache_read_input_token_cost": 1.75e-07, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "openai", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.4e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.2-chat-latest", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "gpt-5.2-2025-12-11": { "cache_read_input_token_cost": 1.75e-07, "cache_read_input_token_cost_batches": 8.75e-08, @@ -33964,6 +34180,37 @@ "supports_xhigh_reasoning_effort": false, "supports_minimal_reasoning_effort": true }, + "gpt-5-chat-latest": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5-chat-latest", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "gpt-5.3-codex": { "cache_read_input_token_cost": 1.75e-07, "cache_read_input_token_cost_priority": 3.5e-07, @@ -34007,6 +34254,37 @@ "supports_xhigh_reasoning_effort": false, "supports_minimal_reasoning_effort": true }, + "gpt-5.3-chat-latest": { + "cache_read_input_token_cost": 1.75e-07, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "openai", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.4e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.3-chat-latest", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "gpt-5-mini": { "cache_read_input_token_cost": 2.5e-08, "cache_read_input_token_cost_batches": 1.25e-08, @@ -38841,6 +39119,37 @@ "supports_vision": true, "supports_web_search": true }, + "o3-deep-research": { + "cache_read_input_token_cost": 2.5e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "openai", + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "max_tokens": 100000, + "mode": "responses", + "output_cost_per_token": 4e-05, + "source": "https://developers.openai.com/api/docs/models/o3-deep-research", + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_parallel_function_calling": false, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "supports_vision": true, + "supports_pdf_input": true + }, "o3-2025-04-16": { "cache_read_input_token_cost": 5e-07, "cache_read_input_token_cost_flex": 2.5e-07, @@ -39039,6 +39348,37 @@ "supports_vision": true, "supports_web_search": true }, + "o4-mini-deep-research": { + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 2e-06, + "litellm_provider": "openai", + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "max_tokens": 100000, + "mode": "responses", + "output_cost_per_token": 8e-06, + "source": "https://developers.openai.com/api/docs/models/o4-mini-deep-research", + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_parallel_function_calling": false, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "supports_vision": true, + "supports_pdf_input": true + }, "o4-mini-2025-04-16": { "cache_read_input_token_cost": 2.75e-07, "cache_read_input_token_cost_flex": 1.38e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 24dfef170c3..50bcc6f71bf 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -32091,6 +32091,37 @@ "supports_xhigh_reasoning_effort": false, "supports_minimal_reasoning_effort": true }, + "gpt-5-codex": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 272000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5-codex", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, "gpt-5.1": { "cache_read_input_token_cost": 1.25e-07, "cache_read_input_token_cost_batches": 6.25e-08, @@ -32143,6 +32174,129 @@ "supports_xhigh_reasoning_effort": false, "supports_minimal_reasoning_effort": false }, + "gpt-5.1-codex-mini": { + "cache_read_input_token_cost": 2.5e-08, + "input_cost_per_token": 2.5e-07, + "litellm_provider": "openai", + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 2e-06, + "source": "https://developers.openai.com/api/docs/models/gpt-5.1-codex-mini", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "gpt-5.1-codex-max": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.1-codex-max", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, + "gpt-5.1-codex": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.1-codex", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, + "gpt-5.1-chat-latest": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.1-chat-latest", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "gpt-5.1-2025-11-13": { "cache_read_input_token_cost": 1.25e-07, "cache_read_input_token_cost_batches": 6.25e-08, @@ -32248,6 +32402,68 @@ "supports_xhigh_reasoning_effort": true, "supports_minimal_reasoning_effort": false }, + "gpt-5.2-codex": { + "cache_read_input_token_cost": 1.75e-07, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "openai", + "max_input_tokens": 272000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "responses", + "output_cost_per_token": 1.4e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.2-codex", + "supported_endpoints": [ + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true + }, + "gpt-5.2-chat-latest": { + "cache_read_input_token_cost": 1.75e-07, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "openai", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.4e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.2-chat-latest", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "gpt-5.2-2025-12-11": { "cache_read_input_token_cost": 1.75e-07, "cache_read_input_token_cost_batches": 8.75e-08, @@ -33964,6 +34180,37 @@ "supports_xhigh_reasoning_effort": false, "supports_minimal_reasoning_effort": true }, + "gpt-5-chat-latest": { + "cache_read_input_token_cost": 1.25e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "openai", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5-chat-latest", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "gpt-5.3-codex": { "cache_read_input_token_cost": 1.75e-07, "cache_read_input_token_cost_priority": 3.5e-07, @@ -34007,6 +34254,37 @@ "supports_xhigh_reasoning_effort": false, "supports_minimal_reasoning_effort": true }, + "gpt-5.3-chat-latest": { + "cache_read_input_token_cost": 1.75e-07, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "openai", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.4e-05, + "source": "https://developers.openai.com/api/docs/models/gpt-5.3-chat-latest", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, "gpt-5-mini": { "cache_read_input_token_cost": 2.5e-08, "cache_read_input_token_cost_batches": 1.25e-08, @@ -38841,6 +39119,37 @@ "supports_vision": true, "supports_web_search": true }, + "o3-deep-research": { + "cache_read_input_token_cost": 2.5e-06, + "input_cost_per_token": 1e-05, + "litellm_provider": "openai", + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "max_tokens": 100000, + "mode": "responses", + "output_cost_per_token": 4e-05, + "source": "https://developers.openai.com/api/docs/models/o3-deep-research", + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_parallel_function_calling": false, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "supports_vision": true, + "supports_pdf_input": true + }, "o3-2025-04-16": { "cache_read_input_token_cost": 5e-07, "cache_read_input_token_cost_flex": 2.5e-07, @@ -39039,6 +39348,37 @@ "supports_vision": true, "supports_web_search": true }, + "o4-mini-deep-research": { + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 2e-06, + "litellm_provider": "openai", + "max_input_tokens": 200000, + "max_output_tokens": 100000, + "max_tokens": 100000, + "mode": "responses", + "output_cost_per_token": 8e-06, + "source": "https://developers.openai.com/api/docs/models/o4-mini-deep-research", + "supported_endpoints": [ + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": false, + "supports_native_streaming": true, + "supports_parallel_function_calling": false, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "supports_vision": true, + "supports_pdf_input": true + }, "o4-mini-2025-04-16": { "cache_read_input_token_cost": 2.75e-07, "cache_read_input_token_cost_flex": 1.38e-07, From 8e74bb0d2313da5bb2ed2afe0f8b87e6fce03d94 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:28:34 +0000 Subject: [PATCH 09/96] fix(proxy): document request body and response schemas for the Responses API in OpenAPI (#42802) * fix(proxy): document responses API request and response schemas in openapi Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): namespace colliding openapi defs instead of overwriting existing components Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(proxy): regenerate lazy openapi snapshot and dashboard schema types Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): require model and input in responses schema, document event stream, fix def collision refs Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): mark responses request fields readonly required Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): reuse existing OpenAPI components when a $defs entry has the same shape Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/proxy/_lazy_openapi_snapshot.json | 6942 ++++++++++++++++- .../proxy/common_utils/custom_openapi_spec.py | 119 +- .../proxy/response_api_endpoints/endpoints.py | 27 + litellm/types/llms/openai.py | 4 +- .../test_responses_openapi_schema.py | 50 +- .../common_utils/test_custom_openapi_spec.py | 198 +- .../response_api_endpoints/test_endpoints.py | 35 + ui/litellm-dashboard/src/lib/http/schema.d.ts | 4362 ++++++++++- 8 files changed, 11654 insertions(+), 83 deletions(-) diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 71dbf0da239..0b43c3864ab 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -16183,6 +16183,673 @@ "llm_passthrough": { "components": { "schemas": { + "AcknowledgedSafetyCheck": { + "additionalProperties": true, + "description": "A pending safety check for the computer call.", + "properties": { + "code": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Code" + }, + "id": { + "title": "Id", + "type": "string" + }, + "message": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Message" + } + }, + "required": [ + "id" + ], + "title": "AcknowledgedSafetyCheck", + "type": "object" + }, + "Action": { + "additionalProperties": true, + "description": "The shell commands and limits that describe how to run the tool call.", + "properties": { + "commands": { + "items": { + "type": "string" + }, + "title": "Commands", + "type": "array" + }, + "max_output_length": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Max Output Length" + }, + "timeout_ms": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Timeout Ms" + } + }, + "required": [ + "commands" + ], + "title": "Action", + "type": "object" + }, + "ActionClick": { + "additionalProperties": true, + "description": "A click action.", + "properties": { + "button": { + "enum": [ + "left", + "right", + "wheel", + "back", + "forward" + ], + "title": "Button", + "type": "string" + }, + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "type": { + "const": "click", + "title": "Type", + "type": "string" + }, + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "button", + "type", + "x", + "y" + ], + "title": "ActionClick", + "type": "object" + }, + "ActionDoubleClick": { + "additionalProperties": true, + "description": "A double click action.", + "properties": { + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "type": { + "const": "double_click", + "title": "Type", + "type": "string" + }, + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "type", + "x", + "y" + ], + "title": "ActionDoubleClick", + "type": "object" + }, + "ActionDrag": { + "additionalProperties": true, + "description": "A drag action.", + "properties": { + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "path": { + "items": { + "$ref": "#/components/schemas/ActionDragPath" + }, + "title": "Path", + "type": "array" + }, + "type": { + "const": "drag", + "title": "Type", + "type": "string" + } + }, + "required": [ + "path", + "type" + ], + "title": "ActionDrag", + "type": "object" + }, + "ActionDragPath": { + "additionalProperties": true, + "description": "An x/y coordinate pair, e.g. `{ x: 100, y: 200 }`.", + "properties": { + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "x", + "y" + ], + "title": "ActionDragPath", + "type": "object" + }, + "ActionFind": { + "additionalProperties": true, + "description": "Action type \"find_in_page\": Searches for a pattern within a loaded page.", + "properties": { + "pattern": { + "title": "Pattern", + "type": "string" + }, + "type": { + "const": "find_in_page", + "title": "Type", + "type": "string" + }, + "url": { + "title": "Url", + "type": "string" + } + }, + "required": [ + "pattern", + "type", + "url" + ], + "title": "ActionFind", + "type": "object" + }, + "ActionKeypress": { + "additionalProperties": true, + "description": "A collection of keypresses the model would like to perform.", + "properties": { + "keys": { + "items": { + "type": "string" + }, + "title": "Keys", + "type": "array" + }, + "type": { + "const": "keypress", + "title": "Type", + "type": "string" + } + }, + "required": [ + "keys", + "type" + ], + "title": "ActionKeypress", + "type": "object" + }, + "ActionMove": { + "additionalProperties": true, + "description": "A mouse move action.", + "properties": { + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "type": { + "const": "move", + "title": "Type", + "type": "string" + }, + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "type", + "x", + "y" + ], + "title": "ActionMove", + "type": "object" + }, + "ActionOpenPage": { + "additionalProperties": true, + "description": "Action type \"open_page\" - Opens a specific URL from search results.", + "properties": { + "type": { + "const": "open_page", + "title": "Type", + "type": "string" + }, + "url": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Url" + } + }, + "required": [ + "type" + ], + "title": "ActionOpenPage", + "type": "object" + }, + "ActionScreenshot": { + "additionalProperties": true, + "description": "A screenshot action.", + "properties": { + "type": { + "const": "screenshot", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ActionScreenshot", + "type": "object" + }, + "ActionScroll": { + "additionalProperties": true, + "description": "A scroll action.", + "properties": { + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "scroll_x": { + "title": "Scroll X", + "type": "integer" + }, + "scroll_y": { + "title": "Scroll Y", + "type": "integer" + }, + "type": { + "const": "scroll", + "title": "Type", + "type": "string" + }, + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "scroll_x", + "scroll_y", + "type", + "x", + "y" + ], + "title": "ActionScroll", + "type": "object" + }, + "ActionSearch": { + "additionalProperties": true, + "description": "Action type \"search\" - Performs a web search query.", + "properties": { + "queries": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Queries" + }, + "query": { + "title": "Query", + "type": "string" + }, + "sources": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/ActionSearchSource" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Sources" + }, + "type": { + "const": "search", + "title": "Type", + "type": "string" + } + }, + "required": [ + "query", + "type" + ], + "title": "ActionSearch", + "type": "object" + }, + "ActionSearchSource": { + "additionalProperties": true, + "description": "A source used in the search.", + "properties": { + "type": { + "const": "url", + "title": "Type", + "type": "string" + }, + "url": { + "title": "Url", + "type": "string" + } + }, + "required": [ + "type", + "url" + ], + "title": "ActionSearchSource", + "type": "object" + }, + "ActionType": { + "additionalProperties": true, + "description": "An action to type in text.", + "properties": { + "text": { + "title": "Text", + "type": "string" + }, + "type": { + "const": "type", + "title": "Type", + "type": "string" + } + }, + "required": [ + "text", + "type" + ], + "title": "ActionType", + "type": "object" + }, + "ActionWait": { + "additionalProperties": true, + "description": "A wait action.", + "properties": { + "type": { + "const": "wait", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ActionWait", + "type": "object" + }, + "AnnotationContainerFileCitation": { + "additionalProperties": true, + "description": "A citation for a container file used to generate a model response.", + "properties": { + "container_id": { + "title": "Container Id", + "type": "string" + }, + "end_index": { + "title": "End Index", + "type": "integer" + }, + "file_id": { + "title": "File Id", + "type": "string" + }, + "filename": { + "title": "Filename", + "type": "string" + }, + "start_index": { + "title": "Start Index", + "type": "integer" + }, + "type": { + "const": "container_file_citation", + "title": "Type", + "type": "string" + } + }, + "required": [ + "container_id", + "end_index", + "file_id", + "filename", + "start_index", + "type" + ], + "title": "AnnotationContainerFileCitation", + "type": "object" + }, + "AnnotationFileCitation": { + "additionalProperties": true, + "description": "A citation to a file.", + "properties": { + "file_id": { + "title": "File Id", + "type": "string" + }, + "filename": { + "title": "Filename", + "type": "string" + }, + "index": { + "title": "Index", + "type": "integer" + }, + "type": { + "const": "file_citation", + "title": "Type", + "type": "string" + } + }, + "required": [ + "file_id", + "filename", + "index", + "type" + ], + "title": "AnnotationFileCitation", + "type": "object" + }, + "AnnotationFilePath": { + "additionalProperties": true, + "description": "A path to a file.", + "properties": { + "file_id": { + "title": "File Id", + "type": "string" + }, + "index": { + "title": "Index", + "type": "integer" + }, + "type": { + "const": "file_path", + "title": "Type", + "type": "string" + } + }, + "required": [ + "file_id", + "index", + "type" + ], + "title": "AnnotationFilePath", + "type": "object" + }, + "AnnotationURLCitation": { + "additionalProperties": true, + "description": "A citation for a web resource used to generate a model response.", + "properties": { + "end_index": { + "title": "End Index", + "type": "integer" + }, + "start_index": { + "title": "Start Index", + "type": "integer" + }, + "title": { + "title": "Title", + "type": "string" + }, + "type": { + "const": "url_citation", + "title": "Type", + "type": "string" + }, + "url": { + "title": "Url", + "type": "string" + } + }, + "required": [ + "end_index", + "start_index", + "title", + "type", + "url" + ], + "title": "AnnotationURLCitation", + "type": "object" + }, + "ApplyPatchTool": { + "additionalProperties": true, + "description": "Allows the assistant to create, delete, or update files using unified diffs.", + "properties": { + "type": { + "const": "apply_patch", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ApplyPatchTool", + "type": "object" + }, "Body_image_edit_api_openai_deployments__model__images_edits_post": { "properties": { "image": { @@ -16249,6 +16916,787 @@ "title": "Body_image_edit_api_openai_deployments__model__images_edits_post", "type": "object" }, + "CachedTokensDetails": { + "properties": { + "audio_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Audio Tokens" + }, + "image_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Image Tokens" + }, + "text_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Text Tokens" + } + }, + "title": "CachedTokensDetails", + "type": "object" + }, + "Click": { + "additionalProperties": true, + "description": "A click action.", + "properties": { + "button": { + "enum": [ + "left", + "right", + "wheel", + "back", + "forward" + ], + "title": "Button", + "type": "string" + }, + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "type": { + "const": "click", + "title": "Type", + "type": "string" + }, + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "button", + "type", + "x", + "y" + ], + "title": "Click", + "type": "object" + }, + "CodeInterpreter": { + "additionalProperties": true, + "description": "A tool that runs Python code to help generate a response to a prompt.", + "properties": { + "container": { + "anyOf": [ + { + "type": "string" + }, + { + "$ref": "#/components/schemas/CodeInterpreterContainerCodeInterpreterToolAuto" + } + ], + "title": "Container" + }, + "type": { + "const": "code_interpreter", + "title": "Type", + "type": "string" + } + }, + "required": [ + "container", + "type" + ], + "title": "CodeInterpreter", + "type": "object" + }, + "CodeInterpreterContainerCodeInterpreterToolAuto": { + "additionalProperties": true, + "description": "Configuration for a code interpreter container.\n\nOptionally specify the IDs of the files to run the code on.", + "properties": { + "file_ids": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "File Ids" + }, + "memory_limit": { + "anyOf": [ + { + "enum": [ + "1g", + "4g", + "16g", + "64g" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Memory Limit" + }, + "network_policy": { + "anyOf": [ + { + "$ref": "#/components/schemas/ContainerNetworkPolicyDisabled" + }, + { + "$ref": "#/components/schemas/ContainerNetworkPolicyAllowlist" + }, + { + "type": "null" + } + ], + "title": "Network Policy" + }, + "type": { + "const": "auto", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "CodeInterpreterContainerCodeInterpreterToolAuto", + "type": "object" + }, + "ComparisonFilter": { + "additionalProperties": true, + "description": "A filter used to compare a specified attribute key to a given value using a defined comparison operation.", + "properties": { + "key": { + "title": "Key", + "type": "string" + }, + "type": { + "enum": [ + "eq", + "ne", + "gt", + "gte", + "lt", + "lte", + "in", + "nin" + ], + "title": "Type", + "type": "string" + }, + "value": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "number" + }, + { + "type": "boolean" + }, + { + "items": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "number" + } + ] + }, + "type": "array" + } + ], + "title": "Value" + } + }, + "required": [ + "key", + "type", + "value" + ], + "title": "ComparisonFilter", + "type": "object" + }, + "CompoundFilter": { + "additionalProperties": true, + "description": "Combine multiple filters using `and` or `or`.", + "properties": { + "filters": { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/ComparisonFilter" + }, + {} + ] + }, + "title": "Filters", + "type": "array" + }, + "type": { + "enum": [ + "and", + "or" + ], + "title": "Type", + "type": "string" + } + }, + "required": [ + "filters", + "type" + ], + "title": "CompoundFilter", + "type": "object" + }, + "ComputerTool": { + "additionalProperties": true, + "description": "A tool that controls a virtual computer.\n\nLearn more about the [computer tool](https://platform.openai.com/docs/guides/tools-computer-use).", + "properties": { + "type": { + "const": "computer", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ComputerTool", + "type": "object" + }, + "ComputerUsePreviewTool": { + "additionalProperties": true, + "description": "A tool that controls a virtual computer.\n\nLearn more about the [computer tool](https://platform.openai.com/docs/guides/tools-computer-use).", + "properties": { + "display_height": { + "title": "Display Height", + "type": "integer" + }, + "display_width": { + "title": "Display Width", + "type": "integer" + }, + "environment": { + "enum": [ + "windows", + "mac", + "linux", + "ubuntu", + "browser" + ], + "title": "Environment", + "type": "string" + }, + "type": { + "const": "computer_use_preview", + "title": "Type", + "type": "string" + } + }, + "required": [ + "display_height", + "display_width", + "environment", + "type" + ], + "title": "ComputerUsePreviewTool", + "type": "object" + }, + "ContainerAuto": { + "additionalProperties": true, + "properties": { + "file_ids": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "File Ids" + }, + "memory_limit": { + "anyOf": [ + { + "enum": [ + "1g", + "4g", + "16g", + "64g" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Memory Limit" + }, + "network_policy": { + "anyOf": [ + { + "$ref": "#/components/schemas/ContainerNetworkPolicyDisabled" + }, + { + "$ref": "#/components/schemas/ContainerNetworkPolicyAllowlist" + }, + { + "type": "null" + } + ], + "title": "Network Policy" + }, + "skills": { + "anyOf": [ + { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/SkillReference" + }, + { + "$ref": "#/components/schemas/InlineSkill" + } + ] + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Skills" + }, + "type": { + "const": "container_auto", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ContainerAuto", + "type": "object" + }, + "ContainerNetworkPolicyAllowlist": { + "additionalProperties": true, + "properties": { + "allowed_domains": { + "items": { + "type": "string" + }, + "title": "Allowed Domains", + "type": "array" + }, + "domain_secrets": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/ContainerNetworkPolicyDomainSecret" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Domain Secrets" + }, + "type": { + "const": "allowlist", + "title": "Type", + "type": "string" + } + }, + "required": [ + "allowed_domains", + "type" + ], + "title": "ContainerNetworkPolicyAllowlist", + "type": "object" + }, + "ContainerNetworkPolicyDisabled": { + "additionalProperties": true, + "properties": { + "type": { + "const": "disabled", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ContainerNetworkPolicyDisabled", + "type": "object" + }, + "ContainerNetworkPolicyDomainSecret": { + "additionalProperties": true, + "properties": { + "domain": { + "title": "Domain", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "value": { + "title": "Value", + "type": "string" + } + }, + "required": [ + "domain", + "name", + "value" + ], + "title": "ContainerNetworkPolicyDomainSecret", + "type": "object" + }, + "ContainerReference": { + "additionalProperties": true, + "properties": { + "container_id": { + "title": "Container Id", + "type": "string" + }, + "type": { + "const": "container_reference", + "title": "Type", + "type": "string" + } + }, + "required": [ + "container_id", + "type" + ], + "title": "ContainerReference", + "type": "object" + }, + "Content": { + "additionalProperties": true, + "description": "Reasoning text from the model.", + "properties": { + "text": { + "title": "Text", + "type": "string" + }, + "type": { + "const": "reasoning_text", + "title": "Type", + "type": "string" + } + }, + "required": [ + "text", + "type" + ], + "title": "Content", + "type": "object" + }, + "CustomTool": { + "additionalProperties": true, + "description": "A custom tool that processes input using a specified format.\n\nLearn more about [custom tools](https://platform.openai.com/docs/guides/function-calling#custom-tools)", + "properties": { + "defer_loading": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Defer Loading" + }, + "description": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Description" + }, + "format": { + "anyOf": [ + { + "$ref": "#/components/schemas/Text" + }, + { + "$ref": "#/components/schemas/Grammar" + }, + { + "type": "null" + } + ], + "title": "Format" + }, + "name": { + "title": "Name", + "type": "string" + }, + "type": { + "const": "custom", + "title": "Type", + "type": "string" + } + }, + "required": [ + "name", + "type" + ], + "title": "CustomTool", + "type": "object" + }, + "CustomToolCallOutputItem": { + "additionalProperties": true, + "description": "A custom/freeform tool call output item (e.g. apply_patch).\n\nMirrors the ``custom_tool_call`` variant of OpenAI's Responses API output.\nUnlike ``OutputFunctionToolCall`` which uses ``arguments`` (JSON string),\nthis uses ``input`` (raw string) for the tool payload.", + "properties": { + "call_id": { + "title": "Call Id", + "type": "string" + }, + "id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Id" + }, + "input": { + "title": "Input", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "status": { + "anyOf": [ + { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Status" + }, + "type": { + "const": "custom_tool_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type", + "call_id", + "name", + "input" + ], + "title": "CustomToolCallOutputItem", + "type": "object" + }, + "DeleteResponseResult": { + "additionalProperties": true, + "description": "Result of a delete response request\n\n{\n \"id\": \"resp_6786a1bec27481909a17d673315b29f6\",\n \"object\": \"response\",\n \"deleted\": true\n}", + "properties": { + "deleted": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Deleted" + }, + "id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Id" + }, + "object": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Object" + } + }, + "required": [ + "id", + "object", + "deleted" + ], + "title": "DeleteResponseResult", + "type": "object" + }, + "DoubleClick": { + "additionalProperties": true, + "description": "A double click action.", + "properties": { + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "type": { + "const": "double_click", + "title": "Type", + "type": "string" + }, + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "type", + "x", + "y" + ], + "title": "DoubleClick", + "type": "object" + }, + "Drag": { + "additionalProperties": true, + "description": "A drag action.", + "properties": { + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "path": { + "items": { + "$ref": "#/components/schemas/DragPath" + }, + "title": "Path", + "type": "array" + }, + "type": { + "const": "drag", + "title": "Type", + "type": "string" + } + }, + "required": [ + "path", + "type" + ], + "title": "Drag", + "type": "object" + }, + "DragPath": { + "additionalProperties": true, + "description": "An x/y coordinate pair, e.g. `{ x: 100, y: 200 }`.", + "properties": { + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "x", + "y" + ], + "title": "DragPath", + "type": "object" + }, "ErrorResponse": { "properties": { "detail": { @@ -16271,6 +17719,339 @@ "title": "ErrorResponse", "type": "object" }, + "FileSearchTool": { + "additionalProperties": true, + "description": "A tool that searches for relevant content from uploaded files.\n\nLearn more about the [file search tool](https://platform.openai.com/docs/guides/tools-file-search).", + "properties": { + "filters": { + "anyOf": [ + { + "$ref": "#/components/schemas/ComparisonFilter" + }, + { + "$ref": "#/components/schemas/CompoundFilter" + }, + { + "type": "null" + } + ], + "title": "Filters" + }, + "max_num_results": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Max Num Results" + }, + "ranking_options": { + "anyOf": [ + { + "$ref": "#/components/schemas/RankingOptions" + }, + { + "type": "null" + } + ] + }, + "type": { + "const": "file_search", + "title": "Type", + "type": "string" + }, + "vector_store_ids": { + "items": { + "type": "string" + }, + "title": "Vector Store Ids", + "type": "array" + } + }, + "required": [ + "type", + "vector_store_ids" + ], + "title": "FileSearchTool", + "type": "object" + }, + "Filters": { + "additionalProperties": true, + "description": "Filters for the search.", + "properties": { + "allowed_domains": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Allowed Domains" + } + }, + "title": "Filters", + "type": "object" + }, + "FunctionShellTool": { + "additionalProperties": true, + "description": "A tool that allows the model to execute shell commands.", + "properties": { + "environment": { + "anyOf": [ + { + "$ref": "#/components/schemas/ContainerAuto" + }, + { + "$ref": "#/components/schemas/LocalEnvironment" + }, + { + "$ref": "#/components/schemas/ContainerReference" + }, + { + "type": "null" + } + ], + "title": "Environment" + }, + "type": { + "const": "shell", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "FunctionShellTool", + "type": "object" + }, + "FunctionTool": { + "additionalProperties": true, + "description": "Defines a function in your own code the model can choose to call.\n\nLearn more about [function calling](https://platform.openai.com/docs/guides/function-calling).", + "properties": { + "defer_loading": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Defer Loading" + }, + "description": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Description" + }, + "name": { + "title": "Name", + "type": "string" + }, + "parameters": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Parameters" + }, + "strict": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Strict" + }, + "type": { + "const": "function", + "title": "Type", + "type": "string" + } + }, + "required": [ + "name", + "type" + ], + "title": "FunctionTool", + "type": "object" + }, + "GenericResponseOutputItem": { + "additionalProperties": true, + "description": "Generic response API output item", + "properties": { + "content": { + "items": { + "$ref": "#/components/schemas/OutputText" + }, + "title": "Content", + "type": "array" + }, + "id": { + "title": "Id", + "type": "string" + }, + "phase": { + "anyOf": [ + { + "enum": [ + "commentary", + "final_answer" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Phase" + }, + "role": { + "title": "Role", + "type": "string" + }, + "status": { + "title": "Status", + "type": "string" + }, + "type": { + "title": "Type", + "type": "string" + } + }, + "required": [ + "type", + "id", + "status", + "role", + "content" + ], + "title": "GenericResponseOutputItem", + "type": "object" + }, + "GenericResponseOutputItemContentAnnotation": { + "additionalProperties": true, + "description": "Annotation for content in a message", + "properties": { + "end_index": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "End Index" + }, + "start_index": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Start Index" + }, + "title": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Title" + }, + "type": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Type" + }, + "url": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Url" + } + }, + "required": [ + "type", + "start_index", + "end_index", + "url", + "title" + ], + "title": "GenericResponseOutputItemContentAnnotation", + "type": "object" + }, + "Grammar": { + "additionalProperties": true, + "description": "A grammar defined by the user.", + "properties": { + "definition": { + "title": "Definition", + "type": "string" + }, + "syntax": { + "enum": [ + "lark", + "regex" + ], + "title": "Syntax", + "type": "string" + }, + "type": { + "const": "grammar", + "title": "Type", + "type": "string" + } + }, + "required": [ + "definition", + "syntax", + "type" + ], + "title": "Grammar", + "type": "object" + }, "HTTPValidationError": { "properties": { "detail": { @@ -16284,6 +18065,1915 @@ "title": "HTTPValidationError", "type": "object" }, + "ImageGeneration": { + "additionalProperties": true, + "description": "A tool that generates images using the GPT image models.", + "properties": { + "action": { + "anyOf": [ + { + "enum": [ + "generate", + "edit", + "auto" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Action" + }, + "background": { + "anyOf": [ + { + "enum": [ + "transparent", + "opaque", + "auto" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Background" + }, + "input_fidelity": { + "anyOf": [ + { + "enum": [ + "high", + "low" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Input Fidelity" + }, + "input_image_mask": { + "anyOf": [ + { + "$ref": "#/components/schemas/ImageGenerationInputImageMask" + }, + { + "type": "null" + } + ] + }, + "model": { + "anyOf": [ + { + "type": "string" + }, + { + "enum": [ + "gpt-image-1", + "gpt-image-1-mini", + "gpt-image-1.5" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Model" + }, + "moderation": { + "anyOf": [ + { + "enum": [ + "auto", + "low" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Moderation" + }, + "output_compression": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Output Compression" + }, + "output_format": { + "anyOf": [ + { + "enum": [ + "png", + "webp", + "jpeg" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Output Format" + }, + "partial_images": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Partial Images" + }, + "quality": { + "anyOf": [ + { + "enum": [ + "low", + "medium", + "high", + "auto" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Quality" + }, + "size": { + "anyOf": [ + { + "enum": [ + "1024x1024", + "1024x1536", + "1536x1024", + "auto" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Size" + }, + "type": { + "const": "image_generation", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ImageGeneration", + "type": "object" + }, + "ImageGenerationCall": { + "additionalProperties": true, + "description": "An image generation request made by the model.", + "properties": { + "id": { + "title": "Id", + "type": "string" + }, + "result": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Result" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "generating", + "failed" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "image_generation_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "status", + "type" + ], + "title": "ImageGenerationCall", + "type": "object" + }, + "ImageGenerationInputImageMask": { + "additionalProperties": true, + "description": "Optional mask for inpainting.\n\nContains `image_url`\n(string, optional) and `file_id` (string, optional).", + "properties": { + "file_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "File Id" + }, + "image_url": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Image Url" + } + }, + "title": "ImageGenerationInputImageMask", + "type": "object" + }, + "IncompleteDetails": { + "additionalProperties": true, + "description": "Details about why the response is incomplete.", + "properties": { + "reason": { + "anyOf": [ + { + "enum": [ + "max_output_tokens", + "content_filter" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Reason" + } + }, + "title": "IncompleteDetails", + "type": "object" + }, + "InlineSkill": { + "additionalProperties": true, + "properties": { + "description": { + "title": "Description", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "source": { + "$ref": "#/components/schemas/InlineSkillSource" + }, + "type": { + "const": "inline", + "title": "Type", + "type": "string" + } + }, + "required": [ + "description", + "name", + "source", + "type" + ], + "title": "InlineSkill", + "type": "object" + }, + "InlineSkillSource": { + "additionalProperties": true, + "description": "Inline skill payload", + "properties": { + "data": { + "title": "Data", + "type": "string" + }, + "media_type": { + "const": "application/zip", + "title": "Media Type", + "type": "string" + }, + "type": { + "const": "base64", + "title": "Type", + "type": "string" + } + }, + "required": [ + "data", + "media_type", + "type" + ], + "title": "InlineSkillSource", + "type": "object" + }, + "InputTokensDetails": { + "additionalProperties": true, + "properties": { + "audio_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Audio Tokens" + }, + "cached_tokens": { + "default": 0, + "title": "Cached Tokens", + "type": "integer" + }, + "cached_tokens_details": { + "anyOf": [ + { + "$ref": "#/components/schemas/CachedTokensDetails" + }, + { + "type": "null" + } + ] + }, + "image_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Image Tokens" + }, + "text_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Text Tokens" + }, + "video_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Video Tokens" + } + }, + "title": "InputTokensDetails", + "type": "object" + }, + "Keypress": { + "additionalProperties": true, + "description": "A collection of keypresses the model would like to perform.", + "properties": { + "keys": { + "items": { + "type": "string" + }, + "title": "Keys", + "type": "array" + }, + "type": { + "const": "keypress", + "title": "Type", + "type": "string" + } + }, + "required": [ + "keys", + "type" + ], + "title": "Keypress", + "type": "object" + }, + "LocalEnvironment": { + "additionalProperties": true, + "properties": { + "skills": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/LocalSkill" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Skills" + }, + "type": { + "const": "local", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "LocalEnvironment", + "type": "object" + }, + "LocalShell": { + "additionalProperties": true, + "description": "A tool that allows the model to execute shell commands in a local environment.", + "properties": { + "type": { + "const": "local_shell", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "LocalShell", + "type": "object" + }, + "LocalShellCall": { + "additionalProperties": true, + "description": "A tool call to run a command on the local shell.", + "properties": { + "action": { + "$ref": "#/components/schemas/LocalShellCallAction" + }, + "call_id": { + "title": "Call Id", + "type": "string" + }, + "id": { + "title": "Id", + "type": "string" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "local_shell_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "action", + "call_id", + "status", + "type" + ], + "title": "LocalShellCall", + "type": "object" + }, + "LocalShellCallAction": { + "additionalProperties": true, + "description": "Execute a shell command on the server.", + "properties": { + "command": { + "items": { + "type": "string" + }, + "title": "Command", + "type": "array" + }, + "env": { + "additionalProperties": { + "type": "string" + }, + "title": "Env", + "type": "object" + }, + "timeout_ms": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Timeout Ms" + }, + "type": { + "const": "exec", + "title": "Type", + "type": "string" + }, + "user": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "User" + }, + "working_directory": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Working Directory" + } + }, + "required": [ + "command", + "env", + "type" + ], + "title": "LocalShellCallAction", + "type": "object" + }, + "LocalShellCallOutput": { + "additionalProperties": true, + "description": "The output of a local shell tool call.", + "properties": { + "id": { + "title": "Id", + "type": "string" + }, + "output": { + "title": "Output", + "type": "string" + }, + "status": { + "anyOf": [ + { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Status" + }, + "type": { + "const": "local_shell_call_output", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "output", + "type" + ], + "title": "LocalShellCallOutput", + "type": "object" + }, + "LocalSkill": { + "additionalProperties": true, + "properties": { + "description": { + "title": "Description", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + } + }, + "required": [ + "description", + "name", + "path" + ], + "title": "LocalSkill", + "type": "object" + }, + "Logprob": { + "additionalProperties": true, + "description": "The log probability of a token.", + "properties": { + "bytes": { + "items": { + "type": "integer" + }, + "title": "Bytes", + "type": "array" + }, + "logprob": { + "title": "Logprob", + "type": "number" + }, + "token": { + "title": "Token", + "type": "string" + }, + "top_logprobs": { + "items": { + "$ref": "#/components/schemas/LogprobTopLogprob" + }, + "title": "Top Logprobs", + "type": "array" + } + }, + "required": [ + "token", + "bytes", + "logprob", + "top_logprobs" + ], + "title": "Logprob", + "type": "object" + }, + "LogprobTopLogprob": { + "additionalProperties": true, + "description": "The top log probability of a token.", + "properties": { + "bytes": { + "items": { + "type": "integer" + }, + "title": "Bytes", + "type": "array" + }, + "logprob": { + "title": "Logprob", + "type": "number" + }, + "token": { + "title": "Token", + "type": "string" + } + }, + "required": [ + "token", + "bytes", + "logprob" + ], + "title": "LogprobTopLogprob", + "type": "object" + }, + "Mcp": { + "additionalProperties": true, + "description": "Give the model access to additional tools via remote Model Context Protocol\n(MCP) servers. [Learn more about MCP](https://platform.openai.com/docs/guides/tools-remote-mcp).", + "properties": { + "allowed_tools": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "$ref": "#/components/schemas/McpAllowedToolsMcpToolFilter" + }, + { + "type": "null" + } + ], + "title": "Allowed Tools" + }, + "authorization": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Authorization" + }, + "connector_id": { + "anyOf": [ + { + "enum": [ + "connector_dropbox", + "connector_gmail", + "connector_googlecalendar", + "connector_googledrive", + "connector_microsoftteams", + "connector_outlookcalendar", + "connector_outlookemail", + "connector_sharepoint" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Connector Id" + }, + "defer_loading": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Defer Loading" + }, + "headers": { + "anyOf": [ + { + "additionalProperties": { + "type": "string" + }, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Headers" + }, + "require_approval": { + "anyOf": [ + { + "$ref": "#/components/schemas/McpRequireApprovalMcpToolApprovalFilter" + }, + { + "enum": [ + "always", + "never" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Require Approval" + }, + "server_description": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Server Description" + }, + "server_label": { + "title": "Server Label", + "type": "string" + }, + "server_url": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Server Url" + }, + "type": { + "const": "mcp", + "title": "Type", + "type": "string" + } + }, + "required": [ + "server_label", + "type" + ], + "title": "Mcp", + "type": "object" + }, + "McpAllowedToolsMcpToolFilter": { + "additionalProperties": true, + "description": "A filter object to specify which tools are allowed.", + "properties": { + "read_only": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Read Only" + }, + "tool_names": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Tool Names" + } + }, + "title": "McpAllowedToolsMcpToolFilter", + "type": "object" + }, + "McpApprovalRequest": { + "additionalProperties": true, + "description": "A request for human approval of a tool invocation.", + "properties": { + "arguments": { + "title": "Arguments", + "type": "string" + }, + "id": { + "title": "Id", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "server_label": { + "title": "Server Label", + "type": "string" + }, + "type": { + "const": "mcp_approval_request", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "arguments", + "name", + "server_label", + "type" + ], + "title": "McpApprovalRequest", + "type": "object" + }, + "McpApprovalResponse": { + "additionalProperties": true, + "description": "A response to an MCP approval request.", + "properties": { + "approval_request_id": { + "title": "Approval Request Id", + "type": "string" + }, + "approve": { + "title": "Approve", + "type": "boolean" + }, + "id": { + "title": "Id", + "type": "string" + }, + "reason": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Reason" + }, + "type": { + "const": "mcp_approval_response", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "approval_request_id", + "approve", + "type" + ], + "title": "McpApprovalResponse", + "type": "object" + }, + "McpCall": { + "additionalProperties": true, + "description": "An invocation of a tool on an MCP server.", + "properties": { + "approval_request_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Approval Request Id" + }, + "arguments": { + "title": "Arguments", + "type": "string" + }, + "error": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Error" + }, + "id": { + "title": "Id", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "output": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Output" + }, + "server_label": { + "title": "Server Label", + "type": "string" + }, + "status": { + "anyOf": [ + { + "enum": [ + "in_progress", + "completed", + "incomplete", + "calling", + "failed" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Status" + }, + "type": { + "const": "mcp_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "arguments", + "name", + "server_label", + "type" + ], + "title": "McpCall", + "type": "object" + }, + "McpListTools": { + "additionalProperties": true, + "description": "A list of tools available on an MCP server.", + "properties": { + "error": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Error" + }, + "id": { + "title": "Id", + "type": "string" + }, + "server_label": { + "title": "Server Label", + "type": "string" + }, + "tools": { + "items": { + "$ref": "#/components/schemas/McpListToolsTool" + }, + "title": "Tools", + "type": "array" + }, + "type": { + "const": "mcp_list_tools", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "server_label", + "tools", + "type" + ], + "title": "McpListTools", + "type": "object" + }, + "McpListToolsTool": { + "additionalProperties": true, + "description": "A tool available on an MCP server.", + "properties": { + "annotations": { + "anyOf": [ + {}, + { + "type": "null" + } + ], + "title": "Annotations" + }, + "description": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Description" + }, + "input_schema": { + "title": "Input Schema" + }, + "name": { + "title": "Name", + "type": "string" + } + }, + "required": [ + "input_schema", + "name" + ], + "title": "McpListToolsTool", + "type": "object" + }, + "McpRequireApprovalMcpToolApprovalFilter": { + "additionalProperties": true, + "description": "Specify which of the MCP server's tools require approval.\n\nCan be\n`always`, `never`, or a filter object associated with tools\nthat require approval.", + "properties": { + "always": { + "anyOf": [ + { + "$ref": "#/components/schemas/McpRequireApprovalMcpToolApprovalFilterAlways" + }, + { + "type": "null" + } + ] + }, + "never": { + "anyOf": [ + { + "$ref": "#/components/schemas/McpRequireApprovalMcpToolApprovalFilterNever" + }, + { + "type": "null" + } + ] + } + }, + "title": "McpRequireApprovalMcpToolApprovalFilter", + "type": "object" + }, + "McpRequireApprovalMcpToolApprovalFilterAlways": { + "additionalProperties": true, + "description": "A filter object to specify which tools are allowed.", + "properties": { + "read_only": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Read Only" + }, + "tool_names": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Tool Names" + } + }, + "title": "McpRequireApprovalMcpToolApprovalFilterAlways", + "type": "object" + }, + "McpRequireApprovalMcpToolApprovalFilterNever": { + "additionalProperties": true, + "description": "A filter object to specify which tools are allowed.", + "properties": { + "read_only": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Read Only" + }, + "tool_names": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Tool Names" + } + }, + "title": "McpRequireApprovalMcpToolApprovalFilterNever", + "type": "object" + }, + "Move": { + "additionalProperties": true, + "description": "A mouse move action.", + "properties": { + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "type": { + "const": "move", + "title": "Type", + "type": "string" + }, + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "type", + "x", + "y" + ], + "title": "Move", + "type": "object" + }, + "NamespaceTool": { + "additionalProperties": true, + "description": "Groups function/custom tools under a shared namespace.", + "properties": { + "description": { + "title": "Description", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "tools": { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/ToolFunction" + }, + { + "$ref": "#/components/schemas/CustomTool" + } + ] + }, + "title": "Tools", + "type": "array" + }, + "type": { + "const": "namespace", + "title": "Type", + "type": "string" + } + }, + "required": [ + "description", + "name", + "tools", + "type" + ], + "title": "NamespaceTool", + "type": "object" + }, + "OperationCreateFile": { + "additionalProperties": true, + "description": "Instruction describing how to create a file via the apply_patch tool.", + "properties": { + "diff": { + "title": "Diff", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + }, + "type": { + "const": "create_file", + "title": "Type", + "type": "string" + } + }, + "required": [ + "diff", + "path", + "type" + ], + "title": "OperationCreateFile", + "type": "object" + }, + "OperationDeleteFile": { + "additionalProperties": true, + "description": "Instruction describing how to delete a file via the apply_patch tool.", + "properties": { + "path": { + "title": "Path", + "type": "string" + }, + "type": { + "const": "delete_file", + "title": "Type", + "type": "string" + } + }, + "required": [ + "path", + "type" + ], + "title": "OperationDeleteFile", + "type": "object" + }, + "OperationUpdateFile": { + "additionalProperties": true, + "description": "Instruction describing how to update a file via the apply_patch tool.", + "properties": { + "diff": { + "title": "Diff", + "type": "string" + }, + "path": { + "title": "Path", + "type": "string" + }, + "type": { + "const": "update_file", + "title": "Type", + "type": "string" + } + }, + "required": [ + "diff", + "path", + "type" + ], + "title": "OperationUpdateFile", + "type": "object" + }, + "Output": { + "additionalProperties": true, + "description": "The content of a shell tool call output that was emitted.", + "properties": { + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "outcome": { + "anyOf": [ + { + "$ref": "#/components/schemas/OutputOutcomeTimeout" + }, + { + "$ref": "#/components/schemas/OutputOutcomeExit" + } + ], + "title": "Outcome" + }, + "stderr": { + "title": "Stderr", + "type": "string" + }, + "stdout": { + "title": "Stdout", + "type": "string" + } + }, + "required": [ + "outcome", + "stderr", + "stdout" + ], + "title": "Output", + "type": "object" + }, + "OutputCodeInterpreterCall": { + "additionalProperties": true, + "description": "A code interpreter / code execution call output", + "properties": { + "code": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Code" + }, + "container_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Container Id" + }, + "id": { + "title": "Id", + "type": "string" + }, + "outputs": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/OutputCodeInterpreterCallLog" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Outputs" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete", + "failed" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "code_interpreter_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type", + "id", + "code", + "container_id", + "status", + "outputs" + ], + "title": "OutputCodeInterpreterCall", + "type": "object" + }, + "OutputCodeInterpreterCallLog": { + "additionalProperties": true, + "description": "Log output from a code interpreter call", + "properties": { + "logs": { + "title": "Logs", + "type": "string" + }, + "type": { + "const": "logs", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type", + "logs" + ], + "title": "OutputCodeInterpreterCallLog", + "type": "object" + }, + "OutputFunctionToolCall": { + "additionalProperties": true, + "description": "A tool call to run a function", + "properties": { + "arguments": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Arguments" + }, + "call_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Call Id" + }, + "id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Id" + }, + "name": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Name" + }, + "phase": { + "anyOf": [ + { + "enum": [ + "commentary", + "final_answer" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Phase" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Type" + } + }, + "required": [ + "arguments", + "call_id", + "name", + "type", + "id", + "status" + ], + "title": "OutputFunctionToolCall", + "type": "object" + }, + "OutputImage": { + "additionalProperties": true, + "description": "The image output from the code interpreter.", + "properties": { + "type": { + "const": "image", + "title": "Type", + "type": "string" + }, + "url": { + "title": "Url", + "type": "string" + } + }, + "required": [ + "type", + "url" + ], + "title": "OutputImage", + "type": "object" + }, + "OutputImageGenerationCall": { + "additionalProperties": true, + "description": "An image generation call output", + "properties": { + "id": { + "title": "Id", + "type": "string" + }, + "result": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Result" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete", + "failed" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "image_generation_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type", + "id", + "status", + "result" + ], + "title": "OutputImageGenerationCall", + "type": "object" + }, + "OutputLogs": { + "additionalProperties": true, + "description": "The logs output from the code interpreter.", + "properties": { + "logs": { + "title": "Logs", + "type": "string" + }, + "type": { + "const": "logs", + "title": "Type", + "type": "string" + } + }, + "required": [ + "logs", + "type" + ], + "title": "OutputLogs", + "type": "object" + }, + "OutputOutcomeExit": { + "additionalProperties": true, + "description": "Indicates that the shell commands finished and returned an exit code.", + "properties": { + "exit_code": { + "title": "Exit Code", + "type": "integer" + }, + "type": { + "const": "exit", + "title": "Type", + "type": "string" + } + }, + "required": [ + "exit_code", + "type" + ], + "title": "OutputOutcomeExit", + "type": "object" + }, + "OutputOutcomeTimeout": { + "additionalProperties": true, + "description": "Indicates that the shell call exceeded its configured time limit.", + "properties": { + "type": { + "const": "timeout", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "OutputOutcomeTimeout", + "type": "object" + }, + "OutputText": { + "additionalProperties": true, + "description": "Text output content from an assistant message", + "properties": { + "annotations": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/GenericResponseOutputItemContentAnnotation" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Annotations" + }, + "text": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Text" + }, + "type": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Type" + } + }, + "required": [ + "type", + "text", + "annotations" + ], + "title": "OutputText", + "type": "object" + }, + "OutputTokensDetails": { + "additionalProperties": true, + "properties": { + "audio_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Audio Tokens" + }, + "reasoning_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Reasoning Tokens" + }, + "text_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Text Tokens" + } + }, + "title": "OutputTokensDetails", + "type": "object" + }, + "PendingSafetyCheck": { + "additionalProperties": true, + "description": "A pending safety check for the computer call.", + "properties": { + "code": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Code" + }, + "id": { + "title": "Id", + "type": "string" + }, + "message": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Message" + } + }, + "required": [ + "id" + ], + "title": "PendingSafetyCheck", + "type": "object" + }, + "RankingOptions": { + "additionalProperties": true, + "description": "Ranking options for search.", + "properties": { + "hybrid_search": { + "anyOf": [ + { + "$ref": "#/components/schemas/RankingOptionsHybridSearch" + }, + { + "type": "null" + } + ] + }, + "ranker": { + "anyOf": [ + { + "enum": [ + "auto", + "default-2024-11-15" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Ranker" + }, + "score_threshold": { + "anyOf": [ + { + "type": "number" + }, + { + "type": "null" + } + ], + "title": "Score Threshold" + } + }, + "title": "RankingOptions", + "type": "object" + }, + "RankingOptionsHybridSearch": { + "additionalProperties": true, + "description": "Weights that control how reciprocal rank fusion balances semantic embedding matches versus sparse keyword matches when hybrid search is enabled.", + "properties": { + "embedding_weight": { + "title": "Embedding Weight", + "type": "number" + }, + "text_weight": { + "title": "Text Weight", + "type": "number" + } + }, + "required": [ + "embedding_weight", + "text_weight" + ], + "title": "RankingOptionsHybridSearch", + "type": "object" + }, "RealtimeClientSecretResponse": { "description": "Response from POST /v1/realtime/client_secrets.\n\nBoth the top-level `value` and `session.client_secret.value`\nwill contain the encrypted token instead of the raw ephemeral key.\nThe `session` field is kept as a raw dict so unknown fields pass through.", "properties": { @@ -16341,6 +20031,2978 @@ "title": "RealtimeTranscriptionSessionResponse", "type": "object" }, + "ResponseAPIUsage": { + "additionalProperties": true, + "properties": { + "cost": { + "anyOf": [ + { + "type": "number" + }, + { + "type": "null" + } + ], + "title": "Cost" + }, + "input_tokens": { + "title": "Input Tokens", + "type": "integer" + }, + "input_tokens_details": { + "anyOf": [ + { + "$ref": "#/components/schemas/InputTokensDetails" + }, + { + "type": "null" + } + ] + }, + "output_tokens": { + "title": "Output Tokens", + "type": "integer" + }, + "output_tokens_details": { + "anyOf": [ + { + "$ref": "#/components/schemas/OutputTokensDetails" + }, + { + "type": "null" + } + ] + }, + "total_tokens": { + "title": "Total Tokens", + "type": "integer" + } + }, + "required": [ + "input_tokens", + "output_tokens", + "total_tokens" + ], + "title": "ResponseAPIUsage", + "type": "object" + }, + "ResponseApplyPatchToolCall": { + "additionalProperties": true, + "description": "A tool call that applies file diffs by creating, deleting, or updating files.", + "properties": { + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "id": { + "title": "Id", + "type": "string" + }, + "operation": { + "anyOf": [ + { + "$ref": "#/components/schemas/OperationCreateFile" + }, + { + "$ref": "#/components/schemas/OperationDeleteFile" + }, + { + "$ref": "#/components/schemas/OperationUpdateFile" + } + ], + "title": "Operation" + }, + "status": { + "enum": [ + "in_progress", + "completed" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "apply_patch_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "call_id", + "operation", + "status", + "type" + ], + "title": "ResponseApplyPatchToolCall", + "type": "object" + }, + "ResponseApplyPatchToolCallOutput": { + "additionalProperties": true, + "description": "The output emitted by an apply patch tool call.", + "properties": { + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "id": { + "title": "Id", + "type": "string" + }, + "output": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Output" + }, + "status": { + "enum": [ + "completed", + "failed" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "apply_patch_call_output", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "call_id", + "status", + "type" + ], + "title": "ResponseApplyPatchToolCallOutput", + "type": "object" + }, + "ResponseCodeInterpreterToolCall": { + "additionalProperties": true, + "description": "A tool call to run code.", + "properties": { + "code": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Code" + }, + "container_id": { + "title": "Container Id", + "type": "string" + }, + "id": { + "title": "Id", + "type": "string" + }, + "outputs": { + "anyOf": [ + { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/OutputLogs" + }, + { + "$ref": "#/components/schemas/OutputImage" + } + ] + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Outputs" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete", + "interpreting", + "failed" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "code_interpreter_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "container_id", + "status", + "type" + ], + "title": "ResponseCodeInterpreterToolCall", + "type": "object" + }, + "ResponseCompactionItem": { + "additionalProperties": true, + "description": "A compaction item generated by the [`v1/responses/compact` API](https://platform.openai.com/docs/api-reference/responses/compact).", + "properties": { + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "encrypted_content": { + "title": "Encrypted Content", + "type": "string" + }, + "id": { + "title": "Id", + "type": "string" + }, + "type": { + "const": "compaction", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "encrypted_content", + "type" + ], + "title": "ResponseCompactionItem", + "type": "object" + }, + "ResponseComputerToolCall": { + "additionalProperties": true, + "description": "A tool call to a computer use tool.\n\nSee the\n[computer use guide](https://platform.openai.com/docs/guides/tools-computer-use) for more information.", + "properties": { + "action": { + "anyOf": [ + { + "$ref": "#/components/schemas/ActionClick" + }, + { + "$ref": "#/components/schemas/ActionDoubleClick" + }, + { + "$ref": "#/components/schemas/ActionDrag" + }, + { + "$ref": "#/components/schemas/ActionKeypress" + }, + { + "$ref": "#/components/schemas/ActionMove" + }, + { + "$ref": "#/components/schemas/ActionScreenshot" + }, + { + "$ref": "#/components/schemas/ActionScroll" + }, + { + "$ref": "#/components/schemas/ActionType" + }, + { + "$ref": "#/components/schemas/ActionWait" + }, + { + "type": "null" + } + ], + "title": "Action" + }, + "actions": { + "anyOf": [ + { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/Click" + }, + { + "$ref": "#/components/schemas/DoubleClick" + }, + { + "$ref": "#/components/schemas/Drag" + }, + { + "$ref": "#/components/schemas/Keypress" + }, + { + "$ref": "#/components/schemas/Move" + }, + { + "$ref": "#/components/schemas/Screenshot" + }, + { + "$ref": "#/components/schemas/Scroll" + }, + { + "$ref": "#/components/schemas/Type" + }, + { + "$ref": "#/components/schemas/Wait" + } + ] + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Actions" + }, + "call_id": { + "title": "Call Id", + "type": "string" + }, + "id": { + "title": "Id", + "type": "string" + }, + "pending_safety_checks": { + "items": { + "$ref": "#/components/schemas/PendingSafetyCheck" + }, + "title": "Pending Safety Checks", + "type": "array" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "computer_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "call_id", + "pending_safety_checks", + "status", + "type" + ], + "title": "ResponseComputerToolCall", + "type": "object" + }, + "ResponseComputerToolCallOutputItem": { + "additionalProperties": true, + "properties": { + "acknowledged_safety_checks": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/AcknowledgedSafetyCheck" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Acknowledged Safety Checks" + }, + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "id": { + "title": "Id", + "type": "string" + }, + "output": { + "$ref": "#/components/schemas/ResponseComputerToolCallOutputScreenshot" + }, + "status": { + "enum": [ + "completed", + "incomplete", + "failed", + "in_progress" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "computer_call_output", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "call_id", + "output", + "status", + "type" + ], + "title": "ResponseComputerToolCallOutputItem", + "type": "object" + }, + "ResponseComputerToolCallOutputScreenshot": { + "additionalProperties": true, + "description": "A computer screenshot image used with the computer use tool.", + "properties": { + "file_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "File Id" + }, + "image_url": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Image Url" + }, + "type": { + "const": "computer_screenshot", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ResponseComputerToolCallOutputScreenshot", + "type": "object" + }, + "ResponseContainerReference": { + "additionalProperties": true, + "description": "Represents a container created with /v1/containers.", + "properties": { + "container_id": { + "title": "Container Id", + "type": "string" + }, + "type": { + "const": "container_reference", + "title": "Type", + "type": "string" + } + }, + "required": [ + "container_id", + "type" + ], + "title": "ResponseContainerReference", + "type": "object" + }, + "ResponseCustomToolCall": { + "additionalProperties": true, + "description": "A call to a custom tool created by the model.", + "properties": { + "call_id": { + "title": "Call Id", + "type": "string" + }, + "id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Id" + }, + "input": { + "title": "Input", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "namespace": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Namespace" + }, + "type": { + "const": "custom_tool_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "call_id", + "input", + "name", + "type" + ], + "title": "ResponseCustomToolCall", + "type": "object" + }, + "ResponseCustomToolCallItem": { + "additionalProperties": true, + "description": "A call to a custom tool created by the model.", + "properties": { + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "id": { + "title": "Id", + "type": "string" + }, + "input": { + "title": "Input", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "namespace": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Namespace" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "custom_tool_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "call_id", + "input", + "name", + "type", + "id", + "status" + ], + "title": "ResponseCustomToolCallItem", + "type": "object" + }, + "ResponseCustomToolCallOutputItem": { + "additionalProperties": true, + "description": "The output of a custom tool call from your code, being sent back to the model.", + "properties": { + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "id": { + "title": "Id", + "type": "string" + }, + "output": { + "anyOf": [ + { + "type": "string" + }, + { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseInputText" + }, + { + "$ref": "#/components/schemas/ResponseInputImage" + }, + { + "$ref": "#/components/schemas/ResponseInputFile" + } + ] + }, + "type": "array" + } + ], + "title": "Output" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "custom_tool_call_output", + "title": "Type", + "type": "string" + } + }, + "required": [ + "call_id", + "output", + "type", + "id", + "status" + ], + "title": "ResponseCustomToolCallOutputItem", + "type": "object" + }, + "ResponseFileSearchToolCall": { + "additionalProperties": true, + "description": "The results of a file search tool call.\n\nSee the\n[file search guide](https://platform.openai.com/docs/guides/tools-file-search) for more information.", + "properties": { + "id": { + "title": "Id", + "type": "string" + }, + "queries": { + "items": { + "type": "string" + }, + "title": "Queries", + "type": "array" + }, + "results": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/Result" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Results" + }, + "status": { + "enum": [ + "in_progress", + "searching", + "completed", + "incomplete", + "failed" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "file_search_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "queries", + "status", + "type" + ], + "title": "ResponseFileSearchToolCall", + "type": "object" + }, + "ResponseFormatJSONObject": { + "additionalProperties": true, + "description": "JSON object response format.\n\nAn older method of generating JSON responses.\nUsing `json_schema` is recommended for models that support it. Note that the\nmodel will not generate JSON without a system or user message instructing it\nto do so.", + "properties": { + "type": { + "const": "json_object", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ResponseFormatJSONObject", + "type": "object" + }, + "ResponseFormatText": { + "additionalProperties": true, + "description": "Default response format. Used to generate text responses.", + "properties": { + "type": { + "const": "text", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ResponseFormatText", + "type": "object" + }, + "ResponseFormatTextJSONSchemaConfigParam": { + "additionalProperties": true, + "description": "JSON Schema response format.\n\nUsed to generate structured JSON responses.\nLearn more about [Structured Outputs](https://platform.openai.com/docs/guides/structured-outputs).", + "properties": { + "description": { + "title": "Description", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "schema": { + "additionalProperties": true, + "title": "Schema", + "type": "object" + }, + "strict": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Strict" + }, + "type": { + "const": "json_schema", + "title": "Type", + "type": "string" + } + }, + "required": [ + "name", + "schema", + "type" + ], + "title": "ResponseFormatTextJSONSchemaConfigParam", + "type": "object" + }, + "ResponseFunctionShellToolCall": { + "additionalProperties": true, + "description": "A tool call that executes one or more shell commands in a managed environment.", + "properties": { + "action": { + "$ref": "#/components/schemas/Action" + }, + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "environment": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseLocalEnvironment" + }, + { + "$ref": "#/components/schemas/ResponseContainerReference" + }, + { + "type": "null" + } + ], + "title": "Environment" + }, + "id": { + "title": "Id", + "type": "string" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "shell_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "action", + "call_id", + "status", + "type" + ], + "title": "ResponseFunctionShellToolCall", + "type": "object" + }, + "ResponseFunctionShellToolCallOutput": { + "additionalProperties": true, + "description": "The output of a shell tool call that was emitted.", + "properties": { + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "id": { + "title": "Id", + "type": "string" + }, + "max_output_length": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Max Output Length" + }, + "output": { + "items": { + "$ref": "#/components/schemas/Output" + }, + "title": "Output", + "type": "array" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "shell_call_output", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "call_id", + "output", + "status", + "type" + ], + "title": "ResponseFunctionShellToolCallOutput", + "type": "object" + }, + "ResponseFunctionToolCall": { + "additionalProperties": true, + "description": "A tool call to run a function.\n\nSee the\n[function calling guide](https://platform.openai.com/docs/guides/function-calling) for more information.", + "properties": { + "arguments": { + "title": "Arguments", + "type": "string" + }, + "call_id": { + "title": "Call Id", + "type": "string" + }, + "id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Id" + }, + "name": { + "title": "Name", + "type": "string" + }, + "namespace": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Namespace" + }, + "status": { + "anyOf": [ + { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Status" + }, + "type": { + "const": "function_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "arguments", + "call_id", + "name", + "type" + ], + "title": "ResponseFunctionToolCall", + "type": "object" + }, + "ResponseFunctionToolCallItem": { + "additionalProperties": true, + "description": "A tool call to run a function.\n\nSee the\n[function calling guide](https://platform.openai.com/docs/guides/function-calling) for more information.", + "properties": { + "arguments": { + "title": "Arguments", + "type": "string" + }, + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "id": { + "title": "Id", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "namespace": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Namespace" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "function_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "arguments", + "call_id", + "name", + "type", + "id", + "status" + ], + "title": "ResponseFunctionToolCallItem", + "type": "object" + }, + "ResponseFunctionToolCallOutputItem": { + "additionalProperties": true, + "properties": { + "call_id": { + "title": "Call Id", + "type": "string" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "id": { + "title": "Id", + "type": "string" + }, + "output": { + "anyOf": [ + { + "type": "string" + }, + { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseInputText" + }, + { + "$ref": "#/components/schemas/ResponseInputImage" + }, + { + "$ref": "#/components/schemas/ResponseInputFile" + } + ] + }, + "type": "array" + } + ], + "title": "Output" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "function_call_output", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "call_id", + "output", + "status", + "type" + ], + "title": "ResponseFunctionToolCallOutputItem", + "type": "object" + }, + "ResponseFunctionWebSearch": { + "additionalProperties": true, + "description": "The results of a web search tool call.\n\nSee the\n[web search guide](https://platform.openai.com/docs/guides/tools-web-search) for more information.", + "properties": { + "action": { + "anyOf": [ + { + "$ref": "#/components/schemas/ActionSearch" + }, + { + "$ref": "#/components/schemas/ActionOpenPage" + }, + { + "$ref": "#/components/schemas/ActionFind" + } + ], + "title": "Action" + }, + "id": { + "title": "Id", + "type": "string" + }, + "status": { + "enum": [ + "in_progress", + "searching", + "completed", + "failed" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "web_search_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "action", + "status", + "type" + ], + "title": "ResponseFunctionWebSearch", + "type": "object" + }, + "ResponseInputFile": { + "additionalProperties": true, + "description": "A file input to the model.", + "properties": { + "detail": { + "anyOf": [ + { + "enum": [ + "high", + "low" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Detail" + }, + "file_data": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "File Data" + }, + "file_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "File Id" + }, + "file_url": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "File Url" + }, + "filename": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Filename" + }, + "type": { + "const": "input_file", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ResponseInputFile", + "type": "object" + }, + "ResponseInputImage": { + "additionalProperties": true, + "description": "An image input to the model.\n\nLearn about [image inputs](https://platform.openai.com/docs/guides/vision).", + "properties": { + "detail": { + "enum": [ + "low", + "high", + "auto", + "original" + ], + "title": "Detail", + "type": "string" + }, + "file_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "File Id" + }, + "image_url": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Image Url" + }, + "type": { + "const": "input_image", + "title": "Type", + "type": "string" + } + }, + "required": [ + "detail", + "type" + ], + "title": "ResponseInputImage", + "type": "object" + }, + "ResponseInputMessageItem": { + "additionalProperties": true, + "properties": { + "content": { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseInputText" + }, + { + "$ref": "#/components/schemas/ResponseInputImage" + }, + { + "$ref": "#/components/schemas/ResponseInputFile" + } + ] + }, + "title": "Content", + "type": "array" + }, + "id": { + "title": "Id", + "type": "string" + }, + "role": { + "enum": [ + "user", + "system", + "developer" + ], + "title": "Role", + "type": "string" + }, + "status": { + "anyOf": [ + { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Status" + }, + "type": { + "const": "message", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "content", + "role", + "type" + ], + "title": "ResponseInputMessageItem", + "type": "object" + }, + "ResponseInputText": { + "additionalProperties": true, + "description": "A text input to the model.", + "properties": { + "text": { + "title": "Text", + "type": "string" + }, + "type": { + "const": "input_text", + "title": "Type", + "type": "string" + } + }, + "required": [ + "text", + "type" + ], + "title": "ResponseInputText", + "type": "object" + }, + "ResponseItemList": { + "additionalProperties": true, + "description": "A list of Response items.", + "properties": { + "data": { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseInputMessageItem" + }, + { + "$ref": "#/components/schemas/ResponseOutputMessage" + }, + { + "$ref": "#/components/schemas/ResponseFileSearchToolCall" + }, + { + "$ref": "#/components/schemas/ResponseComputerToolCall" + }, + { + "$ref": "#/components/schemas/ResponseComputerToolCallOutputItem" + }, + { + "$ref": "#/components/schemas/ResponseFunctionWebSearch" + }, + { + "$ref": "#/components/schemas/ResponseFunctionToolCallItem" + }, + { + "$ref": "#/components/schemas/ResponseFunctionToolCallOutputItem" + }, + { + "$ref": "#/components/schemas/ResponseToolSearchCall" + }, + { + "$ref": "#/components/schemas/ResponseToolSearchOutputItem" + }, + { + "$ref": "#/components/schemas/ResponseReasoningItem" + }, + { + "$ref": "#/components/schemas/ResponseCompactionItem" + }, + { + "$ref": "#/components/schemas/ImageGenerationCall" + }, + { + "$ref": "#/components/schemas/ResponseCodeInterpreterToolCall" + }, + { + "$ref": "#/components/schemas/LocalShellCall" + }, + { + "$ref": "#/components/schemas/LocalShellCallOutput" + }, + { + "$ref": "#/components/schemas/ResponseFunctionShellToolCall" + }, + { + "$ref": "#/components/schemas/ResponseFunctionShellToolCallOutput" + }, + { + "$ref": "#/components/schemas/ResponseApplyPatchToolCall" + }, + { + "$ref": "#/components/schemas/ResponseApplyPatchToolCallOutput" + }, + { + "$ref": "#/components/schemas/McpListTools" + }, + { + "$ref": "#/components/schemas/McpApprovalRequest" + }, + { + "$ref": "#/components/schemas/McpApprovalResponse" + }, + { + "$ref": "#/components/schemas/McpCall" + }, + { + "$ref": "#/components/schemas/ResponseCustomToolCallItem" + }, + { + "$ref": "#/components/schemas/ResponseCustomToolCallOutputItem" + } + ] + }, + "title": "Data", + "type": "array" + }, + "first_id": { + "title": "First Id", + "type": "string" + }, + "has_more": { + "title": "Has More", + "type": "boolean" + }, + "last_id": { + "title": "Last Id", + "type": "string" + }, + "object": { + "const": "list", + "title": "Object", + "type": "string" + } + }, + "required": [ + "data", + "first_id", + "has_more", + "last_id", + "object" + ], + "title": "ResponseItemList", + "type": "object" + }, + "ResponseLocalEnvironment": { + "additionalProperties": true, + "description": "Represents the use of a local environment to perform shell actions.", + "properties": { + "type": { + "const": "local", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ResponseLocalEnvironment", + "type": "object" + }, + "ResponseOutputMessage": { + "additionalProperties": true, + "description": "An output message from the model.", + "properties": { + "content": { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseOutputText" + }, + { + "$ref": "#/components/schemas/ResponseOutputRefusal" + } + ] + }, + "title": "Content", + "type": "array" + }, + "id": { + "title": "Id", + "type": "string" + }, + "phase": { + "anyOf": [ + { + "enum": [ + "commentary", + "final_answer" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Phase" + }, + "role": { + "const": "assistant", + "title": "Role", + "type": "string" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "message", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "content", + "role", + "status", + "type" + ], + "title": "ResponseOutputMessage", + "type": "object" + }, + "ResponseOutputRefusal": { + "additionalProperties": true, + "description": "A refusal from the model.", + "properties": { + "refusal": { + "title": "Refusal", + "type": "string" + }, + "type": { + "const": "refusal", + "title": "Type", + "type": "string" + } + }, + "required": [ + "refusal", + "type" + ], + "title": "ResponseOutputRefusal", + "type": "object" + }, + "ResponseOutputText": { + "additionalProperties": true, + "description": "A text output from the model.", + "properties": { + "annotations": { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/AnnotationFileCitation" + }, + { + "$ref": "#/components/schemas/AnnotationURLCitation" + }, + { + "$ref": "#/components/schemas/AnnotationContainerFileCitation" + }, + { + "$ref": "#/components/schemas/AnnotationFilePath" + } + ] + }, + "title": "Annotations", + "type": "array" + }, + "logprobs": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/Logprob" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Logprobs" + }, + "text": { + "title": "Text", + "type": "string" + }, + "type": { + "const": "output_text", + "title": "Type", + "type": "string" + } + }, + "required": [ + "annotations", + "text", + "type" + ], + "title": "ResponseOutputText", + "type": "object" + }, + "ResponseReasoningItem": { + "additionalProperties": true, + "description": "A description of the chain of thought used by a reasoning model while generating\na response. Be sure to include these items in your `input` to the Responses API\nfor subsequent turns of a conversation if you are manually\n[managing context](https://platform.openai.com/docs/guides/conversation-state).", + "properties": { + "content": { + "anyOf": [ + { + "items": { + "$ref": "#/components/schemas/Content" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Content" + }, + "encrypted_content": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Encrypted Content" + }, + "id": { + "title": "Id", + "type": "string" + }, + "status": { + "anyOf": [ + { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Status" + }, + "summary": { + "items": { + "$ref": "#/components/schemas/Summary" + }, + "title": "Summary", + "type": "array" + }, + "type": { + "const": "reasoning", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "summary", + "type" + ], + "title": "ResponseReasoningItem", + "type": "object" + }, + "ResponseTextConfigParam": { + "additionalProperties": true, + "description": "Configuration options for a text response from the model.\n\nCan be plain\ntext or structured JSON data. Learn more:\n- [Text inputs and outputs](https://platform.openai.com/docs/guides/text)\n- [Structured Outputs](https://platform.openai.com/docs/guides/structured-outputs)", + "properties": { + "format": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseFormatText" + }, + { + "$ref": "#/components/schemas/ResponseFormatTextJSONSchemaConfigParam" + }, + { + "$ref": "#/components/schemas/ResponseFormatJSONObject" + } + ], + "title": "Format" + }, + "verbosity": { + "anyOf": [ + { + "enum": [ + "low", + "medium", + "high" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Verbosity" + } + }, + "title": "ResponseTextConfigParam", + "type": "object" + }, + "ResponseToolSearchCall": { + "additionalProperties": true, + "properties": { + "arguments": { + "title": "Arguments" + }, + "call_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Call Id" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "execution": { + "enum": [ + "server", + "client" + ], + "title": "Execution", + "type": "string" + }, + "id": { + "title": "Id", + "type": "string" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "type": { + "const": "tool_search_call", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "arguments", + "execution", + "status", + "type" + ], + "title": "ResponseToolSearchCall", + "type": "object" + }, + "ResponseToolSearchOutputItem": { + "additionalProperties": true, + "properties": { + "call_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Call Id" + }, + "created_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Created By" + }, + "execution": { + "enum": [ + "server", + "client" + ], + "title": "Execution", + "type": "string" + }, + "id": { + "title": "Id", + "type": "string" + }, + "status": { + "enum": [ + "in_progress", + "completed", + "incomplete" + ], + "title": "Status", + "type": "string" + }, + "tools": { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/FunctionTool" + }, + { + "$ref": "#/components/schemas/FileSearchTool" + }, + { + "$ref": "#/components/schemas/ComputerTool" + }, + { + "$ref": "#/components/schemas/ComputerUsePreviewTool" + }, + { + "$ref": "#/components/schemas/WebSearchTool" + }, + { + "$ref": "#/components/schemas/Mcp" + }, + { + "$ref": "#/components/schemas/CodeInterpreter" + }, + { + "$ref": "#/components/schemas/ImageGeneration" + }, + { + "$ref": "#/components/schemas/LocalShell" + }, + { + "$ref": "#/components/schemas/FunctionShellTool" + }, + { + "$ref": "#/components/schemas/CustomTool" + }, + { + "$ref": "#/components/schemas/NamespaceTool" + }, + { + "$ref": "#/components/schemas/ToolSearchTool" + }, + { + "$ref": "#/components/schemas/WebSearchPreviewTool" + }, + { + "$ref": "#/components/schemas/ApplyPatchTool" + } + ] + }, + "title": "Tools", + "type": "array" + }, + "type": { + "const": "tool_search_output", + "title": "Type", + "type": "string" + } + }, + "required": [ + "id", + "execution", + "status", + "tools", + "type" + ], + "title": "ResponseToolSearchOutputItem", + "type": "object" + }, + "ResponsesAPIResponse": { + "additionalProperties": true, + "properties": { + "created_at": { + "title": "Created At", + "type": "integer" + }, + "error": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Error" + }, + "id": { + "title": "Id", + "type": "string" + }, + "incomplete_details": { + "anyOf": [ + { + "$ref": "#/components/schemas/IncompleteDetails" + }, + { + "type": "null" + } + ] + }, + "instructions": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Instructions" + }, + "max_output_tokens": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Max Output Tokens" + }, + "metadata": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Metadata" + }, + "model": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Model" + }, + "object": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Object" + }, + "output": { + "anyOf": [ + { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseOutputMessage" + }, + { + "$ref": "#/components/schemas/ResponseFileSearchToolCall" + }, + { + "$ref": "#/components/schemas/ResponseFunctionToolCall" + }, + { + "$ref": "#/components/schemas/ResponseFunctionToolCallOutputItem" + }, + { + "$ref": "#/components/schemas/ResponseFunctionWebSearch" + }, + { + "$ref": "#/components/schemas/ResponseComputerToolCall" + }, + { + "$ref": "#/components/schemas/ResponseComputerToolCallOutputItem" + }, + { + "$ref": "#/components/schemas/ResponseReasoningItem" + }, + { + "$ref": "#/components/schemas/ResponseToolSearchCall" + }, + { + "$ref": "#/components/schemas/ResponseToolSearchOutputItem" + }, + { + "$ref": "#/components/schemas/ResponseCompactionItem" + }, + { + "$ref": "#/components/schemas/ImageGenerationCall" + }, + { + "$ref": "#/components/schemas/ResponseCodeInterpreterToolCall" + }, + { + "$ref": "#/components/schemas/LocalShellCall" + }, + { + "$ref": "#/components/schemas/LocalShellCallOutput" + }, + { + "$ref": "#/components/schemas/ResponseFunctionShellToolCall" + }, + { + "$ref": "#/components/schemas/ResponseFunctionShellToolCallOutput" + }, + { + "$ref": "#/components/schemas/ResponseApplyPatchToolCall" + }, + { + "$ref": "#/components/schemas/ResponseApplyPatchToolCallOutput" + }, + { + "$ref": "#/components/schemas/McpCall" + }, + { + "$ref": "#/components/schemas/McpListTools" + }, + { + "$ref": "#/components/schemas/McpApprovalRequest" + }, + { + "$ref": "#/components/schemas/McpApprovalResponse" + }, + { + "$ref": "#/components/schemas/ResponseCustomToolCall" + }, + { + "$ref": "#/components/schemas/ResponseCustomToolCallOutputItem" + }, + { + "additionalProperties": true, + "type": "object" + } + ] + }, + "type": "array" + }, + { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/GenericResponseOutputItem" + }, + { + "$ref": "#/components/schemas/OutputCodeInterpreterCall" + }, + { + "$ref": "#/components/schemas/OutputFunctionToolCall" + }, + { + "$ref": "#/components/schemas/OutputImageGenerationCall" + }, + { + "$ref": "#/components/schemas/ResponseFunctionToolCall" + }, + { + "$ref": "#/components/schemas/ResponseFunctionWebSearch" + }, + { + "$ref": "#/components/schemas/CustomToolCallOutputItem" + } + ] + }, + "type": "array" + } + ], + "title": "Output" + }, + "parallel_tool_calls": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Parallel Tool Calls" + }, + "previous_response_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Previous Response Id" + }, + "reasoning": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Reasoning" + }, + "status": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Status" + }, + "store": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Store" + }, + "temperature": { + "anyOf": [ + { + "type": "number" + }, + { + "type": "null" + } + ], + "title": "Temperature" + }, + "text": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseTextConfigParam" + }, + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Text" + }, + "tool_choice": { + "anyOf": [ + { + "enum": [ + "none", + "auto", + "required" + ], + "type": "string" + }, + { + "$ref": "#/components/schemas/ToolChoiceAllowedParam" + }, + { + "$ref": "#/components/schemas/ToolChoiceTypesParam" + }, + { + "$ref": "#/components/schemas/ToolChoiceFunctionParam" + }, + { + "$ref": "#/components/schemas/ToolChoiceMcpParam" + }, + { + "$ref": "#/components/schemas/ToolChoiceCustomParam" + }, + { + "$ref": "#/components/schemas/ToolChoiceApplyPatchParam" + }, + { + "$ref": "#/components/schemas/ToolChoiceShellParam" + }, + { + "type": "null" + } + ], + "title": "Tool Choice" + }, + "tools": { + "anyOf": [ + { + "items": { + "anyOf": [ + { + "$ref": "#/components/schemas/FunctionTool" + }, + { + "$ref": "#/components/schemas/FileSearchTool" + }, + { + "$ref": "#/components/schemas/ComputerTool" + }, + { + "$ref": "#/components/schemas/ComputerUsePreviewTool" + }, + { + "$ref": "#/components/schemas/WebSearchTool" + }, + { + "$ref": "#/components/schemas/Mcp" + }, + { + "$ref": "#/components/schemas/CodeInterpreter" + }, + { + "$ref": "#/components/schemas/ImageGeneration" + }, + { + "$ref": "#/components/schemas/LocalShell" + }, + { + "$ref": "#/components/schemas/FunctionShellTool" + }, + { + "$ref": "#/components/schemas/CustomTool" + }, + { + "$ref": "#/components/schemas/NamespaceTool" + }, + { + "$ref": "#/components/schemas/ToolSearchTool" + }, + { + "$ref": "#/components/schemas/WebSearchPreviewTool" + }, + { + "$ref": "#/components/schemas/ApplyPatchTool" + } + ] + }, + "type": "array" + }, + { + "items": { + "$ref": "#/components/schemas/ResponseFunctionToolCall" + }, + "type": "array" + }, + { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Tools" + }, + "top_p": { + "anyOf": [ + { + "type": "number" + }, + { + "type": "null" + } + ], + "title": "Top P" + }, + "truncation": { + "anyOf": [ + { + "enum": [ + "auto", + "disabled" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Truncation" + }, + "usage": { + "anyOf": [ + { + "$ref": "#/components/schemas/ResponseAPIUsage" + }, + { + "type": "null" + } + ] + }, + "user": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "User" + } + }, + "required": [ + "id", + "created_at", + "output" + ], + "title": "ResponsesAPIResponse", + "type": "object" + }, + "Result": { + "additionalProperties": true, + "properties": { + "attributes": { + "anyOf": [ + { + "additionalProperties": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "number" + }, + { + "type": "boolean" + } + ] + }, + "type": "object" + }, + { + "type": "null" + } + ], + "title": "Attributes" + }, + "file_id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "File Id" + }, + "filename": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Filename" + }, + "score": { + "anyOf": [ + { + "type": "number" + }, + { + "type": "null" + } + ], + "title": "Score" + }, + "text": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Text" + } + }, + "title": "Result", + "type": "object" + }, + "Screenshot": { + "additionalProperties": true, + "description": "A screenshot action.", + "properties": { + "type": { + "const": "screenshot", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "Screenshot", + "type": "object" + }, + "Scroll": { + "additionalProperties": true, + "description": "A scroll action.", + "properties": { + "keys": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Keys" + }, + "scroll_x": { + "title": "Scroll X", + "type": "integer" + }, + "scroll_y": { + "title": "Scroll Y", + "type": "integer" + }, + "type": { + "const": "scroll", + "title": "Type", + "type": "string" + }, + "x": { + "title": "X", + "type": "integer" + }, + "y": { + "title": "Y", + "type": "integer" + } + }, + "required": [ + "scroll_x", + "scroll_y", + "type", + "x", + "y" + ], + "title": "Scroll", + "type": "object" + }, + "SkillReference": { + "additionalProperties": true, + "properties": { + "skill_id": { + "title": "Skill Id", + "type": "string" + }, + "type": { + "const": "skill_reference", + "title": "Type", + "type": "string" + }, + "version": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Version" + } + }, + "required": [ + "skill_id", + "type" + ], + "title": "SkillReference", + "type": "object" + }, + "Summary": { + "additionalProperties": true, + "description": "A summary text from the model.", + "properties": { + "text": { + "title": "Text", + "type": "string" + }, + "type": { + "const": "summary_text", + "title": "Type", + "type": "string" + } + }, + "required": [ + "text", + "type" + ], + "title": "Summary", + "type": "object" + }, + "Text": { + "additionalProperties": true, + "description": "Unconstrained free-form text.", + "properties": { + "type": { + "const": "text", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "Text", + "type": "object" + }, + "ToolChoiceAllowedParam": { + "additionalProperties": true, + "description": "Constrains the tools available to the model to a pre-defined set.", + "properties": { + "mode": { + "enum": [ + "auto", + "required" + ], + "title": "Mode", + "type": "string" + }, + "tools": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "title": "Tools", + "type": "array" + }, + "type": { + "const": "allowed_tools", + "title": "Type", + "type": "string" + } + }, + "required": [ + "mode", + "tools", + "type" + ], + "title": "ToolChoiceAllowedParam", + "type": "object" + }, + "ToolChoiceApplyPatchParam": { + "additionalProperties": true, + "description": "Forces the model to call the apply_patch tool when executing a tool call.", + "properties": { + "type": { + "const": "apply_patch", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ToolChoiceApplyPatchParam", + "type": "object" + }, + "ToolChoiceCustomParam": { + "additionalProperties": true, + "description": "Use this option to force the model to call a specific custom tool.", + "properties": { + "name": { + "title": "Name", + "type": "string" + }, + "type": { + "const": "custom", + "title": "Type", + "type": "string" + } + }, + "required": [ + "name", + "type" + ], + "title": "ToolChoiceCustomParam", + "type": "object" + }, + "ToolChoiceFunctionParam": { + "additionalProperties": true, + "description": "Use this option to force the model to call a specific function.", + "properties": { + "name": { + "title": "Name", + "type": "string" + }, + "type": { + "const": "function", + "title": "Type", + "type": "string" + } + }, + "required": [ + "name", + "type" + ], + "title": "ToolChoiceFunctionParam", + "type": "object" + }, + "ToolChoiceMcpParam": { + "additionalProperties": true, + "description": "Use this option to force the model to call a specific tool on a remote MCP server.", + "properties": { + "name": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Name" + }, + "server_label": { + "title": "Server Label", + "type": "string" + }, + "type": { + "const": "mcp", + "title": "Type", + "type": "string" + } + }, + "required": [ + "server_label", + "type" + ], + "title": "ToolChoiceMcpParam", + "type": "object" + }, + "ToolChoiceShellParam": { + "additionalProperties": true, + "description": "Forces the model to call the shell tool when a tool call is required.", + "properties": { + "type": { + "const": "shell", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ToolChoiceShellParam", + "type": "object" + }, + "ToolChoiceTypesParam": { + "additionalProperties": true, + "description": "Indicates that the model should use a built-in tool to generate a response.\n[Learn more about built-in tools](https://platform.openai.com/docs/guides/tools).", + "properties": { + "type": { + "enum": [ + "file_search", + "web_search_preview", + "computer", + "computer_use_preview", + "computer_use", + "web_search_preview_2025_03_11", + "image_generation", + "code_interpreter" + ], + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ToolChoiceTypesParam", + "type": "object" + }, + "ToolFunction": { + "additionalProperties": true, + "properties": { + "defer_loading": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Defer Loading" + }, + "description": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Description" + }, + "name": { + "title": "Name", + "type": "string" + }, + "parameters": { + "anyOf": [ + {}, + { + "type": "null" + } + ], + "title": "Parameters" + }, + "strict": { + "anyOf": [ + { + "type": "boolean" + }, + { + "type": "null" + } + ], + "title": "Strict" + }, + "type": { + "const": "function", + "title": "Type", + "type": "string" + } + }, + "required": [ + "name", + "type" + ], + "title": "ToolFunction", + "type": "object" + }, + "ToolSearchTool": { + "additionalProperties": true, + "description": "Hosted or BYOT tool search configuration for deferred tools.", + "properties": { + "description": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Description" + }, + "execution": { + "anyOf": [ + { + "enum": [ + "server", + "client" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Execution" + }, + "parameters": { + "anyOf": [ + {}, + { + "type": "null" + } + ], + "title": "Parameters" + }, + "type": { + "const": "tool_search", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "ToolSearchTool", + "type": "object" + }, + "Type": { + "additionalProperties": true, + "description": "An action to type in text.", + "properties": { + "text": { + "title": "Text", + "type": "string" + }, + "type": { + "const": "type", + "title": "Type", + "type": "string" + } + }, + "required": [ + "text", + "type" + ], + "title": "Type", + "type": "object" + }, "ValidationError": { "properties": { "ctx": { @@ -16380,6 +23042,264 @@ ], "title": "ValidationError", "type": "object" + }, + "Wait": { + "additionalProperties": true, + "description": "A wait action.", + "properties": { + "type": { + "const": "wait", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "Wait", + "type": "object" + }, + "WebSearchPreviewTool": { + "additionalProperties": true, + "description": "This tool searches the web for relevant results to use in a response.\n\nLearn more about the [web search tool](https://platform.openai.com/docs/guides/tools-web-search).", + "properties": { + "search_content_types": { + "anyOf": [ + { + "items": { + "enum": [ + "text", + "image" + ], + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Search Content Types" + }, + "search_context_size": { + "anyOf": [ + { + "enum": [ + "low", + "medium", + "high" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Search Context Size" + }, + "type": { + "enum": [ + "web_search_preview", + "web_search_preview_2025_03_11" + ], + "title": "Type", + "type": "string" + }, + "user_location": { + "anyOf": [ + { + "$ref": "#/components/schemas/openai__types__responses__web_search_preview_tool__UserLocation" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "type" + ], + "title": "WebSearchPreviewTool", + "type": "object" + }, + "WebSearchTool": { + "additionalProperties": true, + "description": "Search the Internet for sources related to the prompt.\n\nLearn more about the\n[web search tool](https://platform.openai.com/docs/guides/tools-web-search).", + "properties": { + "filters": { + "anyOf": [ + { + "$ref": "#/components/schemas/Filters" + }, + { + "type": "null" + } + ] + }, + "search_context_size": { + "anyOf": [ + { + "enum": [ + "low", + "medium", + "high" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Search Context Size" + }, + "type": { + "enum": [ + "web_search", + "web_search_2025_08_26" + ], + "title": "Type", + "type": "string" + }, + "user_location": { + "anyOf": [ + { + "$ref": "#/components/schemas/openai__types__responses__web_search_tool__UserLocation" + }, + { + "type": "null" + } + ] + } + }, + "required": [ + "type" + ], + "title": "WebSearchTool", + "type": "object" + }, + "openai__types__responses__web_search_preview_tool__UserLocation": { + "additionalProperties": true, + "description": "The user's location.", + "properties": { + "city": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "City" + }, + "country": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Country" + }, + "region": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Region" + }, + "timezone": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Timezone" + }, + "type": { + "const": "approximate", + "title": "Type", + "type": "string" + } + }, + "required": [ + "type" + ], + "title": "UserLocation", + "type": "object" + }, + "openai__types__responses__web_search_tool__UserLocation": { + "additionalProperties": true, + "description": "The approximate location of the user.", + "properties": { + "city": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "City" + }, + "country": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Country" + }, + "region": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Region" + }, + "timezone": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Timezone" + }, + "type": { + "anyOf": [ + { + "const": "approximate", + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Type" + } + }, + "title": "UserLocation", + "type": "object" } } }, @@ -20253,7 +27173,15 @@ "200": { "content": { "application/json": { - "schema": {} + "schema": { + "$ref": "#/components/schemas/ResponsesAPIResponse" + } + }, + "text/event-stream": { + "schema": { + "description": "Server sent events when stream=true", + "type": "string" + } } }, "description": "Successful Response" @@ -20339,7 +27267,9 @@ "200": { "content": { "application/json": { - "schema": {} + "schema": { + "$ref": "#/components/schemas/DeleteResponseResult" + } } }, "description": "Successful Response" @@ -20383,7 +27313,9 @@ "200": { "content": { "application/json": { - "schema": {} + "schema": { + "$ref": "#/components/schemas/ResponsesAPIResponse" + } } }, "description": "Successful Response" @@ -20475,7 +27407,9 @@ "200": { "content": { "application/json": { - "schema": {} + "schema": { + "$ref": "#/components/schemas/ResponseItemList" + } } }, "description": "Successful Response" diff --git a/litellm/proxy/common_utils/custom_openapi_spec.py b/litellm/proxy/common_utils/custom_openapi_spec.py index 2a20e7b07ce..c1d4442893a 100644 --- a/litellm/proxy/common_utils/custom_openapi_spec.py +++ b/litellm/proxy/common_utils/custom_openapi_spec.py @@ -1,5 +1,8 @@ from collections.abc import Mapping, Sequence -from typing import Final, TypeAlias, Union +from types import MappingProxyType +from typing import Final, TypeAlias, Union, cast + +from pydantic import TypeAdapter from litellm._logging import verbose_proxy_logger @@ -28,7 +31,7 @@ class CustomOpenAPISpec: "/openai/deployments/{model}/embeddings", ] - RESPONSES_API_PATHS = ["/v1/responses", "/responses"] + RESPONSES_API_PATHS = ["/v1/responses", "/responses", "/openai/v1/responses"] @staticmethod def _as_object(node: JsonValue) -> JsonObject: @@ -44,26 +47,18 @@ class CustomOpenAPISpec: return CustomOpenAPISpec._as_object(components.setdefault("schemas", {})) @staticmethod - def get_pydantic_schema(model_class) -> JsonObject | None: + def get_pydantic_schema(model_class: type) -> JsonObject | None: """ - Get JSON schema from a Pydantic model, handling both v1 and v2 APIs. + Get JSON schema for a request or response model class, including TypedDicts. Args: - model_class: Pydantic model class + model_class: Pydantic model class or TypedDict Returns: JSON schema dict or None if failed """ try: - # Try Pydantic v2 method first - return model_class.model_json_schema() - except AttributeError: - try: - # Fallback to Pydantic v1 method - return model_class.schema() - except AttributeError: - # If both methods fail, return None - return None + return cast(JsonObject, TypeAdapter(model_class).json_schema()) # cast-ok: pydantic returns dict[str, Any] except Exception as e: # FastAPI 0.120+ may fail schema generation for certain types (e.g., openai.Timeout) # Log the error and return None to skip schema generation for this model @@ -83,13 +78,18 @@ class CustomOpenAPISpec: # Ensure components/schemas structure exists _ = CustomOpenAPISpec._components_schemas(openapi_schema) - # Add the schema - CustomOpenAPISpec._move_defs_to_components(openapi_schema, {schema_name: schema_def}) + defs: Final[Mapping[str, JsonValue]] = ( + CustomOpenAPISpec._as_object(schema_def["$defs"]) if "$defs" in schema_def else MappingProxyType({}) + ) + renames: Final = CustomOpenAPISpec._move_defs_to_components(openapi_schema, defs, schema_name) + schemas: Final = CustomOpenAPISpec._components_schemas(openapi_schema) + schemas[schema_name] = CustomOpenAPISpec._rewrite_defs_refs(schema_def, renames) @staticmethod def _expanded_request_field(field_name: str, field_def: JsonValue) -> JsonValue: expanded: Final = CustomOpenAPISpec._rewrite_defs_refs( - CustomOpenAPISpec._expand_field_definition(CustomOpenAPISpec._as_object(field_def)) + CustomOpenAPISpec._expand_field_definition(CustomOpenAPISpec._as_object(field_def)), + MappingProxyType({}), ) if field_name != "messages": return expanded @@ -127,13 +127,6 @@ class CustomOpenAPISpec: schema_properties = CustomOpenAPISpec._as_object(actual_schema.get("properties")) required_fields = actual_schema.get("required", []) - # Extract $defs and add them to components/schemas - # This fixes Pydantic v2 $defs not being resolvable in Swagger/OpenAPI - if "$defs" in actual_schema: - CustomOpenAPISpec._move_defs_to_components( - openapi_schema, CustomOpenAPISpec._as_object(actual_schema["$defs"]) - ) - # Create an expanded inline schema instead of just a $ref # This makes Swagger UI show all individual fields in the request body editor expanded_schema: JsonObject = { @@ -161,7 +154,9 @@ class CustomOpenAPISpec: ] @staticmethod - def _move_defs_to_components(openapi_schema: JsonObject, defs: Mapping[str, JsonValue]) -> None: + def _move_defs_to_components( + openapi_schema: JsonObject, defs: Mapping[str, JsonValue], namespace: str + ) -> Mapping[str, str]: """ Move $defs from Pydantic v2 schema to OpenAPI components/schemas. This makes the definitions resolvable in Swagger/OpenAPI viewers. @@ -169,36 +164,68 @@ class CustomOpenAPISpec: Args: openapi_schema: The OpenAPI schema dict to modify defs: The $defs dictionary from Pydantic schema + namespace: Prefix used to rename defs that would overwrite an existing component + + Returns: + Map of original def names to renamed component names for collision cases """ - if not defs: - return - - # Ensure components/schemas exists schemas: Final = CustomOpenAPISpec._components_schemas(openapi_schema) - - # Add each definition to components/schemas + renames: Final = CustomOpenAPISpec._fixed_renames(schemas, defs, namespace, MappingProxyType({})) for def_name, def_schema in defs.items(): - # Recursively rewrite any nested $defs references within this definition - schemas[def_name] = CustomOpenAPISpec._rewrite_defs_refs(def_schema) - - # If this definition also has $defs, process them recursively - def_object = CustomOpenAPISpec._as_object(def_schema) - if "$defs" in def_object: - CustomOpenAPISpec._move_defs_to_components( - openapi_schema, CustomOpenAPISpec._as_object(def_object["$defs"]) - ) + if def_name in schemas and def_name not in renames: + continue + schemas[renames.get(def_name, def_name)] = CustomOpenAPISpec._rewrite_defs_refs(def_schema, renames) + return renames @staticmethod - def _rewritten_defs_entry(key: str, value: JsonValue) -> JsonValue: + def _def_collisions( + schemas: JsonObject, defs: Mapping[str, JsonValue], namespace: str, renames: Mapping[str, str] + ) -> Mapping[str, str]: + return MappingProxyType( + { + name: f"{namespace}_{name}" + for name, d in defs.items() + if name in schemas + and not CustomOpenAPISpec._same_shape(schemas[name], CustomOpenAPISpec._rewrite_defs_refs(d, renames)) + } + ) + + @staticmethod + def _same_shape(existing: JsonValue, incoming: JsonValue) -> bool: + if existing == incoming: + return True + existing_obj: Final = CustomOpenAPISpec._as_object(existing) + incoming_obj: Final = CustomOpenAPISpec._as_object(incoming) + existing_props: Final = CustomOpenAPISpec._as_object(existing_obj.get("properties")) + incoming_props: Final = CustomOpenAPISpec._as_object(incoming_obj.get("properties")) + if not existing_props or not incoming_props: + return False + return existing_props.keys() == incoming_props.keys() and frozenset( + x for x in CustomOpenAPISpec._as_array(existing_obj.get("required")) if isinstance(x, str) + ) == frozenset(x for x in CustomOpenAPISpec._as_array(incoming_obj.get("required")) if isinstance(x, str)) + + @staticmethod + def _fixed_renames( + schemas: JsonObject, defs: Mapping[str, JsonValue], namespace: str, renames: Mapping[str, str] + ) -> Mapping[str, str]: + next_renames: Final = MappingProxyType( + {**renames, **CustomOpenAPISpec._def_collisions(schemas, defs, namespace, renames)} + ) + if next_renames == renames: + return renames + return CustomOpenAPISpec._fixed_renames(schemas, defs, namespace, next_renames) + + @staticmethod + def _rewritten_defs_entry(key: str, value: JsonValue, renames: Mapping[str, str]) -> JsonValue: if key == "$ref" and isinstance(value, str) and value.startswith("#/$defs/"): # Rewrite the reference to use components/schemas def_name: Final = value.replace("#/$defs/", "") - return f"#/components/schemas/{def_name}" + return f"#/components/schemas/{renames.get(def_name, def_name)}" # Recursively process nested structures - return CustomOpenAPISpec._rewrite_defs_refs(value) + return CustomOpenAPISpec._rewrite_defs_refs(value, renames) @staticmethod - def _rewrite_defs_refs(schema: JsonValue) -> JsonValue: + def _rewrite_defs_refs(schema: JsonValue, renames: Mapping[str, str]) -> JsonValue: """ Recursively rewrite $ref values from #/$defs/... to #/components/schemas/... This converts Pydantic v2 references to OpenAPI-compatible references. @@ -211,12 +238,12 @@ class CustomOpenAPISpec: """ if isinstance(schema, dict): return { - key: CustomOpenAPISpec._rewritten_defs_entry(key, value) + key: CustomOpenAPISpec._rewritten_defs_entry(key, value, renames) for key, value in schema.items() if key != "$defs" } if isinstance(schema, list): - return [CustomOpenAPISpec._rewrite_defs_refs(item) for item in schema] + return [CustomOpenAPISpec._rewrite_defs_refs(item, renames) for item in schema] return schema @staticmethod diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 69c7f0a09ed..ca4008dba3d 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -12,6 +12,7 @@ from uuid import uuid4 import fastapi from fastapi import APIRouter, Depends, HTTPException, Request, Response from fastapi.responses import JSONResponse +from openai.types.responses import ResponseItemList from openai.types.responses.response_create_params import ResponseInputParam from pydantic import BaseModel, ConfigDict, ValidationError from starlette.websockets import WebSocket, WebSocketDisconnect @@ -48,6 +49,20 @@ if TYPE_CHECKING: router: Final = APIRouter() +_ResponseDocSchemas = dict[int | str, dict[str, Any]] # pyright: ignore[reportExplicitAny] # fastapi's responses kwarg + +RESPONSES_API_RESPONSE_SCHEMAS: Final[_ResponseDocSchemas] = {200: {"model": ResponsesAPIResponse}} +RESPONSES_API_CREATE_RESPONSE_SCHEMAS: Final[_ResponseDocSchemas] = { + 200: { + "model": ResponsesAPIResponse, + "content": { + "text/event-stream": {"schema": {"type": "string", "description": "Server sent events when stream=true"}} + }, + } +} +DELETE_RESPONSE_SCHEMAS: Final[_ResponseDocSchemas] = {200: {"model": DeleteResponseResult}} +RESPONSE_ITEM_LIST_SCHEMAS: Final[_ResponseDocSchemas] = {200: {"model": ResponseItemList}} + _user_api_key_auth_dep: Final = Depends(user_api_key_auth) _RESPONSES_TAGS: Final[list[str | Enum]] = ["responses"] # mutable-ok: fastapi's route signature requires list tags @@ -181,16 +196,19 @@ async def _resolve_cursor_model_variant_before_auth(request: Request) -> None: "/v1/responses", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSES_API_CREATE_RESPONSE_SCHEMAS, ) @router.post( "/responses", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSES_API_CREATE_RESPONSE_SCHEMAS, ) @router.post( "/openai/v1/responses", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSES_API_CREATE_RESPONSE_SCHEMAS, ) async def responses_api( request: Request, @@ -664,16 +682,19 @@ async def cursor_chat_completions( "/v1/responses/{response_id}", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSES_API_RESPONSE_SCHEMAS, ) @router.get( "/responses/{response_id}", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSES_API_RESPONSE_SCHEMAS, ) @router.get( "/openai/v1/responses/{response_id}", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSES_API_RESPONSE_SCHEMAS, ) async def get_response( response_id: str, @@ -777,16 +798,19 @@ async def get_response( "/v1/responses/{response_id}", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=DELETE_RESPONSE_SCHEMAS, ) @router.delete( "/responses/{response_id}", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=DELETE_RESPONSE_SCHEMAS, ) @router.delete( "/openai/v1/responses/{response_id}", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=DELETE_RESPONSE_SCHEMAS, ) async def delete_response( response_id: str, @@ -883,16 +907,19 @@ async def delete_response( "/v1/responses/{response_id}/input_items", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSE_ITEM_LIST_SCHEMAS, ) @router.get( "/responses/{response_id}/input_items", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSE_ITEM_LIST_SCHEMAS, ) @router.get( "/openai/v1/responses/{response_id}/input_items", dependencies=[Depends(user_api_key_auth)], tags=["responses"], + responses=RESPONSE_ITEM_LIST_SCHEMAS, ) async def get_response_input_items( response_id: str, diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index 599b1a76249..6e7e9da3498 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -1301,8 +1301,8 @@ class ResponsesAPIOptionalRequestParams(TypedDict, total=False): class ResponsesAPIRequestParams(ResponsesAPIOptionalRequestParams, total=False): """TypedDict for request parameters supported by the responses API.""" - input: str | ResponseInputParam - model: str + input: Required[ReadOnly[str | ResponseInputParam]] + model: Required[ReadOnly[str]] class OutputTokensDetails(BaseLiteLLMOpenAIResponseObject): diff --git a/tests/integration/compatibility/test_responses_openapi_schema.py b/tests/integration/compatibility/test_responses_openapi_schema.py index 65039bb5f2e..c8c51eccd8c 100644 --- a/tests/integration/compatibility/test_responses_openapi_schema.py +++ b/tests/integration/compatibility/test_responses_openapi_schema.py @@ -1,21 +1,43 @@ -import pytest -from integration._support.client import Gateway, object_value +from typing import Final + +from integration._support.client import Gateway, object_value, string_value from pydantic import JsonValue -def _assert_responses_post_is_documented(openapi: dict[str, JsonValue]) -> None: - post: dict[str, JsonValue] = object_value(object_value(object_value(openapi["paths"])["/v1/responses"])["post"]) - body: dict[str, JsonValue] = object_value(post["requestBody"]) - schema: dict[str, JsonValue] = object_value( - object_value(object_value(body["content"])["application/json"])["schema"] - ) - properties: dict[str, JsonValue] = object_value(schema.get("properties")) - assert "model" in properties and "input" in properties, schema - ok: dict[str, JsonValue] = object_value(object_value(object_value(post)["responses"])["200"]) - assert "schema" in object_value(object_value(ok["content"])["application/json"]), ok +def _operation(openapi: dict[str, JsonValue], path: str, method: str) -> dict[str, JsonValue]: + return object_value(object_value(object_value(openapi["paths"])[path])[method]) + + +def _ok_schema_properties(openapi: dict[str, JsonValue], operation: dict[str, JsonValue]) -> dict[str, JsonValue]: + ok: Final = object_value(object_value(operation["responses"])["200"]) + schema: Final = object_value(object_value(object_value(ok["content"])["application/json"])["schema"]) + if "$ref" in schema: + name: Final = string_value(schema["$ref"]).rsplit("/", 1)[-1] + return object_value(object_value(object_value(object_value(openapi["components"])["schemas"])[name])["properties"]) + assert "properties" in schema, ok + return object_value(schema["properties"]) def test_v1_responses_post_declares_a_request_body_and_response_schema(gateway: Gateway) -> None: - pytest.skip("BUG: POST /v1/responses takes a raw Request, so /openapi.json documents no body or response schema") openapi: dict[str, JsonValue] = gateway.get("/openapi.json") - _assert_responses_post_is_documented(openapi) + post: Final = _operation(openapi, "/v1/responses", "post") + body: Final = object_value(post["requestBody"]) + schema: Final = object_value(object_value(object_value(body["content"])["application/json"])["schema"]) + properties: Final = object_value(schema.get("properties")) + assert {"model", "input", "instructions", "tools", "previous_response_id", "background", "stream"} <= set( + properties + ), sorted(properties) + assert {"id", "object", "output", "usage"} <= set(_ok_schema_properties(openapi, post)), post["responses"] + assert "tool_calls" in object_value( + object_value(object_value(object_value(openapi["components"])["schemas"])["Message"])["properties"] + ) + + +def test_v1_responses_by_id_routes_declare_response_schemas(gateway: Gateway) -> None: + openapi: dict[str, JsonValue] = gateway.get("/openapi.json") + get: Final = _operation(openapi, "/v1/responses/{response_id}", "get") + assert {"id", "object", "output"} <= set(_ok_schema_properties(openapi, get)), get["responses"] + delete: Final = _operation(openapi, "/v1/responses/{response_id}", "delete") + assert {"id", "object", "deleted"} <= set(_ok_schema_properties(openapi, delete)), delete["responses"] + items: Final = _operation(openapi, "/v1/responses/{response_id}/input_items", "get") + assert {"data", "object", "has_more"} <= set(_ok_schema_properties(openapi, items)), items["responses"] diff --git a/tests/test_litellm/proxy/common_utils/test_custom_openapi_spec.py b/tests/test_litellm/proxy/common_utils/test_custom_openapi_spec.py index 5d3100bcc64..485dbcef2f6 100644 --- a/tests/test_litellm/proxy/common_utils/test_custom_openapi_spec.py +++ b/tests/test_litellm/proxy/common_utils/test_custom_openapi_spec.py @@ -153,7 +153,7 @@ def test_move_defs_to_components(): }, } - CustomOpenAPISpec._move_defs_to_components(openapi_schema=openapi_schema, defs=defs) + CustomOpenAPISpec._move_defs_to_components(openapi_schema=openapi_schema, defs=defs, namespace="Req") assert "components" in openapi_schema assert "schemas" in openapi_schema["components"] @@ -185,7 +185,7 @@ def test_rewrite_defs_refs(): }, } - rewritten = CustomOpenAPISpec._rewrite_defs_refs(schema=schema) + rewritten = CustomOpenAPISpec._rewrite_defs_refs(schema=schema, renames={}) assert "$defs" not in rewritten assert ( @@ -196,3 +196,197 @@ def test_rewrite_defs_refs(): rewritten["properties"]["messages"]["items"]["anyOf"][1]["$ref"] == "#/components/schemas/AssistantMessage" ) + + +def test_get_pydantic_schema_generates_schema_for_responses_request_typed_dict(): + from litellm.types.llms.openai import ResponsesAPIRequestParams + + schema = CustomOpenAPISpec.get_pydantic_schema(ResponsesAPIRequestParams) + + assert schema is not None + properties = schema["properties"] + assert isinstance(properties, dict) + for field in ( + "model", + "input", + "instructions", + "tools", + "previous_response_id", + "background", + "stream", + ): + assert field in properties + + +def test_responses_api_paths_covers_all_three_routes(): + assert CustomOpenAPISpec.RESPONSES_API_PATHS == [ + "/v1/responses", + "/responses", + "/openai/v1/responses", + ] + + +def test_add_schema_to_components_renames_colliding_def_instead_of_overwriting(): + openapi = { + "components": { + "schemas": { + "Message": {"type": "object", "properties": {"content": {"type": "string"}}}, + } + } + } + + CustomOpenAPISpec.add_schema_to_components( + openapi, + "Req", + { + "type": "object", + "properties": {"m": {"$ref": "#/$defs/Message"}, "n": {"$ref": "#/$defs/Other"}}, + "$defs": { + "Message": {"type": "object", "properties": {"role": {"type": "string"}}}, + "Other": {"type": "integer"}, + }, + }, + ) + + schemas = openapi["components"]["schemas"] + assert schemas["Message"] == {"type": "object", "properties": {"content": {"type": "string"}}} + assert schemas["Req_Message"] == {"type": "object", "properties": {"role": {"type": "string"}}} + assert schemas["Other"] == {"type": "integer"} + assert schemas["Req"]["properties"]["m"]["$ref"] == "#/components/schemas/Req_Message" + assert schemas["Req"]["properties"]["n"]["$ref"] == "#/components/schemas/Other" + assert "$defs" not in schemas["Req"] + + +def test_add_schema_to_components_keeps_name_for_identical_existing_def(): + openapi = {"components": {"schemas": {"Same": {"type": "integer"}}}} + + CustomOpenAPISpec.add_schema_to_components( + openapi, + "Req", + { + "type": "object", + "properties": {"s": {"$ref": "#/$defs/Same"}}, + "$defs": {"Same": {"type": "integer"}}, + }, + ) + + schemas = openapi["components"]["schemas"] + assert "Req_Same" not in schemas + assert schemas["Req"]["properties"]["s"]["$ref"] == "#/components/schemas/Same" + + +def test_responses_request_params_schema_requires_model_and_input(): + from litellm.types.llms.openai import ResponsesAPIRequestParams + + schema = CustomOpenAPISpec.get_pydantic_schema(ResponsesAPIRequestParams) + + assert schema is not None + assert set(schema["required"]) == {"model", "input"} + + +def test_move_defs_to_components_renames_defs_whose_refs_point_at_renamed_defs(): + openapi = { + "components": { + "schemas": { + "Inner": {"type": "string"}, + "Wrapper": {"$ref": "#/components/schemas/Inner"}, + } + } + } + + renames = CustomOpenAPISpec._move_defs_to_components( + openapi, + { + "Inner": {"type": "integer"}, + "Wrapper": {"$ref": "#/$defs/Inner"}, + }, + "NS", + ) + + schemas = openapi["components"]["schemas"] + assert renames == {"Inner": "NS_Inner", "Wrapper": "NS_Wrapper"} + assert schemas["Inner"] == {"type": "string"} + assert schemas["NS_Inner"] == {"type": "integer"} + assert schemas["Wrapper"] == {"$ref": "#/components/schemas/Inner"} + assert schemas["NS_Wrapper"] == {"$ref": "#/components/schemas/NS_Inner"} + + +def test_add_schema_to_components_keeps_name_for_same_shape_existing_def(): + openapi = { + "components": { + "schemas": { + "Block": { + "type": "object", + "properties": {"type": {"type": "string"}, "x": {"type": "string"}}, + "required": ["type", "x"], + "additionalProperties": True, + }, + } + } + } + + CustomOpenAPISpec.add_schema_to_components( + openapi, + "Req", + { + "type": "object", + "properties": {"b": {"$ref": "#/$defs/Block"}}, + "$defs": { + "Block": { + "type": "object", + "properties": {"type": {"type": "string"}, "x": {"type": "string"}}, + "required": ["type", "x"], + }, + }, + }, + ) + + schemas = openapi["components"]["schemas"] + assert "Req_Block" not in schemas + assert schemas["Block"]["additionalProperties"] is True + assert schemas["Req"]["properties"]["b"]["$ref"] == "#/components/schemas/Block" + + +def test_add_schema_to_components_renames_def_with_different_required_set(): + openapi = { + "components": { + "schemas": { + "Block": { + "type": "object", + "properties": { + "keys": {"type": "array"}, + "type": {"type": "string"}, + "x": {"type": "string"}, + "y": {"type": "string"}, + }, + "required": ["type", "x", "y"], + }, + } + } + } + + CustomOpenAPISpec.add_schema_to_components( + openapi, + "Req", + { + "type": "object", + "properties": {"b": {"$ref": "#/$defs/Block"}}, + "$defs": { + "Block": { + "type": "object", + "properties": { + "keys": {"type": "array"}, + "type": {"type": "string"}, + "x": {"type": "string"}, + "y": {"type": "string"}, + }, + "required": ["keys", "type", "x", "y"], + }, + }, + }, + ) + + schemas = openapi["components"]["schemas"] + assert schemas["Block"]["required"] == ["type", "x", "y"] + assert schemas["Req_Block"]["required"] == ["keys", "type", "x", "y"] + assert schemas["Req"]["properties"]["b"]["$ref"] == "#/components/schemas/Req_Block" diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 4153bf7d7ee..e684aa55b33 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -2427,3 +2427,38 @@ class TestResponsesInputTokens: assert response.status_code == 429, response.text assert response.json()["error"]["message"] == "rate limited" + + +def test_responses_routes_document_response_models_in_openapi_schema(): + from typing import cast + + from fastapi import FastAPI + + from litellm.proxy.response_api_endpoints.endpoints import router + + def as_object(value: object) -> dict[str, object]: + assert isinstance(value, dict) + return cast(dict[str, object], value) + + openapi_app = FastAPI() + openapi_app.include_router(router) + openapi: Final = cast(dict[str, object], openapi_app.openapi()) + + def ok_200_properties(path: str, method: str) -> dict[str, object]: + operation: Final = as_object(as_object(as_object(openapi)["paths"])[path])[method] + schema: Final = as_object( + as_object( + as_object(as_object(as_object(as_object(operation)["responses"])["200"])["content"])["application/json"] + )["schema"] + ) + ref: Final = schema["$ref"] + assert isinstance(ref, str) + component: Final = ref.rsplit("/", 1)[-1] + return as_object( + as_object(as_object(as_object(as_object(openapi)["components"])["schemas"])[component])["properties"] + ) + + assert "output" in ok_200_properties("/v1/responses", "post") + assert "output" in ok_200_properties("/v1/responses/{response_id}", "get") + assert "deleted" in ok_200_properties("/v1/responses/{response_id}", "delete") + assert "data" in ok_200_properties("/v1/responses/{response_id}/input_items", "get") diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index e52a17390a7..2938e65cfde 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -23700,6 +23700,270 @@ export interface components { /** Description */ description?: string | null; }; + /** + * AcknowledgedSafetyCheck + * @description A pending safety check for the computer call. + */ + AcknowledgedSafetyCheck: { + /** Code */ + code?: string | null; + /** Id */ + id: string; + /** Message */ + message?: string | null; + } & { + [key: string]: unknown; + }; + /** + * Action + * @description The shell commands and limits that describe how to run the tool call. + */ + Action: { + /** Commands */ + commands: string[]; + /** Max Output Length */ + max_output_length?: number | null; + /** Timeout Ms */ + timeout_ms?: number | null; + } & { + [key: string]: unknown; + }; + /** + * ActionClick + * @description A click action. + */ + ActionClick: { + /** + * Button + * @enum {string} + */ + button: "left" | "right" | "wheel" | "back" | "forward"; + /** Keys */ + keys?: string[] | null; + /** + * Type + * @constant + */ + type: "click"; + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; + /** + * ActionDoubleClick + * @description A double click action. + */ + ActionDoubleClick: { + /** Keys */ + keys?: string[] | null; + /** + * Type + * @constant + */ + type: "double_click"; + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; + /** + * ActionDrag + * @description A drag action. + */ + ActionDrag: { + /** Keys */ + keys?: string[] | null; + /** Path */ + path: components["schemas"]["ActionDragPath"][]; + /** + * Type + * @constant + */ + type: "drag"; + } & { + [key: string]: unknown; + }; + /** + * ActionDragPath + * @description An x/y coordinate pair, e.g. `{ x: 100, y: 200 }`. + */ + ActionDragPath: { + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; + /** + * ActionFind + * @description Action type "find_in_page": Searches for a pattern within a loaded page. + */ + ActionFind: { + /** Pattern */ + pattern: string; + /** + * Type + * @constant + */ + type: "find_in_page"; + /** Url */ + url: string; + } & { + [key: string]: unknown; + }; + /** + * ActionKeypress + * @description A collection of keypresses the model would like to perform. + */ + ActionKeypress: { + /** Keys */ + keys: string[]; + /** + * Type + * @constant + */ + type: "keypress"; + } & { + [key: string]: unknown; + }; + /** + * ActionMove + * @description A mouse move action. + */ + ActionMove: { + /** Keys */ + keys?: string[] | null; + /** + * Type + * @constant + */ + type: "move"; + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; + /** + * ActionOpenPage + * @description Action type "open_page" - Opens a specific URL from search results. + */ + ActionOpenPage: { + /** + * Type + * @constant + */ + type: "open_page"; + /** Url */ + url?: string | null; + } & { + [key: string]: unknown; + }; + /** + * ActionScreenshot + * @description A screenshot action. + */ + ActionScreenshot: { + /** + * Type + * @constant + */ + type: "screenshot"; + } & { + [key: string]: unknown; + }; + /** + * ActionScroll + * @description A scroll action. + */ + ActionScroll: { + /** Keys */ + keys?: string[] | null; + /** Scroll X */ + scroll_x: number; + /** Scroll Y */ + scroll_y: number; + /** + * Type + * @constant + */ + type: "scroll"; + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; + /** + * ActionSearch + * @description Action type "search" - Performs a web search query. + */ + ActionSearch: { + /** Queries */ + queries?: string[] | null; + /** Query */ + query: string; + /** Sources */ + sources?: components["schemas"]["ActionSearchSource"][] | null; + /** + * Type + * @constant + */ + type: "search"; + } & { + [key: string]: unknown; + }; + /** + * ActionSearchSource + * @description A source used in the search. + */ + ActionSearchSource: { + /** + * Type + * @constant + */ + type: "url"; + /** Url */ + url: string; + } & { + [key: string]: unknown; + }; + /** + * ActionType + * @description An action to type in text. + */ + ActionType: { + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "type"; + } & { + [key: string]: unknown; + }; + /** + * ActionWait + * @description A wait action. + */ + ActionWait: { + /** + * Type + * @constant + */ + type: "wait"; + } & { + [key: string]: unknown; + }; /** * ActiveUsersAnalyticsResponse * @description Response for active users analytics @@ -24070,6 +24334,86 @@ export interface components { /** Index Permissions */ index_permissions: ("read" | "write")[]; }; + /** + * AnnotationContainerFileCitation + * @description A citation for a container file used to generate a model response. + */ + AnnotationContainerFileCitation: { + /** Container Id */ + container_id: string; + /** End Index */ + end_index: number; + /** File Id */ + file_id: string; + /** Filename */ + filename: string; + /** Start Index */ + start_index: number; + /** + * Type + * @constant + */ + type: "container_file_citation"; + } & { + [key: string]: unknown; + }; + /** + * AnnotationFileCitation + * @description A citation to a file. + */ + AnnotationFileCitation: { + /** File Id */ + file_id: string; + /** Filename */ + filename: string; + /** Index */ + index: number; + /** + * Type + * @constant + */ + type: "file_citation"; + } & { + [key: string]: unknown; + }; + /** + * AnnotationFilePath + * @description A path to a file. + */ + AnnotationFilePath: { + /** File Id */ + file_id: string; + /** Index */ + index: number; + /** + * Type + * @constant + */ + type: "file_path"; + } & { + [key: string]: unknown; + }; + /** + * AnnotationURLCitation + * @description A citation for a web resource used to generate a model response. + */ + AnnotationURLCitation: { + /** End Index */ + end_index: number; + /** Start Index */ + start_index: number; + /** Title */ + title: string; + /** + * Type + * @constant + */ + type: "url_citation"; + /** Url */ + url: string; + } & { + [key: string]: unknown; + }; /** ApplyGuardrailRequest */ ApplyGuardrailRequest: { /** Entities */ @@ -24099,6 +24443,117 @@ export interface components { /** Response Text */ response_text: string; }; + /** + * ApplyPatchCall + * @description A tool call representing a request to create, delete, or update files using diff patches. + */ + ApplyPatchCall: { + /** Call Id */ + call_id: string; + /** Id */ + id?: string | null; + /** Operation */ + operation: components["schemas"]["ApplyPatchCallOperationCreateFile"] | components["schemas"]["ApplyPatchCallOperationDeleteFile"] | components["schemas"]["ApplyPatchCallOperationUpdateFile"]; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed"; + /** + * Type + * @constant + */ + type: "apply_patch_call"; + }; + /** + * ApplyPatchCallOperationCreateFile + * @description Instruction for creating a new file via the apply_patch tool. + */ + ApplyPatchCallOperationCreateFile: { + /** Diff */ + diff: string; + /** Path */ + path: string; + /** + * Type + * @constant + */ + type: "create_file"; + }; + /** + * ApplyPatchCallOperationDeleteFile + * @description Instruction for deleting an existing file via the apply_patch tool. + */ + ApplyPatchCallOperationDeleteFile: { + /** Path */ + path: string; + /** + * Type + * @constant + */ + type: "delete_file"; + }; + /** + * ApplyPatchCallOperationUpdateFile + * @description Instruction for updating an existing file via the apply_patch tool. + */ + ApplyPatchCallOperationUpdateFile: { + /** Diff */ + diff: string; + /** Path */ + path: string; + /** + * Type + * @constant + */ + type: "update_file"; + }; + /** + * ApplyPatchCallOutput + * @description The streamed output emitted by an apply patch tool call. + */ + ApplyPatchCallOutput: { + /** Call Id */ + call_id: string; + /** Id */ + id?: string | null; + /** Output */ + output?: string | null; + /** + * Status + * @enum {string} + */ + status: "completed" | "failed"; + /** + * Type + * @constant + */ + type: "apply_patch_call_output"; + }; + /** + * ApplyPatchTool + * @description Allows the assistant to create, delete, or update files using unified diffs. + */ + ApplyPatchTool: { + /** + * Type + * @constant + */ + type: "apply_patch"; + } & { + [key: string]: unknown; + }; + /** + * ApplyPatchToolParam + * @description Allows the assistant to create, delete, or update files using unified diffs. + */ + ApplyPatchToolParam: { + /** + * Type + * @constant + */ + type: "apply_patch"; + }; /** * AttachmentImpactResponse * @description Response for estimating the impact of a policy attachment. @@ -26039,6 +26494,15 @@ export interface components { */ uncached_input_tokens: number; }; + /** CachedTokensDetails */ + CachedTokensDetails: { + /** Audio Tokens */ + audio_tokens?: number | null; + /** Image Tokens */ + image_tokens?: number | null; + /** Text Tokens */ + text_tokens?: number | null; + }; /** * CallTypes * @enum {string} @@ -26298,6 +26762,8 @@ export interface components { * @constant */ type: "ephemeral"; + } & { + [key: string]: unknown; }; /** ChatCompletionCustomToolCallPayload */ ChatCompletionCustomToolCallPayload: { @@ -26428,6 +26894,8 @@ export interface components { * @constant */ type: "reasoning"; + } & { + [key: string]: unknown; }; /** ChatCompletionReasoningSummaryTextBlock */ ChatCompletionReasoningSummaryTextBlock: { @@ -26438,6 +26906,8 @@ export interface components { * @constant */ type: "summary_text"; + } & { + [key: string]: unknown; }; /** ChatCompletionRedactedThinkingBlock */ ChatCompletionRedactedThinkingBlock: { @@ -26452,6 +26922,8 @@ export interface components { * @constant */ type: "redacted_thinking"; + } & { + [key: string]: unknown; }; /** ChatCompletionSystemMessage */ ChatCompletionSystemMessage: { @@ -26492,6 +26964,8 @@ export interface components { * @constant */ type: "thinking"; + } & { + [key: string]: unknown; }; /** ChatCompletionTokenLogprob */ ChatCompletionTokenLogprob: { @@ -26801,6 +27275,30 @@ export interface components { */ max_images: number; }; + /** + * Click + * @description A click action. + */ + Click: { + /** + * Button + * @enum {string} + */ + button: "left" | "right" | "wheel" | "back" | "forward"; + /** Keys */ + keys?: string[] | null; + /** + * Type + * @constant + */ + type: "click"; + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; /** * CloudZeroExportRequest * @description Request model for CloudZero export operations @@ -26933,6 +27431,59 @@ export interface components { */ timezone?: string | null; }; + /** + * CodeInterpreter + * @description A tool that runs Python code to help generate a response to a prompt. + */ + CodeInterpreter: { + /** Container */ + container: string | components["schemas"]["CodeInterpreterContainerCodeInterpreterToolAuto"]; + /** + * Type + * @constant + */ + type: "code_interpreter"; + } & { + [key: string]: unknown; + }; + /** + * CodeInterpreterContainerCodeInterpreterToolAuto + * @description Configuration for a code interpreter container. + * + * Optionally specify the IDs of the files to run the code on. + */ + CodeInterpreterContainerCodeInterpreterToolAuto: { + /** File Ids */ + file_ids?: string[] | null; + /** Memory Limit */ + memory_limit?: ("1g" | "4g" | "16g" | "64g") | null; + /** Network Policy */ + network_policy?: components["schemas"]["ContainerNetworkPolicyDisabled"] | components["schemas"]["ContainerNetworkPolicyAllowlist"] | null; + /** + * Type + * @constant + */ + type: "auto"; + } & { + [key: string]: unknown; + }; + /** + * ComparisonFilter + * @description A filter used to compare a specified attribute key to a given value using a defined comparison operation. + */ + ComparisonFilter: { + /** Key */ + key: string; + /** + * Type + * @enum {string} + */ + type: "eq" | "ne" | "gt" | "gte" | "lt" | "lte" | "in" | "nin"; + /** Value */ + value: string | number | boolean | (string | number)[]; + } & { + [key: string]: unknown; + }; /** * ComplexityRouterConfigValidationRequest * @description A complexity-router config to validate without saving, so a form can surface the @@ -27038,6 +27589,114 @@ export interface components { /** Regulation */ regulation: string; }; + /** + * CompoundFilter + * @description Combine multiple filters using `and` or `or`. + */ + CompoundFilter: { + /** Filters */ + filters: (components["schemas"]["ComparisonFilter"] | unknown)[]; + /** + * Type + * @enum {string} + */ + type: "and" | "or"; + } & { + [key: string]: unknown; + }; + /** + * ComputerCallOutput + * @description The output of a computer tool call. + */ + ComputerCallOutput: { + /** Acknowledged Safety Checks */ + acknowledged_safety_checks?: components["schemas"]["ComputerCallOutputAcknowledgedSafetyCheck"][] | null; + /** Call Id */ + call_id: string; + /** Id */ + id?: string | null; + output: components["schemas"]["ResponseComputerToolCallOutputScreenshotParam"]; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "computer_call_output"; + }; + /** + * ComputerCallOutputAcknowledgedSafetyCheck + * @description A pending safety check for the computer call. + */ + ComputerCallOutputAcknowledgedSafetyCheck: { + /** Code */ + code?: string | null; + /** Id */ + id: string; + /** Message */ + message?: string | null; + }; + /** + * ComputerTool + * @description A tool that controls a virtual computer. + * + * Learn more about the [computer tool](https://platform.openai.com/docs/guides/tools-computer-use). + */ + ComputerTool: { + /** + * Type + * @constant + */ + type: "computer"; + } & { + [key: string]: unknown; + }; + /** + * ComputerUsePreviewTool + * @description A tool that controls a virtual computer. + * + * Learn more about the [computer tool](https://platform.openai.com/docs/guides/tools-computer-use). + */ + ComputerUsePreviewTool: { + /** Display Height */ + display_height: number; + /** Display Width */ + display_width: number; + /** + * Environment + * @enum {string} + */ + environment: "windows" | "mac" | "linux" | "ubuntu" | "browser"; + /** + * Type + * @constant + */ + type: "computer_use_preview"; + } & { + [key: string]: unknown; + }; + /** + * ComputerUsePreviewToolParam + * @description A tool that controls a virtual computer. + * + * Learn more about the [computer tool](https://platform.openai.com/docs/guides/tools-computer-use). + */ + ComputerUsePreviewToolParam: { + /** Display Height */ + display_height: number; + /** Display Width */ + display_width: number; + /** + * Environment + * @enum {string} + */ + environment: "windows" | "mac" | "linux" | "ubuntu" | "browser"; + /** + * Type + * @constant + */ + type: "computer_use_preview"; + }; /** ConfigFieldDelete */ ConfigFieldDelete: { /** @@ -27710,6 +28369,141 @@ export interface components { /** Api Base */ api_base: string; }; + /** ContainerAuto */ + ContainerAuto: { + /** File Ids */ + file_ids?: string[] | null; + /** Memory Limit */ + memory_limit?: ("1g" | "4g" | "16g" | "64g") | null; + /** Network Policy */ + network_policy?: components["schemas"]["ContainerNetworkPolicyDisabled"] | components["schemas"]["ContainerNetworkPolicyAllowlist"] | null; + /** Skills */ + skills?: (components["schemas"]["SkillReference"] | components["schemas"]["InlineSkill"])[] | null; + /** + * Type + * @constant + */ + type: "container_auto"; + } & { + [key: string]: unknown; + }; + /** ContainerAutoParam */ + ContainerAutoParam: { + /** File Ids */ + file_ids?: string[]; + /** Memory Limit */ + memory_limit?: ("1g" | "4g" | "16g" | "64g") | null; + /** Network Policy */ + network_policy?: components["schemas"]["ContainerNetworkPolicyDisabledParam"] | components["schemas"]["ContainerNetworkPolicyAllowlistParam"]; + /** Skills */ + skills?: (components["schemas"]["SkillReferenceParam"] | components["schemas"]["InlineSkillParam"])[]; + /** + * Type + * @constant + */ + type: "container_auto"; + }; + /** ContainerNetworkPolicyAllowlist */ + ContainerNetworkPolicyAllowlist: { + /** Allowed Domains */ + allowed_domains: string[]; + /** Domain Secrets */ + domain_secrets?: components["schemas"]["ContainerNetworkPolicyDomainSecret"][] | null; + /** + * Type + * @constant + */ + type: "allowlist"; + } & { + [key: string]: unknown; + }; + /** ContainerNetworkPolicyAllowlistParam */ + ContainerNetworkPolicyAllowlistParam: { + /** Allowed Domains */ + allowed_domains: string[]; + /** Domain Secrets */ + domain_secrets?: components["schemas"]["ContainerNetworkPolicyDomainSecretParam"][]; + /** + * Type + * @constant + */ + type: "allowlist"; + }; + /** ContainerNetworkPolicyDisabled */ + ContainerNetworkPolicyDisabled: { + /** + * Type + * @constant + */ + type: "disabled"; + } & { + [key: string]: unknown; + }; + /** ContainerNetworkPolicyDisabledParam */ + ContainerNetworkPolicyDisabledParam: { + /** + * Type + * @constant + */ + type: "disabled"; + }; + /** ContainerNetworkPolicyDomainSecret */ + ContainerNetworkPolicyDomainSecret: { + /** Domain */ + domain: string; + /** Name */ + name: string; + /** Value */ + value: string; + } & { + [key: string]: unknown; + }; + /** ContainerNetworkPolicyDomainSecretParam */ + ContainerNetworkPolicyDomainSecretParam: { + /** Domain */ + domain: string; + /** Name */ + name: string; + /** Value */ + value: string; + }; + /** ContainerReference */ + ContainerReference: { + /** Container Id */ + container_id: string; + /** + * Type + * @constant + */ + type: "container_reference"; + } & { + [key: string]: unknown; + }; + /** ContainerReferenceParam */ + ContainerReferenceParam: { + /** Container Id */ + container_id: string; + /** + * Type + * @constant + */ + type: "container_reference"; + }; + /** + * Content + * @description Reasoning text from the model. + */ + Content: { + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "reasoning_text"; + } & { + [key: string]: unknown; + }; /** * ContentFilterAction * @description Action to take when content filter detects a match @@ -27806,6 +28600,17 @@ export interface components { */ trigger_ratio: number; }; + /** + * ContextManagementEntry + * @description Context management configuration entry for a request. + * See https://developers.openai.com/api/docs/guides/compaction. + */ + ContextManagementEntry: { + /** Compact Threshold */ + compact_threshold?: number; + /** Type */ + type?: string; + }; /** * CoordinationRedisNode * @description A single startup node of a cluster-mode Redis used for proxy coordination. @@ -28261,6 +29066,77 @@ export interface components { /** Weight */ weight: number; }; + /** + * CustomTool + * @description A custom tool that processes input using a specified format. + * + * Learn more about [custom tools](https://platform.openai.com/docs/guides/function-calling#custom-tools) + */ + CustomTool: { + /** Defer Loading */ + defer_loading?: boolean | null; + /** Description */ + description?: string | null; + /** Format */ + format?: components["schemas"]["Text"] | components["schemas"]["Grammar"] | null; + /** Name */ + name: string; + /** + * Type + * @constant + */ + type: "custom"; + } & { + [key: string]: unknown; + }; + /** + * CustomToolCallOutputItem + * @description A custom/freeform tool call output item (e.g. apply_patch). + * + * Mirrors the ``custom_tool_call`` variant of OpenAI's Responses API output. + * Unlike ``OutputFunctionToolCall`` which uses ``arguments`` (JSON string), + * this uses ``input`` (raw string) for the tool payload. + */ + CustomToolCallOutputItem: { + /** Call Id */ + call_id: string; + /** Id */ + id?: string | null; + /** Input */ + input: string; + /** Name */ + name: string; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "custom_tool_call"; + } & { + [key: string]: unknown; + }; + /** + * CustomToolParam + * @description A custom tool that processes input using a specified format. + * + * Learn more about [custom tools](https://platform.openai.com/docs/guides/function-calling#custom-tools) + */ + CustomToolParam: { + /** Defer Loading */ + defer_loading?: boolean; + /** Description */ + description?: string; + /** Format */ + format?: components["schemas"]["Text"] | components["schemas"]["Grammar"]; + /** Name */ + name: string; + /** + * Type + * @constant + */ + type: "custom"; + }; /** * CustomerResponse * @description Customer object returned by the /customer read+write endpoints. @@ -28616,6 +29492,26 @@ export interface components { /** Project Ids */ project_ids: string[]; }; + /** + * DeleteResponseResult + * @description Result of a delete response request + * + * { + * "id": "resp_6786a1bec27481909a17d673315b29f6", + * "object": "response", + * "deleted": true + * } + */ + DeleteResponseResult: { + /** Deleted */ + deleted: boolean | null; + /** Id */ + id: string | null; + /** Object */ + object: string | null; + } & { + [key: string]: unknown; + }; /** * DeleteSkillResponse * @description Response from deleting a skill @@ -28713,6 +29609,54 @@ export interface components { */ type: "text"; }; + /** + * DoubleClick + * @description A double click action. + */ + DoubleClick: { + /** Keys */ + keys?: string[] | null; + /** + * Type + * @constant + */ + type: "double_click"; + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; + /** + * Drag + * @description A drag action. + */ + Drag: { + /** Keys */ + keys?: string[] | null; + /** Path */ + path: components["schemas"]["DragPath"][]; + /** + * Type + * @constant + */ + type: "drag"; + } & { + [key: string]: unknown; + }; + /** + * DragPath + * @description An x/y coordinate pair, e.g. `{ x: 100, y: 200 }`. + */ + DragPath: { + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; /** DynamoDBArgs */ DynamoDBArgs: { /** Assume Role Aws Role Name */ @@ -28767,6 +29711,30 @@ export interface components { /** Write Capacity Units */ write_capacity_units?: number | null; }; + /** + * EasyInputMessageParam + * @description A message input to the model with a role indicating instruction following + * hierarchy. Instructions given with the `developer` or `system` role take + * precedence over instructions given with the `user` role. Messages with the + * `assistant` role are presumed to have been generated by the model in previous + * interactions. + */ + EasyInputMessageParam: { + /** Content */ + content: string | (components["schemas"]["ResponseInputTextParam"] | components["schemas"]["ResponseInputImageParam"] | components["schemas"]["ResponseInputFileParam"])[]; + /** Phase */ + phase?: ("commentary" | "final_answer") | null; + /** + * Role + * @enum {string} + */ + role: "user" | "assistant" | "system" | "developer"; + /** + * Type + * @constant + */ + type?: "message"; + }; /** * EmailEvent * @enum {string} @@ -29079,6 +30047,58 @@ export interface components { /** Stored In Db */ stored_in_db: boolean | null; }; + /** + * FileSearchTool + * @description A tool that searches for relevant content from uploaded files. + * + * Learn more about the [file search tool](https://platform.openai.com/docs/guides/tools-file-search). + */ + FileSearchTool: { + /** Filters */ + filters?: components["schemas"]["ComparisonFilter"] | components["schemas"]["CompoundFilter"] | null; + /** Max Num Results */ + max_num_results?: number | null; + ranking_options?: components["schemas"]["RankingOptions"] | null; + /** + * Type + * @constant + */ + type: "file_search"; + /** Vector Store Ids */ + vector_store_ids: string[]; + } & { + [key: string]: unknown; + }; + /** + * FileSearchToolParam + * @description A tool that searches for relevant content from uploaded files. + * + * Learn more about the [file search tool](https://platform.openai.com/docs/guides/tools-file-search). + */ + FileSearchToolParam: { + /** Filters */ + filters?: components["schemas"]["ComparisonFilter"] | components["schemas"]["CompoundFilter"] | null; + /** Max Num Results */ + max_num_results?: number; + ranking_options?: components["schemas"]["RankingOptions"]; + /** + * Type + * @constant + */ + type: "file_search"; + /** Vector Store Ids */ + vector_store_ids: string[]; + }; + /** + * Filters + * @description Filters for the search. + */ + Filters: { + /** Allowed Domains */ + allowed_domains?: string[] | null; + } & { + [key: string]: unknown; + }; /** FunctionCall */ FunctionCall: { /** Arguments */ @@ -29088,6 +30108,105 @@ export interface components { } & { [key: string]: unknown; }; + /** + * FunctionCallOutput + * @description The output of a function tool call. + */ + FunctionCallOutput: { + /** Call Id */ + call_id: string; + /** Id */ + id?: string | null; + /** Output */ + output: string | (components["schemas"]["ResponseInputTextContentParam"] | components["schemas"]["ResponseInputImageContentParam"] | components["schemas"]["ResponseInputFileContentParam"])[]; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "function_call_output"; + }; + /** + * FunctionShellTool + * @description A tool that allows the model to execute shell commands. + */ + FunctionShellTool: { + /** Environment */ + environment?: components["schemas"]["ContainerAuto"] | components["schemas"]["LocalEnvironment"] | components["schemas"]["ContainerReference"] | null; + /** + * Type + * @constant + */ + type: "shell"; + } & { + [key: string]: unknown; + }; + /** + * FunctionShellToolParam + * @description A tool that allows the model to execute shell commands. + */ + FunctionShellToolParam: { + /** Environment */ + environment?: components["schemas"]["ContainerAutoParam"] | components["schemas"]["LocalEnvironmentParam"] | components["schemas"]["ContainerReferenceParam"] | null; + /** + * Type + * @constant + */ + type: "shell"; + }; + /** + * FunctionTool + * @description Defines a function in your own code the model can choose to call. + * + * Learn more about [function calling](https://platform.openai.com/docs/guides/function-calling). + */ + FunctionTool: { + /** Defer Loading */ + defer_loading?: boolean | null; + /** Description */ + description?: string | null; + /** Name */ + name: string; + /** Parameters */ + parameters?: { + [key: string]: unknown; + } | null; + /** Strict */ + strict?: boolean | null; + /** + * Type + * @constant + */ + type: "function"; + } & { + [key: string]: unknown; + }; + /** + * FunctionToolParam + * @description Defines a function in your own code the model can choose to call. + * + * Learn more about [function calling](https://platform.openai.com/docs/guides/function-calling). + */ + FunctionToolParam: { + /** Defer Loading */ + defer_loading?: boolean; + /** Description */ + description?: string | null; + /** Name */ + name: string; + /** Parameters */ + parameters: { + [key: string]: unknown; + } | null; + /** Strict */ + strict: boolean | null; + /** + * Type + * @constant + */ + type: "function"; + }; /** FuseHarnessPreset */ FuseHarnessPreset: { /** Id */ @@ -29537,6 +30656,44 @@ export interface components { /** Tools */ tools?: components["schemas"]["ChatCompletionToolParam"][]; }; + /** + * GenericResponseOutputItem + * @description Generic response API output item + */ + GenericResponseOutputItem: { + /** Content */ + content: components["schemas"]["OutputText"][]; + /** Id */ + id: string; + /** Phase */ + phase?: ("commentary" | "final_answer") | null; + /** Role */ + role: string; + /** Status */ + status: string; + /** Type */ + type: string; + } & { + [key: string]: unknown; + }; + /** + * GenericResponseOutputItemContentAnnotation + * @description Annotation for content in a message + */ + GenericResponseOutputItemContentAnnotation: { + /** End Index */ + end_index: number | null; + /** Start Index */ + start_index: number | null; + /** Title */ + title: string | null; + /** Type */ + type: string | null; + /** Url */ + url: string | null; + } & { + [key: string]: unknown; + }; /** * GetTeamMemberPermissionsResponse * @description Response to get the team member permissions for a team @@ -29561,6 +30718,26 @@ export interface components { /** Starttime */ startTime?: string | null; }; + /** + * Grammar + * @description A grammar defined by the user. + */ + Grammar: { + /** Definition */ + definition: string; + /** + * Syntax + * @enum {string} + */ + syntax: "lark" | "regex"; + /** + * Type + * @constant + */ + type: "grammar"; + } & { + [key: string]: unknown; + }; /** Guardrail */ Guardrail: { /** Created At */ @@ -29764,6 +30941,77 @@ export interface components { /** Ip */ ip: string; }; + /** + * ImageGeneration + * @description A tool that generates images using the GPT image models. + */ + ImageGeneration: { + /** Action */ + action?: ("generate" | "edit" | "auto") | null; + /** Background */ + background?: ("transparent" | "opaque" | "auto") | null; + /** Input Fidelity */ + input_fidelity?: ("high" | "low") | null; + input_image_mask?: components["schemas"]["ImageGenerationInputImageMask"] | null; + /** Model */ + model?: string | ("gpt-image-1" | "gpt-image-1-mini" | "gpt-image-1.5") | null; + /** Moderation */ + moderation?: ("auto" | "low") | null; + /** Output Compression */ + output_compression?: number | null; + /** Output Format */ + output_format?: ("png" | "webp" | "jpeg") | null; + /** Partial Images */ + partial_images?: number | null; + /** Quality */ + quality?: ("low" | "medium" | "high" | "auto") | null; + /** Size */ + size?: ("1024x1024" | "1024x1536" | "1536x1024" | "auto") | null; + /** + * Type + * @constant + */ + type: "image_generation"; + } & { + [key: string]: unknown; + }; + /** + * ImageGenerationCall + * @description An image generation request made by the model. + */ + ImageGenerationCall: { + /** Id */ + id: string; + /** Result */ + result?: string | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "generating" | "failed"; + /** + * Type + * @constant + */ + type: "image_generation_call"; + } & { + [key: string]: unknown; + }; + /** + * ImageGenerationInputImageMask + * @description Optional mask for inpainting. + * + * Contains `image_url` + * (string, optional) and `file_id` (string, optional). + */ + ImageGenerationInputImageMask: { + /** File Id */ + file_id?: string | null; + /** Image Url */ + image_url?: string | null; + } & { + [key: string]: unknown; + }; /** ImageURLListItem */ ImageURLListItem: { image_url: components["schemas"]["ImageURLObject"]; @@ -29786,6 +31034,16 @@ export interface components { } & { [key: string]: unknown; }; + /** + * IncompleteDetails + * @description Details about why the response is incomplete. + */ + IncompleteDetails: { + /** Reason */ + reason?: ("max_output_tokens" | "content_filter") | null; + } & { + [key: string]: unknown; + }; /** IndexCreateLiteLLMParams */ IndexCreateLiteLLMParams: { /** Vector Store Index */ @@ -29814,6 +31072,72 @@ export interface components { */ object: "list"; }; + /** InlineSkill */ + InlineSkill: { + /** Description */ + description: string; + /** Name */ + name: string; + source: components["schemas"]["InlineSkillSource"]; + /** + * Type + * @constant + */ + type: "inline"; + } & { + [key: string]: unknown; + }; + /** InlineSkillParam */ + InlineSkillParam: { + /** Description */ + description: string; + /** Name */ + name: string; + source: components["schemas"]["InlineSkillSourceParam"]; + /** + * Type + * @constant + */ + type: "inline"; + }; + /** + * InlineSkillSource + * @description Inline skill payload + */ + InlineSkillSource: { + /** Data */ + data: string; + /** + * Media Type + * @constant + */ + media_type: "application/zip"; + /** + * Type + * @constant + */ + type: "base64"; + } & { + [key: string]: unknown; + }; + /** + * InlineSkillSourceParam + * @description Inline skill payload + */ + InlineSkillSourceParam: { + /** Data */ + data: string; + /** + * Media Type + * @constant + */ + media_type: "application/zip"; + /** + * Type + * @constant + */ + type: "base64"; + }; /** InputAudio */ InputAudio: { /** Data */ @@ -29824,6 +31148,25 @@ export interface components { */ format: "wav" | "mp3"; }; + /** InputTokensDetails */ + InputTokensDetails: { + /** Audio Tokens */ + audio_tokens?: number | null; + /** + * Cached Tokens + * @default 0 + */ + cached_tokens: number; + cached_tokens_details?: components["schemas"]["CachedTokensDetails"] | null; + /** Image Tokens */ + image_tokens?: number | null; + /** Text Tokens */ + text_tokens?: number | null; + /** Video Tokens */ + video_tokens?: number | null; + } & { + [key: string]: unknown; + }; /** * InternalUserSettingsResponse * @description Response model for internal user settings @@ -29894,6 +31237,16 @@ export interface components { /** Is Accepted */ is_accepted: boolean; }; + /** + * ItemReference + * @description An internal identifier for an item to reference. + */ + ItemReference: { + /** Id */ + id: string; + /** Type */ + type?: "item_reference" | null; + }; /** JWTKeyMappingResponse */ JWTKeyMappingResponse: { /** @@ -30072,6 +31425,21 @@ export interface components { /** Tpm Limit Type */ tpm_limit_type?: ("guaranteed_throughput" | "best_effort_throughput" | "dynamic") | null; }; + /** + * Keypress + * @description A collection of keypresses the model would like to perform. + */ + Keypress: { + /** Keys */ + keys: string[]; + /** + * Type + * @constant + */ + type: "keypress"; + } & { + [key: string]: unknown; + }; /** * KeywordTierRule * @description A deterministic override: if any keyword matches, route to this tier. @@ -33208,6 +34576,128 @@ export interface components { * @enum {string} */ LitellmUserRoles: "proxy_admin" | "proxy_admin_viewer" | "org_admin" | "internal_user" | "internal_user_viewer" | "team" | "customer"; + /** LocalEnvironment */ + LocalEnvironment: { + /** Skills */ + skills?: components["schemas"]["LocalSkill"][] | null; + /** + * Type + * @constant + */ + type: "local"; + } & { + [key: string]: unknown; + }; + /** LocalEnvironmentParam */ + LocalEnvironmentParam: { + /** Skills */ + skills?: components["schemas"]["LocalSkillParam"][]; + /** + * Type + * @constant + */ + type: "local"; + }; + /** + * LocalShell + * @description A tool that allows the model to execute shell commands in a local environment. + */ + LocalShell: { + /** + * Type + * @constant + */ + type: "local_shell"; + } & { + [key: string]: unknown; + }; + /** + * LocalShellCall + * @description A tool call to run a command on the local shell. + */ + LocalShellCall: { + action: components["schemas"]["LocalShellCallAction"]; + /** Call Id */ + call_id: string; + /** Id */ + id: string; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "local_shell_call"; + } & { + [key: string]: unknown; + }; + /** + * LocalShellCallAction + * @description Execute a shell command on the server. + */ + LocalShellCallAction: { + /** Command */ + command: string[]; + /** Env */ + env: { + [key: string]: string; + }; + /** Timeout Ms */ + timeout_ms?: number | null; + /** + * Type + * @constant + */ + type: "exec"; + /** User */ + user?: string | null; + /** Working Directory */ + working_directory?: string | null; + } & { + [key: string]: unknown; + }; + /** + * LocalShellCallOutput + * @description The output of a local shell tool call. + */ + LocalShellCallOutput: { + /** Id */ + id: string; + /** Output */ + output: string; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "local_shell_call_output"; + } & { + [key: string]: unknown; + }; + /** LocalSkill */ + LocalSkill: { + /** Description */ + description: string; + /** Name */ + name: string; + /** Path */ + path: string; + } & { + [key: string]: unknown; + }; + /** LocalSkillParam */ + LocalSkillParam: { + /** Description */ + description: string; + /** Name */ + name: string; + /** Path */ + path: string; + }; /** LoggingCallbackStatus */ LoggingCallbackStatus: { /** Callbacks */ @@ -33220,6 +34710,36 @@ export interface components { */ status?: "healthy" | "unhealthy"; }; + /** + * Logprob + * @description The log probability of a token. + */ + Logprob: { + /** Bytes */ + bytes: number[]; + /** Logprob */ + logprob: number; + /** Token */ + token: string; + /** Top Logprobs */ + top_logprobs: components["schemas"]["LogprobTopLogprob"][]; + } & { + [key: string]: unknown; + }; + /** + * LogprobTopLogprob + * @description The top log probability of a token. + */ + LogprobTopLogprob: { + /** Bytes */ + bytes: number[]; + /** Logprob */ + logprob: number; + /** Token */ + token: string; + } & { + [key: string]: unknown; + }; /** * MCPAllowedClient * @description One entry of `general_settings.mcp_allowed_clients`. @@ -33726,6 +35246,198 @@ export interface components { /** Mcp Server Ids */ mcp_server_ids: string[]; }; + /** + * Mcp + * @description Give the model access to additional tools via remote Model Context Protocol + * (MCP) servers. [Learn more about MCP](https://platform.openai.com/docs/guides/tools-remote-mcp). + */ + Mcp: { + /** Allowed Tools */ + allowed_tools?: string[] | components["schemas"]["McpAllowedToolsMcpToolFilter"] | null; + /** Authorization */ + authorization?: string | null; + /** Connector Id */ + connector_id?: ("connector_dropbox" | "connector_gmail" | "connector_googlecalendar" | "connector_googledrive" | "connector_microsoftteams" | "connector_outlookcalendar" | "connector_outlookemail" | "connector_sharepoint") | null; + /** Defer Loading */ + defer_loading?: boolean | null; + /** Headers */ + headers?: { + [key: string]: string; + } | null; + /** Require Approval */ + require_approval?: components["schemas"]["McpRequireApprovalMcpToolApprovalFilter"] | ("always" | "never") | null; + /** Server Description */ + server_description?: string | null; + /** Server Label */ + server_label: string; + /** Server Url */ + server_url?: string | null; + /** + * Type + * @constant + */ + type: "mcp"; + } & { + [key: string]: unknown; + }; + /** + * McpAllowedToolsMcpToolFilter + * @description A filter object to specify which tools are allowed. + */ + McpAllowedToolsMcpToolFilter: { + /** Read Only */ + read_only?: boolean | null; + /** Tool Names */ + tool_names?: string[] | null; + } & { + [key: string]: unknown; + }; + /** + * McpApprovalRequest + * @description A request for human approval of a tool invocation. + */ + McpApprovalRequest: { + /** Arguments */ + arguments: string; + /** Id */ + id: string; + /** Name */ + name: string; + /** Server Label */ + server_label: string; + /** + * Type + * @constant + */ + type: "mcp_approval_request"; + } & { + [key: string]: unknown; + }; + /** + * McpApprovalResponse + * @description A response to an MCP approval request. + */ + McpApprovalResponse: { + /** Approval Request Id */ + approval_request_id: string; + /** Approve */ + approve: boolean; + /** Id */ + id: string; + /** Reason */ + reason?: string | null; + /** + * Type + * @constant + */ + type: "mcp_approval_response"; + } & { + [key: string]: unknown; + }; + /** + * McpCall + * @description An invocation of a tool on an MCP server. + */ + McpCall: { + /** Approval Request Id */ + approval_request_id?: string | null; + /** Arguments */ + arguments: string; + /** Error */ + error?: string | null; + /** Id */ + id: string; + /** Name */ + name: string; + /** Output */ + output?: string | null; + /** Server Label */ + server_label: string; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete" | "calling" | "failed") | null; + /** + * Type + * @constant + */ + type: "mcp_call"; + } & { + [key: string]: unknown; + }; + /** + * McpListTools + * @description A list of tools available on an MCP server. + */ + McpListTools: { + /** Error */ + error?: string | null; + /** Id */ + id: string; + /** Server Label */ + server_label: string; + /** Tools */ + tools: components["schemas"]["McpListToolsTool"][]; + /** + * Type + * @constant + */ + type: "mcp_list_tools"; + } & { + [key: string]: unknown; + }; + /** + * McpListToolsTool + * @description A tool available on an MCP server. + */ + McpListToolsTool: { + /** Annotations */ + annotations?: unknown | null; + /** Description */ + description?: string | null; + /** Input Schema */ + input_schema: unknown; + /** Name */ + name: string; + } & { + [key: string]: unknown; + }; + /** + * McpRequireApprovalMcpToolApprovalFilter + * @description Specify which of the MCP server's tools require approval. + * + * Can be + * `always`, `never`, or a filter object associated with tools + * that require approval. + */ + McpRequireApprovalMcpToolApprovalFilter: { + always?: components["schemas"]["McpRequireApprovalMcpToolApprovalFilterAlways"] | null; + never?: components["schemas"]["McpRequireApprovalMcpToolApprovalFilterNever"] | null; + } & { + [key: string]: unknown; + }; + /** + * McpRequireApprovalMcpToolApprovalFilterAlways + * @description A filter object to specify which tools are allowed. + */ + McpRequireApprovalMcpToolApprovalFilterAlways: { + /** Read Only */ + read_only?: boolean | null; + /** Tool Names */ + tool_names?: string[] | null; + } & { + [key: string]: unknown; + }; + /** + * McpRequireApprovalMcpToolApprovalFilterNever + * @description A filter object to specify which tools are allowed. + */ + McpRequireApprovalMcpToolApprovalFilterNever: { + /** Read Only */ + read_only?: boolean | null; + /** Tool Names */ + tool_names?: string[] | null; + } & { + [key: string]: unknown; + }; /** Member */ Member: { /** @@ -34053,6 +35765,25 @@ export interface components { } & { [key: string]: unknown; }; + /** + * Move + * @description A mouse move action. + */ + Move: { + /** Keys */ + keys?: string[] | null; + /** + * Type + * @constant + */ + type: "move"; + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; /** * MutualTLSSecurityScheme * @description Defines a security scheme using mTLS authentication. @@ -34066,6 +35797,42 @@ export interface components { */ type: "mutualTLS"; }; + /** + * NamespaceTool + * @description Groups function/custom tools under a shared namespace. + */ + NamespaceTool: { + /** Description */ + description: string; + /** Name */ + name: string; + /** Tools */ + tools: (components["schemas"]["ToolFunction"] | components["schemas"]["CustomTool"])[]; + /** + * Type + * @constant + */ + type: "namespace"; + } & { + [key: string]: unknown; + }; + /** + * NamespaceToolParam + * @description Groups function/custom tools under a shared namespace. + */ + NamespaceToolParam: { + /** Description */ + description: string; + /** Name */ + name: string; + /** Tools */ + tools: (components["schemas"]["ToolFunction"] | components["schemas"]["CustomToolParam"])[]; + /** + * Type + * @constant + */ + type: "namespace"; + }; /** * NewCustomerRequest * @description Create a new customer, allocate a budget to them @@ -35039,6 +36806,55 @@ export interface components { */ type: "openIdConnect"; }; + /** + * OperationCreateFile + * @description Instruction describing how to create a file via the apply_patch tool. + */ + OperationCreateFile: { + /** Diff */ + diff: string; + /** Path */ + path: string; + /** + * Type + * @constant + */ + type: "create_file"; + } & { + [key: string]: unknown; + }; + /** + * OperationDeleteFile + * @description Instruction describing how to delete a file via the apply_patch tool. + */ + OperationDeleteFile: { + /** Path */ + path: string; + /** + * Type + * @constant + */ + type: "delete_file"; + } & { + [key: string]: unknown; + }; + /** + * OperationUpdateFile + * @description Instruction describing how to update a file via the apply_patch tool. + */ + OperationUpdateFile: { + /** Diff */ + diff: string; + /** Path */ + path: string; + /** + * Type + * @constant + */ + type: "update_file"; + } & { + [key: string]: unknown; + }; /** OrgMember */ OrgMember: { /** @@ -35136,6 +36952,217 @@ export interface components { /** Tpm Limit */ tpm_limit?: number | null; }; + /** + * OutcomeExit + * @description Indicates that the shell commands finished and returned an exit code. + */ + OutcomeExit: { + /** Exit Code */ + exit_code: number; + /** + * Type + * @constant + */ + type: "exit"; + }; + /** + * OutcomeTimeout + * @description Indicates that the shell call exceeded its configured time limit. + */ + OutcomeTimeout: { + /** + * Type + * @constant + */ + type: "timeout"; + }; + /** + * Output + * @description The content of a shell tool call output that was emitted. + */ + Output: { + /** Created By */ + created_by?: string | null; + /** Outcome */ + outcome: components["schemas"]["OutputOutcomeTimeout"] | components["schemas"]["OutputOutcomeExit"]; + /** Stderr */ + stderr: string; + /** Stdout */ + stdout: string; + } & { + [key: string]: unknown; + }; + /** + * OutputCodeInterpreterCall + * @description A code interpreter / code execution call output + */ + OutputCodeInterpreterCall: { + /** Code */ + code: string | null; + /** Container Id */ + container_id: string | null; + /** Id */ + id: string; + /** Outputs */ + outputs: components["schemas"]["OutputCodeInterpreterCallLog"][] | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete" | "failed"; + /** + * Type + * @constant + */ + type: "code_interpreter_call"; + } & { + [key: string]: unknown; + }; + /** + * OutputCodeInterpreterCallLog + * @description Log output from a code interpreter call + */ + OutputCodeInterpreterCallLog: { + /** Logs */ + logs: string; + /** + * Type + * @constant + */ + type: "logs"; + } & { + [key: string]: unknown; + }; + /** + * OutputFunctionToolCall + * @description A tool call to run a function + */ + OutputFunctionToolCall: { + /** Arguments */ + arguments: string | null; + /** Call Id */ + call_id: string | null; + /** Id */ + id: string | null; + /** Name */ + name: string | null; + /** Phase */ + phase?: ("commentary" | "final_answer") | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** Type */ + type: string | null; + } & { + [key: string]: unknown; + }; + /** + * OutputImage + * @description The image output from the code interpreter. + */ + OutputImage: { + /** + * Type + * @constant + */ + type: "image"; + /** Url */ + url: string; + } & { + [key: string]: unknown; + }; + /** + * OutputImageGenerationCall + * @description An image generation call output + */ + OutputImageGenerationCall: { + /** Id */ + id: string; + /** Result */ + result: string | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete" | "failed"; + /** + * Type + * @constant + */ + type: "image_generation_call"; + } & { + [key: string]: unknown; + }; + /** + * OutputLogs + * @description The logs output from the code interpreter. + */ + OutputLogs: { + /** Logs */ + logs: string; + /** + * Type + * @constant + */ + type: "logs"; + } & { + [key: string]: unknown; + }; + /** + * OutputOutcomeExit + * @description Indicates that the shell commands finished and returned an exit code. + */ + OutputOutcomeExit: { + /** Exit Code */ + exit_code: number; + /** + * Type + * @constant + */ + type: "exit"; + } & { + [key: string]: unknown; + }; + /** + * OutputOutcomeTimeout + * @description Indicates that the shell call exceeded its configured time limit. + */ + OutputOutcomeTimeout: { + /** + * Type + * @constant + */ + type: "timeout"; + } & { + [key: string]: unknown; + }; + /** + * OutputText + * @description Text output content from an assistant message + */ + OutputText: { + /** Annotations */ + annotations: components["schemas"]["GenericResponseOutputItemContentAnnotation"][] | null; + /** Text */ + text: string | null; + /** Type */ + type: string | null; + } & { + [key: string]: unknown; + }; + /** OutputTokensDetails */ + OutputTokensDetails: { + /** Audio Tokens */ + audio_tokens?: number | null; + /** Reasoning Tokens */ + reasoning_tokens?: number | null; + /** Text Tokens */ + text_tokens?: number | null; + } & { + [key: string]: unknown; + }; /** * PageLinks * @description Hypermedia for a paginated list. No `first`/`last`: without a total count the last page is unknown. @@ -35437,6 +37464,20 @@ export interface components { /** Tpm Limit */ tpm_limit?: number | null; }; + /** + * PendingSafetyCheck + * @description A pending safety check for the computer call. + */ + PendingSafetyCheck: { + /** Code */ + code?: string | null; + /** Id */ + id: string; + /** Message */ + message?: string | null; + } & { + [key: string]: unknown; + }; /** * PerTestingCriteriaResult * @description Results for a specific testing criteria @@ -36319,6 +38360,19 @@ export interface components { prompt_id: string; prompt_info?: components["schemas"]["PromptInfo"] | null; }; + /** PromptCacheOptions */ + PromptCacheOptions: { + /** + * Mode + * @enum {string} + */ + mode?: "implicit" | "explicit"; + /** + * Ttl + * @constant + */ + ttl?: "30m"; + }; /** PromptCachingRequest */ PromptCachingRequest: { /** Cache Creation Tokens */ @@ -36412,6 +38466,20 @@ export interface components { } & { [key: string]: unknown; }; + /** + * PromptObject + * @description Reference to a stored prompt template. + */ + PromptObject: { + /** Id */ + id: string; + /** Variables */ + variables?: { + [key: string]: unknown; + } | null; + /** Version */ + version?: string | null; + }; /** PromptSpec */ PromptSpec: { /** Created At */ @@ -36702,6 +38770,31 @@ export interface components { }; } | null; }; + /** + * RankingOptions + * @description Ranking options for search. + */ + RankingOptions: { + hybrid_search?: components["schemas"]["RankingOptionsHybridSearch"] | null; + /** Ranker */ + ranker?: ("auto" | "default-2024-11-15") | null; + /** Score Threshold */ + score_threshold?: number | null; + } & { + [key: string]: unknown; + }; + /** + * RankingOptionsHybridSearch + * @description Weights that control how reciprocal rank fusion balances semantic embedding matches versus sparse keyword matches when hybrid search is enabled. + */ + RankingOptionsHybridSearch: { + /** Embedding Weight */ + embedding_weight: number; + /** Text Weight */ + text_weight: number; + } & { + [key: string]: unknown; + }; /** RawRequestTypedDict */ RawRequestTypedDict: { /** Error */ @@ -36750,6 +38843,21 @@ export interface components { } & { [key: string]: unknown; }; + /** + * Reasoning + * @description **gpt-5 and o-series models only** + * + * Configuration options for + * [reasoning models](https://platform.openai.com/docs/guides/reasoning). + */ + Reasoning: { + /** Effort */ + effort?: ("none" | "minimal" | "low" | "medium" | "high" | "xhigh") | null; + /** Generate Summary */ + generate_summary?: ("auto" | "concise" | "detailed") | null; + /** Summary */ + summary?: ("auto" | "concise" | "detailed") | null; + }; /** RegenerateKeyRequest */ RegenerateKeyRequest: { /** Access Group Ids */ @@ -37420,10 +39528,1489 @@ export interface components { /** Reset To */ reset_to: number; }; + /** ResponseAPIUsage */ + ResponseAPIUsage: { + /** Cost */ + cost?: number | null; + /** Input Tokens */ + input_tokens: number; + input_tokens_details?: components["schemas"]["InputTokensDetails"] | null; + /** Output Tokens */ + output_tokens: number; + output_tokens_details?: components["schemas"]["OutputTokensDetails"] | null; + /** Total Tokens */ + total_tokens: number; + } & { + [key: string]: unknown; + }; + /** + * ResponseApplyPatchToolCall + * @description A tool call that applies file diffs by creating, deleting, or updating files. + */ + ResponseApplyPatchToolCall: { + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Id */ + id: string; + /** Operation */ + operation: components["schemas"]["OperationCreateFile"] | components["schemas"]["OperationDeleteFile"] | components["schemas"]["OperationUpdateFile"]; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed"; + /** + * Type + * @constant + */ + type: "apply_patch_call"; + } & { + [key: string]: unknown; + }; + /** + * ResponseApplyPatchToolCallOutput + * @description The output emitted by an apply patch tool call. + */ + ResponseApplyPatchToolCallOutput: { + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Id */ + id: string; + /** Output */ + output?: string | null; + /** + * Status + * @enum {string} + */ + status: "completed" | "failed"; + /** + * Type + * @constant + */ + type: "apply_patch_call_output"; + } & { + [key: string]: unknown; + }; + /** + * ResponseCodeInterpreterToolCall + * @description A tool call to run code. + */ + ResponseCodeInterpreterToolCall: { + /** Code */ + code?: string | null; + /** Container Id */ + container_id: string; + /** Id */ + id: string; + /** Outputs */ + outputs?: (components["schemas"]["OutputLogs"] | components["schemas"]["OutputImage"])[] | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete" | "interpreting" | "failed"; + /** + * Type + * @constant + */ + type: "code_interpreter_call"; + } & { + [key: string]: unknown; + }; + /** + * ResponseCodeInterpreterToolCallParam + * @description A tool call to run code. + */ + ResponseCodeInterpreterToolCallParam: { + /** Code */ + code: string | null; + /** Container Id */ + container_id: string; + /** Id */ + id: string; + /** Outputs */ + outputs: (components["schemas"]["OutputLogs"] | components["schemas"]["OutputImage"])[] | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete" | "interpreting" | "failed"; + /** + * Type + * @constant + */ + type: "code_interpreter_call"; + }; + /** + * ResponseCompactionItem + * @description A compaction item generated by the [`v1/responses/compact` API](https://platform.openai.com/docs/api-reference/responses/compact). + */ + ResponseCompactionItem: { + /** Created By */ + created_by?: string | null; + /** Encrypted Content */ + encrypted_content: string; + /** Id */ + id: string; + /** + * Type + * @constant + */ + type: "compaction"; + } & { + [key: string]: unknown; + }; + /** + * ResponseCompactionItemParamParam + * @description A compaction item generated by the [`v1/responses/compact` API](https://platform.openai.com/docs/api-reference/responses/compact). + */ + ResponseCompactionItemParamParam: { + /** Encrypted Content */ + encrypted_content: string; + /** Id */ + id?: string | null; + /** + * Type + * @constant + */ + type: "compaction"; + }; + /** + * ResponseComputerToolCall + * @description A tool call to a computer use tool. + * + * See the + * [computer use guide](https://platform.openai.com/docs/guides/tools-computer-use) for more information. + */ + ResponseComputerToolCall: { + /** Action */ + action?: components["schemas"]["ActionClick"] | components["schemas"]["ActionDoubleClick"] | components["schemas"]["ActionDrag"] | components["schemas"]["ActionKeypress"] | components["schemas"]["ActionMove"] | components["schemas"]["ActionScreenshot"] | components["schemas"]["ActionScroll"] | components["schemas"]["ActionType"] | components["schemas"]["ActionWait"] | null; + /** Actions */ + actions?: (components["schemas"]["Click"] | components["schemas"]["DoubleClick"] | components["schemas"]["Drag"] | components["schemas"]["Keypress"] | components["schemas"]["Move"] | components["schemas"]["Screenshot"] | components["schemas"]["Scroll"] | components["schemas"]["Type"] | components["schemas"]["Wait"])[] | null; + /** Call Id */ + call_id: string; + /** Id */ + id: string; + /** Pending Safety Checks */ + pending_safety_checks: components["schemas"]["PendingSafetyCheck"][]; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "computer_call"; + } & { + [key: string]: unknown; + }; + /** ResponseComputerToolCallOutputItem */ + ResponseComputerToolCallOutputItem: { + /** Acknowledged Safety Checks */ + acknowledged_safety_checks?: components["schemas"]["AcknowledgedSafetyCheck"][] | null; + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Id */ + id: string; + output: components["schemas"]["ResponseComputerToolCallOutputScreenshot"]; + /** + * Status + * @enum {string} + */ + status: "completed" | "incomplete" | "failed" | "in_progress"; + /** + * Type + * @constant + */ + type: "computer_call_output"; + } & { + [key: string]: unknown; + }; + /** + * ResponseComputerToolCallOutputScreenshot + * @description A computer screenshot image used with the computer use tool. + */ + ResponseComputerToolCallOutputScreenshot: { + /** File Id */ + file_id?: string | null; + /** Image Url */ + image_url?: string | null; + /** + * Type + * @constant + */ + type: "computer_screenshot"; + } & { + [key: string]: unknown; + }; + /** + * ResponseComputerToolCallOutputScreenshotParam + * @description A computer screenshot image used with the computer use tool. + */ + ResponseComputerToolCallOutputScreenshotParam: { + /** File Id */ + file_id?: string; + /** Image Url */ + image_url?: string; + /** + * Type + * @constant + */ + type: "computer_screenshot"; + }; + /** + * ResponseComputerToolCallParam + * @description A tool call to a computer use tool. + * + * See the + * [computer use guide](https://platform.openai.com/docs/guides/tools-computer-use) for more information. + */ + ResponseComputerToolCallParam: { + /** Action */ + action?: components["schemas"]["ActionClick"] | components["schemas"]["ResponsesAPIRequestParams_ActionDoubleClick"] | components["schemas"]["ActionDrag"] | components["schemas"]["ActionKeypress"] | components["schemas"]["ActionMove"] | components["schemas"]["ActionScreenshot"] | components["schemas"]["ActionScroll"] | components["schemas"]["ActionType"] | components["schemas"]["ActionWait"]; + /** Actions */ + actions?: (components["schemas"]["Click"] | components["schemas"]["ResponsesAPIRequestParams_DoubleClick"] | components["schemas"]["Drag"] | components["schemas"]["Keypress"] | components["schemas"]["Move"] | components["schemas"]["Screenshot"] | components["schemas"]["Scroll"] | components["schemas"]["Type"] | components["schemas"]["Wait"])[]; + /** Call Id */ + call_id: string; + /** Id */ + id: string; + /** Pending Safety Checks */ + pending_safety_checks: components["schemas"]["PendingSafetyCheck"][]; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "computer_call"; + }; + /** + * ResponseContainerReference + * @description Represents a container created with /v1/containers. + */ + ResponseContainerReference: { + /** Container Id */ + container_id: string; + /** + * Type + * @constant + */ + type: "container_reference"; + } & { + [key: string]: unknown; + }; + /** + * ResponseCustomToolCall + * @description A call to a custom tool created by the model. + */ + ResponseCustomToolCall: { + /** Call Id */ + call_id: string; + /** Id */ + id?: string | null; + /** Input */ + input: string; + /** Name */ + name: string; + /** Namespace */ + namespace?: string | null; + /** + * Type + * @constant + */ + type: "custom_tool_call"; + } & { + [key: string]: unknown; + }; + /** + * ResponseCustomToolCallItem + * @description A call to a custom tool created by the model. + */ + ResponseCustomToolCallItem: { + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Id */ + id: string; + /** Input */ + input: string; + /** Name */ + name: string; + /** Namespace */ + namespace?: string | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "custom_tool_call"; + } & { + [key: string]: unknown; + }; + /** + * ResponseCustomToolCallOutputItem + * @description The output of a custom tool call from your code, being sent back to the model. + */ + ResponseCustomToolCallOutputItem: { + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Id */ + id: string; + /** Output */ + output: string | (components["schemas"]["ResponseInputText"] | components["schemas"]["ResponseInputImage"] | components["schemas"]["ResponseInputFile"])[]; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "custom_tool_call_output"; + } & { + [key: string]: unknown; + }; + /** + * ResponseCustomToolCallOutputParam + * @description The output of a custom tool call from your code, being sent back to the model. + */ + ResponseCustomToolCallOutputParam: { + /** Call Id */ + call_id: string; + /** Id */ + id?: string; + /** Output */ + output: string | (components["schemas"]["ResponseInputTextParam"] | components["schemas"]["ResponseInputImageParam"] | components["schemas"]["ResponseInputFileParam"])[]; + /** + * Type + * @constant + */ + type: "custom_tool_call_output"; + }; + /** + * ResponseCustomToolCallParam + * @description A call to a custom tool created by the model. + */ + ResponseCustomToolCallParam: { + /** Call Id */ + call_id: string; + /** Id */ + id?: string; + /** Input */ + input: string; + /** Name */ + name: string; + /** Namespace */ + namespace?: string; + /** + * Type + * @constant + */ + type: "custom_tool_call"; + }; + /** + * ResponseFileSearchToolCall + * @description The results of a file search tool call. + * + * See the + * [file search guide](https://platform.openai.com/docs/guides/tools-file-search) for more information. + */ + ResponseFileSearchToolCall: { + /** Id */ + id: string; + /** Queries */ + queries: string[]; + /** Results */ + results?: components["schemas"]["Result"][] | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "searching" | "completed" | "incomplete" | "failed"; + /** + * Type + * @constant + */ + type: "file_search_call"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFileSearchToolCallParam + * @description The results of a file search tool call. + * + * See the + * [file search guide](https://platform.openai.com/docs/guides/tools-file-search) for more information. + */ + ResponseFileSearchToolCallParam: { + /** Id */ + id: string; + /** Queries */ + queries: string[]; + /** Results */ + results?: components["schemas"]["Result"][] | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "searching" | "completed" | "incomplete" | "failed"; + /** + * Type + * @constant + */ + type: "file_search_call"; + }; + /** + * ResponseFormatJSONObject + * @description JSON object response format. + * + * An older method of generating JSON responses. + * Using `json_schema` is recommended for models that support it. Note that the + * model will not generate JSON without a system or user message instructing it + * to do so. + */ + ResponseFormatJSONObject: { + /** + * Type + * @constant + */ + type: "json_object"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFormatText + * @description Default response format. Used to generate text responses. + */ + ResponseFormatText: { + /** + * Type + * @constant + */ + type: "text"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFormatTextJSONSchemaConfigParam + * @description JSON Schema response format. + * + * Used to generate structured JSON responses. + * Learn more about [Structured Outputs](https://platform.openai.com/docs/guides/structured-outputs). + */ + ResponseFormatTextJSONSchemaConfigParam: { + /** Description */ + description?: string; + /** Name */ + name: string; + /** Schema */ + schema: { + [key: string]: unknown; + }; + /** Strict */ + strict?: boolean | null; + /** + * Type + * @constant + */ + type: "json_schema"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFunctionShellCallOutputContentParam + * @description Captured stdout and stderr for a portion of a shell tool call output. + */ + ResponseFunctionShellCallOutputContentParam: { + /** Outcome */ + outcome: components["schemas"]["OutcomeTimeout"] | components["schemas"]["OutcomeExit"]; + /** Stderr */ + stderr: string; + /** Stdout */ + stdout: string; + }; + /** + * ResponseFunctionShellToolCall + * @description A tool call that executes one or more shell commands in a managed environment. + */ + ResponseFunctionShellToolCall: { + action: components["schemas"]["Action"]; + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Environment */ + environment?: components["schemas"]["ResponseLocalEnvironment"] | components["schemas"]["ResponseContainerReference"] | null; + /** Id */ + id: string; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "shell_call"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFunctionShellToolCallOutput + * @description The output of a shell tool call that was emitted. + */ + ResponseFunctionShellToolCallOutput: { + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Id */ + id: string; + /** Max Output Length */ + max_output_length?: number | null; + /** Output */ + output: components["schemas"]["Output"][]; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "shell_call_output"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFunctionToolCall + * @description A tool call to run a function. + * + * See the + * [function calling guide](https://platform.openai.com/docs/guides/function-calling) for more information. + */ + ResponseFunctionToolCall: { + /** Arguments */ + arguments: string; + /** Call Id */ + call_id: string; + /** Id */ + id?: string | null; + /** Name */ + name: string; + /** Namespace */ + namespace?: string | null; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "function_call"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFunctionToolCallItem + * @description A tool call to run a function. + * + * See the + * [function calling guide](https://platform.openai.com/docs/guides/function-calling) for more information. + */ + ResponseFunctionToolCallItem: { + /** Arguments */ + arguments: string; + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Id */ + id: string; + /** Name */ + name: string; + /** Namespace */ + namespace?: string | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "function_call"; + } & { + [key: string]: unknown; + }; + /** ResponseFunctionToolCallOutputItem */ + ResponseFunctionToolCallOutputItem: { + /** Call Id */ + call_id: string; + /** Created By */ + created_by?: string | null; + /** Id */ + id: string; + /** Output */ + output: string | (components["schemas"]["ResponseInputText"] | components["schemas"]["ResponseInputImage"] | components["schemas"]["ResponseInputFile"])[]; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "function_call_output"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFunctionToolCallParam + * @description A tool call to run a function. + * + * See the + * [function calling guide](https://platform.openai.com/docs/guides/function-calling) for more information. + */ + ResponseFunctionToolCallParam: { + /** Arguments */ + arguments: string; + /** Call Id */ + call_id: string; + /** Id */ + id?: string; + /** Name */ + name: string; + /** Namespace */ + namespace?: string; + /** + * Status + * @enum {string} + */ + status?: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "function_call"; + }; + /** + * ResponseFunctionWebSearch + * @description The results of a web search tool call. + * + * See the + * [web search guide](https://platform.openai.com/docs/guides/tools-web-search) for more information. + */ + ResponseFunctionWebSearch: { + /** Action */ + action: components["schemas"]["ActionSearch"] | components["schemas"]["ActionOpenPage"] | components["schemas"]["ActionFind"]; + /** Id */ + id: string; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "searching" | "completed" | "failed"; + /** + * Type + * @constant + */ + type: "web_search_call"; + } & { + [key: string]: unknown; + }; + /** + * ResponseFunctionWebSearchParam + * @description The results of a web search tool call. + * + * See the + * [web search guide](https://platform.openai.com/docs/guides/tools-web-search) for more information. + */ + ResponseFunctionWebSearchParam: { + /** Action */ + action: components["schemas"]["ActionSearch"] | components["schemas"]["ActionOpenPage"] | components["schemas"]["ActionFind"]; + /** Id */ + id: string; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "searching" | "completed" | "failed"; + /** + * Type + * @constant + */ + type: "web_search_call"; + }; + /** + * ResponseInputFile + * @description A file input to the model. + */ + ResponseInputFile: { + /** Detail */ + detail?: ("high" | "low") | null; + /** File Data */ + file_data?: string | null; + /** File Id */ + file_id?: string | null; + /** File Url */ + file_url?: string | null; + /** Filename */ + filename?: string | null; + /** + * Type + * @constant + */ + type: "input_file"; + } & { + [key: string]: unknown; + }; + /** + * ResponseInputFileContentParam + * @description A file input to the model. + */ + ResponseInputFileContentParam: { + /** + * Detail + * @enum {string} + */ + detail?: "low" | "high"; + /** File Data */ + file_data?: string | null; + /** File Id */ + file_id?: string | null; + /** File Url */ + file_url?: string | null; + /** Filename */ + filename?: string | null; + /** + * Type + * @constant + */ + type: "input_file"; + }; + /** + * ResponseInputFileParam + * @description A file input to the model. + */ + ResponseInputFileParam: { + /** + * Detail + * @enum {string} + */ + detail?: "low" | "high"; + /** File Data */ + file_data?: string; + /** File Id */ + file_id?: string | null; + /** File Url */ + file_url?: string; + /** Filename */ + filename?: string; + /** + * Type + * @constant + */ + type: "input_file"; + }; + /** + * ResponseInputImage + * @description An image input to the model. + * + * Learn about [image inputs](https://platform.openai.com/docs/guides/vision). + */ + ResponseInputImage: { + /** + * Detail + * @enum {string} + */ + detail: "low" | "high" | "auto" | "original"; + /** File Id */ + file_id?: string | null; + /** Image Url */ + image_url?: string | null; + /** + * Type + * @constant + */ + type: "input_image"; + } & { + [key: string]: unknown; + }; + /** + * ResponseInputImageContentParam + * @description An image input to the model. + * + * Learn about [image inputs](https://platform.openai.com/docs/guides/vision) + */ + ResponseInputImageContentParam: { + /** Detail */ + detail?: ("low" | "high" | "auto" | "original") | null; + /** File Id */ + file_id?: string | null; + /** Image Url */ + image_url?: string | null; + /** + * Type + * @constant + */ + type: "input_image"; + }; + /** + * ResponseInputImageParam + * @description An image input to the model. + * + * Learn about [image inputs](https://platform.openai.com/docs/guides/vision). + */ + ResponseInputImageParam: { + /** + * Detail + * @enum {string} + */ + detail: "low" | "high" | "auto" | "original"; + /** File Id */ + file_id?: string | null; + /** Image Url */ + image_url?: string | null; + /** + * Type + * @constant + */ + type: "input_image"; + }; + /** ResponseInputMessageItem */ + ResponseInputMessageItem: { + /** Content */ + content: (components["schemas"]["ResponseInputText"] | components["schemas"]["ResponseInputImage"] | components["schemas"]["ResponseInputFile"])[]; + /** Id */ + id: string; + /** + * Role + * @enum {string} + */ + role: "user" | "system" | "developer"; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "message"; + } & { + [key: string]: unknown; + }; + /** + * ResponseInputText + * @description A text input to the model. + */ + ResponseInputText: { + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "input_text"; + } & { + [key: string]: unknown; + }; + /** + * ResponseInputTextContentParam + * @description A text input to the model. + */ + ResponseInputTextContentParam: { + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "input_text"; + }; + /** + * ResponseInputTextParam + * @description A text input to the model. + */ + ResponseInputTextParam: { + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "input_text"; + }; + /** + * ResponseItemList + * @description A list of Response items. + */ + ResponseItemList: { + /** Data */ + data: (components["schemas"]["ResponseInputMessageItem"] | components["schemas"]["ResponseOutputMessage"] | components["schemas"]["ResponseFileSearchToolCall"] | components["schemas"]["ResponseComputerToolCall"] | components["schemas"]["ResponseComputerToolCallOutputItem"] | components["schemas"]["ResponseFunctionWebSearch"] | components["schemas"]["ResponseFunctionToolCallItem"] | components["schemas"]["ResponseFunctionToolCallOutputItem"] | components["schemas"]["ResponseToolSearchCall"] | components["schemas"]["ResponseToolSearchOutputItem"] | components["schemas"]["ResponseReasoningItem"] | components["schemas"]["ResponseCompactionItem"] | components["schemas"]["ImageGenerationCall"] | components["schemas"]["ResponseCodeInterpreterToolCall"] | components["schemas"]["LocalShellCall"] | components["schemas"]["LocalShellCallOutput"] | components["schemas"]["ResponseFunctionShellToolCall"] | components["schemas"]["ResponseFunctionShellToolCallOutput"] | components["schemas"]["ResponseApplyPatchToolCall"] | components["schemas"]["ResponseApplyPatchToolCallOutput"] | components["schemas"]["McpListTools"] | components["schemas"]["McpApprovalRequest"] | components["schemas"]["McpApprovalResponse"] | components["schemas"]["McpCall"] | components["schemas"]["ResponseCustomToolCallItem"] | components["schemas"]["ResponseCustomToolCallOutputItem"])[]; + /** First Id */ + first_id: string; + /** Has More */ + has_more: boolean; + /** Last Id */ + last_id: string; + /** + * Object + * @constant + */ + object: "list"; + } & { + [key: string]: unknown; + }; /** ResponseLiteLLM_ManagedVectorStore */ ResponseLiteLLM_ManagedVectorStore: { vector_store?: components["schemas"]["LiteLLM_ManagedVectorStoresTable"]; }; + /** + * ResponseLocalEnvironment + * @description Represents the use of a local environment to perform shell actions. + */ + ResponseLocalEnvironment: { + /** + * Type + * @constant + */ + type: "local"; + } & { + [key: string]: unknown; + }; + /** + * ResponseOutputMessage + * @description An output message from the model. + */ + ResponseOutputMessage: { + /** Content */ + content: (components["schemas"]["ResponseOutputText"] | components["schemas"]["ResponseOutputRefusal"])[]; + /** Id */ + id: string; + /** Phase */ + phase?: ("commentary" | "final_answer") | null; + /** + * Role + * @constant + */ + role: "assistant"; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "message"; + } & { + [key: string]: unknown; + }; + /** + * ResponseOutputMessageParam + * @description An output message from the model. + */ + ResponseOutputMessageParam: { + /** Content */ + content: (components["schemas"]["ResponseOutputTextParam"] | components["schemas"]["ResponseOutputRefusalParam"])[]; + /** Id */ + id: string; + /** Phase */ + phase?: ("commentary" | "final_answer") | null; + /** + * Role + * @constant + */ + role: "assistant"; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "message"; + }; + /** + * ResponseOutputRefusal + * @description A refusal from the model. + */ + ResponseOutputRefusal: { + /** Refusal */ + refusal: string; + /** + * Type + * @constant + */ + type: "refusal"; + } & { + [key: string]: unknown; + }; + /** + * ResponseOutputRefusalParam + * @description A refusal from the model. + */ + ResponseOutputRefusalParam: { + /** Refusal */ + refusal: string; + /** + * Type + * @constant + */ + type: "refusal"; + }; + /** + * ResponseOutputText + * @description A text output from the model. + */ + ResponseOutputText: { + /** Annotations */ + annotations: (components["schemas"]["AnnotationFileCitation"] | components["schemas"]["AnnotationURLCitation"] | components["schemas"]["AnnotationContainerFileCitation"] | components["schemas"]["AnnotationFilePath"])[]; + /** Logprobs */ + logprobs?: components["schemas"]["Logprob"][] | null; + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "output_text"; + } & { + [key: string]: unknown; + }; + /** + * ResponseOutputTextParam + * @description A text output from the model. + */ + ResponseOutputTextParam: { + /** Annotations */ + annotations: (components["schemas"]["AnnotationFileCitation"] | components["schemas"]["AnnotationURLCitation"] | components["schemas"]["AnnotationContainerFileCitation"] | components["schemas"]["AnnotationFilePath"])[]; + /** Logprobs */ + logprobs?: components["schemas"]["Logprob"][]; + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "output_text"; + }; + /** + * ResponseReasoningItem + * @description A description of the chain of thought used by a reasoning model while generating + * a response. Be sure to include these items in your `input` to the Responses API + * for subsequent turns of a conversation if you are manually + * [managing context](https://platform.openai.com/docs/guides/conversation-state). + */ + ResponseReasoningItem: { + /** Content */ + content?: components["schemas"]["Content"][] | null; + /** Encrypted Content */ + encrypted_content?: string | null; + /** Id */ + id: string; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** Summary */ + summary: components["schemas"]["Summary"][]; + /** + * Type + * @constant + */ + type: "reasoning"; + } & { + [key: string]: unknown; + }; + /** + * ResponseReasoningItemParam + * @description A description of the chain of thought used by a reasoning model while generating + * a response. Be sure to include these items in your `input` to the Responses API + * for subsequent turns of a conversation if you are manually + * [managing context](https://platform.openai.com/docs/guides/conversation-state). + */ + ResponseReasoningItemParam: { + /** Content */ + content?: components["schemas"]["Content"][]; + /** Encrypted Content */ + encrypted_content?: string | null; + /** Id */ + id: string; + /** + * Status + * @enum {string} + */ + status?: "in_progress" | "completed" | "incomplete"; + /** Summary */ + summary: components["schemas"]["Summary"][]; + /** + * Type + * @constant + */ + type: "reasoning"; + }; + /** + * ResponseTextConfigParam + * @description Configuration options for a text response from the model. + * + * Can be plain + * text or structured JSON data. Learn more: + * - [Text inputs and outputs](https://platform.openai.com/docs/guides/text) + * - [Structured Outputs](https://platform.openai.com/docs/guides/structured-outputs) + */ + ResponseTextConfigParam: { + /** Format */ + format?: components["schemas"]["ResponseFormatText"] | components["schemas"]["ResponseFormatTextJSONSchemaConfigParam"] | components["schemas"]["ResponseFormatJSONObject"]; + /** Verbosity */ + verbosity?: ("low" | "medium" | "high") | null; + } & { + [key: string]: unknown; + }; + /** ResponseToolSearchCall */ + ResponseToolSearchCall: { + /** Arguments */ + arguments: unknown; + /** Call Id */ + call_id?: string | null; + /** Created By */ + created_by?: string | null; + /** + * Execution + * @enum {string} + */ + execution: "server" | "client"; + /** Id */ + id: string; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type: "tool_search_call"; + } & { + [key: string]: unknown; + }; + /** ResponseToolSearchOutputItem */ + ResponseToolSearchOutputItem: { + /** Call Id */ + call_id?: string | null; + /** Created By */ + created_by?: string | null; + /** + * Execution + * @enum {string} + */ + execution: "server" | "client"; + /** Id */ + id: string; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "incomplete"; + /** Tools */ + tools: (components["schemas"]["FunctionTool"] | components["schemas"]["FileSearchTool"] | components["schemas"]["ComputerTool"] | components["schemas"]["ComputerUsePreviewTool"] | components["schemas"]["WebSearchTool"] | components["schemas"]["Mcp"] | components["schemas"]["CodeInterpreter"] | components["schemas"]["ImageGeneration"] | components["schemas"]["LocalShell"] | components["schemas"]["FunctionShellTool"] | components["schemas"]["CustomTool"] | components["schemas"]["NamespaceTool"] | components["schemas"]["ToolSearchTool"] | components["schemas"]["WebSearchPreviewTool"] | components["schemas"]["ApplyPatchTool"])[]; + /** + * Type + * @constant + */ + type: "tool_search_output"; + } & { + [key: string]: unknown; + }; + /** ResponseToolSearchOutputItemParamParam */ + ResponseToolSearchOutputItemParamParam: { + /** Call Id */ + call_id?: string | null; + /** + * Execution + * @enum {string} + */ + execution?: "server" | "client"; + /** Id */ + id?: string | null; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** Tools */ + tools: (components["schemas"]["FunctionToolParam"] | components["schemas"]["FileSearchToolParam"] | components["schemas"]["openai__types__responses__computer_tool_param__ComputerToolParam"] | components["schemas"]["ComputerUsePreviewToolParam"] | components["schemas"]["WebSearchToolParam"] | components["schemas"]["Mcp"] | components["schemas"]["CodeInterpreter"] | components["schemas"]["ImageGeneration"] | components["schemas"]["LocalShell"] | components["schemas"]["FunctionShellToolParam"] | components["schemas"]["CustomToolParam"] | components["schemas"]["NamespaceToolParam"] | components["schemas"]["ToolSearchToolParam"] | components["schemas"]["WebSearchPreviewToolParam"] | components["schemas"]["ApplyPatchToolParam"])[]; + /** + * Type + * @constant + */ + type: "tool_search_output"; + }; + /** + * ResponsesAPIRequestParams + * @description TypedDict for request parameters supported by the responses API. + */ + ResponsesAPIRequestParams: { + /** Background */ + background?: boolean | null; + /** Context Management */ + context_management?: components["schemas"]["ContextManagementEntry"][] | null; + /** Include */ + include?: ("file_search_call.results" | "web_search_call.results" | "web_search_call.action.sources" | "message.input_image.image_url" | "computer_call_output.output.image_url" | "code_interpreter_call.outputs" | "reasoning.encrypted_content" | "message.output_text.logprobs")[] | null; + /** Input */ + input: string | (components["schemas"]["EasyInputMessageParam"] | components["schemas"]["ResponsesAPIRequestParams_Message"] | components["schemas"]["ResponseOutputMessageParam"] | components["schemas"]["ResponseFileSearchToolCallParam"] | components["schemas"]["ResponseComputerToolCallParam"] | components["schemas"]["ComputerCallOutput"] | components["schemas"]["ResponseFunctionWebSearchParam"] | components["schemas"]["ResponseFunctionToolCallParam"] | components["schemas"]["FunctionCallOutput"] | components["schemas"]["ToolSearchCall"] | components["schemas"]["ResponseToolSearchOutputItemParamParam"] | components["schemas"]["ResponseReasoningItemParam"] | components["schemas"]["ResponseCompactionItemParamParam"] | components["schemas"]["ResponsesAPIRequestParams_ImageGenerationCall"] | components["schemas"]["ResponseCodeInterpreterToolCallParam"] | components["schemas"]["LocalShellCall"] | components["schemas"]["LocalShellCallOutput"] | components["schemas"]["ShellCall"] | components["schemas"]["ShellCallOutput"] | components["schemas"]["ApplyPatchCall"] | components["schemas"]["ApplyPatchCallOutput"] | components["schemas"]["McpListTools"] | components["schemas"]["McpApprovalRequest"] | components["schemas"]["ResponsesAPIRequestParams_McpApprovalResponse"] | components["schemas"]["McpCall"] | components["schemas"]["ResponseCustomToolCallOutputParam"] | components["schemas"]["ResponseCustomToolCallParam"] | components["schemas"]["ItemReference"])[]; + /** Instructions */ + instructions?: string | null; + /** Max Output Tokens */ + max_output_tokens?: number | null; + /** Max Tool Calls */ + max_tool_calls?: number | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model */ + model: string; + /** Parallel Tool Calls */ + parallel_tool_calls?: boolean | null; + /** Partial Images */ + partial_images?: number | null; + /** Previous Response Id */ + previous_response_id?: string | null; + prompt?: components["schemas"]["PromptObject"] | null; + /** Prompt Cache Key */ + prompt_cache_key?: string | null; + prompt_cache_options?: components["schemas"]["PromptCacheOptions"] | null; + /** Prompt Cache Retention */ + prompt_cache_retention?: string | null; + reasoning?: components["schemas"]["Reasoning"] | null; + /** Safety Identifier */ + safety_identifier?: string | null; + /** Service Tier */ + service_tier?: string | null; + /** Store */ + store?: boolean | null; + /** Stream */ + stream?: boolean | null; + stream_options?: components["schemas"]["ResponsesAPIStreamOptions"] | null; + /** Temperature */ + temperature?: number | null; + text?: components["schemas"]["ResponseTextConfigParam"] | null; + /** Tool Choice */ + tool_choice?: ("none" | "auto" | "required") | components["schemas"]["ToolChoiceAllowedParam"] | components["schemas"]["ToolChoiceTypesParam"] | components["schemas"]["ToolChoiceFunctionParam"] | components["schemas"]["ToolChoiceMcpParam"] | components["schemas"]["ToolChoiceCustomParam"] | components["schemas"]["ToolChoiceApplyPatchParam"] | components["schemas"]["ToolChoiceShellParam"] | null; + /** Tools */ + tools?: (components["schemas"]["FunctionToolParam"] | components["schemas"]["FileSearchToolParam"] | components["schemas"]["openai__types__responses__computer_tool_param__ComputerToolParam"] | components["schemas"]["ComputerUsePreviewToolParam"] | components["schemas"]["WebSearchToolParam"] | components["schemas"]["Mcp"] | components["schemas"]["CodeInterpreter"] | components["schemas"]["ImageGeneration"] | components["schemas"]["LocalShell"] | components["schemas"]["FunctionShellToolParam"] | components["schemas"]["CustomToolParam"] | components["schemas"]["NamespaceToolParam"] | components["schemas"]["ToolSearchToolParam"] | components["schemas"]["WebSearchPreviewToolParam"] | components["schemas"]["ApplyPatchToolParam"] | components["schemas"]["litellm__types__llms__openai__ComputerToolParam"] | components["schemas"]["ShellToolParam"])[] | null; + /** Top Logprobs */ + top_logprobs?: number | null; + /** Top P */ + top_p?: number | null; + /** Truncation */ + truncation?: ("auto" | "disabled") | null; + /** User */ + user?: string | null; + }; + /** + * ActionDoubleClick + * @description A double click action. + */ + ResponsesAPIRequestParams_ActionDoubleClick: { + /** Keys */ + keys: string[] | null; + /** + * Type + * @constant + */ + type: "double_click"; + /** X */ + x: number; + /** Y */ + y: number; + }; + /** + * DoubleClick + * @description A double click action. + */ + ResponsesAPIRequestParams_DoubleClick: { + /** Keys */ + keys: string[] | null; + /** + * Type + * @constant + */ + type: "double_click"; + /** X */ + x: number; + /** Y */ + y: number; + }; + /** + * ImageGenerationCall + * @description An image generation request made by the model. + */ + ResponsesAPIRequestParams_ImageGenerationCall: { + /** Id */ + id: string; + /** Result */ + result: string | null; + /** + * Status + * @enum {string} + */ + status: "in_progress" | "completed" | "generating" | "failed"; + /** + * Type + * @constant + */ + type: "image_generation_call"; + }; + /** + * McpApprovalResponse + * @description A response to an MCP approval request. + */ + ResponsesAPIRequestParams_McpApprovalResponse: { + /** Approval Request Id */ + approval_request_id: string; + /** Approve */ + approve: boolean; + /** Id */ + id?: string | null; + /** Reason */ + reason?: string | null; + /** + * Type + * @constant + */ + type: "mcp_approval_response"; + }; + /** + * Message + * @description A message input to the model with a role indicating instruction following + * hierarchy. Instructions given with the `developer` or `system` role take + * precedence over instructions given with the `user` role. + */ + ResponsesAPIRequestParams_Message: { + /** Content */ + content: (components["schemas"]["ResponseInputTextParam"] | components["schemas"]["ResponseInputImageParam"] | components["schemas"]["ResponseInputFileParam"])[]; + /** + * Role + * @enum {string} + */ + role: "user" | "system" | "developer"; + /** + * Status + * @enum {string} + */ + status?: "in_progress" | "completed" | "incomplete"; + /** + * Type + * @constant + */ + type?: "message"; + }; + /** ResponsesAPIResponse */ + ResponsesAPIResponse: { + /** Created At */ + created_at: number; + /** Error */ + error?: { + [key: string]: unknown; + } | null; + /** Id */ + id: string; + incomplete_details?: components["schemas"]["IncompleteDetails"] | null; + /** Instructions */ + instructions?: string | null; + /** Max Output Tokens */ + max_output_tokens?: number | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model */ + model?: string | null; + /** Object */ + object?: string | null; + /** Output */ + output: (components["schemas"]["ResponseOutputMessage"] | components["schemas"]["ResponseFileSearchToolCall"] | components["schemas"]["ResponseFunctionToolCall"] | components["schemas"]["ResponseFunctionToolCallOutputItem"] | components["schemas"]["ResponseFunctionWebSearch"] | components["schemas"]["ResponseComputerToolCall"] | components["schemas"]["ResponseComputerToolCallOutputItem"] | components["schemas"]["ResponseReasoningItem"] | components["schemas"]["ResponseToolSearchCall"] | components["schemas"]["ResponseToolSearchOutputItem"] | components["schemas"]["ResponseCompactionItem"] | components["schemas"]["ImageGenerationCall"] | components["schemas"]["ResponseCodeInterpreterToolCall"] | components["schemas"]["LocalShellCall"] | components["schemas"]["LocalShellCallOutput"] | components["schemas"]["ResponseFunctionShellToolCall"] | components["schemas"]["ResponseFunctionShellToolCallOutput"] | components["schemas"]["ResponseApplyPatchToolCall"] | components["schemas"]["ResponseApplyPatchToolCallOutput"] | components["schemas"]["McpCall"] | components["schemas"]["McpListTools"] | components["schemas"]["McpApprovalRequest"] | components["schemas"]["McpApprovalResponse"] | components["schemas"]["ResponseCustomToolCall"] | components["schemas"]["ResponseCustomToolCallOutputItem"] | { + [key: string]: unknown; + })[] | (components["schemas"]["GenericResponseOutputItem"] | components["schemas"]["OutputCodeInterpreterCall"] | components["schemas"]["OutputFunctionToolCall"] | components["schemas"]["OutputImageGenerationCall"] | components["schemas"]["ResponseFunctionToolCall"] | components["schemas"]["ResponseFunctionWebSearch"] | components["schemas"]["CustomToolCallOutputItem"])[]; + /** Parallel Tool Calls */ + parallel_tool_calls?: boolean | null; + /** Previous Response Id */ + previous_response_id?: string | null; + /** Reasoning */ + reasoning?: { + [key: string]: unknown; + } | null; + /** Status */ + status?: string | null; + /** Store */ + store?: boolean | null; + /** Temperature */ + temperature?: number | null; + /** Text */ + text?: components["schemas"]["ResponseTextConfigParam"] | { + [key: string]: unknown; + } | null; + /** Tool Choice */ + tool_choice?: ("none" | "auto" | "required") | components["schemas"]["ToolChoiceAllowedParam"] | components["schemas"]["ToolChoiceTypesParam"] | components["schemas"]["ToolChoiceFunctionParam"] | components["schemas"]["ToolChoiceMcpParam"] | components["schemas"]["ToolChoiceCustomParam"] | components["schemas"]["ToolChoiceApplyPatchParam"] | components["schemas"]["ToolChoiceShellParam"] | null; + /** Tools */ + tools?: (components["schemas"]["FunctionTool"] | components["schemas"]["FileSearchTool"] | components["schemas"]["ComputerTool"] | components["schemas"]["ComputerUsePreviewTool"] | components["schemas"]["WebSearchTool"] | components["schemas"]["Mcp"] | components["schemas"]["CodeInterpreter"] | components["schemas"]["ImageGeneration"] | components["schemas"]["LocalShell"] | components["schemas"]["FunctionShellTool"] | components["schemas"]["CustomTool"] | components["schemas"]["NamespaceTool"] | components["schemas"]["ToolSearchTool"] | components["schemas"]["WebSearchPreviewTool"] | components["schemas"]["ApplyPatchTool"])[] | components["schemas"]["ResponseFunctionToolCall"][] | { + [key: string]: unknown; + }[] | null; + /** Top P */ + top_p?: number | null; + /** Truncation */ + truncation?: ("auto" | "disabled") | null; + usage?: components["schemas"]["ResponseAPIUsage"] | null; + /** User */ + user?: string | null; + } & { + [key: string]: unknown; + }; + /** ResponsesAPIStreamOptions */ + ResponsesAPIStreamOptions: { + /** Include Obfuscation */ + include_obfuscation?: boolean; + }; + /** Result */ + Result: { + /** Attributes */ + attributes?: { + [key: string]: string | number | boolean; + } | null; + /** File Id */ + file_id?: string | null; + /** Filename */ + filename?: string | null; + /** Score */ + score?: number | null; + /** Text */ + text?: string | null; + } & { + [key: string]: unknown; + }; /** * ResultCounts * @description Result counts for a run @@ -38086,6 +41673,42 @@ export interface components { */ window_seconds: number; }; + /** + * Screenshot + * @description A screenshot action. + */ + Screenshot: { + /** + * Type + * @constant + */ + type: "screenshot"; + } & { + [key: string]: unknown; + }; + /** + * Scroll + * @description A scroll action. + */ + Scroll: { + /** Keys */ + keys?: string[] | null; + /** Scroll X */ + scroll_x: number; + /** Scroll Y */ + scroll_y: number; + /** + * Type + * @constant + */ + type: "scroll"; + /** X */ + x: number; + /** Y */ + y: number; + } & { + [key: string]: unknown; + }; /** * SearchTool * @description Search tool configuration. @@ -38408,6 +42031,72 @@ export interface components { /** Turn Count */ turn_count: number; }; + /** + * ShellCall + * @description A tool representing a request to execute one or more shell commands. + */ + ShellCall: { + action: components["schemas"]["ShellCallAction"]; + /** Call Id */ + call_id: string; + /** Environment */ + environment?: components["schemas"]["LocalEnvironmentParam"] | components["schemas"]["ContainerReferenceParam"] | null; + /** Id */ + id?: string | null; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "shell_call"; + }; + /** + * ShellCallAction + * @description The shell commands and limits that describe how to run the tool call. + */ + ShellCallAction: { + /** Commands */ + commands: string[]; + /** Max Output Length */ + max_output_length?: number | null; + /** Timeout Ms */ + timeout_ms?: number | null; + }; + /** + * ShellCallOutput + * @description The streamed output items emitted by a shell tool call. + */ + ShellCallOutput: { + /** Call Id */ + call_id: string; + /** Id */ + id?: string | null; + /** Max Output Length */ + max_output_length?: number | null; + /** Output */ + output: components["schemas"]["ResponseFunctionShellCallOutputContentParam"][]; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "shell_call_output"; + }; + /** + * ShellToolParam + * @description Shell tool for Responses API: run commands in hosted containers or local runtime. + * See https://developers.openai.com/api/docs/guides/tools-shell. + */ + ShellToolParam: { + /** Environment */ + environment: { + [key: string]: unknown; + }; + /** Type */ + type: "shell" | string; + }; /** * Skill * @description Represents a skill from the Anthropic Skills API @@ -38435,6 +42124,32 @@ export interface components { /** Updated At */ updated_at: string; }; + /** SkillReference */ + SkillReference: { + /** Skill Id */ + skill_id: string; + /** + * Type + * @constant + */ + type: "skill_reference"; + /** Version */ + version?: string | null; + } & { + [key: string]: unknown; + }; + /** SkillReferenceParam */ + SkillReferenceParam: { + /** Skill Id */ + skill_id: string; + /** + * Type + * @constant + */ + type: "skill_reference"; + /** Version */ + version?: string; + }; /** SpendAnalyticsPaginatedResponse */ SpendAnalyticsPaginatedResponse: { metadata?: components["schemas"]["DailySpendMetadata"]; @@ -38761,6 +42476,21 @@ export interface components { /** Model */ model?: string | null; }; + /** + * Summary + * @description A summary text from the model. + */ + Summary: { + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "summary_text"; + } & { + [key: string]: unknown; + }; /** * SupportedDBObjectType * @description Supported database object types for fine-grained DB storage control. @@ -39650,6 +43380,19 @@ export interface components { [key: string]: unknown; }; }; + /** + * Text + * @description Unconstrained free-form text. + */ + Text: { + /** + * Type + * @constant + */ + type: "text"; + } & { + [key: string]: unknown; + }; /** TierCohortStatistic */ TierCohortStatistic: { /** Cohort */ @@ -39770,12 +43513,141 @@ export interface components { /** Total Tokens */ total_tokens: number; }; + /** + * ToolChoiceAllowedParam + * @description Constrains the tools available to the model to a pre-defined set. + */ + ToolChoiceAllowedParam: { + /** + * Mode + * @enum {string} + */ + mode: "auto" | "required"; + /** Tools */ + tools: { + [key: string]: unknown; + }[]; + /** + * Type + * @constant + */ + type: "allowed_tools"; + } & { + [key: string]: unknown; + }; + /** + * ToolChoiceApplyPatchParam + * @description Forces the model to call the apply_patch tool when executing a tool call. + */ + ToolChoiceApplyPatchParam: { + /** + * Type + * @constant + */ + type: "apply_patch"; + } & { + [key: string]: unknown; + }; + /** + * ToolChoiceCustomParam + * @description Use this option to force the model to call a specific custom tool. + */ + ToolChoiceCustomParam: { + /** Name */ + name: string; + /** + * Type + * @constant + */ + type: "custom"; + } & { + [key: string]: unknown; + }; + /** + * ToolChoiceFunctionParam + * @description Use this option to force the model to call a specific function. + */ + ToolChoiceFunctionParam: { + /** Name */ + name: string; + /** + * Type + * @constant + */ + type: "function"; + } & { + [key: string]: unknown; + }; + /** + * ToolChoiceMcpParam + * @description Use this option to force the model to call a specific tool on a remote MCP server. + */ + ToolChoiceMcpParam: { + /** Name */ + name?: string | null; + /** Server Label */ + server_label: string; + /** + * Type + * @constant + */ + type: "mcp"; + } & { + [key: string]: unknown; + }; + /** + * ToolChoiceShellParam + * @description Forces the model to call the shell tool when a tool call is required. + */ + ToolChoiceShellParam: { + /** + * Type + * @constant + */ + type: "shell"; + } & { + [key: string]: unknown; + }; + /** + * ToolChoiceTypesParam + * @description Indicates that the model should use a built-in tool to generate a response. + * [Learn more about built-in tools](https://platform.openai.com/docs/guides/tools). + */ + ToolChoiceTypesParam: { + /** + * Type + * @enum {string} + */ + type: "file_search" | "web_search_preview" | "computer" | "computer_use_preview" | "computer_use" | "web_search_preview_2025_03_11" | "image_generation" | "code_interpreter"; + } & { + [key: string]: unknown; + }; /** ToolDetailResponse */ ToolDetailResponse: { /** Overrides */ overrides?: components["schemas"]["ToolPolicyOverrideRow"][]; tool: components["schemas"]["LiteLLM_ToolTableRow"]; }; + /** ToolFunction */ + ToolFunction: { + /** Defer Loading */ + defer_loading?: boolean | null; + /** Description */ + description?: string | null; + /** Name */ + name: string; + /** Parameters */ + parameters?: unknown | null; + /** Strict */ + strict?: boolean | null; + /** + * Type + * @constant + */ + type: "function"; + } & { + [key: string]: unknown; + }; /** ToolListResponse */ ToolListResponse: { /** Tools */ @@ -39886,6 +43758,66 @@ export interface components { /** Updated */ updated: boolean; }; + /** ToolSearchCall */ + ToolSearchCall: { + /** Arguments */ + arguments: unknown; + /** Call Id */ + call_id?: string | null; + /** + * Execution + * @enum {string} + */ + execution?: "server" | "client"; + /** Id */ + id?: string | null; + /** Status */ + status?: ("in_progress" | "completed" | "incomplete") | null; + /** + * Type + * @constant + */ + type: "tool_search_call"; + }; + /** + * ToolSearchTool + * @description Hosted or BYOT tool search configuration for deferred tools. + */ + ToolSearchTool: { + /** Description */ + description?: string | null; + /** Execution */ + execution?: ("server" | "client") | null; + /** Parameters */ + parameters?: unknown | null; + /** + * Type + * @constant + */ + type: "tool_search"; + } & { + [key: string]: unknown; + }; + /** + * ToolSearchToolParam + * @description Hosted or BYOT tool search configuration for deferred tools. + */ + ToolSearchToolParam: { + /** Description */ + description?: string | null; + /** + * Execution + * @enum {string} + */ + execution?: "server" | "client"; + /** Parameters */ + parameters?: unknown | null; + /** + * Type + * @constant + */ + type: "tool_search"; + }; /** * ToolSpendDailyEntry * @description Spend attributed to one tool on one UTC day. @@ -40040,6 +43972,21 @@ export interface components { [key: string]: unknown; }; }; + /** + * Type + * @description An action to type in text. + */ + Type: { + /** Text */ + text: string; + /** + * Type + * @constant + */ + type: "type"; + } & { + [key: string]: unknown; + }; /** * UISettingsResponse * @description Response model for UI settings @@ -41921,6 +45868,19 @@ export interface components { /** Vector Store Name */ vector_store_name?: string | null; }; + /** + * Wait + * @description A wait action. + */ + Wait: { + /** + * Type + * @constant + */ + type: "wait"; + } & { + [key: string]: unknown; + }; /** * WebSearchInterceptionSettings * @description Configuration for server-side web search interception @@ -41968,6 +45928,88 @@ export interface components { [key: string]: unknown; }; }; + /** + * WebSearchPreviewTool + * @description This tool searches the web for relevant results to use in a response. + * + * Learn more about the [web search tool](https://platform.openai.com/docs/guides/tools-web-search). + */ + WebSearchPreviewTool: { + /** Search Content Types */ + search_content_types?: ("text" | "image")[] | null; + /** Search Context Size */ + search_context_size?: ("low" | "medium" | "high") | null; + /** + * Type + * @enum {string} + */ + type: "web_search_preview" | "web_search_preview_2025_03_11"; + user_location?: components["schemas"]["openai__types__responses__web_search_preview_tool__UserLocation"] | null; + } & { + [key: string]: unknown; + }; + /** + * WebSearchPreviewToolParam + * @description This tool searches the web for relevant results to use in a response. + * + * Learn more about the [web search tool](https://platform.openai.com/docs/guides/tools-web-search). + */ + WebSearchPreviewToolParam: { + /** Search Content Types */ + search_content_types?: ("text" | "image")[]; + /** + * Search Context Size + * @enum {string} + */ + search_context_size?: "low" | "medium" | "high"; + /** + * Type + * @enum {string} + */ + type: "web_search_preview" | "web_search_preview_2025_03_11"; + user_location?: components["schemas"]["openai__types__responses__web_search_preview_tool_param__UserLocation"] | null; + }; + /** + * WebSearchTool + * @description Search the Internet for sources related to the prompt. + * + * Learn more about the + * [web search tool](https://platform.openai.com/docs/guides/tools-web-search). + */ + WebSearchTool: { + filters?: components["schemas"]["Filters"] | null; + /** Search Context Size */ + search_context_size?: ("low" | "medium" | "high") | null; + /** + * Type + * @enum {string} + */ + type: "web_search" | "web_search_2025_08_26"; + user_location?: components["schemas"]["openai__types__responses__web_search_tool__UserLocation"] | null; + } & { + [key: string]: unknown; + }; + /** + * WebSearchToolParam + * @description Search the Internet for sources related to the prompt. + * + * Learn more about the + * [web search tool](https://platform.openai.com/docs/guides/tools-web-search). + */ + WebSearchToolParam: { + filters?: components["schemas"]["Filters"] | null; + /** + * Search Context Size + * @enum {string} + */ + search_context_size?: "low" | "medium" | "high"; + /** + * Type + * @enum {string} + */ + type: "web_search" | "web_search_2025_08_26"; + user_location?: components["schemas"]["openai__types__responses__web_search_tool_param__UserLocation"] | null; + }; /** WorkerRegistryEntry */ WorkerRegistryEntry: { /** Name */ @@ -42051,6 +46093,17 @@ export interface components { } & { [key: string]: unknown; }; + /** ComputerToolParam */ + litellm__types__llms__openai__ComputerToolParam: { + /** Display Height */ + display_height: number; + /** Display Width */ + display_width: number; + /** Environment */ + environment: ("mac" | "windows" | "ubuntu" | "browser") | string; + /** Type */ + type: "computer_use_preview" | string; + }; /** ModelInfo */ litellm__types__router__ModelInfo: { /** Access Windows */ @@ -42120,6 +46173,96 @@ export interface components { } & { [key: string]: unknown; }; + /** + * ComputerToolParam + * @description A tool that controls a virtual computer. + * + * Learn more about the [computer tool](https://platform.openai.com/docs/guides/tools-computer-use). + */ + openai__types__responses__computer_tool_param__ComputerToolParam: { + /** + * Type + * @constant + */ + type: "computer"; + }; + /** + * UserLocation + * @description The user's location. + */ + openai__types__responses__web_search_preview_tool__UserLocation: { + /** City */ + city?: string | null; + /** Country */ + country?: string | null; + /** Region */ + region?: string | null; + /** Timezone */ + timezone?: string | null; + /** + * Type + * @constant + */ + type: "approximate"; + } & { + [key: string]: unknown; + }; + /** + * UserLocation + * @description The user's location. + */ + openai__types__responses__web_search_preview_tool_param__UserLocation: { + /** City */ + city?: string | null; + /** Country */ + country?: string | null; + /** Region */ + region?: string | null; + /** Timezone */ + timezone?: string | null; + /** + * Type + * @constant + */ + type: "approximate"; + }; + /** + * UserLocation + * @description The approximate location of the user. + */ + openai__types__responses__web_search_tool__UserLocation: { + /** City */ + city?: string | null; + /** Country */ + country?: string | null; + /** Region */ + region?: string | null; + /** Timezone */ + timezone?: string | null; + /** Type */ + type?: "approximate" | null; + } & { + [key: string]: unknown; + }; + /** + * UserLocation + * @description The approximate location of the user. + */ + openai__types__responses__web_search_tool_param__UserLocation: { + /** City */ + city?: string | null; + /** Country */ + country?: string | null; + /** Region */ + region?: string | null; + /** Timezone */ + timezone?: string | null; + /** + * Type + * @constant + */ + type?: "approximate"; + }; /** updateDeployment */ updateDeployment: { /** Blocked */ @@ -56413,7 +60556,69 @@ export interface operations { path?: never; cookie?: never; }; - requestBody?: never; + requestBody: { + content: { + "application/json": { + /** Background */ + background?: boolean | null; + /** Context Management */ + context_management?: components["schemas"]["ContextManagementEntry"][] | null; + /** Include */ + include?: ("file_search_call.results" | "web_search_call.results" | "web_search_call.action.sources" | "message.input_image.image_url" | "computer_call_output.output.image_url" | "code_interpreter_call.outputs" | "reasoning.encrypted_content" | "message.output_text.logprobs")[] | null; + /** Input */ + input: string | (components["schemas"]["EasyInputMessageParam"] | components["schemas"]["ResponsesAPIRequestParams_Message"] | components["schemas"]["ResponseOutputMessageParam"] | components["schemas"]["ResponseFileSearchToolCallParam"] | components["schemas"]["ResponseComputerToolCallParam"] | components["schemas"]["ComputerCallOutput"] | components["schemas"]["ResponseFunctionWebSearchParam"] | components["schemas"]["ResponseFunctionToolCallParam"] | components["schemas"]["FunctionCallOutput"] | components["schemas"]["ToolSearchCall"] | components["schemas"]["ResponseToolSearchOutputItemParamParam"] | components["schemas"]["ResponseReasoningItemParam"] | components["schemas"]["ResponseCompactionItemParamParam"] | components["schemas"]["ResponsesAPIRequestParams_ImageGenerationCall"] | components["schemas"]["ResponseCodeInterpreterToolCallParam"] | components["schemas"]["LocalShellCall"] | components["schemas"]["LocalShellCallOutput"] | components["schemas"]["ShellCall"] | components["schemas"]["ShellCallOutput"] | components["schemas"]["ApplyPatchCall"] | components["schemas"]["ApplyPatchCallOutput"] | components["schemas"]["McpListTools"] | components["schemas"]["McpApprovalRequest"] | components["schemas"]["ResponsesAPIRequestParams_McpApprovalResponse"] | components["schemas"]["McpCall"] | components["schemas"]["ResponseCustomToolCallOutputParam"] | components["schemas"]["ResponseCustomToolCallParam"] | components["schemas"]["ItemReference"])[]; + /** Instructions */ + instructions?: string | null; + /** Max Output Tokens */ + max_output_tokens?: number | null; + /** Max Tool Calls */ + max_tool_calls?: number | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model */ + model: string; + /** Parallel Tool Calls */ + parallel_tool_calls?: boolean | null; + /** Partial Images */ + partial_images?: number | null; + /** Previous Response Id */ + previous_response_id?: string | null; + prompt?: components["schemas"]["PromptObject"] | null; + /** Prompt Cache Key */ + prompt_cache_key?: string | null; + prompt_cache_options?: components["schemas"]["PromptCacheOptions"] | null; + /** Prompt Cache Retention */ + prompt_cache_retention?: string | null; + reasoning?: components["schemas"]["Reasoning"] | null; + /** Safety Identifier */ + safety_identifier?: string | null; + /** Service Tier */ + service_tier?: string | null; + /** Store */ + store?: boolean | null; + /** Stream */ + stream?: boolean | null; + stream_options?: components["schemas"]["ResponsesAPIStreamOptions"] | null; + /** Temperature */ + temperature?: number | null; + text?: components["schemas"]["ResponseTextConfigParam"] | null; + /** Tool Choice */ + tool_choice?: ("none" | "auto" | "required") | components["schemas"]["ToolChoiceAllowedParam"] | components["schemas"]["ToolChoiceTypesParam"] | components["schemas"]["ToolChoiceFunctionParam"] | components["schemas"]["ToolChoiceMcpParam"] | components["schemas"]["ToolChoiceCustomParam"] | components["schemas"]["ToolChoiceApplyPatchParam"] | components["schemas"]["ToolChoiceShellParam"] | null; + /** Tools */ + tools?: (components["schemas"]["FunctionToolParam"] | components["schemas"]["FileSearchToolParam"] | components["schemas"]["openai__types__responses__computer_tool_param__ComputerToolParam"] | components["schemas"]["ComputerUsePreviewToolParam"] | components["schemas"]["WebSearchToolParam"] | components["schemas"]["Mcp"] | components["schemas"]["CodeInterpreter"] | components["schemas"]["ImageGeneration"] | components["schemas"]["LocalShell"] | components["schemas"]["FunctionShellToolParam"] | components["schemas"]["CustomToolParam"] | components["schemas"]["NamespaceToolParam"] | components["schemas"]["ToolSearchToolParam"] | components["schemas"]["WebSearchPreviewToolParam"] | components["schemas"]["ApplyPatchToolParam"] | components["schemas"]["litellm__types__llms__openai__ComputerToolParam"] | components["schemas"]["ShellToolParam"])[] | null; + /** Top Logprobs */ + top_logprobs?: number | null; + /** Top P */ + top_p?: number | null; + /** Truncation */ + truncation?: ("auto" | "disabled") | null; + /** User */ + user?: string | null; + }; + }; + }; responses: { /** @description Successful Response */ 200: { @@ -56421,7 +60626,8 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponsesAPIResponse"]; + "text/event-stream": string; }; }; }; @@ -56483,7 +60689,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponsesAPIResponse"]; }; }; /** @description Validation Error */ @@ -56514,7 +60720,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["DeleteResponseResult"]; }; }; /** @description Validation Error */ @@ -56576,7 +60782,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponseItemList"]; }; }; /** @description Validation Error */ @@ -59607,7 +63813,69 @@ export interface operations { path?: never; cookie?: never; }; - requestBody?: never; + requestBody: { + content: { + "application/json": { + /** Background */ + background?: boolean | null; + /** Context Management */ + context_management?: components["schemas"]["ContextManagementEntry"][] | null; + /** Include */ + include?: ("file_search_call.results" | "web_search_call.results" | "web_search_call.action.sources" | "message.input_image.image_url" | "computer_call_output.output.image_url" | "code_interpreter_call.outputs" | "reasoning.encrypted_content" | "message.output_text.logprobs")[] | null; + /** Input */ + input: string | (components["schemas"]["EasyInputMessageParam"] | components["schemas"]["ResponsesAPIRequestParams_Message"] | components["schemas"]["ResponseOutputMessageParam"] | components["schemas"]["ResponseFileSearchToolCallParam"] | components["schemas"]["ResponseComputerToolCallParam"] | components["schemas"]["ComputerCallOutput"] | components["schemas"]["ResponseFunctionWebSearchParam"] | components["schemas"]["ResponseFunctionToolCallParam"] | components["schemas"]["FunctionCallOutput"] | components["schemas"]["ToolSearchCall"] | components["schemas"]["ResponseToolSearchOutputItemParamParam"] | components["schemas"]["ResponseReasoningItemParam"] | components["schemas"]["ResponseCompactionItemParamParam"] | components["schemas"]["ResponsesAPIRequestParams_ImageGenerationCall"] | components["schemas"]["ResponseCodeInterpreterToolCallParam"] | components["schemas"]["LocalShellCall"] | components["schemas"]["LocalShellCallOutput"] | components["schemas"]["ShellCall"] | components["schemas"]["ShellCallOutput"] | components["schemas"]["ApplyPatchCall"] | components["schemas"]["ApplyPatchCallOutput"] | components["schemas"]["McpListTools"] | components["schemas"]["McpApprovalRequest"] | components["schemas"]["ResponsesAPIRequestParams_McpApprovalResponse"] | components["schemas"]["McpCall"] | components["schemas"]["ResponseCustomToolCallOutputParam"] | components["schemas"]["ResponseCustomToolCallParam"] | components["schemas"]["ItemReference"])[]; + /** Instructions */ + instructions?: string | null; + /** Max Output Tokens */ + max_output_tokens?: number | null; + /** Max Tool Calls */ + max_tool_calls?: number | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model */ + model: string; + /** Parallel Tool Calls */ + parallel_tool_calls?: boolean | null; + /** Partial Images */ + partial_images?: number | null; + /** Previous Response Id */ + previous_response_id?: string | null; + prompt?: components["schemas"]["PromptObject"] | null; + /** Prompt Cache Key */ + prompt_cache_key?: string | null; + prompt_cache_options?: components["schemas"]["PromptCacheOptions"] | null; + /** Prompt Cache Retention */ + prompt_cache_retention?: string | null; + reasoning?: components["schemas"]["Reasoning"] | null; + /** Safety Identifier */ + safety_identifier?: string | null; + /** Service Tier */ + service_tier?: string | null; + /** Store */ + store?: boolean | null; + /** Stream */ + stream?: boolean | null; + stream_options?: components["schemas"]["ResponsesAPIStreamOptions"] | null; + /** Temperature */ + temperature?: number | null; + text?: components["schemas"]["ResponseTextConfigParam"] | null; + /** Tool Choice */ + tool_choice?: ("none" | "auto" | "required") | components["schemas"]["ToolChoiceAllowedParam"] | components["schemas"]["ToolChoiceTypesParam"] | components["schemas"]["ToolChoiceFunctionParam"] | components["schemas"]["ToolChoiceMcpParam"] | components["schemas"]["ToolChoiceCustomParam"] | components["schemas"]["ToolChoiceApplyPatchParam"] | components["schemas"]["ToolChoiceShellParam"] | null; + /** Tools */ + tools?: (components["schemas"]["FunctionToolParam"] | components["schemas"]["FileSearchToolParam"] | components["schemas"]["openai__types__responses__computer_tool_param__ComputerToolParam"] | components["schemas"]["ComputerUsePreviewToolParam"] | components["schemas"]["WebSearchToolParam"] | components["schemas"]["Mcp"] | components["schemas"]["CodeInterpreter"] | components["schemas"]["ImageGeneration"] | components["schemas"]["LocalShell"] | components["schemas"]["FunctionShellToolParam"] | components["schemas"]["CustomToolParam"] | components["schemas"]["NamespaceToolParam"] | components["schemas"]["ToolSearchToolParam"] | components["schemas"]["WebSearchPreviewToolParam"] | components["schemas"]["ApplyPatchToolParam"] | components["schemas"]["litellm__types__llms__openai__ComputerToolParam"] | components["schemas"]["ShellToolParam"])[] | null; + /** Top Logprobs */ + top_logprobs?: number | null; + /** Top P */ + top_p?: number | null; + /** Truncation */ + truncation?: ("auto" | "disabled") | null; + /** User */ + user?: string | null; + }; + }; + }; responses: { /** @description Successful Response */ 200: { @@ -59615,7 +63883,8 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponsesAPIResponse"]; + "text/event-stream": string; }; }; }; @@ -59677,7 +63946,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponsesAPIResponse"]; }; }; /** @description Validation Error */ @@ -59708,7 +63977,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["DeleteResponseResult"]; }; }; /** @description Validation Error */ @@ -59770,7 +64039,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponseItemList"]; }; }; /** @description Validation Error */ @@ -68721,7 +72990,69 @@ export interface operations { path?: never; cookie?: never; }; - requestBody?: never; + requestBody: { + content: { + "application/json": { + /** Background */ + background?: boolean | null; + /** Context Management */ + context_management?: components["schemas"]["ContextManagementEntry"][] | null; + /** Include */ + include?: ("file_search_call.results" | "web_search_call.results" | "web_search_call.action.sources" | "message.input_image.image_url" | "computer_call_output.output.image_url" | "code_interpreter_call.outputs" | "reasoning.encrypted_content" | "message.output_text.logprobs")[] | null; + /** Input */ + input: string | (components["schemas"]["EasyInputMessageParam"] | components["schemas"]["ResponsesAPIRequestParams_Message"] | components["schemas"]["ResponseOutputMessageParam"] | components["schemas"]["ResponseFileSearchToolCallParam"] | components["schemas"]["ResponseComputerToolCallParam"] | components["schemas"]["ComputerCallOutput"] | components["schemas"]["ResponseFunctionWebSearchParam"] | components["schemas"]["ResponseFunctionToolCallParam"] | components["schemas"]["FunctionCallOutput"] | components["schemas"]["ToolSearchCall"] | components["schemas"]["ResponseToolSearchOutputItemParamParam"] | components["schemas"]["ResponseReasoningItemParam"] | components["schemas"]["ResponseCompactionItemParamParam"] | components["schemas"]["ResponsesAPIRequestParams_ImageGenerationCall"] | components["schemas"]["ResponseCodeInterpreterToolCallParam"] | components["schemas"]["LocalShellCall"] | components["schemas"]["LocalShellCallOutput"] | components["schemas"]["ShellCall"] | components["schemas"]["ShellCallOutput"] | components["schemas"]["ApplyPatchCall"] | components["schemas"]["ApplyPatchCallOutput"] | components["schemas"]["McpListTools"] | components["schemas"]["McpApprovalRequest"] | components["schemas"]["ResponsesAPIRequestParams_McpApprovalResponse"] | components["schemas"]["McpCall"] | components["schemas"]["ResponseCustomToolCallOutputParam"] | components["schemas"]["ResponseCustomToolCallParam"] | components["schemas"]["ItemReference"])[]; + /** Instructions */ + instructions?: string | null; + /** Max Output Tokens */ + max_output_tokens?: number | null; + /** Max Tool Calls */ + max_tool_calls?: number | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model */ + model: string; + /** Parallel Tool Calls */ + parallel_tool_calls?: boolean | null; + /** Partial Images */ + partial_images?: number | null; + /** Previous Response Id */ + previous_response_id?: string | null; + prompt?: components["schemas"]["PromptObject"] | null; + /** Prompt Cache Key */ + prompt_cache_key?: string | null; + prompt_cache_options?: components["schemas"]["PromptCacheOptions"] | null; + /** Prompt Cache Retention */ + prompt_cache_retention?: string | null; + reasoning?: components["schemas"]["Reasoning"] | null; + /** Safety Identifier */ + safety_identifier?: string | null; + /** Service Tier */ + service_tier?: string | null; + /** Store */ + store?: boolean | null; + /** Stream */ + stream?: boolean | null; + stream_options?: components["schemas"]["ResponsesAPIStreamOptions"] | null; + /** Temperature */ + temperature?: number | null; + text?: components["schemas"]["ResponseTextConfigParam"] | null; + /** Tool Choice */ + tool_choice?: ("none" | "auto" | "required") | components["schemas"]["ToolChoiceAllowedParam"] | components["schemas"]["ToolChoiceTypesParam"] | components["schemas"]["ToolChoiceFunctionParam"] | components["schemas"]["ToolChoiceMcpParam"] | components["schemas"]["ToolChoiceCustomParam"] | components["schemas"]["ToolChoiceApplyPatchParam"] | components["schemas"]["ToolChoiceShellParam"] | null; + /** Tools */ + tools?: (components["schemas"]["FunctionToolParam"] | components["schemas"]["FileSearchToolParam"] | components["schemas"]["openai__types__responses__computer_tool_param__ComputerToolParam"] | components["schemas"]["ComputerUsePreviewToolParam"] | components["schemas"]["WebSearchToolParam"] | components["schemas"]["Mcp"] | components["schemas"]["CodeInterpreter"] | components["schemas"]["ImageGeneration"] | components["schemas"]["LocalShell"] | components["schemas"]["FunctionShellToolParam"] | components["schemas"]["CustomToolParam"] | components["schemas"]["NamespaceToolParam"] | components["schemas"]["ToolSearchToolParam"] | components["schemas"]["WebSearchPreviewToolParam"] | components["schemas"]["ApplyPatchToolParam"] | components["schemas"]["litellm__types__llms__openai__ComputerToolParam"] | components["schemas"]["ShellToolParam"])[] | null; + /** Top Logprobs */ + top_logprobs?: number | null; + /** Top P */ + top_p?: number | null; + /** Truncation */ + truncation?: ("auto" | "disabled") | null; + /** User */ + user?: string | null; + }; + }; + }; responses: { /** @description Successful Response */ 200: { @@ -68729,7 +73060,8 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponsesAPIResponse"]; + "text/event-stream": string; }; }; }; @@ -68791,7 +73123,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponsesAPIResponse"]; }; }; /** @description Validation Error */ @@ -68822,7 +73154,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["DeleteResponseResult"]; }; }; /** @description Validation Error */ @@ -68884,7 +73216,7 @@ export interface operations { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": components["schemas"]["ResponseItemList"]; }; }; /** @description Validation Error */ From b431d12cf07b58611267a63c8ef6a8fd66863298 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:49:20 -0700 Subject: [PATCH 10/96] fix(models): add the June 1, 2026 retirement date to the vertex_ai gemini-2.0-flash rows (#42850) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 2 ++ model_prices_and_context_window.json | 2 ++ 2 files changed, 4 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 50bcc6f71bf..135f1215351 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -74696,6 +74696,7 @@ "supports_web_search": false }, "vertex_ai/gemini-2.0-flash": { + "deprecation_date": "2026-06-01", "input_cost_per_audio_token": 1e-06, "input_cost_per_audio_token_batches": 5e-07, "input_cost_per_character": 3.75e-08, @@ -74708,6 +74709,7 @@ "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" }, "vertex_ai/gemini-2.0-flash-lite": { + "deprecation_date": "2026-06-01", "input_cost_per_audio_token": 7.5e-08, "input_cost_per_audio_token_batches": 3.75e-08, "input_cost_per_character": 1.875e-08, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 50bcc6f71bf..135f1215351 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -74696,6 +74696,7 @@ "supports_web_search": false }, "vertex_ai/gemini-2.0-flash": { + "deprecation_date": "2026-06-01", "input_cost_per_audio_token": 1e-06, "input_cost_per_audio_token_batches": 5e-07, "input_cost_per_character": 3.75e-08, @@ -74708,6 +74709,7 @@ "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" }, "vertex_ai/gemini-2.0-flash-lite": { + "deprecation_date": "2026-06-01", "input_cost_per_audio_token": 7.5e-08, "input_cost_per_audio_token_batches": 3.75e-08, "input_cost_per_character": 1.875e-08, From 320b40645c48d77b6d5ffae0f5ca0cfee73442a4 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:49:23 -0500 Subject: [PATCH 11/96] fix(caching): stamp provider on sync cache-hit logs so responses spend logs record provider (#42830) * fix(caching): stamp provider on sync cache-hit logs so responses spend logs record provider Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(caching): tighten sync cache-hit provider regression docstring Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(caching): drop redundant docstring on sync cache-hit provider test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/caching/caching_handler.py | 1 + .../caching/test_caching_handler.py | 42 +++++++++++++++++++ 2 files changed, 43 insertions(+) diff --git a/litellm/caching/caching_handler.py b/litellm/caching/caching_handler.py index 4afd0e7caaa..0887b8bb897 100644 --- a/litellm/caching/caching_handler.py +++ b/litellm/caching/caching_handler.py @@ -429,6 +429,7 @@ class LLMCachingHandler: kwargs=kwargs, cached_result=cached_result, is_async=False, + custom_llm_provider=custom_llm_provider, ) if not _should_defer_streaming_cache_hit_callbacks(cached_result=cached_result): diff --git a/tests/test_litellm/caching/test_caching_handler.py b/tests/test_litellm/caching/test_caching_handler.py index 6956a6932d5..e5a7f1540ca 100644 --- a/tests/test_litellm/caching/test_caching_handler.py +++ b/tests/test_litellm/caching/test_caching_handler.py @@ -591,6 +591,48 @@ async def test_embedding_cache_hit_sets_custom_llm_provider_on_logging_obj(): assert logging_obj.model_call_details["custom_llm_provider"] == "openai" +def test_sync_stream_responses_cache_hit_sets_custom_llm_provider_on_logging_obj(monkeypatch): + import litellm + from litellm.caching.caching import Cache + from litellm.types.utils import CallTypes + + monkeypatch.setattr(litellm, "cache", Cache(type="local")) + kwargs = {"model": "azure/gpt-5.4-mini", "input": "hello", "stream": True} + cached_response = { + "id": "resp_sync_stream", + "created_at": int(time.time()), + "status": "completed", + "model": "gpt-5.4-mini", + "object": "response", + "output": [ + { + "type": "message", + "id": "msg_sync_stream", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "hi", "annotations": []}], + } + ], + } + litellm.cache.add_cache(json.dumps(cached_response), **kwargs) + handler = LLMCachingHandler(original_function=litellm.responses, request_kwargs=kwargs, start_time=datetime.now()) + logging_obj = _build_logging_obj(CallTypes.responses.value, stream=True) + + hit = handler._sync_get_cache( + model="azure/gpt-5.4-mini", + original_function=litellm.responses, + logging_obj=logging_obj, + start_time=datetime.now(), + call_type=CallTypes.responses.value, + kwargs=kwargs, + args=(), + ) + + assert hit.cached_result is not None + assert logging_obj.model_call_details["custom_llm_provider"] == "azure" + assert logging_obj.model_call_details["litellm_params"]["custom_llm_provider"] == "azure" + + def test_request_kwargs_does_not_retain_logging_obj(): """ The caching handler lives on logging_obj._llm_caching_handler, so keeping From a314fe858fda651975dcf671ed9f8a26d954cbd8 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:50:59 -0700 Subject: [PATCH 12/96] feat(models): add 39 together_ai chat rows priced by the Together models API (#42851) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 356 ++++++++++++++++++ model_prices_and_context_window.json | 356 ++++++++++++++++++ 2 files changed, 712 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 135f1215351..acb59e20231 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -67719,6 +67719,362 @@ "output_cost_per_token": 0.0, "source": "https://api.together.ai/v1/models" }, + "together_ai/NousResearch/Nous-Hermes-2-Mixtral-8x7B-DPO": { + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/QwQ-32B": { + "input_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2-72B-Instruct": { + "input_cost_per_token": 9e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 9e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2-VL-72B-Instruct": { + "input_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2.5-72B-Instruct-Turbo": { + "input_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2.5-Coder-32B-Instruct": { + "input_cost_per_token": 8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2.5-VL-72B-Instruct": { + "input_cost_per_token": 1.95e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 8e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": { + "input_cost_per_token": 2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-Coder-Next-FP8": { + "input_cost_per_token": 5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-Next-80B-A3B-Instruct": { + "input_cost_per_token": 1.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-Next-80B-A3B-Thinking": { + "input_cost_per_token": 1.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-VL-32B-Instruct": { + "input_cost_per_token": 5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-VL-8B-Instruct": { + "input_cost_per_token": 1.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 6.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3.5-397B-A17B": { + "cache_read_input_token_cost": 3.5e-07, + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 3.6e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": { + "input_cost_per_token": 2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B": { + "input_cost_per_token": 1.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B": { + "input_cost_per_token": 1.6e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.6e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-V3.1": { + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.7e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/deepseek-coder-33b-instruct": { + "input_cost_per_token": 8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/google/gemma-2-27b-it": { + "input_cost_per_token": 8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/google/gemma-4-31B-it": { + "input_cost_per_token": 3.9e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 9.7e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Llama-3-8b-chat-hf": { + "input_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 2e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Llama-4-Scout-17B-16E-Instruct": { + "input_cost_per_token": 1.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 1048576, + "max_tokens": 1048576, + "mode": "chat", + "output_cost_per_token": 5.9e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Meta-Llama-3-70B-Instruct-Turbo": { + "input_cost_per_token": 8.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 8.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Meta-Llama-3-8B-Instruct": { + "input_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 2e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": { + "input_cost_per_token": 8.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 8.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { + "input_cost_per_token": 1.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/mistralai/Mistral-7B-Instruct-v0.1": { + "input_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 2e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/mistralai/Mistral-Small-24B-Instruct-2501": { + "input_cost_per_token": 1e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1": { + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/moonshotai/Kimi-K2.6": { + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 4.5e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/moonshotai/Kimi-K2.7-Code": { + "cache_read_input_token_cost": 1.9e-07, + "input_cost_per_token": 9.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 4e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/nvidia/Llama-3.1-Nemotron-70B-Instruct-HF": { + "input_cost_per_token": 8.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 8.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/nvidia/nemotron-3-ultra-550b-a55b": { + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 512288, + "max_tokens": 512288, + "mode": "chat", + "output_cost_per_token": 3.6e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/openai/gpt-oss-20b": { + "input_cost_per_token": 5e-08, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 2e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/zai-org/GLM-4.5-Air-FP8": { + "input_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.1e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/zai-org/GLM-4.7": { + "input_cost_per_token": 4.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "max_tokens": 202752, + "mode": "chat", + "output_cost_per_token": 2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/zai-org/GLM-5": { + "input_cost_per_token": 1e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "max_tokens": 202752, + "mode": "chat", + "output_cost_per_token": 3.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/zai-org/GLM-5.1": { + "cache_read_input_token_cost": 2.6e-07, + "input_cost_per_token": 1.4e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "max_tokens": 202752, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + "source": "https://api.together.ai/v1/models" + }, "azure/eu/codex-mini": { "deprecation_date": "2026-11-15", "cache_read_input_token_cost": 4.13e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 135f1215351..acb59e20231 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -67719,6 +67719,362 @@ "output_cost_per_token": 0.0, "source": "https://api.together.ai/v1/models" }, + "together_ai/NousResearch/Nous-Hermes-2-Mixtral-8x7B-DPO": { + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/QwQ-32B": { + "input_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2-72B-Instruct": { + "input_cost_per_token": 9e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 9e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2-VL-72B-Instruct": { + "input_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2.5-72B-Instruct-Turbo": { + "input_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2.5-Coder-32B-Instruct": { + "input_cost_per_token": 8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen2.5-VL-72B-Instruct": { + "input_cost_per_token": 1.95e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 8e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": { + "input_cost_per_token": 2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-Coder-Next-FP8": { + "input_cost_per_token": 5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 1.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-Next-80B-A3B-Instruct": { + "input_cost_per_token": 1.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-Next-80B-A3B-Thinking": { + "input_cost_per_token": 1.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-VL-32B-Instruct": { + "input_cost_per_token": 5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3-VL-8B-Instruct": { + "input_cost_per_token": 1.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 6.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/Qwen/Qwen3.5-397B-A17B": { + "cache_read_input_token_cost": 3.5e-07, + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 3.6e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": { + "input_cost_per_token": 2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B": { + "input_cost_per_token": 1.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B": { + "input_cost_per_token": 1.6e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.6e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/DeepSeek-V3.1": { + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.7e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/deepseek-ai/deepseek-coder-33b-instruct": { + "input_cost_per_token": 8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/google/gemma-2-27b-it": { + "input_cost_per_token": 8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/google/gemma-4-31B-it": { + "input_cost_per_token": 3.9e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 9.7e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Llama-3-8b-chat-hf": { + "input_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 2e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Llama-4-Scout-17B-16E-Instruct": { + "input_cost_per_token": 1.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 1048576, + "max_tokens": 1048576, + "mode": "chat", + "output_cost_per_token": 5.9e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Meta-Llama-3-70B-Instruct-Turbo": { + "input_cost_per_token": 8.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 8.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Meta-Llama-3-8B-Instruct": { + "input_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 2e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": { + "input_cost_per_token": 8.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 8.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { + "input_cost_per_token": 1.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/mistralai/Mistral-7B-Instruct-v0.1": { + "input_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 2e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/mistralai/Mistral-Small-24B-Instruct-2501": { + "input_cost_per_token": 1e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1": { + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/moonshotai/Kimi-K2.6": { + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 1.2e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 4.5e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/moonshotai/Kimi-K2.7-Code": { + "cache_read_input_token_cost": 1.9e-07, + "input_cost_per_token": 9.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 262144, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 4e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/nvidia/Llama-3.1-Nemotron-70B-Instruct-HF": { + "input_cost_per_token": 8.8e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 8.8e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/nvidia/nemotron-3-ultra-550b-a55b": { + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 6e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 512288, + "max_tokens": 512288, + "mode": "chat", + "output_cost_per_token": 3.6e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/openai/gpt-oss-20b": { + "input_cost_per_token": 5e-08, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 2e-07, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/zai-org/GLM-4.5-Air-FP8": { + "input_cost_per_token": 2e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 1.1e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/zai-org/GLM-4.7": { + "input_cost_per_token": 4.5e-07, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "max_tokens": 202752, + "mode": "chat", + "output_cost_per_token": 2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/zai-org/GLM-5": { + "input_cost_per_token": 1e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "max_tokens": 202752, + "mode": "chat", + "output_cost_per_token": 3.2e-06, + "source": "https://api.together.ai/v1/models" + }, + "together_ai/zai-org/GLM-5.1": { + "cache_read_input_token_cost": 2.6e-07, + "input_cost_per_token": 1.4e-06, + "litellm_provider": "together_ai", + "max_input_tokens": 202752, + "max_tokens": 202752, + "mode": "chat", + "output_cost_per_token": 4.4e-06, + "source": "https://api.together.ai/v1/models" + }, "azure/eu/codex-mini": { "deprecation_date": "2026-11-15", "cache_read_input_token_cost": 4.13e-07, From d4b0a547b22ebc76658dc24e6b93cc4175d0f14b Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:51:13 -0700 Subject: [PATCH 13/96] chore(models): add deprecation_date to claude-mythos-preview from the Anthropic deprecations page (#42845) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 1 + model_prices_and_context_window.json | 1 + 2 files changed, 2 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index acb59e20231..e0da868c7dd 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -58821,6 +58821,7 @@ "source": "https://platform.claude.com/docs/en/about-claude/pricing" }, "claude-mythos-preview": { + "deprecation_date": "2026-06-09", "supports_anthropic_compaction": true, "cache_creation_input_token_cost": 1.25e-05, "cache_creation_input_token_cost_above_1hr": 2e-05, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index acb59e20231..e0da868c7dd 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -58821,6 +58821,7 @@ "source": "https://platform.claude.com/docs/en/about-claude/pricing" }, "claude-mythos-preview": { + "deprecation_date": "2026-06-09", "supports_anthropic_compaction": true, "cache_creation_input_token_cost": 1.25e-05, "cache_creation_input_token_cost_above_1hr": 2e-05, From 6d5e87b71b32a090ad7ce71d569bd3832ec21b3a Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:52:06 -0700 Subject: [PATCH 14/96] test(integration): assert /v1/responses usage reports Anthropic system cache write then read (#42855) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...est_anthropic_system_cache_control_wire.py | 72 +++++++++++++++++-- 1 file changed, 66 insertions(+), 6 deletions(-) diff --git a/tests/integration/providers/test_anthropic_system_cache_control_wire.py b/tests/integration/providers/test_anthropic_system_cache_control_wire.py index cdac76158e6..22aeb9f8f0f 100644 --- a/tests/integration/providers/test_anthropic_system_cache_control_wire.py +++ b/tests/integration/providers/test_anthropic_system_cache_control_wire.py @@ -52,9 +52,7 @@ def test_chat_completions_system_block_list_carries_cache_control_to_anthropic_s "messages": [ { "role": "system", - "content": [ - {"type": "text", "text": policy, "cache_control": {"type": "ephemeral"}} - ], + "content": [{"type": "text", "text": policy, "cache_control": {"type": "ephemeral"}}], }, {"role": "user", "content": "hi"}, ], @@ -112,9 +110,7 @@ def test_responses_system_input_item_carries_cache_control_to_anthropic_system(g "input": [ { "role": "system", - "content": [ - {"type": "input_text", "text": policy, "cache_control": {"type": "ephemeral"}} - ], + "content": [{"type": "input_text", "text": policy, "cache_control": {"type": "ephemeral"}}], }, {"role": "user", "content": "hi"}, ], @@ -126,3 +122,67 @@ def test_responses_system_input_item_carries_cache_control_to_anthropic_system(g assert any(item.get("type") == "message" for item in payload.get("output", []) if isinstance(item, dict)) assert len(wire.drain()) == 1 + +def _anthropic_usage_reply(identity: str, cache_creation: int, cache_read: int) -> bytes: + return json.dumps( + { + "id": identity, + "type": "message", + "role": "assistant", + "model": _MODEL, + "content": [{"type": "text", "text": "done"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": { + "input_tokens": 3, + "output_tokens": 1, + "cache_creation_input_tokens": cache_creation, + "cache_read_input_tokens": cache_read, + }, + } + ).encode() + + +def test_responses_usage_reports_anthropic_system_cache_write_then_read(gateway: Gateway) -> None: + identity: Final = f"responses-system-cache-usage-{uuid.uuid4().hex}" + policy: Final = f"policy {identity}" + replies: Final = iter( + ( + _anthropic_usage_reply(identity, cache_creation=1200, cache_read=0), + _anthropic_usage_reply(identity, cache_creation=0, cache_read=1200), + ) + ) + + def respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages" + _assert_system_block(_JSON_OBJECT.validate_json(request.body), policy) + return Reply(body=next(replies)) + + def input_tokens_details(model: str, user_turn: str) -> JsonValue: + response: Final = gateway.request( + "POST", + "/v1/responses", + { + "model": model, + "input": [ + { + "role": "system", + "content": [{"type": "input_text", "text": policy, "cache_control": {"type": "ephemeral"}}], + }, + {"role": "user", "content": user_turn}, + ], + }, + ) + assert response.status_code == 200, response.text + usage: Final = _JSON_OBJECT.validate_json(response.content)["usage"] + assert isinstance(usage, dict), response.text + return usage["input_tokens_details"] + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"anthropic/{_MODEL}", api_base=wire.url, api_key=_API_KEY) + first: Final = input_tokens_details(model, "first turn") + second: Final = input_tokens_details(model, "second turn") + assert len(wire.drain()) == 2 + assert isinstance(first, dict) and isinstance(second, dict), (first, second) + assert (first["cache_write_tokens"], first["cached_tokens"]) == (1200, 0), first + assert (second.get("cache_write_tokens", 0), second["cached_tokens"]) == (0, 1200), second From c21f7822274062f5ca7111cde686bdd854518b31 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:54:09 -0700 Subject: [PATCH 15/96] fix(models): add the sora-2-pro shutdown date to the sora-2-pro-high-res rows (#42846) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 2 ++ model_prices_and_context_window.json | 2 ++ 2 files changed, 4 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e0da868c7dd..f53fa4ee820 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -51184,6 +51184,7 @@ }, "openai/sora-2-pro-high-res": { "litellm_provider": "openai", + "deprecation_date": "2026-09-24", "mode": "video_generation", "output_cost_per_video_per_second": 0.5, "source": "https://platform.openai.com/docs/api-reference/videos", @@ -55392,6 +55393,7 @@ }, "sora-2-pro-high-res": { "litellm_provider": "openai", + "deprecation_date": "2026-09-24", "mode": "video_generation", "output_cost_per_video_per_second": 0.5, "source": "https://developers.openai.com/api/docs/pricing", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e0da868c7dd..f53fa4ee820 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -51184,6 +51184,7 @@ }, "openai/sora-2-pro-high-res": { "litellm_provider": "openai", + "deprecation_date": "2026-09-24", "mode": "video_generation", "output_cost_per_video_per_second": 0.5, "source": "https://platform.openai.com/docs/api-reference/videos", @@ -55392,6 +55393,7 @@ }, "sora-2-pro-high-res": { "litellm_provider": "openai", + "deprecation_date": "2026-09-24", "mode": "video_generation", "output_cost_per_video_per_second": 0.5, "source": "https://developers.openai.com/api/docs/pricing", From 545afadba5d8aafadbfbe049f84df98470207f47 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:57:49 -0700 Subject: [PATCH 16/96] feat(models): add openrouter/openai/gpt-oss-120b:batch from the OpenRouter models API (#42847) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 19 +++++++++++++++++++ model_prices_and_context_window.json | 19 +++++++++++++++++++ 2 files changed, 38 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index f53fa4ee820..19cbae5c235 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41934,6 +41934,25 @@ "supports_vision": false, "supports_web_search": false }, + "openrouter/openai/gpt-oss-120b:batch": { + "input_cost_per_token": 2.96e-08, + "litellm_provider": "openrouter", + "max_input_tokens": 131072, + "max_output_tokens": 117964, + "max_tokens": 117964, + "mode": "chat", + "output_cost_per_token": 1.36e-07, + "source": "https://openrouter.ai/api/v1/models", + "supports_audio_input": false, + "supports_function_calling": true, + "supports_pdf_input": false, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": false, + "supports_web_search": false + }, "openrouter/openai/gpt-oss-20b": { "cache_read_input_token_cost": 3e-08, "input_cost_per_token": 1.8e-08, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index f53fa4ee820..19cbae5c235 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41934,6 +41934,25 @@ "supports_vision": false, "supports_web_search": false }, + "openrouter/openai/gpt-oss-120b:batch": { + "input_cost_per_token": 2.96e-08, + "litellm_provider": "openrouter", + "max_input_tokens": 131072, + "max_output_tokens": 117964, + "max_tokens": 117964, + "mode": "chat", + "output_cost_per_token": 1.36e-07, + "source": "https://openrouter.ai/api/v1/models", + "supports_audio_input": false, + "supports_function_calling": true, + "supports_pdf_input": false, + "supports_prompt_caching": false, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": false, + "supports_web_search": false + }, "openrouter/openai/gpt-oss-20b": { "cache_read_input_token_cost": 3e-08, "input_cost_per_token": 1.8e-08, From 80c39a08c80606a0de39c16d41696412d8289cfe Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:00:47 -0700 Subject: [PATCH 17/96] feat(models): add gemini lyria-realtime-exp row inherited from lyria-3.5 (#42848) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 25 +++++++++++++++++++ model_prices_and_context_window.json | 25 +++++++++++++++++++ 2 files changed, 50 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 19cbae5c235..903700e24cf 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -63760,6 +63760,31 @@ "supports_web_search": false, "output_cost_per_image": 0.08 }, + "gemini/lyria-realtime-exp": { + "input_cost_per_token": 0, + "litellm_provider": "gemini", + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 0, + "source": "https://ai.google.dev/gemini-api/docs/models/lyria-realtime-exp", + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "audio" + ], + "supports_audio_input": false, + "supports_audio_output": true, + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": false, + "supports_vision": false, + "supports_web_search": false, + "output_cost_per_image": 0.08 + }, "perplexity/anthropic/claude-fable-5": { "litellm_provider": "perplexity", "mode": "responses", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 19cbae5c235..903700e24cf 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -63760,6 +63760,31 @@ "supports_web_search": false, "output_cost_per_image": 0.08 }, + "gemini/lyria-realtime-exp": { + "input_cost_per_token": 0, + "litellm_provider": "gemini", + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 0, + "source": "https://ai.google.dev/gemini-api/docs/models/lyria-realtime-exp", + "supported_modalities": [ + "text" + ], + "supported_output_modalities": [ + "audio" + ], + "supports_audio_input": false, + "supports_audio_output": true, + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": false, + "supports_vision": false, + "supports_web_search": false, + "output_cost_per_image": 0.08 + }, "perplexity/anthropic/claude-fable-5": { "litellm_provider": "perplexity", "mode": "responses", From 660e6746e4a6d8fb8915f145944dc4fa7358d8ed Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:16:34 -0700 Subject: [PATCH 18/96] feat(bedrock): add 17 aws-bedrock cost map rows from provider sync (#42852) * feat(bedrock): add 22 aws-bedrock cost map rows from provider sync Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(bedrock): mark mythos-preview regional rows as supporting prompt caching Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(bedrock): drop bare openai.gpt-5.6-sol row shadowing the bedrock_mantle fallback get_model_info checks the bare split_model before bedrock_mantle/, so the new bare key made bedrock_mantle/us-east-2/openai.gpt-5.6-sol resolve to the bedrock_converse row instead of falling back to bedrock_mantle/openai.gpt-5.6-sol Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(bedrock): drop bare OpenAI and xAI keys already covered by bedrock_mantle rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 482 ++++++++++++++++++ model_prices_and_context_window.json | 482 ++++++++++++++++++ 2 files changed, 964 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 903700e24cf..f39c739258b 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -75240,5 +75240,487 @@ "output_cost_per_audio_token": 2e-05, "output_cost_per_token": 2e-05, "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" + }, + "anthropic.claude-mythos-5-1": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_creation_input_token_cost": 1.25e-05, + "cache_read_input_token_cost": 2.5e-07, + "output_cost_per_token": 5e-05, + "input_cost_per_token": 1e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "global.anthropic.claude-mythos-5-1": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "cache_read_input_token_cost": 2.5e-07, + "cache_creation_input_token_cost": 1.25e-05, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 5e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "us.anthropic.claude-mythos-5-1": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "input_cost_per_token": 1.1e-05, + "output_cost_per_token": 5.5e-05, + "cache_read_input_token_cost": 2.75e-07, + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "us.anthropic.claude-mythos-5": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "cache_creation_input_token_cost": 1.375e-05, + "input_cost_per_token": 1.1e-05, + "cache_read_input_token_cost": 1.1e-06, + "output_cost_per_token": 5.5e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-fable-5": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "output_cost_per_token": 5.5e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "au.anthropic.claude-fable-5": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "output_cost_per_token": 5.5e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-opus-4-7": { + "bedrock_converse_supports_strict_tools": false, + "supports_adaptive_thinking": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 2048, + "output_cost_per_token": 2.75e-05, + "cache_creation_input_token_cost": 6.875e-06, + "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-opus-4-8": { + "bedrock_converse_supports_strict_tools": false, + "supports_adaptive_thinking": true, + "supports_mid_conversation_system": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 1024, + "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "output_cost_per_token": 2.75e-05, + "cache_read_input_token_cost": 5.5e-07, + "cache_creation_input_token_cost": 6.875e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-opus-5": { + "bedrock_converse_supports_strict_tools": false, + "supports_adaptive_thinking": true, + "supports_mid_conversation_system": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_creation_input_token_cost": 6.875e-06, + "input_cost_per_token": 5.5e-06, + "cache_read_input_token_cost": 5.5e-07, + "output_cost_per_token": 2.75e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-opus-5-5": { + "bedrock_converse_supports_strict_tools": false, + "supports_adaptive_thinking": true, + "supports_mid_conversation_system": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "thinking_always_on": true, + "supports_forced_tool_use": false, + "cache_creation_input_token_cost_above_1hr": 8.8e-06, + "cache_creation_input_token_cost": 5.5e-06, + "input_cost_per_token": 4.4e-06, + "output_cost_per_token": 2.2e-05, + "cache_read_input_token_cost": 2.2e-07, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-sonnet-4-6": { + "supports_adaptive_thinking": true, + "supports_legacy_thinking": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_max_reasoning_effort": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_native_structured_output": true, + "supports_output_config": true, + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 1024, + "output_cost_per_token": 1.65e-05, + "cache_creation_input_token_cost_above_1hr": 6.6e-06, + "cache_read_input_token_cost": 3.3e-07, + "cache_creation_input_token_cost": 4.125e-06, + "input_cost_per_token": 3.3e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-sonnet-5": { + "bedrock_converse_supports_strict_tools": false, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 1024, + "cache_creation_input_token_cost_above_1hr": 4.4e-06, + "cache_creation_input_token_cost": 2.75e-06, + "input_cost_per_token": 2.2e-06, + "output_cost_per_token": 1.1e-05, + "cache_read_input_token_cost": 2.2e-07, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "us.anthropic.claude-mythos-preview": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "thinking_always_on": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_output_config": true, + "cache_creation_input_token_cost_above_1hr": 5.5e-05, + "cache_read_input_token_cost": 2.75e-06, + "cache_creation_input_token_cost": 3.4375e-05, + "output_cost_per_token": 0.0001375, + "input_cost_per_token": 2.75e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-mythos-preview": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "thinking_always_on": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_output_config": true, + "input_cost_per_token": 2.75e-05, + "cache_creation_input_token_cost": 3.4375e-05, + "output_cost_per_token": 0.0001375, + "cache_creation_input_token_cost_above_1hr": 5.5e-05, + "cache_read_input_token_cost": 2.75e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "au.anthropic.claude-mythos-preview": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "thinking_always_on": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_output_config": true, + "input_cost_per_token": 2.75e-05, + "cache_creation_input_token_cost": 3.4375e-05, + "output_cost_per_token": 0.0001375, + "cache_creation_input_token_cost_above_1hr": 5.5e-05, + "cache_read_input_token_cost": 2.75e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "deepseek.r1-v1:0": { + "input_cost_per_token": 1.35e-06, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 5.4e-06, + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_function_calling": false, + "supports_reasoning": true, + "supports_tool_choice": false + }, + "mistral.pixtral-large-2502-v1:0": { + "input_cost_per_token": 2e-06, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 6e-06, + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_function_calling": true, + "supports_tool_choice": false } } diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 903700e24cf..f39c739258b 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -75240,5 +75240,487 @@ "output_cost_per_audio_token": 2e-05, "output_cost_per_token": 2e-05, "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" + }, + "anthropic.claude-mythos-5-1": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_creation_input_token_cost": 1.25e-05, + "cache_read_input_token_cost": 2.5e-07, + "output_cost_per_token": 5e-05, + "input_cost_per_token": 1e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "global.anthropic.claude-mythos-5-1": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "cache_read_input_token_cost": 2.5e-07, + "cache_creation_input_token_cost": 1.25e-05, + "input_cost_per_token": 1e-05, + "output_cost_per_token": 5e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "us.anthropic.claude-mythos-5-1": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "input_cost_per_token": 1.1e-05, + "output_cost_per_token": 5.5e-05, + "cache_read_input_token_cost": 2.75e-07, + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "us.anthropic.claude-mythos-5": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "cache_creation_input_token_cost": 1.375e-05, + "input_cost_per_token": 1.1e-05, + "cache_read_input_token_cost": 1.1e-06, + "output_cost_per_token": 5.5e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-fable-5": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "output_cost_per_token": 5.5e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "au.anthropic.claude-fable-5": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "thinking_always_on": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": true, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "output_cost_per_token": 5.5e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-opus-4-7": { + "bedrock_converse_supports_strict_tools": false, + "supports_adaptive_thinking": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 2048, + "output_cost_per_token": 2.75e-05, + "cache_creation_input_token_cost": 6.875e-06, + "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-opus-4-8": { + "bedrock_converse_supports_strict_tools": false, + "supports_adaptive_thinking": true, + "supports_mid_conversation_system": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 1024, + "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "output_cost_per_token": 2.75e-05, + "cache_read_input_token_cost": 5.5e-07, + "cache_creation_input_token_cost": 6.875e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-opus-5": { + "bedrock_converse_supports_strict_tools": false, + "supports_adaptive_thinking": true, + "supports_mid_conversation_system": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_creation_input_token_cost": 6.875e-06, + "input_cost_per_token": 5.5e-06, + "cache_read_input_token_cost": 5.5e-07, + "output_cost_per_token": 2.75e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-opus-5-5": { + "bedrock_converse_supports_strict_tools": false, + "supports_adaptive_thinking": true, + "supports_mid_conversation_system": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 512, + "thinking_always_on": true, + "supports_forced_tool_use": false, + "cache_creation_input_token_cost_above_1hr": 8.8e-06, + "cache_creation_input_token_cost": 5.5e-06, + "input_cost_per_token": 4.4e-06, + "output_cost_per_token": 2.2e-05, + "cache_read_input_token_cost": 2.2e-07, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-sonnet-4-6": { + "supports_adaptive_thinking": true, + "supports_legacy_thinking": true, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "supports_assistant_prefill": true, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_max_reasoning_effort": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_native_structured_output": true, + "supports_output_config": true, + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 1024, + "output_cost_per_token": 1.65e-05, + "cache_creation_input_token_cost_above_1hr": 6.6e-06, + "cache_read_input_token_cost": 3.3e-07, + "cache_creation_input_token_cost": 4.125e-06, + "input_cost_per_token": 3.3e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-sonnet-5": { + "bedrock_converse_supports_strict_tools": false, + "litellm_provider": "bedrock_converse", + "supports_tool_search": true, + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "supports_adaptive_thinking": true, + "supports_mid_conversation_system": true, + "supports_assistant_prefill": false, + "supports_computer_use": true, + "supports_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_sampling_params": false, + "supports_tool_choice": true, + "supports_vision": true, + "supports_xhigh_reasoning_effort": true, + "supports_native_structured_output": false, + "supports_max_reasoning_effort": true, + "supports_output_config": true, + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_parallel_tool_use_config": true, + "prompt_cache_min_tokens": 1024, + "cache_creation_input_token_cost_above_1hr": 4.4e-06, + "cache_creation_input_token_cost": 2.75e-06, + "input_cost_per_token": 2.2e-06, + "output_cost_per_token": 1.1e-05, + "cache_read_input_token_cost": 2.2e-07, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "us.anthropic.claude-mythos-preview": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "thinking_always_on": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_output_config": true, + "cache_creation_input_token_cost_above_1hr": 5.5e-05, + "cache_read_input_token_cost": 2.75e-06, + "cache_creation_input_token_cost": 3.4375e-05, + "output_cost_per_token": 0.0001375, + "input_cost_per_token": 2.75e-05, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "apac.anthropic.claude-mythos-preview": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "thinking_always_on": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_output_config": true, + "input_cost_per_token": 2.75e-05, + "cache_creation_input_token_cost": 3.4375e-05, + "output_cost_per_token": 0.0001375, + "cache_creation_input_token_cost_above_1hr": 5.5e-05, + "cache_read_input_token_cost": 2.75e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "au.anthropic.claude-mythos-preview": { + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "thinking_always_on": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_output_config": true, + "input_cost_per_token": 2.75e-05, + "cache_creation_input_token_cost": 3.4375e-05, + "output_cost_per_token": 0.0001375, + "cache_creation_input_token_cost_above_1hr": 5.5e-05, + "cache_read_input_token_cost": 2.75e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" + }, + "deepseek.r1-v1:0": { + "input_cost_per_token": 1.35e-06, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 5.4e-06, + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_function_calling": false, + "supports_reasoning": true, + "supports_tool_choice": false + }, + "mistral.pixtral-large-2502-v1:0": { + "input_cost_per_token": 2e-06, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 6e-06, + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_function_calling": true, + "supports_tool_choice": false } } From 153f13b913f8683888b2a81ce41d6d0591a14f50 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:35:22 -0700 Subject: [PATCH 19/96] fix(models): add fireworks deprecation dates for kimi k2.6 fast, kimi k2.7 code fast and glm 5.2 fast us (#42849) * fix(models): add fireworks deprecation dates for kimi k2.6 fast, kimi k2.7 code fast and glm 5.2 fast us Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(models): add the same fireworks deprecation dates to the router twin rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 6 ++++++ model_prices_and_context_window.json | 6 ++++++ 2 files changed, 12 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index f39c739258b..a251ebe7b5c 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -25193,6 +25193,7 @@ }, "fireworks_ai/kimi-k2p6-fast": { "cache_read_input_token_cost": 3e-07, + "deprecation_date": "2026-08-27", "input_cost_per_token": 2e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 262144, @@ -25228,6 +25229,7 @@ }, "fireworks_ai/kimi-k2p7-code-fast": { "cache_read_input_token_cost": 3.8e-07, + "deprecation_date": "2026-08-27", "input_cost_per_token": 1.9e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 262144, @@ -53546,6 +53548,7 @@ }, "fireworks_ai/accounts/fireworks/routers/kimi-k2p6-fast": { "cache_read_input_token_cost": 3e-07, + "deprecation_date": "2026-08-27", "input_cost_per_token": 2e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 262144, @@ -53562,6 +53565,7 @@ }, "fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast": { "cache_read_input_token_cost": 3.8e-07, + "deprecation_date": "2026-08-27", "input_cost_per_token": 1.9e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 262144, @@ -59481,6 +59485,7 @@ }, "fireworks_ai/glm-5p2-fast-us": { "cache_read_input_token_cost": 2.1e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.1e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, @@ -59709,6 +59714,7 @@ }, "fireworks_ai/accounts/fireworks/routers/glm-5p2-fast-us": { "cache_read_input_token_cost": 2.1e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.1e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index f39c739258b..a251ebe7b5c 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -25193,6 +25193,7 @@ }, "fireworks_ai/kimi-k2p6-fast": { "cache_read_input_token_cost": 3e-07, + "deprecation_date": "2026-08-27", "input_cost_per_token": 2e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 262144, @@ -25228,6 +25229,7 @@ }, "fireworks_ai/kimi-k2p7-code-fast": { "cache_read_input_token_cost": 3.8e-07, + "deprecation_date": "2026-08-27", "input_cost_per_token": 1.9e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 262144, @@ -53546,6 +53548,7 @@ }, "fireworks_ai/accounts/fireworks/routers/kimi-k2p6-fast": { "cache_read_input_token_cost": 3e-07, + "deprecation_date": "2026-08-27", "input_cost_per_token": 2e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 262144, @@ -53562,6 +53565,7 @@ }, "fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast": { "cache_read_input_token_cost": 3.8e-07, + "deprecation_date": "2026-08-27", "input_cost_per_token": 1.9e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 262144, @@ -59481,6 +59485,7 @@ }, "fireworks_ai/glm-5p2-fast-us": { "cache_read_input_token_cost": 2.1e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.1e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, @@ -59709,6 +59714,7 @@ }, "fireworks_ai/accounts/fireworks/routers/glm-5p2-fast-us": { "cache_read_input_token_cost": 2.1e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.1e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, From b0980638c5d56492be145ce3da2d1033c5c154d2 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:40:09 -0700 Subject: [PATCH 20/96] fix(model-catalog): declare above_32k cost fields on ModelInfo (#42856) * fix(model-catalog): add above_32k cost fields to ModelInfo round-trip Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model-catalog): drop redundant comments on above_32k fields Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm-rust/crates/model-catalog/src/model_info.rs | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 77a8b768e38..4a56e1112d1 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -245,6 +245,8 @@ pub struct ModelInfo { #[serde(default, skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_above_272k_tokens_priority: Option, #[serde(default, skip_serializing_if = "Option::is_none")] + pub cache_creation_input_token_cost_above_32k_tokens: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_batches: Option, /// Flex service-tier rate for the same-named base field. #[serde(default, skip_serializing_if = "Option::is_none")] @@ -283,6 +285,8 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(default, skip_serializing_if = "Option::is_none")] pub cache_read_input_token_cost_above_272k_tokens_priority: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub cache_read_input_token_cost_above_32k_tokens: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(default, skip_serializing_if = "Option::is_none")] pub cache_read_input_token_cost_above_512k_tokens: Option, @@ -377,6 +381,8 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(default, skip_serializing_if = "Option::is_none")] pub input_cost_per_token_above_272k_tokens_priority: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub input_cost_per_token_above_32k_tokens: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(default, skip_serializing_if = "Option::is_none")] pub input_cost_per_token_above_512k_tokens: Option, @@ -498,6 +504,8 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(default, skip_serializing_if = "Option::is_none")] pub output_cost_per_token_above_272k_tokens_priority: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output_cost_per_token_above_32k_tokens: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(default, skip_serializing_if = "Option::is_none")] pub output_cost_per_token_above_512k_tokens: Option, From a4f69e058c822eeb5821511283a4d138a0d98dd4 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 02:47:29 +0000 Subject: [PATCH 21/96] fix(model_prices): registry audit 2026-09-23, in-region Bedrock Claude prices (#42779) * fix(model_prices): registry audit 2026-09-23, in-region Bedrock Claude and OpenRouter price fixes Absorbs #42698 Co-authored-by: coldStoneSoul Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model_prices): keep registry formatting unchanged Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test: use eu.amazon.nova-pro for regional pricing probe after in-region parity Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test: lock in-region parity for bare Bedrock Claude ids Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test: compare every pricing field for bare Bedrock Claude parity Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model-catalog): add above_32k cost fields to ModelInfo round-trip Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * Revert "fix(model-catalog): add above_32k cost fields to ModelInfo round-trip" This reverts commit c71d5a3de320fc6ff5257486280376d635e10c08. --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: coldStoneSoul --- ...odel_prices_and_context_window_backup.json | 138 +++++++++--------- model_prices_and_context_window.json | 138 +++++++++--------- tests/test_litellm/test_utils.py | 46 +++++- 3 files changed, 176 insertions(+), 146 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a251ebe7b5c..1a2501ab19b 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -812,17 +812,17 @@ "prompt_cache_min_tokens": 2048 }, "anthropic.claude-haiku-4-5-20251001-v1:0": { - "cache_creation_input_token_cost": 1.25e-06, - "cache_creation_input_token_cost_above_1hr": 2e-06, - "cache_read_input_token_cost": 1e-07, - "input_cost_per_token": 1e-06, + "cache_creation_input_token_cost": 1.375e-06, + "cache_creation_input_token_cost_above_1hr": 2.2e-06, + "cache_read_input_token_cost": 1.1e-07, + "input_cost_per_token": 1.1e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 5e-06, + "output_cost_per_token": 5.5e-06, "source": "https://aws.amazon.com/bedrock/pricing/", "supports_assistant_prefill": true, "supports_computer_use": true, @@ -836,8 +836,8 @@ "supports_native_structured_output": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 4096, - "input_cost_per_token_batches": 5e-07, - "output_cost_per_token_batches": 2.5e-06 + "input_cost_per_token_batches": 5.5e-07, + "output_cost_per_token_batches": 2.75e-06 }, "anthropic.claude-haiku-4-5@20251001": { "cache_creation_input_token_cost": 1.25e-06, @@ -1028,17 +1028,17 @@ "prompt_cache_min_tokens": 1024 }, "anthropic.claude-opus-4-5-20251101-v1:0": { - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1063,17 +1063,17 @@ "anthropic.claude-opus-4-6-v1": { "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1243,17 +1243,17 @@ "anthropic.claude-opus-4-7": { "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1447,16 +1447,16 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-fable-5": { - "cache_creation_input_token_cost": 1.25e-05, - "cache_creation_input_token_cost_above_1hr": 2e-05, - "cache_read_input_token_cost": 1e-06, - "input_cost_per_token": 1e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 5e-05, + "output_cost_per_token": 5.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1485,16 +1485,16 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-fable-5-1": { - "cache_creation_input_token_cost": 1.25e-05, - "cache_creation_input_token_cost_above_1hr": 2e-05, - "cache_read_input_token_cost": 2.5e-07, - "input_cost_per_token": 1e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "cache_read_input_token_cost": 2.75e-07, + "input_cost_per_token": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 5e-05, + "output_cost_per_token": 5.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1757,17 +1757,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1795,17 +1795,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 5e-06, - "cache_creation_input_token_cost_above_1hr": 8e-06, - "cache_read_input_token_cost": 2e-07, - "input_cost_per_token": 4e-06, + "cache_creation_input_token_cost": 5.5e-06, + "cache_creation_input_token_cost_above_1hr": 8.8e-06, + "cache_read_input_token_cost": 2.2e-07, + "input_cost_per_token": 4.4e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2e-05, + "output_cost_per_token": 2.2e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2225,17 +2225,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2494,17 +2494,17 @@ }, "anthropic.claude-sonnet-5": { "bedrock_converse_supports_strict_tools": false, - "cache_creation_input_token_cost": 2.5e-06, - "cache_creation_input_token_cost_above_1hr": 4e-06, - "cache_read_input_token_cost": 2e-07, - "input_cost_per_token": 2e-06, + "cache_creation_input_token_cost": 2.75e-06, + "cache_creation_input_token_cost_above_1hr": 4.4e-06, + "cache_read_input_token_cost": 2.2e-07, + "input_cost_per_token": 2.2e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 1e-05, + "output_cost_per_token": 1.1e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2729,17 +2729,17 @@ "anthropic.claude-sonnet-4-6": { "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_1hr": 6e-06, - "cache_read_input_token_cost": 3e-07, - "input_cost_per_token": 3e-06, + "cache_creation_input_token_cost": 4.125e-06, + "cache_creation_input_token_cost_above_1hr": 6.6e-06, + "cache_read_input_token_cost": 3.3e-07, + "input_cost_per_token": 3.3e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 1.5e-05, + "output_cost_per_token": 1.65e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2970,22 +2970,22 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-sonnet-4-5-20250929-v1:0": { - "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_1hr": 6e-06, - "cache_read_input_token_cost": 3e-07, - "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, - "output_cost_per_token_above_200k_tokens": 2.25e-05, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, - "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.2e-05, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, + "cache_creation_input_token_cost": 4.125e-06, + "cache_creation_input_token_cost_above_1hr": 6.6e-06, + "cache_read_input_token_cost": 3.3e-07, + "input_cost_per_token": 3.3e-06, + "input_cost_per_token_above_200k_tokens": 6.6e-06, + "output_cost_per_token_above_200k_tokens": 2.475e-05, + "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, + "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.32e-05, + "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 1.5e-05, + "output_cost_per_token": 1.65e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -3003,8 +3003,8 @@ "supports_native_structured_output": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "input_cost_per_token_batches": 1.5e-06, - "output_cost_per_token_batches": 7.5e-06, + "input_cost_per_token_batches": 1.65e-06, + "output_cost_per_token_batches": 8.25e-06, "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-v1": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a251ebe7b5c..1a2501ab19b 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -812,17 +812,17 @@ "prompt_cache_min_tokens": 2048 }, "anthropic.claude-haiku-4-5-20251001-v1:0": { - "cache_creation_input_token_cost": 1.25e-06, - "cache_creation_input_token_cost_above_1hr": 2e-06, - "cache_read_input_token_cost": 1e-07, - "input_cost_per_token": 1e-06, + "cache_creation_input_token_cost": 1.375e-06, + "cache_creation_input_token_cost_above_1hr": 2.2e-06, + "cache_read_input_token_cost": 1.1e-07, + "input_cost_per_token": 1.1e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 5e-06, + "output_cost_per_token": 5.5e-06, "source": "https://aws.amazon.com/bedrock/pricing/", "supports_assistant_prefill": true, "supports_computer_use": true, @@ -836,8 +836,8 @@ "supports_native_structured_output": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 4096, - "input_cost_per_token_batches": 5e-07, - "output_cost_per_token_batches": 2.5e-06 + "input_cost_per_token_batches": 5.5e-07, + "output_cost_per_token_batches": 2.75e-06 }, "anthropic.claude-haiku-4-5@20251001": { "cache_creation_input_token_cost": 1.25e-06, @@ -1028,17 +1028,17 @@ "prompt_cache_min_tokens": 1024 }, "anthropic.claude-opus-4-5-20251101-v1:0": { - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1063,17 +1063,17 @@ "anthropic.claude-opus-4-6-v1": { "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1243,17 +1243,17 @@ "anthropic.claude-opus-4-7": { "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1447,16 +1447,16 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-fable-5": { - "cache_creation_input_token_cost": 1.25e-05, - "cache_creation_input_token_cost_above_1hr": 2e-05, - "cache_read_input_token_cost": 1e-06, - "input_cost_per_token": 1e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "cache_read_input_token_cost": 1.1e-06, + "input_cost_per_token": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 5e-05, + "output_cost_per_token": 5.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1485,16 +1485,16 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-fable-5-1": { - "cache_creation_input_token_cost": 1.25e-05, - "cache_creation_input_token_cost_above_1hr": 2e-05, - "cache_read_input_token_cost": 2.5e-07, - "input_cost_per_token": 1e-05, + "cache_creation_input_token_cost": 1.375e-05, + "cache_creation_input_token_cost_above_1hr": 2.2e-05, + "cache_read_input_token_cost": 2.75e-07, + "input_cost_per_token": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 5e-05, + "output_cost_per_token": 5.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1757,17 +1757,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1795,17 +1795,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 5e-06, - "cache_creation_input_token_cost_above_1hr": 8e-06, - "cache_read_input_token_cost": 2e-07, - "input_cost_per_token": 4e-06, + "cache_creation_input_token_cost": 5.5e-06, + "cache_creation_input_token_cost_above_1hr": 8.8e-06, + "cache_read_input_token_cost": 2.2e-07, + "input_cost_per_token": 4.4e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2e-05, + "output_cost_per_token": 2.2e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2225,17 +2225,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_1hr": 1e-05, - "cache_read_input_token_cost": 5e-07, - "input_cost_per_token": 5e-06, + "cache_creation_input_token_cost": 6.875e-06, + "cache_creation_input_token_cost_above_1hr": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "input_cost_per_token": 5.5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.5e-05, + "output_cost_per_token": 2.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2494,17 +2494,17 @@ }, "anthropic.claude-sonnet-5": { "bedrock_converse_supports_strict_tools": false, - "cache_creation_input_token_cost": 2.5e-06, - "cache_creation_input_token_cost_above_1hr": 4e-06, - "cache_read_input_token_cost": 2e-07, - "input_cost_per_token": 2e-06, + "cache_creation_input_token_cost": 2.75e-06, + "cache_creation_input_token_cost_above_1hr": 4.4e-06, + "cache_read_input_token_cost": 2.2e-07, + "input_cost_per_token": 2.2e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 1e-05, + "output_cost_per_token": 1.1e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2729,17 +2729,17 @@ "anthropic.claude-sonnet-4-6": { "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_1hr": 6e-06, - "cache_read_input_token_cost": 3e-07, - "input_cost_per_token": 3e-06, + "cache_creation_input_token_cost": 4.125e-06, + "cache_creation_input_token_cost_above_1hr": 6.6e-06, + "cache_read_input_token_cost": 3.3e-07, + "input_cost_per_token": 3.3e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 1.5e-05, + "output_cost_per_token": 1.65e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2970,22 +2970,22 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-sonnet-4-5-20250929-v1:0": { - "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_1hr": 6e-06, - "cache_read_input_token_cost": 3e-07, - "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, - "output_cost_per_token_above_200k_tokens": 2.25e-05, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, - "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.2e-05, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, + "cache_creation_input_token_cost": 4.125e-06, + "cache_creation_input_token_cost_above_1hr": 6.6e-06, + "cache_read_input_token_cost": 3.3e-07, + "input_cost_per_token": 3.3e-06, + "input_cost_per_token_above_200k_tokens": 6.6e-06, + "output_cost_per_token_above_200k_tokens": 2.475e-05, + "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, + "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.32e-05, + "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 1.5e-05, + "output_cost_per_token": 1.65e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -3003,8 +3003,8 @@ "supports_native_structured_output": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "input_cost_per_token_batches": 1.5e-06, - "output_cost_per_token_batches": 7.5e-06, + "input_cost_per_token_batches": 1.65e-06, + "output_cost_per_token_batches": 8.25e-06, "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-v1": { diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 1c2d86841f7..3fca6935d57 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -1192,22 +1192,52 @@ def test_get_model_info_bedrock_regional_inference_profile_pricing(local_model_c """Regression LIT-4056: with the bedrock/ routing prefix (plain, converse/, or invoke/), the exact regional cost-map entry must win over the region-stripped base entry, matching the unprefixed control form.""" - regional = litellm.model_cost["au.anthropic.claude-opus-4-8"] - base = litellm.model_cost["anthropic.claude-opus-4-8"] + regional = litellm.model_cost["eu.amazon.nova-pro-v1:0"] + base = litellm.model_cost["amazon.nova-pro-v1:0"] assert regional["input_cost_per_token"] > base["input_cost_per_token"] for model in ( - "bedrock/au.anthropic.claude-opus-4-8", - "bedrock/converse/au.anthropic.claude-opus-4-8", - "bedrock/invoke/au.anthropic.claude-opus-4-8", + "bedrock/eu.amazon.nova-pro-v1:0", + "bedrock/converse/eu.amazon.nova-pro-v1:0", + "bedrock/invoke/eu.amazon.nova-pro-v1:0", ): info = litellm.get_model_info(model=model) - assert info["key"] == "au.anthropic.claude-opus-4-8", model + assert info["key"] == "eu.amazon.nova-pro-v1:0", model assert info["input_cost_per_token"] == regional["input_cost_per_token"], model assert info["output_cost_per_token"] == regional["output_cost_per_token"], model - control = litellm.get_model_info(model="au.anthropic.claude-opus-4-8", custom_llm_provider="bedrock") - assert control["key"] == "au.anthropic.claude-opus-4-8" + control = litellm.get_model_info(model="eu.amazon.nova-pro-v1:0", custom_llm_provider="bedrock") + assert control["key"] == "eu.amazon.nova-pro-v1:0" + + +@pytest.mark.parametrize( + "bare_key", + [ + "anthropic.claude-fable-5", + "anthropic.claude-fable-5-1", + "anthropic.claude-haiku-4-5-20251001-v1:0", + "anthropic.claude-opus-4-5-20251101-v1:0", + "anthropic.claude-opus-4-6-v1", + "anthropic.claude-opus-4-7", + "anthropic.claude-opus-4-8", + "anthropic.claude-opus-5", + "anthropic.claude-opus-5-5", + "anthropic.claude-sonnet-4-5-20250929-v1:0", + "anthropic.claude-sonnet-4-6", + "anthropic.claude-sonnet-5", + ], +) +def test_bedrock_bare_claude_id_is_priced_in_region(local_model_cost_map, bare_key): + """A bare Bedrock Claude id is an in-region invocation, so it carries the same + in-region rate as its us. inference profile, not the cheaper global. rate.""" + bare = litellm.model_cost[bare_key] + us = litellm.model_cost[f"us.{bare_key}"] + global_ = litellm.model_cost[f"global.{bare_key}"] + cost_fields = [f for f in bare if "cost" in f] + assert cost_fields + for field in cost_fields: + assert bare[field] == us[field], field + assert bare["input_cost_per_token"] > global_["input_cost_per_token"] def test_get_model_info_bedrock_mantle_region_prefix_falls_back_to_the_mantle_row(local_model_cost_map): From a3d791f34858abd06bba96e09d9c9330b86a9240 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 02:49:01 +0000 Subject: [PATCH 22/96] test: add tag rpm limit and tag budget reset integration coverage (#42859) Adds two integration tests to tests/integration/spend/test_tag_budget_enforcement.py test_key_tag_rpm_limit_rejects_the_second_request_carrying_that_tag proves a key metadata tag_rpm_limit of 1 rejects the second request carrying that tag with 429 while a request carrying a different tag still passes test_tag_budget_duration_resets_spend_and_unblocks_the_tag boots an owned proxy with a 2 to 3 second budget rescheduler, creates a tag with max_budget 0.0001 and budget_duration 5s, observes the spend block, then observes the tag serving again once ResetBudgetJob zeroes the tag spend Mutation evidence get_key_tag_rpm_limit forced to return None: the second tagged request returned 200 instead of 429, test red _queue_budget_linked_resets for uow.tags disabled in _commit_budget_cascade_once: the tag stayed blocked at 422 for the full 70 second recovery window, test red Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_tag_budget_enforcement.py | 98 +++++++++++++++++++ 1 file changed, 98 insertions(+) diff --git a/tests/integration/spend/test_tag_budget_enforcement.py b/tests/integration/spend/test_tag_budget_enforcement.py index e8d4c6438a5..f29ee6c07c8 100644 --- a/tests/integration/spend/test_tag_budget_enforcement.py +++ b/tests/integration/spend/test_tag_budget_enforcement.py @@ -1,7 +1,9 @@ import uuid +from pathlib import Path from typing import Final from integration._support.client import Gateway, eventually +from integration._support.process import owned_proxy def test_spend_over_a_tag_max_budget_rejects_the_next_request(gateway: Gateway) -> None: @@ -58,3 +60,99 @@ def test_spend_over_a_tag_max_budget_rejects_the_next_request(gateway: Gateway) }, ) assert control.status_code == 200, control.text + + +def test_key_tag_rpm_limit_rejects_the_second_request_carrying_that_tag(gateway: Gateway) -> None: + tag: Final = f"tag-rpm-{uuid.uuid4().hex}" + with gateway.scenario() as scenario: + model: Final = scenario.model(input_cost_per_token=0.01, output_cost_per_token=0.01) + key: Final = scenario.key(metadata={"tag_rpm_limit": {tag: 1}}) + first: Final = gateway.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": f"tag rpm {tag}"}], + "metadata": {"tags": [tag]}, + }, + key=key, + ) + assert first.status_code == 200, first.text + second: Final = gateway.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": f"tag rpm {tag}"}], + "metadata": {"tags": [tag]}, + }, + key=key, + ) + assert second.status_code == 429, second.text + assert "rpm" in second.text.lower() or "rate" in second.text.lower(), second.text + control: Final = gateway.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": f"other tag rpm {tag}"}], + "metadata": {"tags": [f"other-{tag}"]}, + }, + key=key, + ) + assert control.status_code == 200, control.text + + +def test_tag_budget_duration_resets_spend_and_unblocks_the_tag(gateway: Gateway, tmp_path: Path) -> None: + tag: Final = f"tag-reset-{uuid.uuid4().hex}" + with ( + owned_proxy( + gateway, + tmp_path, + {"PROXY_BUDGET_RESCHEDULER_MIN_TIME": "2", "PROXY_BUDGET_RESCHEDULER_MAX_TIME": "3"}, + ) as candidate, + candidate.scenario() as scenario, + ): + + def delete_tag() -> None: + candidate.post("/tag/delete", {"name": tag}) + + model: Final = scenario.model(input_cost_per_token=0.01, output_cost_per_token=0.01) + candidate.post("/tag/new", {"name": tag, "max_budget": 0.0001, "budget_duration": "5s"}) + scenario.cleanups.callback(delete_tag) + first: Final = candidate.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": f"tag spend {tag}"}], + "metadata": {"tags": [tag]}, + }, + ) + assert first.status_code == 200, first.text + + def rejection() -> int: + return candidate.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": f"tag reset probe {tag}"}], + "metadata": {"tags": [tag]}, + }, + ).status_code + + status: Final = eventually(rejection, lambda code: code != 200, seconds=70) + assert status in (400, 422, 429), status + blocked: Final = candidate.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": f"tag reset probe {tag}"}], + "metadata": {"tags": [tag]}, + }, + ) + assert "budget" in blocked.text.lower(), blocked.text + recovered: Final = eventually(rejection, lambda code: code == 200, seconds=70) + assert recovered == 200, recovered From 184969df37fcab351fc828dd991ae8ccf068221a Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:28:04 -0700 Subject: [PATCH 23/96] fix(models): add fireworks deprecation date for glm 5.2 fast serverless rows (#42866) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 2 ++ model_prices_and_context_window.json | 2 ++ 2 files changed, 4 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 1a2501ab19b..60ac969781e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -59469,6 +59469,7 @@ }, "fireworks_ai/glm-5p2-fast": { "cache_read_input_token_cost": 2.1e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.1e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, @@ -59698,6 +59699,7 @@ }, "fireworks_ai/accounts/fireworks/routers/glm-5p2-fast": { "cache_read_input_token_cost": 2.1e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.1e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1a2501ab19b..60ac969781e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -59469,6 +59469,7 @@ }, "fireworks_ai/glm-5p2-fast": { "cache_read_input_token_cost": 2.1e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.1e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, @@ -59698,6 +59699,7 @@ }, "fireworks_ai/accounts/fireworks/routers/glm-5p2-fast": { "cache_read_input_token_cost": 2.1e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.1e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, From 58e41e697bd2cdda98701932b5e85143c879e5df Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:37:43 -0700 Subject: [PATCH 24/96] chore(vertex_ai): add deprecation dates for retired claude 3 and jamba 1.5 partner models (#42867) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 13 +++++++++++++ model_prices_and_context_window.json | 13 +++++++++++++ 2 files changed, 26 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 60ac969781e..18c8578cec8 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -47710,6 +47710,7 @@ ] }, "vertex_ai/claude-3-5-haiku": { + "deprecation_date": "2026-07-05", "input_cost_per_token": 1e-06, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47723,6 +47724,7 @@ "supports_tool_choice": true }, "vertex_ai/claude-3-5-haiku@20241022": { + "deprecation_date": "2026-07-05", "input_cost_per_token": 1e-06, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47794,6 +47796,7 @@ "prompt_cache_min_tokens": 4096 }, "vertex_ai/claude-3-5-sonnet": { + "deprecation_date": "2026-02-19", "input_cost_per_token": 3e-06, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47809,6 +47812,7 @@ "supports_vision": true }, "vertex_ai/claude-3-5-sonnet@20240620": { + "deprecation_date": "2026-02-19", "input_cost_per_token": 3e-06, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47823,6 +47827,7 @@ "supports_vision": true }, "vertex_ai/claude-3-haiku": { + "deprecation_date": "2026-08-23", "input_cost_per_token": 2.5e-07, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47836,6 +47841,7 @@ "supports_vision": true }, "vertex_ai/claude-3-haiku@20240307": { + "deprecation_date": "2026-08-23", "input_cost_per_token": 2.5e-07, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47849,6 +47855,7 @@ "supports_vision": true }, "vertex_ai/claude-3-opus": { + "deprecation_date": "2025-08-01", "input_cost_per_token": 1.5e-05, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47862,6 +47869,7 @@ "supports_vision": true }, "vertex_ai/claude-3-opus@20240229": { + "deprecation_date": "2025-08-01", "input_cost_per_token": 1.5e-05, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -49173,6 +49181,7 @@ "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" }, "vertex_ai/jamba-1.5": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-07, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, @@ -49183,6 +49192,7 @@ "supports_tool_choice": true }, "vertex_ai/jamba-1.5-large": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-06, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, @@ -49193,6 +49203,7 @@ "supports_tool_choice": true }, "vertex_ai/jamba-1.5-large@001": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-06, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, @@ -49203,6 +49214,7 @@ "supports_tool_choice": true }, "vertex_ai/jamba-1.5-mini": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-07, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, @@ -49213,6 +49225,7 @@ "supports_tool_choice": true }, "vertex_ai/jamba-1.5-mini@001": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-07, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 60ac969781e..18c8578cec8 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -47710,6 +47710,7 @@ ] }, "vertex_ai/claude-3-5-haiku": { + "deprecation_date": "2026-07-05", "input_cost_per_token": 1e-06, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47723,6 +47724,7 @@ "supports_tool_choice": true }, "vertex_ai/claude-3-5-haiku@20241022": { + "deprecation_date": "2026-07-05", "input_cost_per_token": 1e-06, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47794,6 +47796,7 @@ "prompt_cache_min_tokens": 4096 }, "vertex_ai/claude-3-5-sonnet": { + "deprecation_date": "2026-02-19", "input_cost_per_token": 3e-06, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47809,6 +47812,7 @@ "supports_vision": true }, "vertex_ai/claude-3-5-sonnet@20240620": { + "deprecation_date": "2026-02-19", "input_cost_per_token": 3e-06, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47823,6 +47827,7 @@ "supports_vision": true }, "vertex_ai/claude-3-haiku": { + "deprecation_date": "2026-08-23", "input_cost_per_token": 2.5e-07, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47836,6 +47841,7 @@ "supports_vision": true }, "vertex_ai/claude-3-haiku@20240307": { + "deprecation_date": "2026-08-23", "input_cost_per_token": 2.5e-07, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47849,6 +47855,7 @@ "supports_vision": true }, "vertex_ai/claude-3-opus": { + "deprecation_date": "2025-08-01", "input_cost_per_token": 1.5e-05, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -47862,6 +47869,7 @@ "supports_vision": true }, "vertex_ai/claude-3-opus@20240229": { + "deprecation_date": "2025-08-01", "input_cost_per_token": 1.5e-05, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 200000, @@ -49173,6 +49181,7 @@ "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing" }, "vertex_ai/jamba-1.5": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-07, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, @@ -49183,6 +49192,7 @@ "supports_tool_choice": true }, "vertex_ai/jamba-1.5-large": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-06, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, @@ -49193,6 +49203,7 @@ "supports_tool_choice": true }, "vertex_ai/jamba-1.5-large@001": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-06, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, @@ -49203,6 +49214,7 @@ "supports_tool_choice": true }, "vertex_ai/jamba-1.5-mini": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-07, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, @@ -49213,6 +49225,7 @@ "supports_tool_choice": true }, "vertex_ai/jamba-1.5-mini@001": { + "deprecation_date": "2026-02-27", "input_cost_per_token": 2e-07, "litellm_provider": "vertex_ai-ai21_models", "max_input_tokens": 256000, From 3b715525d3717d636a9c663fcd755c1491f7c2de Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:42:27 -0700 Subject: [PATCH 25/96] test(integration): assert /v1/models reports max_input_tokens and max_output_tokens (#42858) * test(integration): assert /v1/models reports max_input_tokens and max_output_tokens Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): pin published gpt-4o-mini limits instead of reading the cost map in-test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): name the gpt-4o-mini deployment explicitly in the /v1/models test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): cite the source of the pinned gpt-4o-mini limits Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_model_listing_token_limits.py | 29 +++++++++++++++++++ 1 file changed, 29 insertions(+) create mode 100644 tests/integration/pricing/test_model_listing_token_limits.py diff --git a/tests/integration/pricing/test_model_listing_token_limits.py b/tests/integration/pricing/test_model_listing_token_limits.py new file mode 100644 index 00000000000..13702748ec5 --- /dev/null +++ b/tests/integration/pricing/test_model_listing_token_limits.py @@ -0,0 +1,29 @@ +import uuid +from typing import Final + +from integration._support.client import Gateway, object_value +from pydantic import JsonValue + + +def _listed_model(gateway: Gateway, model: str) -> dict[str, JsonValue]: + entries: Final = gateway.get("/v1/models")["data"] + assert isinstance(entries, list) + return next(object_value(entry) for entry in entries if object_value(entry)["id"] == model) + + +def test_v1_models_carries_cost_map_context_window_for_a_known_model(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + model: Final = scenario.model(model="openai/gpt-4o-mini") + listed: Final = _listed_model(gateway, model) + # OpenAI publishes these for gpt-4o-mini: https://platform.openai.com/docs/models/gpt-4o-mini (checked 2026-09-24) + assert listed["max_input_tokens"] == 128000, listed + assert listed["max_output_tokens"] == 16384, listed + + +def test_v1_models_carries_deployment_model_info_limits_for_an_unknown_model(gateway: Gateway) -> None: + unknown: Final = f"openai/custom-{uuid.uuid4().hex}" + with gateway.scenario() as scenario: + model: Final = scenario.model(model=unknown, model_info={"max_input_tokens": 4321, "max_output_tokens": 987}) + listed: Final = _listed_model(gateway, model) + assert listed["max_input_tokens"] == 4321, listed + assert listed["max_output_tokens"] == 987, listed From d248cc591430d6297ac29bcd9a560f20708a157f Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:53:17 -0700 Subject: [PATCH 26/96] fix(fireworks_ai): route firerouter short names and bill pass-through legs at the routed model's rates (#42814) * fix(fireworks_ai): route firerouter short names and bill pass-through legs at the routed model's rates fireworks_ai/firerouter and fireworks_ai/firerouter/ resolve to accounts/fireworks/routers/... instead of a models/ path, and the cost calculator falls back to the routed model's own catalog entry before the Fireworks size buckets so a Claude leg is no longer priced at $0 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(fireworks_ai): bill routed legs under the routed model's own provider Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(fireworks_ai): require the k suffix when parsing tiered input fields Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/fireworks_ai/common_utils.py | 3 +- litellm/llms/fireworks_ai/cost_calculator.py | 9 +- .../test_fireworks_ai_router_slug_wire.py | 117 +++++++++++++++++- .../test_fireworks_ai_common_utils.py | 6 +- .../test_fireworks_ai_cost_calculator.py | 69 +++++++++++ 5 files changed, 199 insertions(+), 5 deletions(-) diff --git a/litellm/llms/fireworks_ai/common_utils.py b/litellm/llms/fireworks_ai/common_utils.py index 21a630a76d7..ae52b89aa58 100644 --- a/litellm/llms/fireworks_ai/common_utils.py +++ b/litellm/llms/fireworks_ai/common_utils.py @@ -59,6 +59,7 @@ def resolve_fireworks_api_key(api_key: str | None) -> str | None: AZURE_FOUNDRY_FIREWORKS_MODEL_ID_PREFIX: Final = "FW-" +FIREROUTER: Final = "firerouter" def resolve_fireworks_resource_name(model: str) -> str: @@ -67,7 +68,7 @@ def resolve_fireworks_resource_name(model: str) -> str: return stripped if stripped.startswith(("routers/", "models/")): return f"accounts/fireworks/{stripped}" - if stripped.endswith("-fast"): + if stripped.endswith("-fast") or stripped == FIREROUTER or stripped.startswith(f"{FIREROUTER}/"): return f"accounts/fireworks/routers/{stripped}" return f"accounts/fireworks/models/{stripped}" diff --git a/litellm/llms/fireworks_ai/cost_calculator.py b/litellm/llms/fireworks_ai/cost_calculator.py index 4b6ca7c9896..d42b4ddbbf7 100644 --- a/litellm/llms/fireworks_ai/cost_calculator.py +++ b/litellm/llms/fireworks_ai/cost_calculator.py @@ -59,6 +59,13 @@ def get_base_model_for_pricing(model_name: str) -> str: def _resolve_model_info(model: str) -> ModelInfo: try: return get_model_info(model=model, custom_llm_provider="fireworks_ai") + except Exception: + return _resolve_routed_model_info(model) + + +def _resolve_routed_model_info(model: str) -> ModelInfo: + try: + return get_model_info(model=model.removeprefix("fireworks_ai/")) except Exception: base_model: Final = get_base_model_for_pricing(model_name=model) return get_model_info(model=base_model, custom_llm_provider="fireworks_ai") @@ -81,7 +88,7 @@ def cost_per_token(model: str, usage: Usage, current_time: datetime | None = Non return generic_cost_per_token( model=model, usage=usage, - custom_llm_provider="fireworks_ai", + custom_llm_provider=model_info["litellm_provider"], model_info=model_info, current_time=current_time, ) diff --git a/tests/integration/providers/test_fireworks_ai_router_slug_wire.py b/tests/integration/providers/test_fireworks_ai_router_slug_wire.py index 4b4ad0b1243..55c83945ec7 100644 --- a/tests/integration/providers/test_fireworks_ai_router_slug_wire.py +++ b/tests/integration/providers/test_fireworks_ai_router_slug_wire.py @@ -1,16 +1,70 @@ import json +import uuid +from pathlib import Path from typing import Final import pytest -from integration._support.client import Gateway +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows from integration._support.wire import Reply, Request, wire_server from pydantic import JsonValue, TypeAdapter _ROUTER_SLUG: Final = "routers/glm-latest" _ROUTER_RESOURCE: Final = "accounts/fireworks/routers/glm-latest" +_FIREROUTER_SLUGS: Final = ("firerouter", "firerouter/kimi-k3/deepseek-v4") _API_KEY: Final = "synthetic-fireworks-key" _PROMPT: Final = "route me through the router" +_COST_MAP_PATH: Final = Path(__file__).resolve().parents[3] / "model_prices_and_context_window.json" _JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +_COST_MAP: Final = TypeAdapter(dict[str, dict[str, object]]) + + +def _positive_rate(entry: dict[str, object], field: str) -> bool: + value: Final = entry.get(field) + return isinstance(value, (int, float)) and value > 0 + + +def _pick_routed_model() -> str: + catalog: Final = _COST_MAP.validate_json(_COST_MAP_PATH.read_bytes()) + return next( + key + for key, entry in catalog.items() + if "/" not in key + and entry.get("litellm_provider") == "anthropic" + and _positive_rate(entry, "input_cost_per_token") + and _positive_rate(entry, "output_cost_per_token") + and f"fireworks_ai/{key}" not in catalog + ) + + +def _catalog_cost(model: str, field: str) -> float: + cost_value: Final = _COST_MAP.validate_json(_COST_MAP_PATH.read_bytes())[model][field] + assert isinstance(cost_value, (int, float)) + return float(cost_value) + + +_ROUTED_MODEL: Final = _pick_routed_model() + + +def _approx(value: float) -> object: + return pytest.approx(value, rel=1e-6) # pyright: ignore[reportUnknownMemberType] # pytest lacks typed approx stubs + + +def _chat_completion(identity: str, model: str, prompt_tokens: int, completion_tokens: int) -> bytes: + return json.dumps( + { + "id": identity, + "object": "chat.completion", + "created": 1, + "model": model, + "choices": [{"index": 0, "message": {"role": "assistant", "content": "routed"}, "finish_reason": "stop"}], + "usage": { + "prompt_tokens": prompt_tokens, + "completion_tokens": completion_tokens, + "total_tokens": prompt_tokens + completion_tokens, + }, + } + ).encode() def _provider_body(request: Request, target: str) -> dict[str, JsonValue]: @@ -82,3 +136,64 @@ def test_fireworks_router_slug_text_completion_sends_router_resource_not_models_ payload: Final = _JSON_OBJECT.validate_json(response.content) assert payload["choices"] == [{"index": 0, "text": "routed", "finish_reason": "stop", "logprobs": None}] assert [(request.method, request.target) for request in wire.drain()] == [("POST", "/completions")] + + +@pytest.mark.parametrize("slug", _FIREROUTER_SLUGS) +def test_fireworks_firerouter_short_name_sends_router_resource_not_models_path(gateway: Gateway, slug: str) -> None: + resource: Final = f"accounts/fireworks/routers/{slug}" + + def respond(request: Request) -> Reply: + body: Final = _provider_body(request, "/chat/completions") + assert body["model"] == resource, body + return Reply(body=_chat_completion(f"fw-{slug}", resource, 5, 1)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model=f"fireworks_ai/{slug}", api_base=wire.url, api_key=_API_KEY) + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": _PROMPT}]}, + ) + assert response.status_code == 200, response.text + payload: Final = _JSON_OBJECT.validate_json(response.content) + assert payload["choices"] == [ + {"finish_reason": "stop", "index": 0, "message": {"role": "assistant", "content": "routed"}} + ] + assert [(request.method, request.target) for request in wire.drain()] == [("POST", "/chat/completions")] + + +def test_fireworks_firerouter_claude_leg_is_charged_at_the_routed_models_own_rate(gateway: Gateway) -> None: + identity: Final = f"fw-firerouter-claude-{uuid.uuid4().hex}" + + def respond(request: Request) -> Reply: + body: Final = _provider_body(request, "/chat/completions") + assert body["model"] == "accounts/fireworks/routers/firerouter", body + assert request.headers["x-anthropic-api-key"] == "synthetic-anthropic-key" + return Reply(body=_chat_completion(identity, _ROUTED_MODEL, 23, 41)) + + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model="fireworks_ai/firerouter", + api_base=wire.url, + api_key=_API_KEY, + extra_headers={"x-anthropic-api-key": "synthetic-anthropic-key"}, + ) + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": _PROMPT}]}, + ) + assert response.status_code == 200, response.text + expected_cost: Final = 23 * _catalog_cost(_ROUTED_MODEL, "input_cost_per_token") + 41 * _catalog_cost( + _ROUTED_MODEL, "output_cost_per_token" + ) + assert expected_cost > 0 + assert float(response.headers["x-litellm-response-cost"]) == _approx(expected_cost) + rows: Final = eventually( + lambda: read_rows('SELECT spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (identity,)), + lambda values: len(values) == 1, + seconds=70, + ) + spend: Final = rows[0]["spend"] + assert isinstance(spend, (int, float, str)) + assert float(spend) == _approx(expected_cost) diff --git a/tests/unit/llms/fireworks_ai/test_fireworks_ai_common_utils.py b/tests/unit/llms/fireworks_ai/test_fireworks_ai_common_utils.py index 16226a3ce74..e505f2ae8a6 100644 --- a/tests/unit/llms/fireworks_ai/test_fireworks_ai_common_utils.py +++ b/tests/unit/llms/fireworks_ai/test_fireworks_ai_common_utils.py @@ -1,7 +1,5 @@ - import pytest - from litellm.llms.fireworks_ai.common_utils import resolve_fireworks_resource_name @@ -16,6 +14,10 @@ from litellm.llms.fireworks_ai.common_utils import resolve_fireworks_resource_na ("glm-4p6", "accounts/fireworks/models/glm-4p6"), ("fireworks_ai/glm-4p6", "accounts/fireworks/models/glm-4p6"), ("kimi-k2p6-fast", "accounts/fireworks/routers/kimi-k2p6-fast"), + ("firerouter", "accounts/fireworks/routers/firerouter"), + ("fireworks_ai/firerouter", "accounts/fireworks/routers/firerouter"), + ("firerouter/kimi-k3/deepseek-v4", "accounts/fireworks/routers/firerouter/kimi-k3/deepseek-v4"), + ("firerouter-v2", "accounts/fireworks/models/firerouter-v2"), ( "accounts/fireworks/routers/glm-latest", "accounts/fireworks/routers/glm-latest", diff --git a/tests/unit/llms/fireworks_ai/test_fireworks_ai_cost_calculator.py b/tests/unit/llms/fireworks_ai/test_fireworks_ai_cost_calculator.py index c6096ba2745..c3aec979aa6 100644 --- a/tests/unit/llms/fireworks_ai/test_fireworks_ai_cost_calculator.py +++ b/tests/unit/llms/fireworks_ai/test_fireworks_ai_cost_calculator.py @@ -1,4 +1,5 @@ import math +import re from collections.abc import Generator from datetime import datetime, timezone from typing import Final @@ -326,3 +327,71 @@ def test_an_entry_without_an_input_rate_gets_no_cache_read_fallback(): assert prompt_cost == 0 assert completion_cost == 200 * 2e-06 + + +ROUTED_MODEL: Final = next( + key + for key, info in litellm.model_cost.items() + if "/" not in key + and info.get("litellm_provider") == "anthropic" + and (info.get("input_cost_per_token") or 0) > 0 + and (info.get("output_cost_per_token") or 0) > 0 + and f"fireworks_ai/{key}" not in litellm.model_cost +) + + +@pytest.mark.parametrize("model", [ROUTED_MODEL, f"fireworks_ai/{ROUTED_MODEL}"]) +def test_a_model_routed_to_another_provider_is_billed_at_that_models_own_rates(model: str): + own_rates: Final = litellm.get_model_info(model=ROUTED_MODEL, custom_llm_provider="anthropic") + usage: Final = _usage(prompt_tokens=23, cached_tokens=0, completion_tokens=41) + + prompt_cost, completion_cost = cost_per_token(model=model, usage=usage) + + assert prompt_cost == pytest.approx(23 * own_rates["input_cost_per_token"]) + assert completion_cost == pytest.approx(41 * own_rates["output_cost_per_token"]) + assert prompt_cost > 0 and completion_cost > 0 + + +def test_an_unknown_fireworks_model_still_falls_back_to_the_parameter_size_bucket(): + prompt_cost, completion_cost = cost_per_token( + model="accounts/fireworks/models/not-in-the-map-13b", + usage=_usage(prompt_tokens=100, cached_tokens=0, completion_tokens=10), + ) + bucket_prompt_cost, bucket_completion_cost = cost_per_token( + model="fireworks-ai-4.1b-to-16b", usage=_usage(prompt_tokens=100, cached_tokens=0, completion_tokens=10) + ) + + assert (prompt_cost, completion_cost) == (bucket_prompt_cost, bucket_completion_cost) + assert prompt_cost > 0 + + +_TIERED_INPUT_PATTERN: Final = re.compile(r"^input_cost_per_token_above_(\d+)k_tokens$") + + +def _threshold_tokens(field: str) -> int: + match: Final = _TIERED_INPUT_PATTERN.match(field) + assert match is not None, field + return int(match.group(1)) * 1000 + + +def test_a_routed_xai_model_keeps_xais_inclusive_token_threshold(): + candidate: Final = next( + ( + (key, field) + for key, info in litellm.model_cost.items() + if info.get("litellm_provider") == "xai" and f"fireworks_ai/{key}" not in litellm.model_cost + for field in info + if _TIERED_INPUT_PATTERN.match(field) + ), + None, + ) + if candidate is None: + pytest.skip("cost map has no xai entry with a tiered input rate") + key, field = candidate + usage: Final = _usage(prompt_tokens=_threshold_tokens(field), cached_tokens=0, completion_tokens=10) + + routed_prompt_cost, routed_completion_cost = cost_per_token(model=f"fireworks_ai/{key}", usage=usage) + + assert (routed_prompt_cost, routed_completion_cost) == generic_cost_per_token( + model=key, usage=usage, custom_llm_provider="xai" + ) From 1f1817ed399cc2129d7503cd504c1b2acba48dbc Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:00:33 -0700 Subject: [PATCH 27/96] fix(models): add azure gpt-4o-mini-transcribe and gpt-4o-mini-tts deprecation dates (#42873) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 2 ++ model_prices_and_context_window.json | 2 ++ 2 files changed, 4 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 18c8578cec8..15807127f13 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -6054,6 +6054,7 @@ "supports_tool_choice": true }, "azure/gpt-4o-mini-transcribe": { + "deprecation_date": "2027-06-15", "input_cost_per_audio_token": 1.25e-06, "input_cost_per_token": 1.25e-06, "litellm_provider": "azure", @@ -6066,6 +6067,7 @@ ] }, "azure/gpt-4o-mini-tts": { + "deprecation_date": "2027-06-15", "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", "mode": "audio_speech", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 18c8578cec8..15807127f13 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -6054,6 +6054,7 @@ "supports_tool_choice": true }, "azure/gpt-4o-mini-transcribe": { + "deprecation_date": "2027-06-15", "input_cost_per_audio_token": 1.25e-06, "input_cost_per_token": 1.25e-06, "litellm_provider": "azure", @@ -6066,6 +6067,7 @@ ] }, "azure/gpt-4o-mini-tts": { + "deprecation_date": "2027-06-15", "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", "mode": "audio_speech", From 871f562f7288a92d91e9e98eab40be69b9f306bb Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:01:10 -0700 Subject: [PATCH 28/96] fix(models): add openai deprecation date for gpt-5-chat-latest and gpt-5-chat (#42872) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 2 ++ model_prices_and_context_window.json | 2 ++ 2 files changed, 4 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 15807127f13..e3fda66723f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -34151,6 +34151,7 @@ }, "gpt-5-chat": { "cache_read_input_token_cost": 1.25e-07, + "deprecation_date": "2026-07-23", "input_cost_per_token": 1.25e-06, "litellm_provider": "openai", "max_input_tokens": 128000, @@ -34186,6 +34187,7 @@ }, "gpt-5-chat-latest": { "cache_read_input_token_cost": 1.25e-07, + "deprecation_date": "2026-07-23", "input_cost_per_token": 1.25e-06, "litellm_provider": "openai", "max_input_tokens": 128000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 15807127f13..e3fda66723f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -34151,6 +34151,7 @@ }, "gpt-5-chat": { "cache_read_input_token_cost": 1.25e-07, + "deprecation_date": "2026-07-23", "input_cost_per_token": 1.25e-06, "litellm_provider": "openai", "max_input_tokens": 128000, @@ -34186,6 +34187,7 @@ }, "gpt-5-chat-latest": { "cache_read_input_token_cost": 1.25e-07, + "deprecation_date": "2026-07-23", "input_cost_per_token": 1.25e-06, "litellm_provider": "openai", "max_input_tokens": 128000, From 15a2bd8b28fe88440e5f4f2ff79723e8363f61e9 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:07:44 -0700 Subject: [PATCH 29/96] fix(models): add fireworks 2026-09-25 deprecation dates for glm 5.2, kimi k2.6, kimi k2.7 code, deepseek v4 and muse glimmer serverless rows (#42874) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../model_prices_and_context_window_backup.json | 14 ++++++++++++++ model_prices_and_context_window.json | 14 ++++++++++++++ 2 files changed, 28 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e3fda66723f..9298900f1bc 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -24588,6 +24588,7 @@ "fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813": { "cache_read_input_token_cost": 4.4e-08, "cache_read_input_token_cost_priority": 5.5e-08, + "deprecation_date": "2026-09-25", "input_cost_per_token": 1.32e-06, "input_cost_per_token_priority": 1.65e-06, "litellm_provider": "fireworks_ai", @@ -24607,6 +24608,7 @@ "fireworks_ai/deepseek-v4-pro-0813": { "cache_read_input_token_cost": 4.4e-08, "cache_read_input_token_cost_priority": 5.5e-08, + "deprecation_date": "2026-09-25", "input_cost_per_token": 1.32e-06, "input_cost_per_token_priority": 1.65e-06, "litellm_provider": "fireworks_ai", @@ -24712,6 +24714,7 @@ "fireworks_ai/accounts/fireworks/models/glm-5p2": { "cache_read_input_token_cost": 1.4e-07, "cache_read_input_token_cost_priority": 1.75e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 1.4e-06, "input_cost_per_token_priority": 1.75e-06, "litellm_provider": "fireworks_ai", @@ -24820,6 +24823,7 @@ "fireworks_ai/accounts/fireworks/models/kimi-k2p6": { "cache_read_input_token_cost": 1.6e-07, "cache_read_input_token_cost_priority": 2.2e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 9.5e-07, "input_cost_per_token_priority": 1.5e-06, "litellm_provider": "fireworks_ai", @@ -24839,6 +24843,7 @@ "fireworks_ai/accounts/fireworks/models/kimi-k2p7-code": { "cache_read_input_token_cost": 1.9e-07, "cache_read_input_token_cost_priority": 2.85e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 9.5e-07, "input_cost_per_token_priority": 1.425e-06, "litellm_provider": "fireworks_ai", @@ -25109,6 +25114,7 @@ "fireworks_ai/glm-5p2": { "cache_read_input_token_cost": 1.4e-07, "cache_read_input_token_cost_priority": 1.75e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 1.4e-06, "input_cost_per_token_priority": 1.75e-06, "litellm_provider": "fireworks_ai", @@ -25177,6 +25183,7 @@ "fireworks_ai/kimi-k2p6": { "cache_read_input_token_cost": 1.6e-07, "cache_read_input_token_cost_priority": 2.2e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 9.5e-07, "input_cost_per_token_priority": 1.5e-06, "litellm_provider": "fireworks_ai", @@ -25213,6 +25220,7 @@ "fireworks_ai/kimi-k2p7-code": { "cache_read_input_token_cost": 1.9e-07, "cache_read_input_token_cost_priority": 2.85e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 9.5e-07, "input_cost_per_token_priority": 1.425e-06, "litellm_provider": "fireworks_ai", @@ -59357,6 +59365,7 @@ "fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731": { "cache_read_input_token_cost": 7e-09, "cache_read_input_token_cost_priority": 8.75e-09, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.2e-07, "input_cost_per_token_priority": 2.75e-07, "litellm_provider": "fireworks_ai", @@ -59395,6 +59404,7 @@ }, "fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-vision-exp": { "cache_read_input_token_cost": 7e-09, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.2e-07, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, @@ -59434,6 +59444,7 @@ "fireworks_ai/deepseek-v4-flash-0731": { "cache_read_input_token_cost": 7e-09, "cache_read_input_token_cost_priority": 8.75e-09, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.2e-07, "input_cost_per_token_priority": 2.75e-07, "litellm_provider": "fireworks_ai", @@ -59472,6 +59483,7 @@ }, "fireworks_ai/deepseek-v4-flash-vision-exp": { "cache_read_input_token_cost": 7e-09, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.2e-07, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, @@ -59603,6 +59615,7 @@ }, "fireworks_ai/muse-glimmer-30b": { "cache_read_input_token_cost": 4e-08, + "deprecation_date": "2026-09-25", "input_cost_per_token": 3.5e-07, "litellm_provider": "fireworks_ai", "max_input_tokens": 131072, @@ -59651,6 +59664,7 @@ }, "fireworks_ai/accounts/fireworks/models/muse-glimmer-30b": { "cache_read_input_token_cost": 4e-08, + "deprecation_date": "2026-09-25", "input_cost_per_token": 3.5e-07, "litellm_provider": "fireworks_ai", "max_input_tokens": 131072, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e3fda66723f..9298900f1bc 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -24588,6 +24588,7 @@ "fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813": { "cache_read_input_token_cost": 4.4e-08, "cache_read_input_token_cost_priority": 5.5e-08, + "deprecation_date": "2026-09-25", "input_cost_per_token": 1.32e-06, "input_cost_per_token_priority": 1.65e-06, "litellm_provider": "fireworks_ai", @@ -24607,6 +24608,7 @@ "fireworks_ai/deepseek-v4-pro-0813": { "cache_read_input_token_cost": 4.4e-08, "cache_read_input_token_cost_priority": 5.5e-08, + "deprecation_date": "2026-09-25", "input_cost_per_token": 1.32e-06, "input_cost_per_token_priority": 1.65e-06, "litellm_provider": "fireworks_ai", @@ -24712,6 +24714,7 @@ "fireworks_ai/accounts/fireworks/models/glm-5p2": { "cache_read_input_token_cost": 1.4e-07, "cache_read_input_token_cost_priority": 1.75e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 1.4e-06, "input_cost_per_token_priority": 1.75e-06, "litellm_provider": "fireworks_ai", @@ -24820,6 +24823,7 @@ "fireworks_ai/accounts/fireworks/models/kimi-k2p6": { "cache_read_input_token_cost": 1.6e-07, "cache_read_input_token_cost_priority": 2.2e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 9.5e-07, "input_cost_per_token_priority": 1.5e-06, "litellm_provider": "fireworks_ai", @@ -24839,6 +24843,7 @@ "fireworks_ai/accounts/fireworks/models/kimi-k2p7-code": { "cache_read_input_token_cost": 1.9e-07, "cache_read_input_token_cost_priority": 2.85e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 9.5e-07, "input_cost_per_token_priority": 1.425e-06, "litellm_provider": "fireworks_ai", @@ -25109,6 +25114,7 @@ "fireworks_ai/glm-5p2": { "cache_read_input_token_cost": 1.4e-07, "cache_read_input_token_cost_priority": 1.75e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 1.4e-06, "input_cost_per_token_priority": 1.75e-06, "litellm_provider": "fireworks_ai", @@ -25177,6 +25183,7 @@ "fireworks_ai/kimi-k2p6": { "cache_read_input_token_cost": 1.6e-07, "cache_read_input_token_cost_priority": 2.2e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 9.5e-07, "input_cost_per_token_priority": 1.5e-06, "litellm_provider": "fireworks_ai", @@ -25213,6 +25220,7 @@ "fireworks_ai/kimi-k2p7-code": { "cache_read_input_token_cost": 1.9e-07, "cache_read_input_token_cost_priority": 2.85e-07, + "deprecation_date": "2026-09-25", "input_cost_per_token": 9.5e-07, "input_cost_per_token_priority": 1.425e-06, "litellm_provider": "fireworks_ai", @@ -59357,6 +59365,7 @@ "fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731": { "cache_read_input_token_cost": 7e-09, "cache_read_input_token_cost_priority": 8.75e-09, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.2e-07, "input_cost_per_token_priority": 2.75e-07, "litellm_provider": "fireworks_ai", @@ -59395,6 +59404,7 @@ }, "fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-vision-exp": { "cache_read_input_token_cost": 7e-09, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.2e-07, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, @@ -59434,6 +59444,7 @@ "fireworks_ai/deepseek-v4-flash-0731": { "cache_read_input_token_cost": 7e-09, "cache_read_input_token_cost_priority": 8.75e-09, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.2e-07, "input_cost_per_token_priority": 2.75e-07, "litellm_provider": "fireworks_ai", @@ -59472,6 +59483,7 @@ }, "fireworks_ai/deepseek-v4-flash-vision-exp": { "cache_read_input_token_cost": 7e-09, + "deprecation_date": "2026-09-25", "input_cost_per_token": 2.2e-07, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, @@ -59603,6 +59615,7 @@ }, "fireworks_ai/muse-glimmer-30b": { "cache_read_input_token_cost": 4e-08, + "deprecation_date": "2026-09-25", "input_cost_per_token": 3.5e-07, "litellm_provider": "fireworks_ai", "max_input_tokens": 131072, @@ -59651,6 +59664,7 @@ }, "fireworks_ai/accounts/fireworks/models/muse-glimmer-30b": { "cache_read_input_token_cost": 4e-08, + "deprecation_date": "2026-09-25", "input_cost_per_token": 3.5e-07, "litellm_provider": "fireworks_ai", "max_input_tokens": 131072, From 7370650d91bea7ecdd04dd7350a6cf4229441048 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:27:25 -0700 Subject: [PATCH 30/96] fix(model_prices): bedrock bare Claude ids priced at the Global SKU (aws-bedrock sync) (#42875) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 162 +++++++++--------- model_prices_and_context_window.json | 162 +++++++++--------- tests/test_litellm/test_utils.py | 10 +- 3 files changed, 167 insertions(+), 167 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 9298900f1bc..059c92f4b4b 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -812,18 +812,18 @@ "prompt_cache_min_tokens": 2048 }, "anthropic.claude-haiku-4-5-20251001-v1:0": { - "cache_creation_input_token_cost": 1.375e-06, - "cache_creation_input_token_cost_above_1hr": 2.2e-06, - "cache_read_input_token_cost": 1.1e-07, - "input_cost_per_token": 1.1e-06, + "cache_creation_input_token_cost": 1.25e-06, + "cache_creation_input_token_cost_above_1hr": 2e-06, + "cache_read_input_token_cost": 1e-07, + "input_cost_per_token": 1e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 5.5e-06, - "source": "https://aws.amazon.com/bedrock/pricing/", + "output_cost_per_token": 5e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "supports_assistant_prefill": true, "supports_computer_use": true, "supports_function_calling": true, @@ -836,8 +836,8 @@ "supports_native_structured_output": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 4096, - "input_cost_per_token_batches": 5.5e-07, - "output_cost_per_token_batches": 2.75e-06 + "input_cost_per_token_batches": 5e-07, + "output_cost_per_token_batches": 2.5e-06 }, "anthropic.claude-haiku-4-5@20251001": { "cache_creation_input_token_cost": 1.25e-06, @@ -1028,17 +1028,17 @@ "prompt_cache_min_tokens": 1024 }, "anthropic.claude-opus-4-5-20251101-v1:0": { - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1058,22 +1058,22 @@ "bedrock_output_config_effort_ceiling": "high", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 4096, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-opus-4-6-v1": { "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1094,7 +1094,7 @@ "bedrock_output_config_effort_ceiling": "max", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 4096, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-opus-4-6-v1": { "supports_adaptive_thinking": true, @@ -1243,17 +1243,17 @@ "anthropic.claude-opus-4-7": { "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1276,7 +1276,7 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 2048, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-mythos-preview": { "input_cost_per_token": 0, @@ -1447,16 +1447,16 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-fable-5": { - "cache_creation_input_token_cost": 1.375e-05, - "cache_creation_input_token_cost_above_1hr": 2.2e-05, - "cache_read_input_token_cost": 1.1e-06, - "input_cost_per_token": 1.1e-05, + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 5.5e-05, + "output_cost_per_token": 5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1482,19 +1482,19 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-fable-5-1": { - "cache_creation_input_token_cost": 1.375e-05, - "cache_creation_input_token_cost_above_1hr": 2.2e-05, - "cache_read_input_token_cost": 2.75e-07, - "input_cost_per_token": 1.1e-05, + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 2.5e-07, + "input_cost_per_token": 1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 5.5e-05, + "output_cost_per_token": 5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1521,7 +1521,7 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-fable-5": { "cache_creation_input_token_cost": 1.25e-05, @@ -1757,17 +1757,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1789,23 +1789,23 @@ "supports_output_config": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-opus-5-5": { "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 5.5e-06, - "cache_creation_input_token_cost_above_1hr": 8.8e-06, - "cache_read_input_token_cost": 2.2e-07, - "input_cost_per_token": 4.4e-06, + "cache_creation_input_token_cost": 5e-06, + "cache_creation_input_token_cost_above_1hr": 8e-06, + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 4e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.2e-05, + "output_cost_per_token": 2e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1827,7 +1827,7 @@ "supports_output_config": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/", + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "thinking_always_on": true, "supports_forced_tool_use": false }, @@ -2225,17 +2225,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2258,7 +2258,7 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-opus-4-8": { "bedrock_converse_supports_strict_tools": false, @@ -2494,17 +2494,17 @@ }, "anthropic.claude-sonnet-5": { "bedrock_converse_supports_strict_tools": false, - "cache_creation_input_token_cost": 2.75e-06, - "cache_creation_input_token_cost_above_1hr": 4.4e-06, - "cache_read_input_token_cost": 2.2e-07, - "input_cost_per_token": 2.2e-06, + "cache_creation_input_token_cost": 2.5e-06, + "cache_creation_input_token_cost_above_1hr": 4e-06, + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 2e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 1.1e-05, + "output_cost_per_token": 1e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2529,7 +2529,7 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-sonnet-5": { "bedrock_converse_supports_strict_tools": false, @@ -2729,17 +2729,17 @@ "anthropic.claude-sonnet-4-6": { "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_1hr": 6.6e-06, - "cache_read_input_token_cost": 3.3e-07, - "input_cost_per_token": 3.3e-06, + "cache_creation_input_token_cost": 3.75e-06, + "cache_creation_input_token_cost_above_1hr": 6e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 1.65e-05, + "output_cost_per_token": 1.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2759,7 +2759,7 @@ "supports_output_config": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-sonnet-4-6": { "supports_adaptive_thinking": true, @@ -2970,22 +2970,22 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-sonnet-4-5-20250929-v1:0": { - "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_1hr": 6.6e-06, - "cache_read_input_token_cost": 3.3e-07, - "input_cost_per_token": 3.3e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, - "output_cost_per_token_above_200k_tokens": 2.475e-05, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, - "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.32e-05, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, + "cache_creation_input_token_cost": 3.75e-06, + "cache_creation_input_token_cost_above_1hr": 6e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "input_cost_per_token_above_200k_tokens": 6e-06, + "output_cost_per_token_above_200k_tokens": 2.25e-05, + "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, + "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.2e-05, + "cache_read_input_token_cost_above_200k_tokens": 6e-07, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 1.65e-05, + "output_cost_per_token": 1.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -3003,9 +3003,9 @@ "supports_native_structured_output": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "input_cost_per_token_batches": 1.65e-06, - "output_cost_per_token_batches": 8.25e-06, - "source": "https://aws.amazon.com/bedrock/pricing/" + "input_cost_per_token_batches": 1.5e-06, + "output_cost_per_token_batches": 7.5e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-v1": { "input_cost_per_token": 8e-06, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 9298900f1bc..059c92f4b4b 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -812,18 +812,18 @@ "prompt_cache_min_tokens": 2048 }, "anthropic.claude-haiku-4-5-20251001-v1:0": { - "cache_creation_input_token_cost": 1.375e-06, - "cache_creation_input_token_cost_above_1hr": 2.2e-06, - "cache_read_input_token_cost": 1.1e-07, - "input_cost_per_token": 1.1e-06, + "cache_creation_input_token_cost": 1.25e-06, + "cache_creation_input_token_cost_above_1hr": 2e-06, + "cache_read_input_token_cost": 1e-07, + "input_cost_per_token": 1e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 5.5e-06, - "source": "https://aws.amazon.com/bedrock/pricing/", + "output_cost_per_token": 5e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "supports_assistant_prefill": true, "supports_computer_use": true, "supports_function_calling": true, @@ -836,8 +836,8 @@ "supports_native_structured_output": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 4096, - "input_cost_per_token_batches": 5.5e-07, - "output_cost_per_token_batches": 2.75e-06 + "input_cost_per_token_batches": 5e-07, + "output_cost_per_token_batches": 2.5e-06 }, "anthropic.claude-haiku-4-5@20251001": { "cache_creation_input_token_cost": 1.25e-06, @@ -1028,17 +1028,17 @@ "prompt_cache_min_tokens": 1024 }, "anthropic.claude-opus-4-5-20251101-v1:0": { - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1058,22 +1058,22 @@ "bedrock_output_config_effort_ceiling": "high", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 4096, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-opus-4-6-v1": { "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1094,7 +1094,7 @@ "bedrock_output_config_effort_ceiling": "max", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 4096, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-opus-4-6-v1": { "supports_adaptive_thinking": true, @@ -1243,17 +1243,17 @@ "anthropic.claude-opus-4-7": { "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1276,7 +1276,7 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 2048, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-mythos-preview": { "input_cost_per_token": 0, @@ -1447,16 +1447,16 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-fable-5": { - "cache_creation_input_token_cost": 1.375e-05, - "cache_creation_input_token_cost_above_1hr": 2.2e-05, - "cache_read_input_token_cost": 1.1e-06, - "input_cost_per_token": 1.1e-05, + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 1e-06, + "input_cost_per_token": 1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 5.5e-05, + "output_cost_per_token": 5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1482,19 +1482,19 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-fable-5-1": { - "cache_creation_input_token_cost": 1.375e-05, - "cache_creation_input_token_cost_above_1hr": 2.2e-05, - "cache_read_input_token_cost": 2.75e-07, - "input_cost_per_token": 1.1e-05, + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_1hr": 2e-05, + "cache_read_input_token_cost": 2.5e-07, + "input_cost_per_token": 1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 5.5e-05, + "output_cost_per_token": 5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1521,7 +1521,7 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-fable-5": { "cache_creation_input_token_cost": 1.25e-05, @@ -1757,17 +1757,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1789,23 +1789,23 @@ "supports_output_config": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-opus-5-5": { "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 5.5e-06, - "cache_creation_input_token_cost_above_1hr": 8.8e-06, - "cache_read_input_token_cost": 2.2e-07, - "input_cost_per_token": 4.4e-06, + "cache_creation_input_token_cost": 5e-06, + "cache_creation_input_token_cost_above_1hr": 8e-06, + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 4e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.2e-05, + "output_cost_per_token": 2e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1827,7 +1827,7 @@ "supports_output_config": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/", + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "thinking_always_on": true, "supports_forced_tool_use": false }, @@ -2225,17 +2225,17 @@ "bedrock_converse_supports_strict_tools": false, "supports_adaptive_thinking": true, "supports_mid_conversation_system": true, - "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_1hr": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token": 5.5e-06, + "cache_creation_input_token_cost": 6.25e-06, + "cache_creation_input_token_cost_above_1hr": 1e-05, + "cache_read_input_token_cost": 5e-07, + "input_cost_per_token": 5e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 2.75e-05, + "output_cost_per_token": 2.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2258,7 +2258,7 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-opus-4-8": { "bedrock_converse_supports_strict_tools": false, @@ -2494,17 +2494,17 @@ }, "anthropic.claude-sonnet-5": { "bedrock_converse_supports_strict_tools": false, - "cache_creation_input_token_cost": 2.75e-06, - "cache_creation_input_token_cost_above_1hr": 4.4e-06, - "cache_read_input_token_cost": 2.2e-07, - "input_cost_per_token": 2.2e-06, + "cache_creation_input_token_cost": 2.5e-06, + "cache_creation_input_token_cost_above_1hr": 4e-06, + "cache_read_input_token_cost": 2e-07, + "input_cost_per_token": 2e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", - "output_cost_per_token": 1.1e-05, + "output_cost_per_token": 1e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2529,7 +2529,7 @@ "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-sonnet-5": { "bedrock_converse_supports_strict_tools": false, @@ -2729,17 +2729,17 @@ "anthropic.claude-sonnet-4-6": { "supports_adaptive_thinking": true, "supports_legacy_thinking": true, - "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_1hr": 6.6e-06, - "cache_read_input_token_cost": 3.3e-07, - "input_cost_per_token": 3.3e-06, + "cache_creation_input_token_cost": 3.75e-06, + "cache_creation_input_token_cost_above_1hr": 6e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 1.65e-05, + "output_cost_per_token": 1.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -2759,7 +2759,7 @@ "supports_output_config": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-sonnet-4-6": { "supports_adaptive_thinking": true, @@ -2970,22 +2970,22 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "anthropic.claude-sonnet-4-5-20250929-v1:0": { - "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_1hr": 6.6e-06, - "cache_read_input_token_cost": 3.3e-07, - "input_cost_per_token": 3.3e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, - "output_cost_per_token_above_200k_tokens": 2.475e-05, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, - "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.32e-05, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, + "cache_creation_input_token_cost": 3.75e-06, + "cache_creation_input_token_cost_above_1hr": 6e-06, + "cache_read_input_token_cost": 3e-07, + "input_cost_per_token": 3e-06, + "input_cost_per_token_above_200k_tokens": 6e-06, + "output_cost_per_token_above_200k_tokens": 2.25e-05, + "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, + "cache_creation_input_token_cost_above_1hr_above_200k_tokens": 1.2e-05, + "cache_read_input_token_cost_above_200k_tokens": 6e-07, "litellm_provider": "bedrock_converse", "supports_tool_search": true, "max_input_tokens": 200000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", - "output_cost_per_token": 1.65e-05, + "output_cost_per_token": 1.5e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -3003,9 +3003,9 @@ "supports_native_structured_output": true, "supports_parallel_tool_use_config": true, "prompt_cache_min_tokens": 1024, - "input_cost_per_token_batches": 1.65e-06, - "output_cost_per_token_batches": 8.25e-06, - "source": "https://aws.amazon.com/bedrock/pricing/" + "input_cost_per_token_batches": 1.5e-06, + "output_cost_per_token_batches": 7.5e-06, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "anthropic.claude-v1": { "input_cost_per_token": 8e-06, diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 3fca6935d57..5dc09db4535 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -1227,17 +1227,17 @@ def test_get_model_info_bedrock_regional_inference_profile_pricing(local_model_c "anthropic.claude-sonnet-5", ], ) -def test_bedrock_bare_claude_id_is_priced_in_region(local_model_cost_map, bare_key): - """A bare Bedrock Claude id is an in-region invocation, so it carries the same - in-region rate as its us. inference profile, not the cheaper global. rate.""" +def test_bedrock_bare_claude_id_is_priced_global(local_model_cost_map, bare_key): + """A bare Bedrock Claude id is billed at the Global SKU, so it carries the same + rate as its global. inference profile and sits below the regional us. rate.""" bare = litellm.model_cost[bare_key] us = litellm.model_cost[f"us.{bare_key}"] global_ = litellm.model_cost[f"global.{bare_key}"] cost_fields = [f for f in bare if "cost" in f] assert cost_fields for field in cost_fields: - assert bare[field] == us[field], field - assert bare["input_cost_per_token"] > global_["input_cost_per_token"] + assert bare[field] == global_[field], field + assert bare["input_cost_per_token"] < us["input_cost_per_token"] def test_get_model_info_bedrock_mantle_region_prefix_falls_back_to_the_mantle_row(local_model_cost_map): From e99c5d30b6238f417d0370b450bded6f7e403692 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:32:27 -0700 Subject: [PATCH 31/96] feat(cost-map): add retired azure gpt-5 chat and o1-preview data zone rows (#42878) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 249 ++++++++++++++++++ model_prices_and_context_window.json | 249 ++++++++++++++++++ 2 files changed, 498 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 059c92f4b4b..2c4de480f8b 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -6627,6 +6627,255 @@ "output_cost_per_token_priority": 2e-05, "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, + "azure/gpt-5-chat": { + "cache_read_input_token_cost": 1.25e-07, + "deprecation_date": "2026-05-13", + "input_cost_per_token": 1.25e-06, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_prompt_caching": true, + "supports_vision": true + }, + "azure/gpt-5.1-chat": { + "cache_read_input_token_cost": 1.25e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.25e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/gpt-5.2-chat": { + "cache_read_input_token_cost": 1.75e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.75e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.4e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/gpt-5.3-chat": { + "cache_read_input_token_cost": 1.75e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.75e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.4e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/us/gpt-5.1-chat": { + "cache_read_input_token_cost": 1.375e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.375e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.1e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/us/gpt-5.2-chat": { + "cache_read_input_token_cost": 1.925e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.925e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.54e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/us/gpt-5.3-chat": { + "cache_read_input_token_cost": 1.925e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.925e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.54e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/eu/gpt-5.1-chat": { + "cache_read_input_token_cost": 1.375e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.375e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.1e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/eu/gpt-5.2-chat": { + "cache_read_input_token_cost": 1.925e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.925e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.54e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/eu/gpt-5.3-chat": { + "cache_read_input_token_cost": 1.925e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.925e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.54e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/us/o1-preview": { + "cache_read_input_token_cost": 8.25e-06, + "deprecation_date": "2025-07-28", + "input_cost_per_token": 1.65e-05, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6.6e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_prompt_caching": true + }, + "azure/eu/o1-preview": { + "cache_read_input_token_cost": 8.25e-06, + "deprecation_date": "2025-07-28", + "input_cost_per_token": 1.65e-05, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6.6e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_prompt_caching": true + }, "azure/gpt-5.1-codex": { "deprecation_date": "2027-05-15", "cache_read_input_token_cost": 1.25e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 059c92f4b4b..2c4de480f8b 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -6627,6 +6627,255 @@ "output_cost_per_token_priority": 2e-05, "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, + "azure/gpt-5-chat": { + "cache_read_input_token_cost": 1.25e-07, + "deprecation_date": "2026-05-13", + "input_cost_per_token": 1.25e-06, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_prompt_caching": true, + "supports_vision": true + }, + "azure/gpt-5.1-chat": { + "cache_read_input_token_cost": 1.25e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.25e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/gpt-5.2-chat": { + "cache_read_input_token_cost": 1.75e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.75e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.4e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/gpt-5.3-chat": { + "cache_read_input_token_cost": 1.75e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.75e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.4e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/us/gpt-5.1-chat": { + "cache_read_input_token_cost": 1.375e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.375e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.1e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/us/gpt-5.2-chat": { + "cache_read_input_token_cost": 1.925e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.925e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.54e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/us/gpt-5.3-chat": { + "cache_read_input_token_cost": 1.925e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.925e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.54e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/eu/gpt-5.1-chat": { + "cache_read_input_token_cost": 1.375e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.375e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.1e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/eu/gpt-5.2-chat": { + "cache_read_input_token_cost": 1.925e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.925e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.54e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/eu/gpt-5.3-chat": { + "cache_read_input_token_cost": 1.925e-07, + "deprecation_date": "2026-06-29", + "input_cost_per_token": 1.925e-06, + "litellm_provider": "azure", + "max_input_tokens": 111616, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_token": 1.54e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/responses" + ], + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure/us/o1-preview": { + "cache_read_input_token_cost": 8.25e-06, + "deprecation_date": "2025-07-28", + "input_cost_per_token": 1.65e-05, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6.6e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_prompt_caching": true + }, + "azure/eu/o1-preview": { + "cache_read_input_token_cost": 8.25e-06, + "deprecation_date": "2025-07-28", + "input_cost_per_token": 1.65e-05, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "chat", + "output_cost_per_token": 6.6e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_prompt_caching": true + }, "azure/gpt-5.1-codex": { "deprecation_date": "2027-05-15", "cache_read_input_token_cost": 1.25e-07, From 9d12c217afec1e16b003b9015e88c010d2b2584f Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:42:23 -0700 Subject: [PATCH 32/96] fix(models): correct gemini robotics er 2 preview audio input price (#42877) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 4 ++-- model_prices_and_context_window.json | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 2c4de480f8b..bcf77bc16d4 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -27530,7 +27530,7 @@ "gemini/gemini-robotics-er-2-preview": { "cache_read_input_token_cost": 1e-07, "cache_read_input_token_cost_batches": 5e-08, - "input_cost_per_audio_token": 2e-06, + "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 1e-06, "input_cost_per_token_batches": 5e-07, "litellm_provider": "gemini", @@ -59160,7 +59160,7 @@ } }, "gemini/gemini-robotics-er-2-streaming-preview": { - "input_cost_per_audio_token": 2e-06, + "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 1e-06, "litellm_provider": "gemini", "mode": "chat", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 2c4de480f8b..bcf77bc16d4 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -27530,7 +27530,7 @@ "gemini/gemini-robotics-er-2-preview": { "cache_read_input_token_cost": 1e-07, "cache_read_input_token_cost_batches": 5e-08, - "input_cost_per_audio_token": 2e-06, + "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 1e-06, "input_cost_per_token_batches": 5e-07, "litellm_provider": "gemini", @@ -59160,7 +59160,7 @@ } }, "gemini/gemini-robotics-er-2-streaming-preview": { - "input_cost_per_audio_token": 2e-06, + "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 1e-06, "litellm_provider": "gemini", "mode": "chat", From b21b20ed13b0cb531bbf4ef56db263983f1511ce Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 21:55:55 -0700 Subject: [PATCH 33/96] feat(vertex_ai): add gemini-3.8-flash-cyber pricing (#42879) * feat(vertex_ai): add gemini-3.8-flash-cyber pricing Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(vertex_ai): mark gemini-3.8-flash-cyber minimal reasoning unsupported Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 92 +++++++++++++++++++ model_prices_and_context_window.json | 92 +++++++++++++++++++ 2 files changed, 184 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index bcf77bc16d4..66b72090631 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -27336,6 +27336,52 @@ "web_search_billing_unit": "per_query", "google_maps_grounding_cost_per_query": 0.014 }, + "vertex_ai/gemini-3.8-flash-cyber": { + "cache_read_input_token_cost": 1.5e-07, + "cache_read_input_token_cost_flex": 7.5e-08, + "cache_read_input_token_cost_priority": 2.7e-07, + "input_cost_per_token": 1.5e-06, + "input_cost_per_token_flex": 7.5e-07, + "input_cost_per_token_priority": 2.7e-06, + "litellm_provider": "vertex_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_reasoning_token": 7.5e-06, + "output_cost_per_token": 7.5e-06, + "output_cost_per_token_flex": 3.75e-06, + "output_cost_per_token_priority": 1.35e-05, + "regional_endpoint_uplift_multiplier": 1.1, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_input": true, + "supports_function_calling": false, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_minimal_reasoning_effort": false, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_url_context": false, + "supports_video_input": true, + "supports_vision": true, + "supports_web_search": false, + "supports_native_streaming": true + }, "vertex_ai/gemini-3.1-pro-preview": { "prompt_cache_min_tokens": 4096, "cache_read_input_token_cost": 2e-07, @@ -29508,6 +29554,52 @@ "web_search_billing_unit": "per_query", "google_maps_grounding_cost_per_query": 0.014 }, + "gemini-3.8-flash-cyber": { + "cache_read_input_token_cost": 1.5e-07, + "cache_read_input_token_cost_flex": 7.5e-08, + "cache_read_input_token_cost_priority": 2.7e-07, + "input_cost_per_token": 1.5e-06, + "input_cost_per_token_flex": 7.5e-07, + "input_cost_per_token_priority": 2.7e-06, + "litellm_provider": "vertex_ai-language-models", + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_reasoning_token": 7.5e-06, + "output_cost_per_token": 7.5e-06, + "output_cost_per_token_flex": 3.75e-06, + "output_cost_per_token_priority": 1.35e-05, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_output": false, + "supports_audio_input": true, + "supports_function_calling": false, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_minimal_reasoning_effort": false, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_url_context": false, + "supports_video_input": true, + "supports_vision": true, + "supports_web_search": false, + "supports_native_streaming": true + }, "gemini/gemini-2.5-pro-preview-tts": { "cache_read_input_token_cost": 1.25e-07, "input_cost_per_audio_token": 7e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index bcf77bc16d4..66b72090631 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -27336,6 +27336,52 @@ "web_search_billing_unit": "per_query", "google_maps_grounding_cost_per_query": 0.014 }, + "vertex_ai/gemini-3.8-flash-cyber": { + "cache_read_input_token_cost": 1.5e-07, + "cache_read_input_token_cost_flex": 7.5e-08, + "cache_read_input_token_cost_priority": 2.7e-07, + "input_cost_per_token": 1.5e-06, + "input_cost_per_token_flex": 7.5e-07, + "input_cost_per_token_priority": 2.7e-06, + "litellm_provider": "vertex_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_reasoning_token": 7.5e-06, + "output_cost_per_token": 7.5e-06, + "output_cost_per_token_flex": 3.75e-06, + "output_cost_per_token_priority": 1.35e-05, + "regional_endpoint_uplift_multiplier": 1.1, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_input": true, + "supports_function_calling": false, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_minimal_reasoning_effort": false, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_url_context": false, + "supports_video_input": true, + "supports_vision": true, + "supports_web_search": false, + "supports_native_streaming": true + }, "vertex_ai/gemini-3.1-pro-preview": { "prompt_cache_min_tokens": 4096, "cache_read_input_token_cost": 2e-07, @@ -29508,6 +29554,52 @@ "web_search_billing_unit": "per_query", "google_maps_grounding_cost_per_query": 0.014 }, + "gemini-3.8-flash-cyber": { + "cache_read_input_token_cost": 1.5e-07, + "cache_read_input_token_cost_flex": 7.5e-08, + "cache_read_input_token_cost_priority": 2.7e-07, + "input_cost_per_token": 1.5e-06, + "input_cost_per_token_flex": 7.5e-07, + "input_cost_per_token_priority": 2.7e-06, + "litellm_provider": "vertex_ai-language-models", + "max_input_tokens": 1048576, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_reasoning_token": 7.5e-06, + "output_cost_per_token": 7.5e-06, + "output_cost_per_token_flex": 3.75e-06, + "output_cost_per_token_priority": 1.35e-05, + "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/completions" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_output": false, + "supports_audio_input": true, + "supports_function_calling": false, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_minimal_reasoning_effort": false, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": false, + "supports_url_context": false, + "supports_video_input": true, + "supports_vision": true, + "supports_web_search": false, + "supports_native_streaming": true + }, "gemini/gemini-2.5-pro-preview-tts": { "cache_read_input_token_cost": 1.25e-07, "input_cost_per_audio_token": 7e-07, From b8154bcbc0907a99b916f07a4b05c0622c4965ee Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:08:23 -0700 Subject: [PATCH 34/96] fix(presidio): stream non-Anthropic raw SSE through the post_call hook unbuffered (#42777) * fix(presidio): stream non-Anthropic raw SSE through the post_call hook unbuffered Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(presidio): keep the pytest.raises block to a single await Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(presidio): move raw SSE format check into a helper to keep hook complexity flat Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(presidio): fold the raw SSE format check into the existing bytes branch to stay within the complexity budget Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(presidio): decide raw SSE stream shape on a complete first frame, not a transport fragment Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): cover presidio post_call streaming for native gemini passthrough and anthropic messages Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(presidio): cap first SSE frame coalescing at 64 KiB so an unterminated first event cannot buffer unbounded Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(presidio): name raw SSE passthrough in the skipped output masking warning Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../guardrails/guardrail_hooks/presidio.py | 54 +- .../test_presidio_streaming_output.py | 538 ++++++++++++++++++ .../guardrail_hooks/test_presidio.py | 143 ++++- 3 files changed, 730 insertions(+), 5 deletions(-) create mode 100644 tests/integration/observability/test_presidio_streaming_output.py diff --git a/litellm/proxy/guardrails/guardrail_hooks/presidio.py b/litellm/proxy/guardrails/guardrail_hooks/presidio.py index eb07c19a580..794bf08729e 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/presidio.py +++ b/litellm/proxy/guardrails/guardrail_hooks/presidio.py @@ -42,6 +42,7 @@ from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.guardrails.anthropic_sse import ( anthropic_sse_chunks_from_response, assemble_anthropic_sse_stream, + is_anthropic_sse_stream, model_response_text, ) from litellm.types.guardrails import ( @@ -93,6 +94,42 @@ def _json_escaped_len(text: str) -> int: return len(json.dumps(text).encode("utf-8")) - 2 # strip the surrounding quotes +_MAX_FIRST_SSE_FRAME_BYTES: Final = 64 * 1024 + + +def _holds_complete_sse_frame(raw: bytes) -> bool: + """Whether ``raw`` holds one blank-line terminated SSE event, or is too large to keep joining.""" + return b"\n\n" in raw or b"\r\n\r\n" in raw or len(raw) >= _MAX_FIRST_SSE_FRAME_BYTES + + +async def _coalesce_first_sse_frame(stream: AsyncIterator[object]) -> AsyncGenerator[object, None]: + """ + Join leading raw ``bytes`` chunks until they hold one complete SSE event, so + the stream shape is decided on a whole frame rather than a transport fragment. + Everything after that first frame is forwarded untouched. + """ + pending = b"" + try: + async for chunk in stream: + if not isinstance(chunk, bytes): + yield chunk + continue + pending += chunk + if _holds_complete_sse_frame(pending): + break + else: + if pending: + yield pending + return + except Exception: + if pending: + yield pending + raise + yield pending + async for chunk in stream: + yield chunk + + class _OPTIONAL_PresidioPIIMasking(CustomGuardrail): user_api_key_cache = None ad_hoc_recognizers: list[str] | None = None @@ -1356,7 +1393,7 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail): all_chunks: list[ModelResponseStream] = [] passthrough_due_to_unknown_stream_shape = False try: - stream: Final = response.__aiter__() + stream: Final = _coalesce_first_sse_frame(response.__aiter__()) async for chunk in stream: if isinstance(chunk, ModelResponseStream): if passthrough_due_to_unknown_stream_shape: @@ -1364,7 +1401,15 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail): else: all_chunks.append(chunk) elif isinstance(chunk, bytes): - if passthrough_due_to_unknown_stream_shape or all_chunks: + first_frame_is_anthropic = ( + not passthrough_due_to_unknown_stream_shape + and not all_chunks + and is_anthropic_sse_stream((chunk,)) + ) + if not first_frame_is_anthropic: + passthrough_due_to_unknown_stream_shape = ( + passthrough_due_to_unknown_stream_shape or not all_chunks + ) yield chunk continue for masked_chunk in await self._mask_anthropic_sse_stream(chunk, stream, request_data): @@ -1387,8 +1432,9 @@ class _OPTIONAL_PresidioPIIMasking(CustomGuardrail): yield chunk if passthrough_due_to_unknown_stream_shape: verbose_proxy_logger.warning( - "Presidio apply_to_output: streaming response contained unknown event objects " - "(e.g. /v1/responses events). Output PII masking was skipped for this response." + "Presidio apply_to_output: streaming response was not a parsed chat completion stream " + "(raw non-Anthropic SSE passthrough or /v1/responses events). " + "Output PII masking was skipped for this response." ) return if not all_chunks: diff --git a/tests/integration/observability/test_presidio_streaming_output.py b/tests/integration/observability/test_presidio_streaming_output.py new file mode 100644 index 00000000000..5bc0a46427b --- /dev/null +++ b/tests/integration/observability/test_presidio_streaming_output.py @@ -0,0 +1,538 @@ +import json +import re +import signal +import threading +import uuid +from collections.abc import Callable, Iterator, Mapping +from concurrent.futures import ThreadPoolExecutor +from contextlib import ExitStack, contextmanager +from dataclasses import dataclass +from pathlib import Path +from typing import Final + +import psutil +import yaml +from integration._support.client import Gateway, eventually +from integration._support.process import OwnedProxy, group_members, owned_proxy_process +from integration._support.wire import Reply, Request, Wire, wire_server +from openai import OpenAI +from pydantic import BaseModel + +PERSON: Final = "John Smith" +MASK: Final = "" +GEMINI_MODEL: Final = "gemini-2.5-flash" + + +def gemini_frame(text: str) -> bytes: + payload: Final = { + "candidates": [{"content": {"parts": [{"text": text}], "role": "model"}, "index": 0}], + "usageMetadata": {"promptTokenCount": 10, "candidatesTokenCount": 5, "totalTokenCount": 15}, + "modelVersion": GEMINI_MODEL, + } + return b"data: " + json.dumps(payload).encode() + b"\r\n\r\n" + + +class GeminiPart(BaseModel): + text: str + + +class GeminiContent(BaseModel): + parts: list[GeminiPart] + + +class GeminiCandidate(BaseModel): + content: GeminiContent + + +class GeminiFrame(BaseModel): + candidates: list[GeminiCandidate] + + +def data_payloads(raw: bytes) -> tuple[dict[str, object], ...]: + """JSON payload of each ``data:`` frame, whatever line ending the sender used.""" + return tuple(json.loads(line[len("data: ") :]) for line in raw.decode().splitlines() if line.startswith("data: ")) + + +def gemini_text(payload: Mapping[str, object]) -> str: + return GeminiFrame.model_validate(payload).candidates[0].content.parts[0].text + + +def gemini_texts(raw: bytes) -> tuple[str, ...]: + return tuple(gemini_text(payload) for payload in data_payloads(raw)) + + +def anthropic_frame(event_type: str, payload: dict[str, object]) -> bytes: + return f"event: {event_type}\ndata: {json.dumps(payload)}\n\n".encode() + + +def anthropic_stream(identity: str, text: str) -> tuple[bytes, ...]: + return ( + anthropic_frame( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [], + "stop_reason": None, + "usage": {"input_tokens": 11, "output_tokens": 0}, + }, + }, + ), + anthropic_frame( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + anthropic_frame( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": text}}, + ), + anthropic_frame("content_block_stop", {"type": "content_block_stop", "index": 0}), + anthropic_frame( + "message_delta", + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 4}, + }, + ), + anthropic_frame("message_stop", {"type": "message_stop"}), + ) + + +def openai_frame(identity: str, delta: dict[str, str], finish: str | None = None) -> bytes: + payload: Final = { + "id": identity, + "object": "chat.completion.chunk", + "created": 1, + "model": "gpt-4o-mini", + "choices": [{"index": 0, "delta": delta, "finish_reason": finish}], + } + return b"data: " + json.dumps(payload).encode() + b"\n\n" + + +def analyzer(request: Request) -> Reply: + assert request.target == "/analyze", request.target + text: Final = json.loads(request.body)["text"] + findings: Final = [ + {"entity_type": "PERSON", "start": match.start(), "end": match.end(), "score": 0.85} + for match in re.finditer(re.escape(PERSON), text) + ] + return Reply(body=json.dumps(findings).encode()) + + +def anonymizer(request: Request) -> Reply: + assert request.target == "/anonymize", request.target + body: Final = json.loads(request.body) + text: Final = body["text"] + items: Final = [ + {"entity_type": "PERSON", "start": item["start"], "end": item["end"], "operator": "replace", "text": MASK} + for item in body["analyzer_results"] + ] + return Reply(body=json.dumps({"text": text.replace(PERSON, MASK), "items": items}).encode()) + + +def broken(request: Request) -> Reply: + return Reply(status=500, body=b'{"error": "scripted outage"}') + + +@dataclass(frozen=True, slots=True) +class Received: + status: int + frames: tuple[bytes, ...] + + @property + def text(self) -> str: + return b"".join(self.frames).decode() + + +@dataclass(frozen=True, slots=True) +class Rig: + proxy: OwnedProxy + upstream: Wire + analyzer: Wire + anonymizer: Wire + guardrail: str + gemini: str + anthropic: str + openai: str + + @property + def gateway(self) -> Gateway: + return self.proxy.gateway + + def stream(self, path: str, body: dict[str, object] | None = None, *, key: str | None = None) -> Received: + with self.gateway.client.stream( + "POST", path, json=body, headers={"Authorization": f"Bearer {key or self.gateway.key}"} + ) as response: + return Received(response.status_code, tuple(response.iter_raw())) + + def gemini_path(self) -> str: + return f"/v1beta/models/{self.gemini}:streamGenerateContent?alt=sse" + + def gemini_body(self) -> dict[str, object]: + return {"contents": [{"role": "user", "parts": [{"text": "who designed it"}]}]} + + def messages_body(self, *, guardrails: tuple[str, ...] | None = None) -> dict[str, object]: + return { + "model": self.anthropic, + "max_tokens": 64, + "stream": True, + "messages": [{"role": "user", "content": "who designed it"}], + **({"guardrails": list(guardrails)} if guardrails is not None else {}), + } + + +def anthropic_text(received: Received) -> str: + events: Final = tuple( + json.loads(line.removeprefix("data: ")) for line in received.text.split("\n") if line.startswith("data: ") + ) + return "".join(event["delta"]["text"] for event in events if event.get("type") == "content_block_delta") + + +@contextmanager +def presidio_rig( + gateway: Gateway, + tmp_path: Path, + provider: Callable[[Request], Reply], + *, + analyze: Callable[[Request], Reply] = analyzer, + anonymize: Callable[[Request], Reply] = anonymizer, + default_on: bool = True, +) -> Iterator[Rig]: + guardrail: Final = "presidio" + uuid.uuid4().hex + with ExitStack() as stack: + upstream: Final = stack.enter_context(wire_server(provider)) + analyze_sink: Final = stack.enter_context(wire_server(analyze)) + anonymize_sink: Final = stack.enter_context(wire_server(anonymize)) + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["guardrails"] = [ + { + "guardrail_name": guardrail, + "litellm_params": { + "guardrail": "presidio", + "mode": "post_call", + "default_on": default_on, + "presidio_analyzer_api_base": analyze_sink.url, + "presidio_anonymizer_api_base": anonymize_sink.url, + "presidio_filter_scope": "output", + }, + } + ] + path: Final = tmp_path / f"{guardrail}.yaml" + path.write_text(yaml.safe_dump(config)) + proxy: Final = stack.enter_context(owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2)) + scenario: Final = stack.enter_context(proxy.gateway.scenario()) + yield Rig( + proxy=proxy, + upstream=upstream, + analyzer=analyze_sink, + anonymizer=anonymize_sink, + guardrail=guardrail, + gemini=scenario.model( + model=f"gemini/{GEMINI_MODEL}", api_base=upstream.url, api_key="synthetic-gemini-key" + ), + anthropic=scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=upstream.url, api_key="synthetic-anthropic-key" + ), + openai=scenario.model(model="openai/gpt-4o-mini", api_base=upstream.url + "/v1", api_key="synthetic-key"), + ) + + +def gemini_provider(reply: Reply) -> Callable[[Request], Reply]: + def provider(request: Request) -> Reply: + assert "streamGenerateContent" in request.target, request.target + return reply + + return provider + + +def test_native_gemini_first_frame_reaches_caller_before_upstream_sends_the_second( + gateway: Gateway, tmp_path: Path +) -> None: + gate: Final = threading.Event() + first: Final = gemini_frame("first ") + second: Final = gemini_frame("second ") + provider: Final = gemini_provider( + Reply(content_type="text/event-stream", chunks=(first, second), gate_after_first=gate) + ) + with presidio_rig(gateway, tmp_path, provider) as rig: + with rig.gateway.client.stream( + "POST", rig.gemini_path(), json=rig.gemini_body(), headers={"Authorization": f"Bearer {rig.gateway.key}"} + ) as response: + assert response.status_code == 200, response.read().decode() + chunks: Final = response.iter_raw() + arrived: Final = next(chunks) + assert gemini_texts(arrived) == ("first ",), f"first chunk while upstream is gated: {arrived!r}" + gate.set() + rest: Final = b"".join(chunks) + assert gemini_texts(rest) == ("second ",), rest + assert len(rig.upstream.drain()) == 1 + assert rig.analyzer.drain() == () and rig.anonymizer.drain() == () + + +def test_native_gemini_frames_received_before_upstream_abort_reach_caller(gateway: Gateway, tmp_path: Path) -> None: + frames: Final = (gemini_frame(f"chunk {index} from {PERSON}. ") for index in range(3)) + provider: Final = gemini_provider( + Reply(content_type="text/event-stream", chunks=tuple(frames), abort_after=2, pause_between_chunks=0.2) + ) + with presidio_rig(gateway, tmp_path, provider) as rig: + received: Final = rig.stream(rig.gemini_path(), rig.gemini_body()) + assert received.status == 200, received.text + *frames_before_abort, trailer = data_payloads(b"".join(received.frames)) + assert [gemini_text(frame) for frame in frames_before_abort] == [ + f"chunk 0 from {PERSON}. ", + f"chunk 1 from {PERSON}. ", + ], received.text + assert "candidates" not in trailer and json.dumps(trailer).count('"code": "500"') == 1, received.text + assert len(rig.upstream.drain()) == 1 + + +def test_native_gemini_first_frame_split_into_transport_fragments_streams_every_byte( + gateway: Gateway, tmp_path: Path +) -> None: + first: Final = gemini_frame(f"fragmented {PERSON}") + second: Final = gemini_frame("whole") + chunks: Final = (first[:7], first[7:19], first[19:], second) + provider: Final = gemini_provider(Reply(content_type="text/event-stream", chunks=chunks)) + with presidio_rig(gateway, tmp_path, provider) as rig: + received: Final = rig.stream(rig.gemini_path(), rig.gemini_body()) + assert received.status == 200, received.text + assert gemini_texts(b"".join(received.frames)) == (f"fragmented {PERSON}", "whole") + + +def test_native_gemini_non_json_frame_passes_through_unchanged(gateway: Gateway, tmp_path: Path) -> None: + frames: Final = (b"data: not json at all\r\n\r\n", gemini_frame("after")) + provider: Final = gemini_provider(Reply(content_type="text/event-stream", chunks=frames)) + with presidio_rig(gateway, tmp_path, provider) as rig: + received: Final = rig.stream(rig.gemini_path(), rig.gemini_body()) + assert received.status == 200, received.text + assert received.text.replace("\r\n", "\n") == b"".join(frames).decode().replace("\r\n", "\n") + + +def test_native_gemini_empty_stream_returns_200_with_no_body(gateway: Gateway, tmp_path: Path) -> None: + provider: Final = gemini_provider(Reply(content_type="text/event-stream", chunks=())) + with presidio_rig(gateway, tmp_path, provider) as rig: + received: Final = rig.stream(rig.gemini_path(), rig.gemini_body()) + assert received.status == 200, received.text + assert received.text == "" + + +def test_native_gemini_streams_while_presidio_analyzer_is_down(gateway: Gateway, tmp_path: Path) -> None: + frames: Final = (gemini_frame(f"{PERSON} one. "), gemini_frame("two.")) + provider: Final = gemini_provider(Reply(content_type="text/event-stream", chunks=frames)) + with presidio_rig(gateway, tmp_path, provider, analyze=broken) as rig: + received: Final = rig.stream(rig.gemini_path(), rig.gemini_body()) + assert received.status == 200, received.text + assert gemini_texts(b"".join(received.frames)) == (f"{PERSON} one. ", "two.") + assert rig.analyzer.drain() == () + + +def test_native_gemini_unauthenticated_request_is_rejected_before_upstream(gateway: Gateway, tmp_path: Path) -> None: + provider: Final = gemini_provider(Reply(content_type="text/event-stream", chunks=(gemini_frame("never"),))) + with presidio_rig(gateway, tmp_path, provider) as rig: + received: Final = rig.stream(rig.gemini_path(), rig.gemini_body(), key="sk-not-a-key") + assert received.status == 401, received.text + assert rig.upstream.drain() == () + + +def anthropic_provider(chunks: tuple[bytes, ...]) -> Callable[[Request], Reply]: + def provider(request: Request) -> Reply: + assert request.target == "/v1/messages", request.target + return Reply(content_type="text/event-stream", chunks=chunks) + + return provider + + +def test_anthropic_messages_stream_masks_person_in_text_delta(gateway: Gateway, tmp_path: Path) -> None: + identity: Final = "msg_" + uuid.uuid4().hex + provider: Final = anthropic_provider(anthropic_stream(identity, f"{PERSON} designed it.")) + with presidio_rig(gateway, tmp_path, provider) as rig: + received: Final = rig.stream("/v1/messages", rig.messages_body()) + assert received.status == 200, received.text + assert anthropic_text(received) == f"{MASK} designed it." + assert PERSON not in received.text + assert identity in received.text + analyzed: Final = rig.analyzer.drain() + anonymized: Final = rig.anonymizer.drain() + assert len(analyzed) == len(anonymized) == 1 + assert json.loads(analyzed[0].body)["text"] == f"{PERSON} designed it." + + +def test_anthropic_messages_first_frame_split_across_transport_chunks_is_still_masked( + gateway: Gateway, tmp_path: Path +) -> None: + identity: Final = "msg_" + uuid.uuid4().hex + whole: Final = anthropic_stream(identity, f"{PERSON} designed it.") + split_at: Final = whole[0].index(b'"message_') + len(b'"message_') + chunks: Final = (whole[0][:split_at], whole[0][split_at:], *whole[1:]) + with presidio_rig(gateway, tmp_path, anthropic_provider(chunks)) as rig: + received: Final = rig.stream("/v1/messages", rig.messages_body()) + assert received.status == 200, received.text + assert anthropic_text(received) == f"{MASK} designed it." + assert received.text.count("event: message_start") == 1 + + +def test_anthropic_messages_stream_fails_closed_when_analyzer_is_down(gateway: Gateway, tmp_path: Path) -> None: + identity: Final = "msg_" + uuid.uuid4().hex + provider: Final = anthropic_provider(anthropic_stream(identity, f"{PERSON} designed it.")) + with presidio_rig(gateway, tmp_path, provider, analyze=broken) as rig: + received: Final = rig.stream("/v1/messages", rig.messages_body()) + assert PERSON not in received.text, received.text + assert "Presidio analyzer" in received.text, received.text + assert rig.anonymizer.drain() == () + + +def test_anthropic_messages_per_request_guardrails_selects_masking(gateway: Gateway, tmp_path: Path) -> None: + identity: Final = "msg_" + uuid.uuid4().hex + provider: Final = anthropic_provider(anthropic_stream(identity, f"{PERSON} designed it.")) + with presidio_rig(gateway, tmp_path, provider, default_on=False) as rig: + unguarded: Final = rig.stream("/v1/messages", rig.messages_body()) + assert unguarded.status == 200, unguarded.text + assert anthropic_text(unguarded) == f"{PERSON} designed it." + assert rig.analyzer.drain() == () + guarded: Final = rig.stream("/v1/messages", rig.messages_body(guardrails=(rig.guardrail,))) + assert guarded.status == 200, guarded.text + assert anthropic_text(guarded) == f"{MASK} designed it." + assert len(rig.analyzer.drain()) == 1 + + +def openai_provider(identity: str) -> Callable[[Request], Reply]: + def provider(request: Request) -> Reply: + assert request.target == "/v1/chat/completions", request.target + if json.loads(request.body).get("stream"): + return Reply( + content_type="text/event-stream", + chunks=( + openai_frame(identity, {"role": "assistant", "content": ""}), + openai_frame(identity, {"content": f"{PERSON} designed"}), + openai_frame(identity, {"content": " it."}, "stop"), + b"data: [DONE]\n\n", + ), + ) + return Reply( + body=json.dumps( + { + "id": identity, + "object": "chat.completion", + "created": 1, + "model": "gpt-4o-mini", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": f"{PERSON} designed it."}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 11, "completion_tokens": 4, "total_tokens": 15}, + } + ).encode() + ) + + return provider + + +def test_chat_completions_openai_sdk_stream_and_non_stream_are_masked(gateway: Gateway, tmp_path: Path) -> None: + identity: Final = "chatcmpl-" + uuid.uuid4().hex + with presidio_rig(gateway, tmp_path, openai_provider(identity)) as rig: + client: Final = OpenAI(api_key=rig.gateway.key, base_url=f"{rig.gateway.client.base_url}/v1", max_retries=0) + streamed: Final = client.chat.completions.create( + model=rig.openai, messages=[{"role": "user", "content": "who designed it"}], stream=True + ) + pieces: Final = tuple( + chunk.choices[0].delta.content for chunk in streamed if chunk.choices and chunk.choices[0].delta.content + ) + assert "".join(pieces) == f"{MASK} designed it.", pieces + whole: Final = client.chat.completions.create( + model=rig.openai, messages=[{"role": "user", "content": "who designed it"}] + ) + assert whole.id == identity + assert whole.choices[0].message.content == f"{MASK} designed it." + assert len(rig.upstream.drain()) == 2 + assert len(rig.analyzer.drain()) == len(rig.anonymizer.drain()) == 2 + + +def test_mixed_burst_survives_anonymizer_outage_and_recovers(gateway: Gateway, tmp_path: Path) -> None: + outage: Final = threading.Event() + + def flaky_anonymizer(request: Request) -> Reply: + return Reply(status=503, body=b'{"error": "scripted outage"}') if outage.is_set() else anonymizer(request) + + def provider(request: Request) -> Reply: + if request.target == "/v1/messages": + identity: Final = "msg_" + json.loads(request.body)["messages"][0]["content"] + return Reply(content_type="text/event-stream", chunks=anthropic_stream(identity, f"{PERSON} designed it.")) + return Reply( + content_type="text/event-stream", + chunks=(gemini_frame(f"{PERSON} "), gemini_frame("designed it.")), + pause_between_chunks=0.05, + ) + + with presidio_rig(gateway, tmp_path, provider, anonymize=flaky_anonymizer) as rig: + + def gemini_call(index: int) -> tuple[str, str, int]: + received: Final = rig.stream(rig.gemini_path(), rig.gemini_body()) + return ( + "gemini", + f"g{index}", + received.status if gemini_texts(b"".join(received.frames)) == (f"{PERSON} ", "designed it.") else -1, + ) + + def anthropic_call(index: int) -> tuple[str, str, int]: + body: Final = {**rig.messages_body(), "messages": [{"role": "user", "content": f"a{index}"}]} + received: Final = rig.stream("/v1/messages", body) + leaked: Final = PERSON in received.text + return ("anthropic", f"a{index}", -1 if leaked else (1 if MASK in received.text else 0)) + + def phase(offset: int) -> tuple[tuple[str, str, int], ...]: + with ThreadPoolExecutor(max_workers=12) as pool: + futures: Final = tuple( + pool.submit(gemini_call if index % 2 == 0 else anthropic_call, offset + index) + for index in range(12) + ) + return tuple(future.result() for future in futures) + + healthy_before: Final = phase(0) + outage.set() + during: Final = phase(100) + outage.clear() + healthy_after: Final = phase(200) + + for name, results in (("before", healthy_before), ("during", during), ("after", healthy_after)): + assert all(status == 200 for kind, _, status in results if kind == "gemini"), (name, results) + assert all(status == 1 for kind, _, status in healthy_before + healthy_after if kind == "anthropic"), ( + healthy_before, + healthy_after, + ) + assert all(status == 0 for kind, _, status in during if kind == "anthropic"), during + identities: Final = tuple(identity for _, identity, _ in healthy_before + during + healthy_after) + assert len(identities) == len(set(identities)) == 36 + + +def test_native_gemini_keeps_streaming_after_one_worker_is_killed(gateway: Gateway, tmp_path: Path) -> None: + frames: Final = (gemini_frame("alive "), gemini_frame("still.")) + provider: Final = gemini_provider(Reply(content_type="text/event-stream", chunks=frames, pause_between_chunks=0.05)) + with presidio_rig(gateway, tmp_path, provider) as rig: + workers: Final = eventually( + lambda: tuple( + member for member in group_members(rig.proxy.process.pid) if member.pid != rig.proxy.process.pid + ), + lambda members: len(members) >= 2, + seconds=30, + ) + victim: Final = workers[0] + with ThreadPoolExecutor(max_workers=8) as pool: + futures: Final = tuple(pool.submit(rig.stream, rig.gemini_path(), rig.gemini_body()) for _ in range(8)) + victim.send_signal(signal.SIGKILL) + psutil.wait_procs((victim,), timeout=10) + first_wave: Final = tuple(future.result() for future in futures) + survivors: Final = tuple(received for received in first_wave if received.status == 200) + assert survivors, [received.text[:200] for received in first_wave] + assert all(gemini_texts(b"".join(received.frames)) == ("alive ", "still.") for received in survivors) + second_wave: Final = tuple(rig.stream(rig.gemini_path(), rig.gemini_body()) for _ in range(6)) + assert all(received.status == 200 for received in second_wave), [r.text[:200] for r in second_wave] + assert all(gemini_texts(b"".join(received.frames)) == ("alive ", "still.") for received in second_wave) + assert rig.proxy.process.poll() is None diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_presidio.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_presidio.py index 89e72debbf5..1b696669724 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_presidio.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_presidio.py @@ -2261,7 +2261,7 @@ async def test_apply_to_output_streaming_mixed_chunks_flushes_and_warns(): assert mock_logger.warning.call_count == 2 warning_messages = [call.args[0] for call in mock_logger.warning.call_args_list] assert any("mixed stream detected" in msg for msg in warning_messages) - assert any("unknown event objects" in msg for msg in warning_messages) + assert any("Output PII masking was skipped" in msg for msg in warning_messages) # --------------------------------------------------------------------------- @@ -2519,6 +2519,147 @@ async def test_apply_to_output_streaming_anthropic_sse_bytes_without_pii_are_for assert collected == byte_chunks +def _gemini_sse(text: str) -> bytes: + payload = {"candidates": [{"content": {"parts": [{"text": text}], "role": "model"}, "index": 0}]} + return f"data: {json.dumps(payload)}\n\n".encode() + + +@pytest.mark.asyncio +async def test_apply_to_output_streaming_gemini_sse_bytes_are_forwarded_incrementally_until_upstream_aborts(): + guardrail = _OPTIONAL_PresidioPIIMasking( + mock_testing=True, + apply_to_output=True, + mock_redacted_text={"text": ""}, + ) + frames = [_gemini_sse("Partial one from John Smith. "), _gemini_sse("Partial two. ")] + collected: list[object] = [] + + async def mock_stream(): + for frame in frames: + yield frame + raise ConnectionError("upstream closed mid-stream") + + async def collect() -> None: + async for chunk in guardrail.async_post_call_streaming_iterator_hook( + user_api_key_dict=UserAPIKeyAuth(api_key="test-key"), + response=mock_stream(), + request_data={}, + ): + collected.append(chunk) + + with pytest.raises(ConnectionError): + await collect() + + assert collected == frames + + +@pytest.mark.asyncio +async def test_apply_to_output_streaming_anthropic_first_frame_split_across_transport_chunks_is_still_masked(): + guardrail = _OPTIONAL_PresidioPIIMasking( + mock_testing=True, + apply_to_output=True, + mock_redacted_text={"text": ""}, + ) + message_start = _anthropic_sse( + "message_start", + {"type": "message_start", "message": {"id": "msg_1", "model": "claude", "content": [], "usage": {}}}, + ) + split_at = message_start.index(b'"message_') + len(b'"message_') + byte_chunks = [ + message_start[:split_at], + message_start[split_at:], + _anthropic_sse( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + _anthropic_sse( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "John Smith"}}, + ), + _anthropic_sse("content_block_stop", {"type": "content_block_stop", "index": 0}), + _anthropic_sse("message_delta", {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {}}), + _anthropic_sse("message_stop", {"type": "message_stop"}), + ] + + async def mock_stream(): + for b in byte_chunks: + yield b + + collected = [] + async for chunk in guardrail.async_post_call_streaming_iterator_hook( + user_api_key_dict=UserAPIKeyAuth(api_key="test-key"), + response=mock_stream(), + request_data={}, + ): + collected.append(chunk) + + joined = b"".join(collected).decode() + assert "John Smith" not in joined, joined + assert "".join(text for _, text in _anthropic_text_deltas(collected)) == "" + assert joined.count("event: message_start") == 1 + + +@pytest.mark.asyncio +async def test_apply_to_output_streaming_gemini_first_frame_split_across_transport_chunks_streams_incrementally(): + guardrail = _OPTIONAL_PresidioPIIMasking( + mock_testing=True, + apply_to_output=True, + mock_redacted_text={"text": ""}, + ) + first = _gemini_sse("Partial one from John Smith. ") + second = _gemini_sse("Partial two. ") + collected: list[object] = [] + + async def mock_stream(): + yield first[:20] + yield first[20:] + yield second + raise ConnectionError("upstream closed mid-stream") + + async def collect() -> None: + async for chunk in guardrail.async_post_call_streaming_iterator_hook( + user_api_key_dict=UserAPIKeyAuth(api_key="test-key"), + response=mock_stream(), + request_data={}, + ): + collected.append(chunk) + + with pytest.raises(ConnectionError): + await collect() + + assert collected == [first, second] + + +@pytest.mark.asyncio +async def test_apply_to_output_streaming_unterminated_first_frame_is_released_once_it_exceeds_the_cap(): + guardrail = _OPTIONAL_PresidioPIIMasking( + mock_testing=True, + apply_to_output=True, + mock_redacted_text={"text": ""}, + ) + piece = b"data: " + b"x" * 1023 + b"\n" + pieces_to_cap = -(-(64 * 1024) // len(piece)) + released_at: list[int] = [] + + async def mock_stream(): + for index in range(pieces_to_cap * 4): + if collected: + released_at.append(index) + yield piece + + collected: list[object] = [] + async for chunk in guardrail.async_post_call_streaming_iterator_hook( + user_api_key_dict=UserAPIKeyAuth(api_key="test-key"), + response=mock_stream(), + request_data={}, + ): + collected.append(chunk) + + assert released_at, "nothing reached the caller before the upstream finished" + assert released_at[0] == pieces_to_cap, released_at[:3] + assert b"".join(collected) == piece * (pieces_to_cap * 4) + + @pytest.mark.asyncio async def test_apply_to_output_streaming_anthropic_sse_bytes_fail_closed_when_presidio_is_unreachable(): """ From 0fb999b613b126d077b7eead27bb709ea449414a Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:18:28 -0700 Subject: [PATCH 35/96] fix(cost-map): halve openrouter deepseek-v4-flash-0731 output price (#42881) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 2 +- model_prices_and_context_window.json | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 66b72090631..6d577fcf41b 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -65505,7 +65505,7 @@ }, "openrouter/deepseek/deepseek-v4-flash-0731": { "input_cost_per_token": 4e-08, - "output_cost_per_token": 6.4e-07, + "output_cost_per_token": 3.2e-07, "cache_read_input_token_cost": 1.6e-08, "litellm_provider": "openrouter", "max_input_tokens": 1310720, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 66b72090631..6d577fcf41b 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -65505,7 +65505,7 @@ }, "openrouter/deepseek/deepseek-v4-flash-0731": { "input_cost_per_token": 4e-08, - "output_cost_per_token": 6.4e-07, + "output_cost_per_token": 3.2e-07, "cache_read_input_token_cost": 1.6e-08, "litellm_provider": "openrouter", "max_input_tokens": 1310720, From 001179a6368b416810b9c47f3bee40861e85df7b Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:21:05 -0700 Subject: [PATCH 36/96] fix(cost-map): sync vertex-ai deprecation dates from Vertex model lifecycle pages (#42882) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...model_prices_and_context_window_backup.json | 18 ++++++++++++++++-- model_prices_and_context_window.json | 18 ++++++++++++++++-- 2 files changed, 32 insertions(+), 4 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6d577fcf41b..e9046f345a1 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -26057,7 +26057,7 @@ "supports_image_size": false }, "gemini-2.5-flash-image": { - "deprecation_date": "2026-10-02", + "deprecation_date": "2027-03-15", "cache_read_input_token_cost": 3e-08, "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -49110,6 +49110,7 @@ }, "vertex_ai/deepseek-ai/deepseek-v3.1-maas": { "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 6e-07, "input_cost_per_token_batches": 3e-07, "litellm_provider": "vertex_ai-deepseek_models", @@ -49131,6 +49132,7 @@ }, "vertex_ai/deepseek-ai/deepseek-v3.2-maas": { "cache_read_input_token_cost": 5.6e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 5.6e-07, "input_cost_per_token_batches": 2.8e-07, "litellm_provider": "vertex_ai-deepseek_models", @@ -49151,6 +49153,7 @@ "supports_tool_choice": true }, "vertex_ai/deepseek-ai/deepseek-r1-0528-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 1.35e-06, "input_cost_per_token_batches": 6.75e-07, "litellm_provider": "vertex_ai-deepseek_models", @@ -49171,7 +49174,7 @@ "supports_tool_choice": true }, "vertex_ai/gemini-2.5-flash-image": { - "deprecation_date": "2026-10-02", + "deprecation_date": "2027-03-15", "cache_read_input_token_cost": 3e-08, "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -49858,6 +49861,7 @@ }, "vertex_ai/minimaxai/minimax-m2-maas": { "cache_read_input_token_cost": 3e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 3e-07, "litellm_provider": "vertex_ai-minimax_models", "max_input_tokens": 196608, @@ -49871,6 +49875,7 @@ }, "vertex_ai/moonshotai/kimi-k2-thinking-maas": { "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 6e-07, "litellm_provider": "vertex_ai-moonshot_models", "max_input_tokens": 256000, @@ -49885,6 +49890,7 @@ }, "vertex_ai/zai-org/glm-4.7-maas": { "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 6e-07, "litellm_provider": "vertex_ai-zai_models", "max_input_tokens": 200000, @@ -49902,6 +49908,7 @@ }, "vertex_ai/zai-org/glm-5-maas": { "cache_read_input_token_cost": 1e-07, + "deprecation_date": "2026-10-21", "input_cost_per_token": 1e-06, "litellm_provider": "vertex_ai-zai_models", "max_input_tokens": 200000, @@ -50067,6 +50074,7 @@ "source": "https://cloud.google.com/generative-ai-app-builder/pricing" }, "vertex_ai/deepseek-ai/deepseek-ocr-maas": { + "deprecation_date": "2026-10-21", "litellm_provider": "vertex_ai", "mode": "ocr", "input_cost_per_token": 3e-07, @@ -50107,6 +50115,7 @@ "supports_reasoning": true }, "vertex_ai/openai/gpt-oss-20b-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 7e-08, "litellm_provider": "vertex_ai-openai_models", "max_input_tokens": 131072, @@ -50237,6 +50246,7 @@ "supports_vision": true }, "vertex_ai/qwen/qwen3-235b-a22b-instruct-2507-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 2.2e-07, "input_cost_per_token_batches": 1.1e-07, "litellm_provider": "vertex_ai-qwen_models", @@ -50256,6 +50266,7 @@ }, "vertex_ai/qwen/qwen3-coder-480b-a35b-instruct-maas": { "cache_read_input_token_cost": 2.2e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 2.2e-07, "input_cost_per_token_batches": 1.1e-07, "litellm_provider": "vertex_ai-qwen_models", @@ -50273,6 +50284,7 @@ "supports_tool_choice": true }, "vertex_ai/qwen/qwen3-next-80b-a3b-instruct-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 1.5e-07, "litellm_provider": "vertex_ai-qwen_models", "max_input_tokens": 262144, @@ -50288,6 +50300,7 @@ "supports_tool_choice": true }, "vertex_ai/qwen/qwen3-next-80b-a3b-thinking-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 1.5e-07, "litellm_provider": "vertex_ai-qwen_models", "max_input_tokens": 262144, @@ -75522,6 +75535,7 @@ "supports_tool_choice": true }, "vertex_ai/meta/llama-3.3-70b-instruct-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 7.2e-07, "input_cost_per_token_batches": 3.6e-07, "litellm_provider": "vertex_ai-llama_models", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6d577fcf41b..e9046f345a1 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -26057,7 +26057,7 @@ "supports_image_size": false }, "gemini-2.5-flash-image": { - "deprecation_date": "2026-10-02", + "deprecation_date": "2027-03-15", "cache_read_input_token_cost": 3e-08, "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -49110,6 +49110,7 @@ }, "vertex_ai/deepseek-ai/deepseek-v3.1-maas": { "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 6e-07, "input_cost_per_token_batches": 3e-07, "litellm_provider": "vertex_ai-deepseek_models", @@ -49131,6 +49132,7 @@ }, "vertex_ai/deepseek-ai/deepseek-v3.2-maas": { "cache_read_input_token_cost": 5.6e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 5.6e-07, "input_cost_per_token_batches": 2.8e-07, "litellm_provider": "vertex_ai-deepseek_models", @@ -49151,6 +49153,7 @@ "supports_tool_choice": true }, "vertex_ai/deepseek-ai/deepseek-r1-0528-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 1.35e-06, "input_cost_per_token_batches": 6.75e-07, "litellm_provider": "vertex_ai-deepseek_models", @@ -49171,7 +49174,7 @@ "supports_tool_choice": true }, "vertex_ai/gemini-2.5-flash-image": { - "deprecation_date": "2026-10-02", + "deprecation_date": "2027-03-15", "cache_read_input_token_cost": 3e-08, "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -49858,6 +49861,7 @@ }, "vertex_ai/minimaxai/minimax-m2-maas": { "cache_read_input_token_cost": 3e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 3e-07, "litellm_provider": "vertex_ai-minimax_models", "max_input_tokens": 196608, @@ -49871,6 +49875,7 @@ }, "vertex_ai/moonshotai/kimi-k2-thinking-maas": { "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 6e-07, "litellm_provider": "vertex_ai-moonshot_models", "max_input_tokens": 256000, @@ -49885,6 +49890,7 @@ }, "vertex_ai/zai-org/glm-4.7-maas": { "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 6e-07, "litellm_provider": "vertex_ai-zai_models", "max_input_tokens": 200000, @@ -49902,6 +49908,7 @@ }, "vertex_ai/zai-org/glm-5-maas": { "cache_read_input_token_cost": 1e-07, + "deprecation_date": "2026-10-21", "input_cost_per_token": 1e-06, "litellm_provider": "vertex_ai-zai_models", "max_input_tokens": 200000, @@ -50067,6 +50074,7 @@ "source": "https://cloud.google.com/generative-ai-app-builder/pricing" }, "vertex_ai/deepseek-ai/deepseek-ocr-maas": { + "deprecation_date": "2026-10-21", "litellm_provider": "vertex_ai", "mode": "ocr", "input_cost_per_token": 3e-07, @@ -50107,6 +50115,7 @@ "supports_reasoning": true }, "vertex_ai/openai/gpt-oss-20b-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 7e-08, "litellm_provider": "vertex_ai-openai_models", "max_input_tokens": 131072, @@ -50237,6 +50246,7 @@ "supports_vision": true }, "vertex_ai/qwen/qwen3-235b-a22b-instruct-2507-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 2.2e-07, "input_cost_per_token_batches": 1.1e-07, "litellm_provider": "vertex_ai-qwen_models", @@ -50256,6 +50266,7 @@ }, "vertex_ai/qwen/qwen3-coder-480b-a35b-instruct-maas": { "cache_read_input_token_cost": 2.2e-08, + "deprecation_date": "2026-10-21", "input_cost_per_token": 2.2e-07, "input_cost_per_token_batches": 1.1e-07, "litellm_provider": "vertex_ai-qwen_models", @@ -50273,6 +50284,7 @@ "supports_tool_choice": true }, "vertex_ai/qwen/qwen3-next-80b-a3b-instruct-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 1.5e-07, "litellm_provider": "vertex_ai-qwen_models", "max_input_tokens": 262144, @@ -50288,6 +50300,7 @@ "supports_tool_choice": true }, "vertex_ai/qwen/qwen3-next-80b-a3b-thinking-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 1.5e-07, "litellm_provider": "vertex_ai-qwen_models", "max_input_tokens": 262144, @@ -75522,6 +75535,7 @@ "supports_tool_choice": true }, "vertex_ai/meta/llama-3.3-70b-instruct-maas": { + "deprecation_date": "2026-10-21", "input_cost_per_token": 7.2e-07, "input_cost_per_token_batches": 3.6e-07, "litellm_provider": "vertex_ai-llama_models", From dfb61e9b48bdadf80eab828c0e2b64d2a1d17445 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:24:37 -0700 Subject: [PATCH 37/96] fix(cost-map): add azure gpt-realtime-mini deprecation date (#42883) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 1 + model_prices_and_context_window.json | 1 + 2 files changed, 2 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e9046f345a1..1d15bea40ac 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -5991,6 +5991,7 @@ "cache_creation_input_audio_token_cost": 3e-07, "cache_read_input_audio_token_cost": 3e-07, "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2027-06-15", "input_cost_per_audio_token": 1e-05, "input_cost_per_image_token": 8e-07, "input_cost_per_token": 6e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e9046f345a1..1d15bea40ac 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -5991,6 +5991,7 @@ "cache_creation_input_audio_token_cost": 3e-07, "cache_read_input_audio_token_cost": 3e-07, "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2027-06-15", "input_cost_per_audio_token": 1e-05, "input_cost_per_image_token": 8e-07, "input_cost_per_token": 6e-07, From 6b1ec4cd3abeae890b52e506cf96aa592b8935af Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:26:59 -0700 Subject: [PATCH 38/96] fix(models): correct fireworks kimi k3 us pricing to the published rate (#42884) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 12 ++++++------ model_prices_and_context_window.json | 12 ++++++------ 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 1d15bea40ac..b75e7ec7430 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -59931,14 +59931,14 @@ "supports_vision": true }, "fireworks_ai/kimi-k3-us": { - "cache_read_input_token_cost": 3.3e-07, - "input_cost_per_token": 3.3e-06, + "cache_read_input_token_cost": 4.5e-07, + "input_cost_per_token": 4.5e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 1.65e-05, + "output_cost_per_token": 2.25e-05, "reasoning_effort_levels": [ "low", "high", @@ -60139,14 +60139,14 @@ "supports_vision": true }, "fireworks_ai/accounts/fireworks/routers/kimi-k3-us": { - "cache_read_input_token_cost": 3.3e-07, - "input_cost_per_token": 3.3e-06, + "cache_read_input_token_cost": 4.5e-07, + "input_cost_per_token": 4.5e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 1.65e-05, + "output_cost_per_token": 2.25e-05, "reasoning_effort_levels": [ "low", "high", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 1d15bea40ac..b75e7ec7430 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -59931,14 +59931,14 @@ "supports_vision": true }, "fireworks_ai/kimi-k3-us": { - "cache_read_input_token_cost": 3.3e-07, - "input_cost_per_token": 3.3e-06, + "cache_read_input_token_cost": 4.5e-07, + "input_cost_per_token": 4.5e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 1.65e-05, + "output_cost_per_token": 2.25e-05, "reasoning_effort_levels": [ "low", "high", @@ -60139,14 +60139,14 @@ "supports_vision": true }, "fireworks_ai/accounts/fireworks/routers/kimi-k3-us": { - "cache_read_input_token_cost": 3.3e-07, - "input_cost_per_token": 3.3e-06, + "cache_read_input_token_cost": 4.5e-07, + "input_cost_per_token": 4.5e-06, "litellm_provider": "fireworks_ai", "max_input_tokens": 1048576, "max_output_tokens": 131072, "max_tokens": 131072, "mode": "chat", - "output_cost_per_token": 1.65e-05, + "output_cost_per_token": 2.25e-05, "reasoning_effort_levels": [ "low", "high", From 0094cff47a9890a434a7d47b774f37e5f779d3fc Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:38:31 -0700 Subject: [PATCH 39/96] fix(cost-map): add azure gpt-realtime-mini-2025-10-06 deprecation date (#42885) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 1 + model_prices_and_context_window.json | 1 + 2 files changed, 2 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index b75e7ec7430..0b606f5db4e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -6025,6 +6025,7 @@ "cache_creation_input_audio_token_cost": 3e-07, "cache_read_input_audio_token_cost": 3e-07, "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2027-04-06", "input_cost_per_audio_token": 1e-05, "input_cost_per_image_token": 8e-07, "input_cost_per_token": 6e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index b75e7ec7430..0b606f5db4e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -6025,6 +6025,7 @@ "cache_creation_input_audio_token_cost": 3e-07, "cache_read_input_audio_token_cost": 3e-07, "cache_read_input_token_cost": 6e-08, + "deprecation_date": "2027-04-06", "input_cost_per_audio_token": 1e-05, "input_cost_per_image_token": 8e-07, "input_cost_per_token": 6e-07, From 251fdf03089be0bb3d27c5306300b0e6df572a4f Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:44:23 -0700 Subject: [PATCH 40/96] fix(logging): scan the exceeded budget wording linearly so a crafted error message cannot stall the proxy (#42778) * fix(logging): bound the exceeded budget regex so a crafted error message cannot stall the proxy Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): audit normalized_error clustering on long messages with a real two worker proxy Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): poll for the budget denial and correlate upstream 503 bursts by request marker Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(logging): replace the bounded exceeded budget regex with a linear scan that keeps the original semantics Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): compare upstream error wording against the decoded message Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): tolerate a reaped worker while listing proxy children Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../litellm_core_utils/error_normalization.py | 14 +- .../test_normalized_error_long_message.py | 463 ++++++++++++++++++ .../test_error_normalization.py | 35 ++ 3 files changed, 510 insertions(+), 2 deletions(-) create mode 100644 tests/integration/spend/test_normalized_error_long_message.py diff --git a/litellm/litellm_core_utils/error_normalization.py b/litellm/litellm_core_utils/error_normalization.py index 2a9ff748899..61eb47a9675 100644 --- a/litellm/litellm_core_utils/error_normalization.py +++ b/litellm/litellm_core_utils/error_normalization.py @@ -61,7 +61,7 @@ class _HasProxyErrorType(Protocol): _MESSAGE_PATTERNS: Final[tuple[tuple[re.Pattern[str], str], ...]] = ( ( - re.compile(r"budget has been exceeded|max budget|exceeded.*budget|crossed budget", re.IGNORECASE), + re.compile(r"budget has been exceeded|max budget|crossed budget", re.IGNORECASE), BUDGET_EXCEEDED, ), (re.compile(r"no healthy deployments?|no deployments available", re.IGNORECASE), NO_HEALTHY_DEPLOYMENTS), @@ -155,6 +155,14 @@ _CLASS_CODE_TABLE: Final[tuple[tuple[tuple[type[BaseException], ...], str], ...] ) +def _exceeded_before_budget(message: str) -> bool: + """Linear-time equivalent of ``re.search(r"exceeded.*budget", message, re.IGNORECASE)``.""" + return any( + (start := line.find("exceeded")) != -1 and line.find("budget", start + len("exceeded")) != -1 + for line in message.lower().split("\n") + ) + + def _classify_by_message(message: str, patterns: tuple[tuple[re.Pattern[str], str], ...]) -> str | None: return next((code for pattern, code in patterns if pattern.search(message)), None) @@ -183,7 +191,9 @@ def normalize_error(exc: Exception | None, status_code: str, message: str) -> st by_proxy_type: Final = _PROXY_ERROR_TYPE_MAP.get(proxy_type) if isinstance(proxy_type, str) else None if by_proxy_type is not None: return by_proxy_type - by_message: Final = _classify_by_message(message, _MESSAGE_PATTERNS) + by_message: Final = ( + BUDGET_EXCEEDED if _exceeded_before_budget(message) else _classify_by_message(message, _MESSAGE_PATTERNS) + ) if by_message is not None: return by_message by_class: Final = _classify_by_class(exc) diff --git a/tests/integration/spend/test_normalized_error_long_message.py b/tests/integration/spend/test_normalized_error_long_message.py new file mode 100644 index 00000000000..accf160b03f --- /dev/null +++ b/tests/integration/spend/test_normalized_error_long_message.py @@ -0,0 +1,463 @@ +import asyncio +import json +import os +import signal +import threading +import time +import uuid +from collections.abc import Callable, Iterator, Mapping +from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait +from contextlib import contextmanager +from dataclasses import dataclass +from hashlib import sha256 +from pathlib import Path +from typing import Final + +import anthropic +import httpx +import openai +import psutil +import pytest +from integration._support.client import Gateway, eventually, object_value, string_value +from integration._support.database import read_rows +from integration._support.process import OwnedProxy, owned_proxy_process +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + +CRAFTED_MODEL: Final = ("exceeded " * 32_000)[:288_000] +HOSTILE_5KB_MODEL: Final = ("exceeded budget " * 400)[:5_000] +FAST_SECONDS: Final = 10.0 +LIVELINESS_MAX_SECONDS: Final = 5.0 +ROW_SECONDS: Final = 70 +CHAT: Final = "/v1/chat/completions" +MESSAGES: Final = "/v1/messages" +RESPONSES: Final = "/v1/responses" + + +def _body(path: str, model: str, marker: str, stream: bool = False) -> dict[str, JsonValue]: + content: Final = f"normalized error audit {marker}" + match path: + case "/v1/messages": + return { + "model": model, + "max_tokens": 8, + "messages": [{"role": "user", "content": content}], + "stream": stream, + } + case "/v1/responses": + return {"model": model, "input": content, "stream": stream} + case _: + return {"model": model, "messages": [{"role": "user", "content": content}], "stream": stream} + + +@dataclass(frozen=True, slots=True) +class _Timed: + response: httpx.Response + seconds: float + + +def _timed_post(client: httpx.Client, path: str, body: Mapping[str, JsonValue], key: str) -> _Timed: + started: Final = time.perf_counter() + response: Final = client.post(path, json=body, headers={"Authorization": f"Bearer {key}"}) + return _Timed(response, time.perf_counter() - started) + + +@contextmanager +def _patient_client(gateway: Gateway) -> Iterator[httpx.Client]: + with httpx.Client(base_url=str(gateway.client.base_url), timeout=120, trust_env=False) as client: + yield client + + +def _error_information(call_id: str) -> dict[str, JsonValue]: + rows: Final = eventually( + lambda: read_rows( + "SELECT status, metadata->'error_information' AS info FROM \"LiteLLM_SpendLogs\" WHERE request_id=%s", + (call_id,), + ), + lambda values: len(values) == 1, + seconds=ROW_SECONDS, + ) + assert rows[0]["status"] == "failure", rows + return object_value(rows[0]["info"]) + + +def _assert_crafted_failure(timed: _Timed, expected_status: int = 400) -> None: + response: Final = timed.response + assert response.status_code == expected_status, response.text[:300] + assert "Invalid model name passed in" in response.text, response.text[:300] + assert timed.seconds < FAST_SECONDS, f"crafted 288 KB model took {timed.seconds:.2f}s" + info: Final = _error_information(response.headers["x-litellm-call-id"]) + assert info["normalized_error"] == "400_INVALID_REQUEST" and info["error_code"] == "400", info + + +@pytest.mark.parametrize("path", [CHAT, MESSAGES, RESPONSES]) +def test_crafted_288kb_model_fails_fast_and_logs_invalid_request(gateway: Gateway, path: str) -> None: + with gateway.scenario() as scenario, _patient_client(gateway) as client: + key: Final = scenario.key() + _assert_crafted_failure(_timed_post(client, path, _body(path, CRAFTED_MODEL, uuid.uuid4().hex), key)) + + +@pytest.mark.parametrize("path", [CHAT, MESSAGES, RESPONSES]) +def test_crafted_288kb_model_with_stream_true_fails_fast(gateway: Gateway, path: str) -> None: + with gateway.scenario() as scenario, _patient_client(gateway) as client: + key: Final = scenario.key() + body: Final = _body(path, CRAFTED_MODEL, uuid.uuid4().hex, stream=True) + _assert_crafted_failure(_timed_post(client, path, body, key)) + + +def test_crafted_288kb_model_through_async_openai_sdk_fails_fast(gateway: Gateway) -> None: + async def call(key: str) -> tuple[openai.BadRequestError, float]: + client: Final = openai.AsyncOpenAI(base_url=f"{gateway.client.base_url}/v1", api_key=key, timeout=120) + started: Final = time.perf_counter() + try: + with pytest.raises(openai.BadRequestError) as raised: + await client.chat.completions.create( + model=CRAFTED_MODEL, messages=[{"role": "user", "content": f"audit {uuid.uuid4().hex}"}] + ) + return raised.value, time.perf_counter() - started + finally: + await client.close() + + with gateway.scenario() as scenario: + error, seconds = asyncio.run(call(scenario.key())) + assert seconds < FAST_SECONDS, f"crafted 288 KB model took {seconds:.2f}s" + assert "Invalid model name passed in" in str(error), str(error)[:300] + info: Final = _error_information(error.response.headers["x-litellm-call-id"]) + assert info["normalized_error"] == "400_INVALID_REQUEST", info + + +def test_crafted_288kb_model_through_anthropic_sdk_fails_fast(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + client: Final = anthropic.Anthropic(base_url=str(gateway.client.base_url), api_key=scenario.key(), timeout=120) + started: Final = time.perf_counter() + with pytest.raises(anthropic.BadRequestError) as raised: + client.messages.create( + model=CRAFTED_MODEL, max_tokens=8, messages=[{"role": "user", "content": f"audit {uuid.uuid4().hex}"}] + ) + seconds: Final = time.perf_counter() - started + assert seconds < FAST_SECONDS, f"crafted 288 KB model took {seconds:.2f}s" + assert "Invalid model name passed in" in str(raised.value), str(raised.value)[:300] + info: Final = _error_information(raised.value.response.headers["x-litellm-call-id"]) + assert info["normalized_error"] == "400_INVALID_REQUEST", info + + +def _poll_liveliness(client: httpx.Client, stop: threading.Event) -> list[float]: + latencies: Final[list[float]] = [] # mutable-ok: thread-local sample buffer drained once by the caller + while not stop.is_set(): + started = time.perf_counter() + assert client.get("/health/liveliness").status_code == 200 + latencies.append(time.perf_counter() - started) + stop.wait(0.1) + return latencies + + +def _cmdline(process: psutil.Process) -> str: + try: + return " ".join(process.cmdline()) + except psutil.Error: + return "" + + +def _worker_pids(owned: OwnedProxy) -> tuple[int, ...]: + return tuple(child.pid for child in psutil.Process(owned.process.pid).children() if "spawn_main" in _cmdline(child)) + + +def test_two_concurrent_crafted_requests_do_not_stall_liveliness_on_a_two_worker_proxy( + gateway: Gateway, tmp_path: Path +) -> None: + with owned_proxy_process(gateway, tmp_path, {}, workers=2) as owned, _patient_client(owned.gateway) as client: + assert len(_worker_pids(owned)) == 2, _worker_pids(owned) + stop: Final = threading.Event() + with ThreadPoolExecutor(max_workers=3) as pool: + liveliness: Final = pool.submit(_poll_liveliness, client, stop) + crafted: Final = tuple( + pool.submit(_timed_post, client, CHAT, _body(CHAT, CRAFTED_MODEL, uuid.uuid4().hex), gateway.key) + for _ in range(2) + ) + results: Final = tuple(future.result() for future in crafted) + stop.set() + latencies: Final = liveliness.result() + for timed in results: + _assert_crafted_failure(timed) + assert latencies and max(latencies) < LIVELINESS_MAX_SECONDS, f"liveliness max {max(latencies):.2f}s" + + +def _completion(request: Request) -> Reply: + body: Final = object_value(json.loads(request.body or b"{}")) + return Reply( + body=json.dumps( + { + "id": "chatcmpl-" + uuid.uuid4().hex, + "object": "chat.completion", + "created": 1, + "model": body.get("model", "unknown"), + "choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 20, "completion_tokens": 20, "total_tokens": 40}, + } + ).encode() + ) + + +def _rate_limited(message: str) -> Callable[[Request], Reply]: + def respond(_request: Request) -> Reply: + return Reply( + status=429, + body=json.dumps({"error": {"message": message, "type": "rate_limit_error", "code": "429"}}).encode(), + ) + + return respond + + +def _budget_denied_row(key: str) -> dict[str, JsonValue]: + rows: Final = eventually( + lambda: read_rows( + "SELECT request_id, metadata->'error_information' AS info FROM \"LiteLLM_SpendLogs\" " + "WHERE api_key=%s AND status='failure'", + (sha256(key.encode()).hexdigest(),), + ), + lambda values: len(values) == 1, + seconds=ROW_SECONDS, + ) + return object_value(rows[0]["info"]) + + +def _exhaust( + gateway: Gateway, client: httpx.Client, model: str, key: str, table: str, column: str, identity: str +) -> None: + first: Final = gateway.chat(model, key=key, text=f"spend {uuid.uuid4().hex}") + assert object_value(first["usage"])["total_tokens"] == 40, first + eventually( + lambda: read_rows(f'SELECT spend FROM "{table}" WHERE {column}=%s', (identity,)), + lambda values: len(values) == 1 and float(string_value(str(values[0]["spend"]))) >= 0.06, + seconds=ROW_SECONDS, + ) + denied: Final = eventually( + lambda: client.post( + CHAT, json=_body(CHAT, model, uuid.uuid4().hex), headers={"Authorization": f"Bearer {key}"} + ), + lambda response: response.status_code in {400, 422}, + seconds=ROW_SECONDS, + ) + assert denied.json()["error"]["type"] == "budget_exceeded", denied.text + info: Final = _budget_denied_row(key) + assert info["normalized_error"] == "429_BUDGET_EXCEEDED", info + assert "budget" in string_value(info["error_message"]).lower(), info + + +def test_exhausted_key_budget_denial_clusters_as_budget_exceeded(gateway: Gateway) -> None: + with gateway.scenario() as scenario, _patient_client(gateway) as client: + model: Final = scenario.model(input_cost_per_token=0.001, output_cost_per_token=0.002) + key: Final = scenario.key(models=[model], max_budget=0.06) + _exhaust(gateway, client, model, key, "LiteLLM_VerificationToken", "token", sha256(key.encode()).hexdigest()) + + +def test_exhausted_team_budget_denial_clusters_as_budget_exceeded(gateway: Gateway) -> None: + with gateway.scenario() as scenario, _patient_client(gateway) as client: + model: Final = scenario.model(input_cost_per_token=0.001, output_cost_per_token=0.002) + team: Final = scenario.team(models=[model], max_budget=0.06) + key: Final = scenario.key(team_id=team, models=[model]) + _exhaust(gateway, client, model, key, "LiteLLM_TeamTable", "team_id", team) + + +def _upstream_failure_row(gateway: Gateway, message: str) -> dict[str, JsonValue]: + with wire_server(_rate_limited(message)) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(api_base=wire.url + "/v1", num_retries=0) + key: Final = scenario.key(models=[model]) + failed: Final = gateway.request("POST", CHAT, _body(CHAT, model, uuid.uuid4().hex), key=key) + assert failed.status_code == 429 and message in failed.json()["error"]["message"], failed.text[:300] + assert len(wire.drain()) == 1 + return _error_information(failed.headers["x-litellm-call-id"]) + + +@pytest.mark.parametrize( + "message", + [ + "Budget has been exceeded! Current cost: 11.0, Max budget: 10.0", + "ExceededBudget: User=audit over budget. Spend=12.5, Budget=10.0", + "Exceeded budget for provider openai: 105.2 >= 100.0", + "exceeded" + "x" * 64 + "budget", + ], +) +def test_upstream_budget_wording_clusters_as_budget_exceeded(gateway: Gateway, message: str) -> None: + info: Final = _upstream_failure_row(gateway, message) + assert info["normalized_error"] == "429_BUDGET_EXCEEDED", info + + +def test_upstream_exceeded_and_budget_65_chars_apart_still_clusters_as_budget_exceeded(gateway: Gateway) -> None: + info: Final = _upstream_failure_row(gateway, "exceeded" + "x" * 65 + "budget") + assert info["normalized_error"] == "429_BUDGET_EXCEEDED", info + + +def test_upstream_exceeded_and_budget_on_different_lines_cluster_by_exception_class(gateway: Gateway) -> None: + info: Final = _upstream_failure_row(gateway, "exceeded the limit\nbudget unaffected") + assert info["normalized_error"] == "429_RATE_LIMIT_EXCEEDED", info + + +def test_hostile_model_values_are_rejected_without_taking_the_proxy_down(gateway: Gateway) -> None: + with gateway.scenario() as scenario, _patient_client(gateway) as client: + key: Final = scenario.key() + hostile_values: Final[tuple[JsonValue, ...]] = (5, ["gpt-4o-mini"]) + for hostile in hostile_values: + rejected = _timed_post(client, CHAT, {"model": hostile, "messages": []}, key) + assert rejected.response.status_code == 400 and "must be a string" in rejected.response.text + assert _error_information(rejected.response.headers["x-litellm-call-id"])["normalized_error"] == ( + "400_INVALID_REQUEST" + ) + empty: Final = _timed_post(client, CHAT, _body(CHAT, "", uuid.uuid4().hex), key) + assert empty.response.status_code == 400, empty.response.text + assert _error_information(empty.response.headers["x-litellm-call-id"])["normalized_error"] == ( + "400_INVALID_REQUEST" + ) + repeated: Final = tuple( + _timed_post(client, CHAT, _body(CHAT, HOSTILE_5KB_MODEL, uuid.uuid4().hex), key) for _ in range(2) + ) + call_ids: Final = tuple(timed.response.headers["x-litellm-call-id"] for timed in repeated) + assert len(set(call_ids)) == 2 and all(timed.response.status_code == 400 for timed in repeated) + assert all(timed.seconds < FAST_SECONDS for timed in repeated), [timed.seconds for timed in repeated] + codes: Final = tuple(_error_information(call_id)["normalized_error"] for call_id in call_ids) + assert len(set(codes)) == 1 and codes[0] in {"400_INVALID_REQUEST", "429_BUDGET_EXCEEDED"}, codes + unauthenticated: Final = client.post(CHAT, json=_body(CHAT, "gpt-4o-mini", "x")) + assert unauthenticated.status_code == 401, unauthenticated.text + assert client.get("/health/liveliness").status_code == 200 + + +@dataclass(frozen=True, slots=True) +class _BurstResult: + label: str + status: int | None + call_id: str | None + response_id: str | None + seconds: float + + +def _burst_call(client: httpx.Client, label: str, path: str, body: Mapping[str, JsonValue], key: str) -> _BurstResult: + started: Final = time.perf_counter() + try: + response: Final = client.post(path, json=body, headers={"Authorization": f"Bearer {key}"}) + except httpx.TransportError: + return _BurstResult(label, None, None, None, time.perf_counter() - started) + identity: Final = object_value(response.json()).get("id") if response.status_code == 200 else None + return _BurstResult( + label, + response.status_code, + response.headers.get("x-litellm-call-id"), + identity if isinstance(identity, str) else None, + time.perf_counter() - started, + ) + + +def _burst( + client: httpx.Client, happy_model: str, happy_key: str, open_key: str, during: Callable[[], None] +) -> tuple[_BurstResult, ...]: + crafted: Final = tuple( + (f"crafted-{path}-{index}", path, _body(path, CRAFTED_MODEL, uuid.uuid4().hex, stream=index % 2 == 1), open_key) + for path in (CHAT, MESSAGES, RESPONSES) + for index in range(4) + ) + happy: Final = tuple( + (f"happy-{index}", CHAT, _body(CHAT, happy_model, f"happy-{index}"), happy_key) for index in range(8) + ) + late: Final = tuple( + (f"late-{index}", CHAT, _body(CHAT, happy_model, f"late-{index}"), happy_key) for index in range(8) + ) + with ( + ThreadPoolExecutor(max_workers=28) as pool, + httpx.Client(base_url=client.base_url, timeout=client.timeout, trust_env=False) as fresh, + ): + first: Final = tuple( + pool.submit(_burst_call, client, label, path, body, key) for label, path, body, key in crafted + happy + ) + wait(first, return_when=FIRST_COMPLETED) + during() + second: Final = tuple( + pool.submit(_burst_call, fresh, label, path, body, key) for label, path, body, key in late + ) + return tuple(future.result() for future in first + second) + + +def _assert_rows_land_exactly_once(results: tuple[_BurstResult, ...], prefix: str) -> None: + landed: Final = tuple(result for result in results if result.label.startswith(prefix) and result.status == 200) + assert landed, results + response_ids: Final = tuple(string_value(result.response_id) for result in landed) + assert len(set(response_ids)) == len(response_ids), response_ids + rows: Final = eventually( + lambda: read_rows( + 'SELECT request_id, status FROM "LiteLLM_SpendLogs" WHERE request_id = ANY(%s::text[])', + ("{" + ",".join(response_ids) + "}",), + ), + lambda values: len(values) == len(response_ids), + seconds=ROW_SECONDS, + ) + assert sorted(string_value(row["request_id"]) for row in rows) == sorted(response_ids), rows + assert all(row["status"] == "success" for row in rows), rows + + +def test_killing_one_worker_mid_burst_leaves_the_other_serving_crafted_and_happy_traffic( + gateway: Gateway, tmp_path: Path +) -> None: + with ( + wire_server(_completion) as wire, + owned_proxy_process(gateway, tmp_path, {}, workers=2) as owned, + owned.gateway.scenario() as scenario, + _patient_client(owned.gateway) as client, + ): + model: Final = scenario.model(api_base=wire.url + "/v1", num_retries=0) + key: Final = scenario.key(models=[model]) + workers: Final = _worker_pids(owned) + assert len(workers) == 2, workers + + def kill_one_worker() -> None: + os.kill(workers[0], signal.SIGKILL) + + results: Final = _burst(client, model, key, scenario.key(), kill_one_worker) + dropped: Final = tuple(result for result in results if result.status is None) + assert len(dropped) < len(results), results + crafted: Final = tuple(result for result in results if result.label.startswith("crafted") and result.status) + assert crafted and all(result.status == 400 and result.seconds < FAST_SECONDS for result in crafted), crafted + late: Final = tuple(result for result in results if result.label.startswith("late")) + assert all(result.status == 200 for result in late), late + _assert_rows_land_exactly_once(results, "late") + survivor: Final = tuple(pid for pid in _worker_pids(owned) if pid != workers[0]) + assert survivor, "no worker left serving" + after: Final = owned.gateway.chat(model, key=key, text=f"after kill {uuid.uuid4().hex}") + assert isinstance(after["id"], str) and after["id"].startswith("chatcmpl-"), after + assert client.get("/health/liveliness").status_code == 200 + + +def test_upstream_returning_503_mid_burst_logs_every_failure_with_its_own_cluster_key( + gateway: Gateway, tmp_path: Path +) -> None: + def overloaded(_request: Request) -> Reply: + return Reply( + status=503, + body=b'{"error":{"message":"Controlled provider outage","type":"server_error","code":"503"}}', + ) + + with ( + wire_server(overloaded) as wire, + owned_proxy_process(gateway, tmp_path, {}, workers=2) as owned, + owned.gateway.scenario() as scenario, + _patient_client(owned.gateway) as client, + ): + model: Final = scenario.model(api_base=wire.url + "/v1", num_retries=0) + key: Final = scenario.key(models=[model]) + results: Final = _burst(client, model, key, scenario.key(), lambda: None) + assert all(result.status is not None for result in results), results + happy: Final = tuple(result for result in results if result.label.startswith("happy")) + assert all(result.status == 503 for result in happy), happy + late: Final = tuple(result for result in results if result.label.startswith("late")) + assert all(result.status == 503 for result in late), late + seen: Final = tuple(request.body.decode() for request in wire.drain()) + assert all(any(f"normalized error audit {result.label}" in body for body in seen) for result in happy + late), ( + seen + ) + crafted: Final = tuple(result for result in results if result.label.startswith("crafted")) + assert all(result.status == 400 and result.seconds < FAST_SECONDS for result in crafted), crafted + codes: Final = { + result.label: _error_information(string_value(result.call_id))["normalized_error"] for result in results + } + assert all(code == "503_PROVIDER_OVERLOADED" for label, code in codes.items() if label.startswith("happy")), ( + codes + ) + assert all(code == "400_INVALID_REQUEST" for label, code in codes.items() if label.startswith("crafted")), codes + assert client.get("/health/liveliness").status_code == 200 diff --git a/tests/test_litellm/litellm_core_utils/test_error_normalization.py b/tests/test_litellm/litellm_core_utils/test_error_normalization.py index 9d5469ddb4c..d65b6d316ac 100644 --- a/tests/test_litellm/litellm_core_utils/test_error_normalization.py +++ b/tests/test_litellm/litellm_core_utils/test_error_normalization.py @@ -1,3 +1,5 @@ +import time + import httpx import pytest @@ -229,3 +231,36 @@ def test_normalized_error_never_embeds_dynamic_parts() -> None: info = StandardLoggingPayloadSetup.get_error_information(exc) assert info["error_message"] == "No team has access to anthropic.claude-sonnet-4-5" assert "claude" not in (info["normalized_error"] or "") + + +def test_repeated_exceeded_in_a_288kb_message_classifies_in_linear_time() -> None: + model = ("exceeded " * 32_000)[:288_000] + message = ( + f"/chat/completions: Invalid model name passed in model={model}. Call `/v1/models` to view available models" + ) + exc = litellm.BadRequestError(message=message, model="unknown-model", llm_provider="openai") + started = time.perf_counter() + code = normalize_error(exc, "400", message) + elapsed = time.perf_counter() - started + assert code == "400_INVALID_REQUEST", code + assert elapsed < 1.0, f"normalize_error took {elapsed:.2f}s on a 288 KB message" + + +@pytest.mark.parametrize( + "message", + [ + "ExceededBudget: User=abc over budget. Spend=12.5, Budget=10.0", + "Exceeded budget for provider openai: 105.2 >= 100.0", + "LiteLLM Team: team-1, exceeded budget for model=gpt-4o-mini", + "ExceededBudget: Key over 1d budget. Spend=3.0, Budget=2.0", + "Budget has been exceeded! Key=sk-... Current cost: 11.0, Max budget: 10.0", + "EXCEEDED " + "x" * 65 + " BuDgEt", + ], +) +def test_real_budget_wordings_still_cluster_as_budget_exceeded(message: str) -> None: + assert normalize_error(Exception(message), "400", message) == "429_BUDGET_EXCEEDED" + + +@pytest.mark.parametrize("message", ["budget then exceeded", "exceeded the limit\nbudget unaffected", "exceededbudge"]) +def test_exceeded_without_a_following_budget_on_the_same_line_is_not_budget(message: str) -> None: + assert normalize_error(Exception(message), "400", message) == "400_INVALID_REQUEST" From 1519032d9042d9e9b3541612def76231082ec50b Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:48:04 -0700 Subject: [PATCH 41/96] fix(proxy): keep deployment labels on cache-hit post_call guardrail rejections (#42780) * fix(proxy): keep deployment labels on cache-hit post_call rejections A post-call failure on a response served from the litellm cache set no first_api_call_start_time, so the failure hook flagged it as rejected before routing and dropped the model_id and provider labels Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): read the cache hit from caching_details in the failure hook model_call_details[cache_hit] is stamped inside the enqueued success handler, so a post-call failure can observe it too early; logging_obj.caching_details is set synchronously before the cached response returns Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): cover cache-hit guardrail reject deployment labels across endpoints, modes and chaos Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): bound the worker-kill reject count by in-flight losses Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(proxy): assert provider and model labels on the cache-hit regression test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/proxy/utils.py | 10 +- .../test_cache_hit_guardrail_metrics.py | 652 ++++++++++++++++++ .../test_cache_hit_guardrail_metrics_chaos.py | 348 ++++++++++ .../test_post_call_failure_hook.py | 60 ++ 4 files changed, 1069 insertions(+), 1 deletion(-) create mode 100644 tests/integration/observability/test_cache_hit_guardrail_metrics.py create mode 100644 tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 78d6c25a336..bc64293c9b3 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -974,6 +974,14 @@ def _failure_usage_to_lift( _EMPTY_LIFT: Final = MappingProxyType({}) +def _reached_deployment(litellm_logging_obj: Logging) -> bool: + """A provider handoff or a cached response both mean the router selected a deployment.""" + caching_details: Final = litellm_logging_obj.caching_details + return litellm_logging_obj.model_call_details.get("first_api_call_start_time") is not None or ( + caching_details is not None and caching_details.get("cache_hit") is True + ) + + def _stamp_deployment_attribution( litellm_params: dict[str, object], model_group: str | None, team_id: str | None, dispatched: bool ) -> Mapping[str, object]: @@ -3325,7 +3333,7 @@ class ProxyLogging: _litellm_params, request_data.get("model"), user_api_key_dict.team_id, - dispatched=litellm_logging_obj.model_call_details.get("first_api_call_start_time") is not None, + dispatched=_reached_deployment(litellm_logging_obj), ) litellm_logging_obj.update_environment_variables( diff --git a/tests/integration/observability/test_cache_hit_guardrail_metrics.py b/tests/integration/observability/test_cache_hit_guardrail_metrics.py new file mode 100644 index 00000000000..888cfd7ab9d --- /dev/null +++ b/tests/integration/observability/test_cache_hit_guardrail_metrics.py @@ -0,0 +1,652 @@ +import asyncio +import json +import subprocess +import uuid +from collections.abc import Callable, Generator, Mapping +from contextlib import contextmanager +from dataclasses import dataclass +from pathlib import Path +from typing import Final + +import anthropic +import httpx +import openai +import pytest +import yaml +from integration._support.client import Gateway, Scenario, eventually, object_value, string_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy_process +from integration._support.wire import Reply, Request, Wire, wire_server +from prometheus_client.parser import text_string_to_metric_families + +GUARDRAIL_PATH: Final = "/beta/litellm_basic_guardrail_api" +DEPLOYMENT_FAILURE: Final = "litellm_deployment_failure_responses_total" +DEPLOYMENT_REQUESTS: Final = "litellm_deployment_total_requests_total" +DEPLOYMENT_STATE: Final = "litellm_deployment_state" +PROXY_FAILED: Final = "litellm_proxy_failed_requests_metric_total" + + +def _chat_sse(marker: str) -> tuple[bytes, ...]: + chunk: Final = { + "id": "chatcmpl_" + marker, + "object": "chat.completion.chunk", + "created": 1, + "model": "gpt-4o-mini", + } + frames: Final = ( + {**chunk, "choices": [{"index": 0, "delta": {"role": "assistant", "content": ""}, "finish_reason": None}]}, + {**chunk, "choices": [{"index": 0, "delta": {"content": "provider control"}, "finish_reason": None}]}, + {**chunk, "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}, + ) + return tuple(f"data: {json.dumps(frame)}".encode() for frame in frames) + (b"data: [DONE]",) + + +def _provider_body(target: str, marker: str, streamed: bool) -> Reply: + match target: + case "/v1/chat/completions": + if streamed: + return Reply(content_type="text/event-stream", chunks=_chat_sse(marker)) + body: dict = { + "id": "chatcmpl_" + marker, + "object": "chat.completion", + "created": 1, + "model": "gpt-4o-mini", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "provider control " + marker}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 11, "completion_tokens": 4, "total_tokens": 15}, + } + case "/v1/messages": + body = { + "id": "msg_" + marker, + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [{"type": "text", "text": "provider control " + marker}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 11, "output_tokens": 4}, + } + case "/v1/responses": + body = { + "id": "resp_" + marker, + "object": "response", + "created_at": 1, + "status": "completed", + "model": "gpt-4o-mini", + "output": [ + { + "type": "message", + "id": "msg_" + marker, + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "provider control " + marker, "annotations": []}], + } + ], + "usage": {"input_tokens": 11, "output_tokens": 4, "total_tokens": 15}, + } + case "/v1/embeddings": + body = { + "object": "list", + "data": [{"object": "embedding", "index": 0, "embedding": [0.1, 0.2, 0.3]}], + "model": "text-embedding-3-small", + "usage": {"prompt_tokens": 3, "total_tokens": 3}, + } + case _: + return Reply(status=404, body=json.dumps({"error": "unexpected provider target " + target}).encode()) + return Reply(body=json.dumps(body).encode()) + + +def _provider(marker: str) -> Callable[[Request], Reply]: + def respond(request: Request) -> Reply: + streamed: Final = b'"stream":true' in request.body.replace(b" ", b"") + return _provider_body(request.target.split("?", 1)[0], marker, streamed) + + return respond + + +def _blocking_sink(request: Request) -> Reply: + assert request.target == GUARDRAIL_PATH, request.target + return Reply(body=json.dumps({"action": "BLOCKED", "blocked_reason": "synthetic block"}).encode()) + + +def _failing_sink(status: int) -> Callable[[Request], Reply]: + def respond(request: Request) -> Reply: + assert request.target == GUARDRAIL_PATH, request.target + return Reply(status=status, body=json.dumps({"error": "synthetic guardrail outage"}).encode()) + + return respond + + +def _first_call_pass_sink() -> Callable[[Request], Reply]: + calls: list[int] = [] # mutable-ok: the wire handler must remember call order across requests + + def respond(request: Request) -> Reply: + assert request.target == GUARDRAIL_PATH, request.target + calls.append(1) + action: dict = ( + {"action": "NONE"} if len(calls) == 1 else {"action": "BLOCKED", "blocked_reason": "synthetic block"} + ) + return Reply(body=json.dumps(action).encode()) + + return respond + + +def _guardrail_config( + tmp_path: Path, + name: str, + sink_url: str, + *, + mode: str = "post_call", + default_on: bool = False, + local_cache: bool = False, + ttl: int | None = None, +) -> Path: + config: dict = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["litellm_settings"]["callbacks"] = ["prometheus"] + if local_cache: + config["litellm_settings"]["cache_params"] = {"type": "local"} + if ttl is not None: + config["litellm_settings"]["cache_params"]["ttl"] = ttl + config["guardrails"] = [ + { + "guardrail_name": name, + "litellm_params": { + "guardrail": "generic_guardrail_api", + "mode": mode, + "default_on": default_on, + "api_base": sink_url, + "api_key": "synthetic-guardrail-key", + }, + } + ] + path: Final = tmp_path / "guardrail.yaml" + path.write_text(yaml.safe_dump(config)) + return path + + +@dataclass(frozen=True, slots=True) +class Rig: + candidate: Gateway + scenario: Scenario + model_name: str + deployment_id: str + guardrail_name: str + policy: Wire + provider: Wire + process: subprocess.Popen[bytes] + + +@contextmanager +def _rig( + gateway: Gateway, + tmp_path: Path, + marker: str, + *, + sink: Callable[[Request], Reply] = _blocking_sink, + mode: str = "post_call", + default_on: bool = False, + local_cache: bool = False, + ttl: int | None = None, + workers: int = 1, + upstream_model: str = "openai/gpt-4o-mini", + api_base_suffix: str = "/v1", + env: Mapping[str, str] | None = None, +) -> Generator[Rig, None, None]: + identity: Final = "guardrail-" + marker + with wire_server(sink) as policy, wire_server(_provider(marker)) as provider: + config: Final = _guardrail_config( + tmp_path, identity, policy.url, mode=mode, default_on=default_on, local_cache=local_cache, ttl=ttl + ) + prom_dir: Final = tmp_path / "prom" + prom_dir.mkdir() + with ( + owned_proxy_process( + gateway, + tmp_path, + {"PROMETHEUS_MULTIPROC_DIR": str(prom_dir), **(env or {})}, + config=config, + workers=workers, + ) as owned, + owned.gateway.scenario() as scenario, + ): + model: Final = scenario.model( + model=upstream_model, api_base=provider.url + api_base_suffix, api_key="synthetic-provider-key" + ) + entries: Final = owned.gateway.get("/model/info")["data"] + assert isinstance(entries, list) + entry: Final = next(item for item in entries if object_value(item)["model_name"] == model) + yield Rig( + owned.gateway, + scenario, + model, + string_value(object_value(object_value(entry)["model_info"])["id"]), + identity, + policy, + provider, + owned.process, + ) + + +def _metric_samples(candidate: Gateway, model_name: str) -> tuple: + response: Final = candidate.client.request( + "GET", "/metrics", headers={"Authorization": f"Bearer {candidate.key}"}, follow_redirects=True + ) + assert response.status_code == 200, f"GET /metrics: {response.status_code} {response.text[:300]}" + return tuple( + sample + for family in text_string_to_metric_families(response.text) + for sample in family.samples + if sample.labels.get("requested_model") == model_name + or (sample.name == DEPLOYMENT_STATE and sample.labels.get("model_id") != "") + ) + + +def _count(samples: tuple, name: str, model_id: str) -> float: + return float( + sum(sample.value for sample in samples if sample.name == name and sample.labels.get("model_id") == model_id) + ) + + +def _populated_failures(samples: tuple, rig: Rig, api_provider: str) -> float: + return float( + sum( + sample.value + for sample in samples + if sample.name == DEPLOYMENT_FAILURE + and sample.labels.get("model_id") == rig.deployment_id + and sample.labels.get("api_provider") == api_provider + and sample.labels.get("litellm_model_name") != "" + ) + ) + + +def _expect_metrics( + rig: Rig, + populated: float, + blank: float, + *, + api_provider: str = "openai", + pf_id: str | None = None, + pf_populated: float | None = None, + pf_blank: float | None = None, +) -> tuple: + expected_id: Final = rig.deployment_id if pf_id is None else pf_id + expected_pf_populated: Final = populated if pf_populated is None else pf_populated + expected_pf_blank: Final = blank if pf_blank is None else pf_blank + + def read() -> tuple: + samples: Final = _metric_samples(rig.candidate, rig.model_name) + satisfied: Final = ( + _populated_failures(samples, rig, api_provider) == populated + and _count(samples, DEPLOYMENT_FAILURE, "") == blank + and _count(samples, PROXY_FAILED, expected_id) == expected_pf_populated + and _count(samples, PROXY_FAILED, "") == expected_pf_blank + ) + return samples if satisfied else () + + return eventually(read, bool, seconds=70) + + +def _spend_rows(call_id: str) -> tuple[dict, ...]: + rows: Final = eventually( + lambda: read_rows( + 'SELECT request_id, custom_llm_provider, model_id, status FROM "LiteLLM_SpendLogs" ' + "WHERE request_id = %s OR request_id LIKE %s", + (call_id, call_id + "\\_%"), + ), + lambda values: len(values) >= 1, + seconds=70, + ) + return tuple(dict(row) for row in rows) + + +def _assert_spend(call_id: str, rig: Rig, api_provider: str = "openai") -> None: + rows: Final = _spend_rows(call_id) + failures: Final = tuple(row for row in rows if row["status"] == "failure") + assert len(failures) == 1, rows + assert (failures[0]["custom_llm_provider"], failures[0]["model_id"]) == (api_provider, rig.deployment_id), rows + + +def _call_id(reject: httpx.Response) -> str: + return reject.headers["x-litellm-call-id"] + + +def _chat_body(model: str, text: str, guardrail: str | None, stream: bool = False) -> dict: + body: dict = {"model": model, "messages": [{"role": "user", "content": text}]} + if stream: + body["stream"] = True + if guardrail is not None: + body["guardrails"] = [guardrail] + return body + + +def test_cache_hit_post_call_reject_keeps_deployment_labels(gateway: Gateway, tmp_path: Path) -> None: + """H1: warm then identical post_call-rejected cache hit keeps populated deployment labels.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "cache hit control h1 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name) + ) + assert reject.status_code == 400, reject.text + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 1, 0) + _assert_spend(_call_id(reject), rig) + + +def test_cache_hit_post_call_reject_keeps_deployment_labels_openai_sdk(gateway: Gateway, tmp_path: Path) -> None: + """H2: same as H1 through the openai AsyncOpenAI client.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "cache hit control h2 " + marker + sdk: Final = openai.AsyncOpenAI( + base_url=str(rig.candidate.client.base_url) + "/v1", + api_key=rig.candidate.key, + http_client=httpx.AsyncClient(trust_env=False, timeout=15), + ) + + async def run() -> int: + await sdk.chat.completions.create(model=rig.model_name, messages=[{"role": "user", "content": text}]) + try: + await sdk.chat.completions.create( + model=rig.model_name, + messages=[{"role": "user", "content": text}], + extra_body={"guardrails": [rig.guardrail_name]}, + ) + return 200 + except openai.BadRequestError: + return 400 + + assert asyncio.run(run()) == 400 + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 1, 0) + + +def test_cache_hit_post_call_reject_streaming(gateway: Gateway, tmp_path: Path) -> None: + """H3: streamed responses are not cached; the reject call hits upstream again and no failure hook fires.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "cache hit control h3 " + marker + warm: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None, stream=True) + ) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name, stream=True) + ) + assert reject.status_code == 200, reject.text + assert rig.provider.received.qsize() == 2, rig.provider.drain() + samples: Final = _metric_samples(rig.candidate, rig.model_name) + assert _populated_failures(samples, rig, "openai") == 0, samples + assert _count(samples, DEPLOYMENT_FAILURE, "") == 0, samples + + +def test_cache_hit_post_call_reject_keeps_deployment_labels_anthropic(gateway: Gateway, tmp_path: Path) -> None: + """H4: /v1/messages cache hit reject through the anthropic SDK.""" + marker: Final = uuid.uuid4().hex + with _rig( + gateway, tmp_path, marker, upstream_model="anthropic/claude-sonnet-4-5-20250929", api_base_suffix="" + ) as rig: + text: Final = "cache hit control h4 " + marker + sdk: Final = anthropic.Anthropic( + base_url=str(rig.candidate.client.base_url), + api_key=rig.candidate.key, + http_client=httpx.Client(trust_env=False, timeout=15), + ) + sdk.messages.create(model=rig.model_name, max_tokens=16, messages=[{"role": "user", "content": text}]) + raised: bool = False # mutable-ok: a flag set inside the except block cannot be Final + try: + sdk.messages.create( + model=rig.model_name, + max_tokens=16, + messages=[{"role": "user", "content": text}], + extra_body={"guardrails": [rig.guardrail_name]}, + ) + except anthropic.BadRequestError: + raised = True + assert raised, "cache-hit post_call guardrail did not reject /v1/messages" + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 1, 0, api_provider="anthropic", pf_id="None") + + +def test_cache_hit_post_call_reject_keeps_deployment_labels_responses(gateway: Gateway, tmp_path: Path) -> None: + """H5: /v1/responses cache hit reject.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "cache hit control h5 " + marker + warm: Final = rig.candidate.request("POST", "/v1/responses", {"model": rig.model_name, "input": text}) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", + "/v1/responses", + {"model": rig.model_name, "input": text, "guardrails": [rig.guardrail_name]}, + ) + assert reject.status_code == 400, reject.text + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 1, 0, pf_id="None") + _assert_spend(_call_id(reject), rig) + + +def test_cache_hit_post_call_reject_embeddings(gateway: Gateway, tmp_path: Path) -> None: + """H6: post_call guardrails do not run on embeddings; the cached response returns 200 unguarded.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "cache hit control h6 " + marker + warm: Final = rig.candidate.request("POST", "/v1/embeddings", {"model": rig.model_name, "input": text}) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", + "/v1/embeddings", + {"model": rig.model_name, "input": text, "guardrails": [rig.guardrail_name]}, + ) + assert reject.status_code == 200, reject.text + assert rig.provider.received.qsize() == 1, rig.provider.drain() + samples: Final = _metric_samples(rig.candidate, rig.model_name) + assert _populated_failures(samples, rig, "openai") == 0, samples + assert _count(samples, DEPLOYMENT_FAILURE, "") == 0, samples + + +def test_cache_hit_during_call_reject_keeps_deployment_labels(gateway: Gateway, tmp_path: Path) -> None: + """H7: during_call guardrail reject on a cache hit.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, mode="during_call") as rig: + text: Final = "cache hit control h7 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name) + ) + assert reject.status_code == 400, reject.text + _expect_metrics(rig, 1, 0) + + +def test_pre_call_reject_on_cache_hit_stays_blank(gateway: Gateway, tmp_path: Path) -> None: + """C1: pre_call reject never reaches the deployment; labels stay blank on both legs.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, mode="pre_call") as rig: + text: Final = "cache hit control c1 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name) + ) + assert reject.status_code == 400, reject.text + _expect_metrics(rig, 0, 1, pf_populated=1, pf_blank=0) + + +def test_post_call_reject_without_cache_keeps_deployment_labels(gateway: Gateway, tmp_path: Path) -> None: + """C2: a real provider call rejected post_call keeps populated labels on both legs.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "non cache control c2 " + marker + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name) + ) + assert reject.status_code == 400, reject.text + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 1, 0) + _assert_spend(_call_id(reject), rig) + + +def test_cache_hit_post_call_reject_default_on(gateway: Gateway, tmp_path: Path) -> None: + """C3: default_on post_call guardrail rejects the cached response (sink passes the warm call).""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, sink=_first_call_pass_sink(), default_on=True) as rig: + text: Final = "cache hit control c3 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert reject.status_code == 400, reject.text + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 1, 0) + + +def test_cache_hit_post_call_reject_key_metadata_guardrails(gateway: Gateway, tmp_path: Path) -> None: + """C4: guardrail attached via key metadata guardrails on a cache hit.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, sink=_first_call_pass_sink()) as rig: + key: Final = rig.candidate.post("/key/generate", {"metadata": {"guardrails": [rig.guardrail_name]}})["key"] + text: Final = "cache hit control c4 " + marker + warm: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None), key=key + ) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None), key=key + ) + assert reject.status_code == 400, reject.text + assert rig.provider.received.qsize() == 1, rig.provider.drain() + assert rig.policy.received.qsize() == 2 + _expect_metrics(rig, 1, 0) + + +def test_cache_hit_post_call_reject_local_cache(gateway: Gateway, tmp_path: Path) -> None: + """C5: same cache-hit reject with cache_params type local.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, local_cache=True) as rig: + text: Final = "cache hit control c5 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name) + ) + assert reject.status_code == 400, reject.text + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 1, 0) + + +@pytest.mark.parametrize("status", (500, 403)) +def test_cache_hit_post_call_guardrail_outage_keeps_deployment_labels( + gateway: Gateway, tmp_path: Path, status: int +) -> None: + """S1/S2: guardrail sink answers 500/403 on the cache-hit call; failure hook still counts as dispatched.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, sink=_failing_sink(status)) as rig: + text: Final = "cache hit control s " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name) + ) + assert reject.status_code >= 400, reject.text + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 0, 0, pf_populated=1, pf_blank=0) + + +def test_two_identical_cache_hit_rejects_increment_populated_series(gateway: Gateway, tmp_path: Path) -> None: + """E1: two identical cache-hit rejects count +2 on the populated series, two spend rows.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "cache hit control e1 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + rejects: Final = tuple( + rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name)) + for _ in range(2) + ) + assert all(response.status_code == 400 for response in rejects), [r.text for r in rejects] + assert rig.provider.received.qsize() == 1, rig.provider.drain() + _expect_metrics(rig, 2, 0) + + +def test_two_identical_cache_hit_rejects_write_matching_spend_rows(gateway: Gateway, tmp_path: Path) -> None: + """E1b: both cache-hit rejects land a failure spend row.""" + pytest.skip("BUG: roughly one in four back-to-back cache-hit rejects never lands its LiteLLM_SpendLogs row") + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "cache hit control e1b " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + rejects: Final = tuple( + rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name)) + for _ in range(2) + ) + assert all(response.status_code == 400 for response in rejects), [r.text for r in rejects] + for response in rejects: + _assert_spend(_call_id(response), rig) + + +def test_cache_hit_reject_after_ttl_expiry_is_a_miss(gateway: Gateway, tmp_path: Path) -> None: + """E2: cache_params ttl=1; post-expiry the same body misses, hits upstream again, labels populated.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, ttl=1) as rig: + text: Final = "cache hit control e2 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + assert rig.provider.received.qsize() == 1 + + rejects: list[int] = [] # mutable-ok: the poll helper must remember how many rejects it issued + + def expired_miss() -> int: + rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name)) + rejects.append(1) + return rig.provider.received.qsize() + + eventually(lambda: expired_miss() == 2, bool, seconds=70) + _expect_metrics(rig, len(rejects), 0) + + +def test_cache_hit_reject_metrics_aggregate_across_workers(gateway: Gateway, tmp_path: Path) -> None: + """E3: workers=2, 8 cache-hit rejects, aggregated /metrics shows +8 on the populated series.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, workers=2) as rig: + text: Final = "cache hit control e3 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + rejects: Final = tuple( + rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name)) + for _ in range(8) + ) + assert all(response.status_code == 400 for response in rejects), [r.text for r in rejects] + _expect_metrics(rig, 8, 0) + + +def test_cache_hit_reject_deployment_metric_set_diff(gateway: Gateway, tmp_path: Path) -> None: + """E4: exact expected label sets on litellm_deployment_* and litellm_proxy_failed_requests_metric.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + text: Final = "cache hit control e4 " + marker + warm: Final = rig.candidate.request("POST", "/v1/chat/completions", _chat_body(rig.model_name, text, None)) + assert warm.status_code == 200, warm.text + reject: Final = rig.candidate.request( + "POST", "/v1/chat/completions", _chat_body(rig.model_name, text, rig.guardrail_name) + ) + assert reject.status_code == 400, reject.text + samples: Final = _expect_metrics(rig, 1, 0) + blank: Final = tuple(sample for sample in samples if sample.labels.get("model_id") == "") + assert blank == (), blank + states: Final = tuple( + sample.value + for sample in samples + if sample.name == DEPLOYMENT_STATE + and sample.labels.get("model_id") == rig.deployment_id + and sample.labels.get("api_base") == "" + ) + assert states == (1.0,), states diff --git a/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py b/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py new file mode 100644 index 00000000000..faaa1fc7325 --- /dev/null +++ b/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py @@ -0,0 +1,348 @@ +import json +import signal +import socket +import subprocess +import threading +import uuid +from collections.abc import Callable, Generator +from concurrent.futures import ThreadPoolExecutor +from contextlib import contextmanager +from pathlib import Path +from typing import Final + +import httpx +import psutil +from integration._support.client import Gateway, eventually, object_value, string_value +from integration._support.process import owned_proxy_process +from integration._support.wire import Reply, Request, wire_server +from prometheus_client.parser import text_string_to_metric_families +from test_cache_hit_guardrail_metrics import ( + DEPLOYMENT_FAILURE, + GUARDRAIL_PATH, + PROXY_FAILED, + Rig, + _blocking_sink, + _chat_body, + _guardrail_config, + _provider, + _rig, +) + +BURST: Final = 10 + + +@contextmanager +def _redis(port: int) -> Generator[subprocess.Popen[bytes], None, None]: + process: Final = subprocess.Popen(["redis-server", "--port", str(port), "--save", ""], stdout=subprocess.DEVNULL) + try: + yield process + finally: + process.kill() + process.wait(timeout=10) + + +def _free_port() -> int: + with socket.socket() as reserve: + reserve.bind(("127.0.0.1", 0)) + return reserve.getsockname()[1] + + +def _deployment_id(candidate: Gateway, model_name: str) -> str: + entries: Final = candidate.get("/model/info")["data"] + assert isinstance(entries, list) + entry: Final = next(item for item in entries if object_value(item)["model_name"] == model_name) + return string_value(object_value(object_value(entry)["model_info"])["id"]) + + +def _stall_sink(stall: threading.Event, release: threading.Event) -> Callable[[Request], Reply]: + def respond(request: Request) -> Reply: + assert request.target == GUARDRAIL_PATH, request.target + if stall.is_set(): + release.wait(timeout=60) + return Reply(body=json.dumps({"action": "BLOCKED", "blocked_reason": "synthetic block"}).encode()) + + return respond + + +def _samples(candidate: Gateway, model_names: tuple[str, ...]) -> tuple: + response: Final = candidate.client.request( + "GET", "/metrics", headers={"Authorization": f"Bearer {candidate.key}"}, follow_redirects=True + ) + assert response.status_code == 200, f"GET /metrics: {response.status_code}" + return tuple( + sample + for family in text_string_to_metric_families(response.text) + for sample in family.samples + if sample.labels.get("requested_model") in model_names + ) + + +def _populated(samples: tuple, deployment_id: str) -> float: + return float( + sum( + sample.value + for sample in samples + if sample.name == DEPLOYMENT_FAILURE and sample.labels.get("model_id") == deployment_id + ) + ) + + +def _blank(samples: tuple) -> float: + return float( + sum( + sample.value + for sample in samples + if sample.name == DEPLOYMENT_FAILURE and sample.labels.get("model_id") == "" + ) + ) + + +def _proxy_failed(samples: tuple) -> float: + return float(sum(sample.value for sample in samples if sample.name == PROXY_FAILED)) + + +def _burst_bodies(rig: Rig, marker: str, anthropic_name: str | None) -> tuple[tuple[str, dict], ...]: + chat: Final = tuple( + ("/v1/chat/completions", _chat_body(rig.model_name, f"burst {marker} {index}", rig.guardrail_name)) + for index in range(BURST) + ) + responses: Final = tuple( + ( + "/v1/responses", + {"model": rig.model_name, "input": f"burst {marker} r{index}", "guardrails": [rig.guardrail_name]}, + ) + for index in range(BURST) + ) + messages: Final = ( + tuple( + ( + "/v1/messages", + { + "model": anthropic_name, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"burst {marker} m{index}"}], + "guardrails": [rig.guardrail_name], + }, + ) + for index in range(BURST) + ) + if anthropic_name is not None + else () + ) + return chat + responses + messages + + +def _warm(rig: Rig, bodies: tuple[tuple[str, dict], ...]) -> None: + for path, body in bodies: + warmed: Final = dict(body) + warmed.pop("guardrails", None) + response: Final = rig.candidate.request("POST", path, warmed) + assert response.status_code == 200, f"warm {path}: {response.status_code} {response.text}" + + +def _fire(rig: Rig, bodies: tuple[tuple[str, dict], ...]) -> tuple[tuple[int, str | None], ...]: + def call(item: tuple[str, dict]) -> tuple[int, str | None]: + path, body = item + try: + response: Final = rig.candidate.request("POST", path, body) + return response.status_code, response.headers.get("x-litellm-call-id") + except httpx.HTTPError: + return -1, None + + with ThreadPoolExecutor(max_workers=8) as pool: + return tuple(pool.map(call, bodies)) + + +def _expect_counted_within( + rig: Rig, model_names: tuple[str, ...], deployment_ids: tuple[str, ...], low: int, high: int +) -> None: + def converged() -> tuple: + samples: Final = _samples(rig.candidate, model_names) + populated: Final = sum(_populated(samples, deployment) for deployment in deployment_ids) + if low <= populated <= high and _blank(samples) == 0: + return samples + return () + + eventually(converged, bool, seconds=70) + + +def _expect_exactly_once(rig: Rig, model_names: tuple[str, ...], deployment_ids: tuple[str, ...], four_xx: int) -> None: + _expect_counted_within(rig, model_names, deployment_ids, four_xx, four_xx) + + +def test_burst_cache_hit_rejects_count_exactly_once(gateway: Gateway, tmp_path: Path) -> None: + """X0: 30 mixed-endpoint cache-hit rejects across two deployments, each counted once.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker) as rig: + anthropic_name: Final = rig.scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=rig.provider.url, api_key="synthetic-provider-key" + ) + anthropic_id: Final = _deployment_id(rig.candidate, anthropic_name) + bodies: Final = _burst_bodies(rig, marker, anthropic_name) + _warm(rig, bodies) + outcomes: Final = _fire(rig, bodies) + rejected: Final = sum(1 for status, _ in outcomes if status >= 400) + assert all(status == 400 for status, _ in outcomes), outcomes + _expect_exactly_once(rig, (rig.model_name, anthropic_name), (rig.deployment_id, anthropic_id), rejected) + + +def test_stalled_guardrail_sink_recovers_and_counts(gateway: Gateway, tmp_path: Path) -> None: + """X1: guardrail sink stalls mid-burst; requests fail exactly once, then recovery counts again.""" + marker: Final = uuid.uuid4().hex + stall: Final = threading.Event() + release: Final = threading.Event() + with _rig(gateway, tmp_path, marker, sink=_stall_sink(stall, release)) as rig: + bodies: Final = _burst_bodies(rig, marker, None) + _warm(rig, bodies) + stall.set() + with ThreadPoolExecutor(max_workers=8) as pool: + futures: Final = tuple( + pool.submit(lambda b: rig.candidate.request("POST", b[0], b[1]), body) for body in bodies + ) + eventually(lambda: rig.policy.received.qsize() >= 5, bool, seconds=30) + release.set() + outcomes: Final = tuple( + (future.result().status_code, future.result().headers.get("x-litellm-call-id")) for future in futures + ) + assert all(status >= 400 for status, _ in outcomes), outcomes + blocked: Final = sum(1 for status, _ in outcomes if status == 400) + outages: Final = sum(1 for status, _ in outcomes if status >= 500) + assert blocked + outages == len(bodies), outcomes + samples: Final = eventually( + lambda: _samples(rig.candidate, (rig.model_name,)), + lambda observed: _proxy_failed(observed) == blocked + outages, + seconds=70, + ) + assert _proxy_failed(samples) == blocked + outages, (samples, outcomes) + follow_up: Final = rig.candidate.request( + "POST", + "/v1/chat/completions", + _chat_body(rig.model_name, "post stall unrelated " + marker, None), + ) + assert follow_up.status_code == 200, follow_up.text + _expect_exactly_once(rig, (rig.model_name,), (rig.deployment_id,), blocked) + + +def test_redis_outage_keeps_serving_in_memory_hits(gateway: Gateway, tmp_path: Path) -> None: + """X2: the redis cache keeps an in-memory shadow, so a redis kill does not stop cache-hit rejects.""" + marker: Final = uuid.uuid4().hex + port: Final = _free_port() + with _redis(port) as redis_one: + with _rig(gateway, tmp_path, marker, env={"REDIS_HOST": "127.0.0.1", "REDIS_PORT": str(port)}) as rig: + bodies: Final = _burst_bodies(rig, marker, None)[:BURST] + _warm(rig, bodies) + reject: Final = rig.candidate.request("POST", *bodies[0]) + assert reject.status_code == 400, reject.text + warmed_hits: Final = rig.provider.received.qsize() + redis_one.kill() + redis_one.wait(timeout=10) + outcomes: Final = _fire(rig, bodies[1:]) + assert all(status == 400 for status, _ in outcomes), outcomes + assert rig.provider.received.qsize() == warmed_hits, ( + "redis outage reached the provider", + warmed_hits, + rig.provider.received.qsize(), + ) + with _redis(port): + recovered: Final = rig.candidate.request( + "POST", + "/v1/chat/completions", + _chat_body(rig.model_name, "x2 rehit " + marker, rig.guardrail_name), + ) + assert recovered.status_code == 400, recovered.text + _expect_exactly_once(rig, (rig.model_name,), (rig.deployment_id,), 1 + len(bodies)) + + +def test_worker_kill_mid_burst_keeps_counting(gateway: Gateway, tmp_path: Path) -> None: + """X3: workers=2, SIGKILL one uvicorn child mid-burst; survivors keep rejecting; the count is answered plus at most the in-flight requests the killed worker had already counted.""" + marker: Final = uuid.uuid4().hex + with _rig(gateway, tmp_path, marker, workers=2) as rig: + bodies: Final = _burst_bodies(rig, marker, None) + _warm(rig, bodies) + children: Final = psutil.Process(rig.process.pid).children(recursive=True) + assert children, "no uvicorn worker children found" + with ThreadPoolExecutor(max_workers=8) as pool: + futures: Final = tuple( + pool.submit(lambda b: rig.candidate.request("POST", b[0], b[1]), body) for body in bodies + ) + eventually(lambda: rig.policy.received.qsize() >= 3, bool, seconds=30) + children[0].send_signal(signal.SIGKILL) + statuses: list[int] = [] # mutable-ok: collect per-request outcomes from concurrent futures + for future in futures: + try: + statuses.append(future.result().status_code) + except httpx.HTTPError: + statuses.append(-1) + answered: Final = sum(1 for status in statuses if status >= 0) + transport_lost: Final = sum(1 for status in statuses if status == -1) + assert all(status == 400 for status in statuses if status >= 0), ( + statuses, + transport_lost, + ) + _expect_counted_within(rig, (rig.model_name,), (rig.deployment_id,), answered, answered + transport_lost) + + +def test_proxy_restart_mid_burst_keeps_counting(gateway: Gateway, tmp_path: Path) -> None: + """X4: restart the owned proxy between the two halves; pre-restart count asserted, then recounted.""" + marker: Final = uuid.uuid4().hex + prom_dir: Final = tmp_path / "prom" + prom_dir.mkdir() + with wire_server(_blocking_sink) as policy, wire_server(_provider(marker)) as provider: + config: Final = _guardrail_config(tmp_path, "guardrail-" + marker, policy.url) + bodies: Final = tuple( + ( + "/v1/chat/completions", + _chat_body("pending-model", f"burst {marker} {index}", "guardrail-" + marker), + ) + for index in range(BURST) + ) + with owned_proxy_process( + gateway, tmp_path, {"PROMETHEUS_MULTIPROC_DIR": str(prom_dir)}, config=config + ) as owned_one: + model: Final = "restart-" + marker + owned_one.gateway.post( + "/model/new", + { + "model_name": model, + "litellm_params": { + "model": "openai/gpt-4o-mini", + "api_base": provider.url + "/v1", + "api_key": "synthetic-provider-key", + }, + }, + ) + deployment: Final = _deployment_id(owned_one.gateway, model) + named: Final = tuple((path, {**body, "model": model}) for path, body in bodies) + first_half, second_half = named[: BURST // 2], named[BURST // 2 :] + for path, body in named: + warmed: Final = dict(body) + warmed.pop("guardrails", None) + assert owned_one.gateway.request("POST", path, warmed).status_code == 200 + outcomes_one: Final = tuple(owned_one.gateway.request("POST", path, body) for path, body in first_half) + assert all(response.status_code == 400 for response in outcomes_one), [r.text for r in outcomes_one] + pre: Final = eventually( + lambda: ( + _populated(_samples(owned_one.gateway, (model,)), deployment), + _blank(_samples(owned_one.gateway, (model,))), + ), + lambda observed: observed[0] == len(first_half) and observed[1] == 0, + seconds=70, + ) + with owned_proxy_process( + gateway, tmp_path, {"PROMETHEUS_MULTIPROC_DIR": str(prom_dir)}, config=config + ) as owned_two: + outcomes_two: Final = tuple(owned_two.gateway.request("POST", path, body) for path, body in second_half) + assert all(response.status_code == 400 for response in outcomes_two), ( + pre, + [(r.status_code, r.text[:200]) for r in outcomes_two], + ) + post: Final = eventually( + lambda: ( + _populated(_samples(owned_two.gateway, (model,)), deployment), + _blank(_samples(owned_two.gateway, (model,))), + ), + lambda observed: observed[0] == len(named) and observed[1] == 0, + seconds=70, + ) + assert post[0] == len(named), (pre, post, outcomes_two) + owned_two.gateway.post("/model/delete", {"id": deployment}) diff --git a/tests/test_litellm/proxy/utils/proxy_logging/test_post_call_failure_hook.py b/tests/test_litellm/proxy/utils/proxy_logging/test_post_call_failure_hook.py index 0046721fd9c..51145ca687b 100644 --- a/tests/test_litellm/proxy/utils/proxy_logging/test_post_call_failure_hook.py +++ b/tests/test_litellm/proxy/utils/proxy_logging/test_post_call_failure_hook.py @@ -18,6 +18,7 @@ from litellm.exceptions import GuardrailRaisedException from litellm.integrations.custom_logger import CustomLogger from litellm.proxy._types import ProxyErrorTypes, UserAPIKeyAuth from litellm.proxy.utils import ProxyLogging +from litellm.types.utils import CachingDetails @pytest.fixture(autouse=True) @@ -249,6 +250,65 @@ async def test_post_call_failure_hook_keeps_router_stamped_metadata_for_post_cal assert kwargs["standard_logging_object"]["model_id"] == "routed-deployment" +@pytest.mark.asyncio +async def test_post_call_failure_hook_keeps_deployment_attribution_for_cache_hit_post_call_failures( + proxy_logging, make_user_api_key_auth, monkeypatch +): + """A post-call guardrail blocks a response served from the litellm cache. No provider call was made, + so ``first_api_call_start_time`` is unset, but the router did pick the deployment: the pre-routing + flag must stay off so ``litellm_deployment_failure_responses`` keeps its model_id and provider labels.""" + from litellm.proxy import proxy_server + + recorded: list[dict] = [] + + class _RecordingLogger(CustomLogger): + async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): + recorded.append(kwargs) + + monkeypatch.setattr( + proxy_server, + "llm_router", + litellm.Router( + model_list=[ + { + "model_name": "internal-model", + "litellm_params": {"model": "openai/gpt-4.1", "api_key": "sk-test"}, + "model_info": {"id": "routed-deployment"}, + } + ] + ), + ) + monkeypatch.setattr(litellm, "callbacks", [_RecordingLogger()]) + proxy_logging.alert_types = [] + + request_data = { + "litellm_call_id": "cache-hit-post-call-guardrail", + "model": "internal-model", + "messages": [{"role": "user", "content": "hi"}], + "metadata": {"model_info": {"id": "routed-deployment"}}, + } + logging_obj, request_data = litellm.utils.function_setup( + original_function="acompletion", rules_obj=litellm.utils.Rules(), start_time=datetime.now(), **request_data + ) + logging_obj.caching_details = CachingDetails(cache_hit=True, cache_duration_ms=1.0) + request_data["litellm_logging_obj"] = logging_obj + + await proxy_logging.post_call_failure_hook( + request_data=request_data, + original_exception=GuardrailRaisedException(guardrail_name="g", message="response blocked"), + user_api_key_dict=make_user_api_key_auth(request_route="/chat/completions"), + route="/chat/completions", + ) + + assert len(recorded) == 1 + kwargs = recorded[0] + assert PROXY_REJECTED_BEFORE_ROUTING_KEY not in kwargs["litellm_params"], kwargs["litellm_params"] + assert kwargs["standard_logging_object"]["model_id"] == "routed-deployment" + assert kwargs["standard_logging_object"]["custom_llm_provider"] == "openai" + assert kwargs["model"] == "internal-model" + assert kwargs["litellm_params"]["custom_llm_provider"] == "openai" + + @pytest.mark.asyncio async def test_post_call_failure_hook_flags_pre_routing_reject_despite_caller_model_info( proxy_logging, make_user_api_key_auth, monkeypatch From 093ceb576d7d9743bbe6ca061791884084143916 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:54:19 -0700 Subject: [PATCH 42/96] fix(proxy): do not requeue a daily spend batch whose commit already left for postgres (#42786) * fix(proxy): do not requeue a daily spend batch whose commit already left for postgres Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): settle an interrupted daily spend commit from the shutdown flush instead of blocking the cancelled tick Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): burst two workers and SIGTERM during daily spend COMMIT, expect exactly once Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/proxy/db/db_spend_update_writer.py | 114 ++++++++- .../daily_spend_update_queue.py | 14 ++ .../integration/spend/test_shutdown_flush.py | 132 +++++++++- .../proxy/db/test_db_spend_update_writer.py | 238 ++++++++++++++++++ 4 files changed, 486 insertions(+), 12 deletions(-) diff --git a/litellm/proxy/db/db_spend_update_writer.py b/litellm/proxy/db/db_spend_update_writer.py index e38214c98a6..d3ac37a3e0f 100644 --- a/litellm/proxy/db/db_spend_update_writer.py +++ b/litellm/proxy/db/db_spend_update_writer.py @@ -12,7 +12,8 @@ import os import random import time import traceback -from collections.abc import Callable, Mapping, Sequence +from collections.abc import Callable, Coroutine, Mapping, Sequence +from contextvars import ContextVar from datetime import datetime, timedelta, timezone from types import MappingProxyType from typing import TYPE_CHECKING, Any, Final, Literal, Protocol, TypeAlias, TypeVar, cast, overload @@ -269,6 +270,80 @@ def _spend_update_tx(prisma_client: PrismaClient) -> _SpendTransactionManager: return tx +_daily_spend_commit_started: Final[ContextVar[asyncio.Event | None]] = ContextVar( + "_daily_spend_commit_started", default=None +) + + +def _mark_daily_spend_commit_started() -> None: + started: Final = _daily_spend_commit_started.get() + if started is not None: + started.set() + + +def _mark_daily_spend_commit_finished() -> None: + started: Final = _daily_spend_commit_started.get() + if started is not None: + started.clear() + + +def _start_daily_spend_commit( + commit_started: asyncio.Event, commit: Callable[[], Coroutine[object, object, None]] +) -> "asyncio.Task[None]": + token: Final = _daily_spend_commit_started.set(commit_started) + try: + return asyncio.ensure_future(commit()) + finally: + _daily_spend_commit_started.reset(token) + + +def _track_interrupted_commit(commits: set[asyncio.Task[None]], settle: Coroutine[object, object, None]) -> None: + task: Final = asyncio.ensure_future(settle) + commits.add(task) + task.add_done_callback(commits.discard) + + +async def _settle_interrupted_commits(commits: set[asyncio.Task[None]]) -> None: + while commits: + await asyncio.wait(tuple(commits)) + + +async def _restore_tag_spend_the_commit_left_behind( + commit_task: "asyncio.Task[None]", + redis_update_buffer: RedisUpdateBuffer, + transactions: dict[str, DailyTagSpendTransaction], +) -> None: + await asyncio.wait({commit_task}) + if commit_task.cancelled() or commit_task.exception() is None: + return + await redis_update_buffer.restore_transactions_to_redis( + daily_tag_spend_update_transactions=transactions, + ) + + +async def _requeue_daily_spend_the_commit_left_behind( + commit_task: "asyncio.Task[None]", + queue: DailySpendUpdateQueue, + entity_type: str, + transactions: dict[str, BaseDailySpendTransaction], +) -> None: + await asyncio.wait({commit_task}) + if commit_task.cancelled() or not transactions: + return + failure: Final = commit_task.exception() + if failure is None: + return + spend_log_error( + "Spend tracking - daily %s spend commit interrupted by shutdown failed. Re-queued %d rows for the " + "shutdown flush. Error: %s", + entity_type, + len(transactions), + str(failure), + exc=failure, + ) + await queue.add_update(transactions) + + # The per-team advisory lock the team endpoints hold while changing a roster (TEAM_ADVISORY_LOCK_SQL), # so the roster check below cannot interleave with their writes. A row lock would deadlock with the # access-group endpoints, which lock a team row after an access-group lock. @@ -391,6 +466,9 @@ class DBSpendUpdateWriter: self.daily_org_spend_update_queue = DailySpendUpdateQueue() self.daily_tag_spend_update_queue = DailySpendUpdateQueue() self.window_spend_update_queue = WindowSpendUpdateQueue() + self.interrupted_tag_commits: set[asyncio.Task[None]] = ( + set() + ) # mutable-ok: same registry as DailySpendUpdateQueue.interrupted_commits async def update_database( # LiteLLM management object fields @@ -1606,17 +1684,24 @@ class DBSpendUpdateWriter: proxy_logging_obj: ProxyLogging, ) -> None: transactions: Final = await queue.flush_and_get_aggregated_daily_spend_update_transactions() - commit_task: Final = asyncio.ensure_future( - commit( + commit_started: Final = asyncio.Event() + commit_task: Final = _start_daily_spend_commit( + commit_started, + lambda: commit( n_retry_times=n_retry_times, prisma_client=prisma_client, proxy_logging_obj=proxy_logging_obj, daily_spend_transactions=cast(dict[str, _DailySpendTransactionT], transactions), - ) + ), ) try: await asyncio.shield(commit_task) except asyncio.CancelledError: + if commit_started.is_set(): + queue.track_interrupted_commit( + _requeue_daily_spend_the_commit_left_behind(commit_task, queue, entity_type, transactions) + ) + raise commit_task.cancel() if transactions: await queue.add_update(transactions) @@ -1841,23 +1926,36 @@ class DBSpendUpdateWriter: The drain is destructive, so a failed commit must push the transactions back for the next tick or their spend is lost permanently. """ + await _settle_interrupted_commits(self.interrupted_tag_commits) daily_tag_spend_update_transactions: Final = ( await self.redis_update_buffer.get_all_daily_tag_spend_update_transactions_from_redis_buffer() ) if not daily_tag_spend_update_transactions: return - commit_task: Final = asyncio.ensure_future( - DBSpendUpdateWriter.update_daily_tag_spend( + commit_started: Final = asyncio.Event() + commit_task: Final = _start_daily_spend_commit( + commit_started, + lambda: DBSpendUpdateWriter.update_daily_tag_spend( n_retry_times=n_retry_times, prisma_client=prisma_client, proxy_logging_obj=proxy_logging_obj, daily_spend_transactions=daily_tag_spend_update_transactions, - ) + ), ) try: await asyncio.shield(commit_task) except BaseException: # noqa: BLE001 # a cancel must restore the drained rows before its rollback returns + if commit_started.is_set(): + _track_interrupted_commit( + self.interrupted_tag_commits, + _restore_tag_spend_the_commit_left_behind( + commit_task, + self.redis_update_buffer, + daily_tag_spend_update_transactions, + ), + ) + raise commit_task.cancel() await self.redis_update_buffer.restore_transactions_to_redis( daily_tag_spend_update_transactions=daily_tag_spend_update_transactions, @@ -2382,6 +2480,8 @@ class DBSpendUpdateWriter: sql, params = build_bulk_upsert(table=table, batch=merged_batch) async with _spend_update_tx(prisma_client) as transaction: await transaction.execute_raw(sql, *params) + _mark_daily_spend_commit_started() + _mark_daily_spend_commit_finished() except Exception as batch_error: if _spend_commit_failure_is_requeue_safe(batch_error): spend_log_error( diff --git a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py index c6381cd070b..f911d5a6767 100644 --- a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py +++ b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py @@ -1,4 +1,5 @@ import asyncio +from collections.abc import Coroutine from copy import deepcopy from typing import Final @@ -57,6 +58,18 @@ class DailySpendUpdateQueue(BaseUpdateQueue): self.update_queue: asyncio.Queue[dict[str, BaseDailySpendTransaction]] = asyncio.Queue( maxsize=LITELLM_ASYNCIO_QUEUE_MAXSIZE ) + self.interrupted_commits: set[asyncio.Task[None]] = ( + set() + ) # mutable-ok: registry of in-flight commit outcomes, entries leave via their done callback + + def track_interrupted_commit(self, settle: Coroutine[object, object, None]) -> None: + task: Final = asyncio.ensure_future(settle) + self.interrupted_commits.add(task) + task.add_done_callback(self.interrupted_commits.discard) + + async def settle_interrupted_commits(self) -> None: + while self.interrupted_commits: + await asyncio.wait(tuple(self.interrupted_commits)) async def add_update(self, update: dict[str, BaseDailySpendTransaction]): """Enqueue an update.""" @@ -81,6 +94,7 @@ class DailySpendUpdateQueue(BaseUpdateQueue): self, ) -> dict[str, BaseDailySpendTransaction]: """Get all updates from the queue and return all updates aggregated by daily_transaction_key. Works for both user and team spend updates.""" + await self.settle_interrupted_commits() updates: Final = await self.flush_all_updates_from_in_memory_queue() if len(updates) > 0: verbose_proxy_logger.info( diff --git a/tests/integration/spend/test_shutdown_flush.py b/tests/integration/spend/test_shutdown_flush.py index d27275f5213..b744c1f4b5f 100644 --- a/tests/integration/spend/test_shutdown_flush.py +++ b/tests/integration/spend/test_shutdown_flush.py @@ -4,6 +4,7 @@ import signal import threading import uuid from collections.abc import Callable, Iterator +from concurrent.futures import ThreadPoolExecutor from contextlib import contextmanager from dataclasses import dataclass from pathlib import Path @@ -13,16 +14,18 @@ import httpx import psycopg import pytest import yaml - from integration._support.client import Gateway, delete_key_if_present, eventually, string_value from integration._support.database import read_rows from integration._support.process import OwnedProxy, owned_proxy_process from integration._support.wire import Reply, Request, wire_server +from psycopg import sql REQUESTS_WHILE_BLOCKED: Final = 6 CANCEL_LOG_LINE: Final = "in-flight scheduled job(s) for shutdown" BATCH_DRAINED_LOG_LINE: Final = f"flushed {REQUESTS_WHILE_BLOCKED} daily spend update items from in-memory queue" MODEL_INSERT_ARRIVED_LOG_LINE: Final = "path=/model/new" +COMMIT_DELAY_SECONDS: Final = 15 +BURST_REQUESTS: Final = 30 def _api_requests(table: str, column: str, identity: str) -> int: @@ -44,6 +47,50 @@ def _waiting_on(table: str) -> int: return waiting +def _committing_daily_user_spend() -> int: + rows: Final = read_rows( + "SELECT count(*)::int AS committing FROM pg_stat_activity " + "WHERE query='COMMIT' AND state='active' AND wait_event='PgSleep' AND pid IN " + "(SELECT pid FROM pg_locks WHERE relation = %s::regclass AND mode='RowExclusiveLock')", + ('"LiteLLM_DailyUserSpend"',), + ) + committing: Final = rows[0]["committing"] + assert isinstance(committing, int) + return committing + + +def _install_slow_commit(user_id: str, fails_once: bool) -> str: + suffix: Final = f"slow_commit_{uuid.uuid4().hex}" + with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection: + connection.execute(sql.SQL("CREATE SEQUENCE {}").format(sql.Identifier(suffix))) + connection.execute( + sql.SQL( + "CREATE FUNCTION {}() RETURNS trigger LANGUAGE plpgsql AS $slow$ " + "BEGIN PERFORM pg_sleep({}); " + "IF {} AND nextval({}) = 1 THEN RAISE EXCEPTION 'integration: first COMMIT fails'; END IF; " + "RETURN NULL; END $slow$" + ).format( + sql.Identifier(suffix), sql.Literal(COMMIT_DELAY_SECONDS), sql.Literal(fails_once), sql.Literal(suffix) + ) + ) + connection.execute( + sql.SQL( + 'CREATE CONSTRAINT TRIGGER {} AFTER INSERT OR UPDATE ON "LiteLLM_DailyUserSpend" ' + "DEFERRABLE INITIALLY DEFERRED FOR EACH ROW WHEN (NEW.user_id = {}) EXECUTE FUNCTION {}()" + ).format(sql.Identifier(suffix), sql.Literal(user_id), sql.Identifier(suffix)) + ) + return suffix + + +def _drop_slow_commit(suffix: str) -> None: + with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection: + connection.execute( + sql.SQL('DROP TRIGGER IF EXISTS {} ON "LiteLLM_DailyUserSpend"').format(sql.Identifier(suffix)) + ) + connection.execute(sql.SQL("DROP FUNCTION IF EXISTS {}()").format(sql.Identifier(suffix))) + connection.execute(sql.SQL("DROP SEQUENCE IF EXISTS {}").format(sql.Identifier(suffix))) + + def _provider(request: Request) -> Reply: if request.method != "POST": return Reply(status=404, body=b'{"error":"not scripted"}') @@ -77,6 +124,17 @@ class _Shutdown: def daily_user_requests(self) -> int: return _api_requests("LiteLLM_DailyUserSpend", "user_id", self.owner) + def spend_logs(self) -> int: + rows: Final = read_rows('SELECT count(*)::int AS total FROM "LiteLLM_SpendLogs" WHERE "user"=%s', (self.owner,)) + total: Final = rows[0]["total"] + assert isinstance(total, int) + return total + + def burst(self, requests: int) -> None: + with ThreadPoolExecutor(max_workers=8) as pool: + for outcome in pool.map(lambda _: self.chat(), range(requests)): + assert outcome is None + def logged(self, line: str, times: int = 1) -> bool: return self.owned.log.read_text(errors="replace").count(line) >= times @@ -121,7 +179,15 @@ def _config_with_pool_limit(tmp_path: Path, pool_limit: int) -> Path: @contextmanager -def _proxy_with_one_seeded_row(gateway: Gateway, tmp_path: Path, pool_limit: int) -> Iterator[_Shutdown]: +def _proxy_with_one_seeded_row( + gateway: Gateway, + tmp_path: Path, + pool_limit: int, + cancel_timeout_seconds: int = 5, + settle_seconds: int = 0, + requests: int = REQUESTS_WHILE_BLOCKED, + workers: int = 1, +) -> Iterator[_Shutdown]: owner: Final = f"integration-owner-{uuid.uuid4().hex}" with gateway.scenario() as scenario, wire_server(_provider) as wire: model: Final = scenario.model(api_base=wire.url + "/v1", num_retries=0) @@ -133,9 +199,10 @@ def _proxy_with_one_seeded_row(gateway: Gateway, tmp_path: Path, pool_limit: int "LITELLM_LOG": "DEBUG", "GRACEFUL_SHUTDOWN_TIMEOUT": "1", "SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS": "1", - "SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS": "5", + "SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS": str(cancel_timeout_seconds), }, config=_config_with_pool_limit(tmp_path, pool_limit), + workers=workers, ) as owned: key: Final = string_value( owned.gateway.post("/key/generate", {"user_id": owner, "team_id": team, "models": [model]})["key"] @@ -145,8 +212,18 @@ def _proxy_with_one_seeded_row(gateway: Gateway, tmp_path: Path, pool_limit: int shutdown.chat() eventually(shutdown.daily_user_requests, lambda total: total == 1, seconds=60) yield shutdown - assert _api_requests("LiteLLM_DailyUserSpend", "user_id", owner) == 1 + REQUESTS_WHILE_BLOCKED - assert _api_requests("LiteLLM_DailyTeamSpend", "team_id", team) == 1 + REQUESTS_WHILE_BLOCKED + written: Final = 1 + requests + if settle_seconds: + eventually( + lambda: ( + _api_requests("LiteLLM_DailyUserSpend", "user_id", owner), + _api_requests("LiteLLM_DailyTeamSpend", "team_id", team), + ), + lambda totals: totals == (written, written), + seconds=settle_seconds, + ) + assert _api_requests("LiteLLM_DailyUserSpend", "user_id", owner) == written + assert _api_requests("LiteLLM_DailyTeamSpend", "team_id", team) == written @pytest.mark.covers("quota_management.spend_tracking.shutdown_cancel_keeps_in_flight_daily_batch") @@ -187,3 +264,48 @@ def test_daily_spend_batch_cancelled_while_waiting_for_a_row_lock_is_written_exa lambda: shutdown.logged(BATCH_DRAINED_LOG_LINE) and _waiting_on("LiteLLM_DailyUserSpend") == 1, holder.rollback, ) + + +@pytest.mark.parametrize( + ("cancel_timeout_seconds", "commit_fails_once"), + [ + pytest.param(60, False, id="cancel_budget_outlives_commit"), + pytest.param(5, False, id="commit_outlives_cancel_budget"), + pytest.param(60, True, id="commit_fails_within_cancel_budget"), + pytest.param(5, True, id="commit_fails_after_cancel_budget"), + ], +) +def test_daily_spend_batch_cancelled_while_postgres_is_committing_it_is_written_exactly_once( + gateway: Gateway, tmp_path: Path, cancel_timeout_seconds: int, commit_fails_once: bool +) -> None: + with ( + _proxy_with_one_seeded_row( + gateway, tmp_path, pool_limit=10, cancel_timeout_seconds=cancel_timeout_seconds, settle_seconds=90 + ) as shutdown, + psycopg.connect(os.environ["DATABASE_URL"]) as memberships, + ): + suffix: Final = _install_slow_commit(shutdown.owner, fails_once=commit_fails_once) + try: + shutdown.chat_while_spend_update_is_blocked(memberships, "LiteLLM_TeamMembership") + memberships.rollback() + shutdown.terminate_once( + lambda: shutdown.logged(BATCH_DRAINED_LOG_LINE) and _committing_daily_user_spend() == 1, + lambda: None, + ) + finally: + _drop_slow_commit(suffix) + + +def test_daily_spend_burst_across_two_workers_survives_shutdown_during_commit_exactly_once( + gateway: Gateway, tmp_path: Path +) -> None: + with _proxy_with_one_seeded_row( + gateway, tmp_path, pool_limit=10, settle_seconds=120, requests=BURST_REQUESTS, workers=2 + ) as shutdown: + suffix: Final = _install_slow_commit(shutdown.owner, fails_once=False) + try: + shutdown.burst(BURST_REQUESTS) + eventually(shutdown.spend_logs, lambda total: total == 1 + BURST_REQUESTS, seconds=60) + shutdown.terminate_once(lambda: _committing_daily_user_spend() >= 1, lambda: None) + finally: + _drop_slow_commit(suffix) diff --git a/tests/test_litellm/proxy/db/test_db_spend_update_writer.py b/tests/test_litellm/proxy/db/test_db_spend_update_writer.py index daa07c8224a..bf3b9aed234 100644 --- a/tests/test_litellm/proxy/db/test_db_spend_update_writer.py +++ b/tests/test_litellm/proxy/db/test_db_spend_update_writer.py @@ -4501,3 +4501,241 @@ async def test_tag_batch_drained_from_redis_and_cancelled_mid_flight_is_restored await asyncio.wait_for(db.rolled_back.wait(), timeout=5) assert db.transaction_outcomes == ["rollback"] assert _daily_upserts(db, "LiteLLM_DailyTagSpend") == [] + + +class _CommittingDailySpendFakeDB(_DailySpendFakeDB): + """Runs the daily upsert at once but holds the COMMIT until released. A COMMIT that has left + the client lands on the server whether or not the client keeps waiting for the reply.""" + + def __init__(self) -> None: + super().__init__(failing_table=None) + self.committing = asyncio.Event() + self.commit_release = asyncio.Event() + self.transaction_outcomes: list[str] = [] + + @asynccontextmanager + async def _tx(self) -> AsyncIterator["_CommittingDailySpendFakeDB"]: + try: + yield self + except BaseException: + self.transaction_outcomes.append("rollback") + raise + self.committing.set() + try: + await self.commit_release.wait() + finally: + self.transaction_outcomes.append("commit") + + +@pytest.mark.parametrize(("queue_name", "entity_type", "entity_id_field", "table"), _DAILY_SPEND_ENTITIES) +@pytest.mark.asyncio +async def test_cancel_that_lands_while_the_daily_batch_is_committing_waits_for_the_commit_and_does_not_requeue_it( + queue_name: str, entity_type: str, entity_id_field: str, table: str +): + """Shutdown cancels the tick after the COMMIT has left for Postgres. The server finishes that + commit whatever the client does, so putting the batch back on the queue makes the final flush + write the same spend a second time. The tick has to wait for the commit's outcome instead.""" + db_writer = DBSpendUpdateWriter() + queue = _DAILY_SPEND_QUEUES[queue_name](db_writer) + await queue.add_update({"key-a": _daily_entity_txn(entity_id_field)}) + await queue.add_update({"key-a": _daily_entity_txn(entity_id_field)}) + db = _CommittingDailySpendFakeDB() + + def flush(prisma_db: _DailySpendFakeDB): + return db_writer._flush_daily_spend_queue( + queue=queue, + entity_type=entity_type, + commit=_DAILY_SPEND_COMMITS[entity_type], + n_retry_times=0, + prisma_client=_WindowSpendFakePrisma(prisma_db), + proxy_logging_obj=MagicMock(), + ) + + tick = asyncio.ensure_future(flush(db)) + await asyncio.wait_for(db.committing.wait(), timeout=5) + tick.cancel() + finished, _ = await asyncio.wait({tick}, timeout=0.2) + assert finished == {tick}, "the cancelled tick must hand the in-flight commit's outcome to the next flush" + with pytest.raises(asyncio.CancelledError): + tick.result() + assert len(queue.interrupted_commits) == 1 + + db.commit_release.set() + await queue.settle_interrupted_commits() + + assert db.transaction_outcomes == ["commit"] + (upsert,) = _daily_upserts(db, table) + assert _row_values(upsert, "api_requests") == [2] + assert queue.update_queue.empty(), "a batch whose COMMIT already left for the server must not be requeued" + + final_db = _DailySpendFakeDB(failing_table=None) + await flush(final_db) + assert _daily_upserts(final_db, table) == [], "the final flush must not write the committed batch again" + + +@pytest.mark.asyncio +async def test_tag_batch_drained_from_redis_and_cancelled_while_committing_is_not_restored(): + """Same in-flight COMMIT as the in-memory path, but the drained rows live in Redis. Restoring + them after the server committed writes the tag spend twice on the next tick.""" + db_writer = DBSpendUpdateWriter() + drained = {"key-a": cast(DailyTagSpendTransaction, _daily_entity_txn("tag"))} + redis_buffer = _DrainedTagRedisBuffer(drained) + db_writer.redis_update_buffer = cast(RedisUpdateBuffer, redis_buffer) + db = _CommittingDailySpendFakeDB() + + tick = asyncio.ensure_future( + db_writer._drain_and_commit_daily_tag_spend_from_redis( + prisma_client=_WindowSpendFakePrisma(db), + n_retry_times=0, + proxy_logging_obj=MagicMock(), + ) + ) + await asyncio.wait_for(db.committing.wait(), timeout=5) + tick.cancel() + finished, _ = await asyncio.wait({tick}, timeout=0.2) + assert finished == {tick}, "the cancelled drain must hand the in-flight commit's outcome to the next drain" + with pytest.raises(asyncio.CancelledError): + tick.result() + assert len(db_writer.interrupted_tag_commits) == 1 + + db.commit_release.set() + (settle,) = tuple(db_writer.interrupted_tag_commits) + await settle + + assert db.transaction_outcomes == ["commit"] + assert redis_buffer.restored == [], ( + "a tag batch whose COMMIT already left for the server must not be restored to Redis" + ) + + +class _CommitFailingDailySpendFakeDB(_CommittingDailySpendFakeDB): + """COMMIT leaves for the server but the reply comes back as a failure.""" + + @asynccontextmanager + async def _tx(self) -> AsyncIterator["_CommitFailingDailySpendFakeDB"]: + yield self + self.committing.set() + await self.commit_release.wait() + self.transaction_outcomes.append("commit_failed") + raise Exception("connection reset") + + +@pytest.mark.asyncio +async def test_cancel_while_committing_requeues_the_batch_when_the_commit_itself_fails(): + """Waiting for the in-flight commit's outcome must not swallow a real commit failure: + the batch still goes back on the queue and the next flush writes it once.""" + db_writer = DBSpendUpdateWriter() + queue = db_writer.daily_spend_update_queue + await queue.add_update({"key-a": _daily_txn()}) + await queue.add_update({"key-a": _daily_txn()}) + db = _CommitFailingDailySpendFakeDB() + + def flush(prisma_db: _DailySpendFakeDB): + return db_writer._flush_daily_spend_queue( + queue=queue, + entity_type="user", + commit=DBSpendUpdateWriter.update_daily_user_spend, + n_retry_times=0, + prisma_client=_WindowSpendFakePrisma(prisma_db), + proxy_logging_obj=MagicMock(), + ) + + tick = asyncio.ensure_future(flush(db)) + await asyncio.wait_for(db.committing.wait(), timeout=5) + tick.cancel() + finished, _ = await asyncio.wait({tick}, timeout=0.2) + assert finished == {tick}, "the cancelled tick must not eat the shutdown budget waiting on the commit" + with pytest.raises(asyncio.CancelledError): + tick.result() + + db.commit_release.set() + await queue.settle_interrupted_commits() + + assert db.transaction_outcomes == ["commit_failed"] + assert not queue.update_queue.empty(), "a batch whose COMMIT came back failed must be requeued" + + final_db = _DailySpendFakeDB(failing_table=None) + await flush(final_db) + (upsert,) = _daily_upserts(final_db, "LiteLLM_DailyUserSpend") + assert _row_values(upsert, "api_requests") == [2] + assert queue.update_queue.empty() + + +@pytest.mark.asyncio +async def test_shutdown_flush_that_lands_before_the_interrupted_commit_resolves_still_writes_a_failed_batch_once(): + """The cancelled tick returns right away, so a COMMIT can still be in flight when the + shutdown flush runs. If that commit later fails, the flush must first settle it, pick the + requeued rows back up, and write them exactly once instead of losing them.""" + db_writer = DBSpendUpdateWriter() + queue = db_writer.daily_spend_update_queue + await queue.add_update({"key-a": _daily_txn()}) + await queue.add_update({"key-a": _daily_txn()}) + db = _CommitFailingDailySpendFakeDB() + + def flush(prisma_db: _DailySpendFakeDB): + return db_writer._flush_daily_spend_queue( + queue=queue, + entity_type="user", + commit=DBSpendUpdateWriter.update_daily_user_spend, + n_retry_times=0, + prisma_client=_WindowSpendFakePrisma(prisma_db), + proxy_logging_obj=MagicMock(), + ) + + tick = asyncio.ensure_future(flush(db)) + await asyncio.wait_for(db.committing.wait(), timeout=5) + tick.cancel() + with pytest.raises(asyncio.CancelledError): + await asyncio.wait_for(tick, timeout=5) + assert db.transaction_outcomes == [], "the COMMIT is still on the wire when the shutdown flush starts" + + final_db = _DailySpendFakeDB(failing_table=None) + shutdown_flush = asyncio.ensure_future(flush(final_db)) + finished, _ = await asyncio.wait({shutdown_flush}, timeout=0.2) + assert finished == set(), "the shutdown flush must wait for the interrupted commit's outcome" + assert _daily_upserts(final_db, "LiteLLM_DailyUserSpend") == [] + + db.commit_release.set() + await asyncio.wait_for(shutdown_flush, timeout=5) + + (upsert,) = _daily_upserts(final_db, "LiteLLM_DailyUserSpend") + assert _row_values(upsert, "api_requests") == [2] + assert queue.update_queue.empty() + + +@pytest.mark.asyncio +async def test_shutdown_drain_that_lands_before_the_interrupted_tag_commit_resolves_restores_a_failed_batch(): + """Same ordering for the Redis tag path: the shutdown drain must settle the interrupted + commit before the destructive drain, or a commit that fails late is never restored.""" + db_writer = DBSpendUpdateWriter() + drained = {"key-a": cast(DailyTagSpendTransaction, _daily_entity_txn("tag"))} + redis_buffer = _DrainedTagRedisBuffer(drained) + db_writer.redis_update_buffer = cast(RedisUpdateBuffer, redis_buffer) + db = _CommitFailingDailySpendFakeDB() + + def drain(prisma_db: _DailySpendFakeDB): + return db_writer._drain_and_commit_daily_tag_spend_from_redis( + prisma_client=_WindowSpendFakePrisma(prisma_db), + n_retry_times=0, + proxy_logging_obj=MagicMock(), + ) + + tick = asyncio.ensure_future(drain(db)) + await asyncio.wait_for(db.committing.wait(), timeout=5) + tick.cancel() + with pytest.raises(asyncio.CancelledError): + await asyncio.wait_for(tick, timeout=5) + assert db.transaction_outcomes == [] + + final_db = _DailySpendFakeDB(failing_table=None) + shutdown_drain = asyncio.ensure_future(drain(final_db)) + finished, _ = await asyncio.wait({shutdown_drain}, timeout=0.2) + assert finished == set(), "the shutdown drain must wait for the interrupted commit's outcome" + assert _daily_upserts(final_db, "LiteLLM_DailyTagSpend") == [] + + db.commit_release.set() + await asyncio.wait_for(shutdown_drain, timeout=5) + + assert redis_buffer.restored == [drained], "a tag batch whose COMMIT came back failed must be restored to Redis" + (upsert,) = _daily_upserts(final_db, "LiteLLM_DailyTagSpend") + assert _row_values(upsert, "api_requests") == [1] From 8f7cea5fcd87229f4768e3cc4628fa68d034c17a Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:21:59 -0700 Subject: [PATCH 43/96] fix(cost-map): sync openrouter prices and add fireworks ember-1 (#42889) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 27 ++++++++++++++++--- model_prices_and_context_window.json | 27 ++++++++++++++++--- 2 files changed, 48 insertions(+), 6 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 0b606f5db4e..e19ddaf72a7 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41253,6 +41253,27 @@ "supports_vision": false, "supports_web_search": false }, + "openrouter/fireworks/ember-1": { + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "cache_read_input_token_cost": 3e-07, + "litellm_provider": "openrouter", + "max_input_tokens": 1048576, + "max_output_tokens": 943718, + "max_tokens": 943718, + "mode": "chat", + "source": "https://openrouter.ai/api/v1/models", + "supports_audio_input": false, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_parallel_function_calling": false, + "supports_pdf_input": false, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_web_search": false + }, "openrouter/google/gemini-2.5-flash": { "cache_creation_input_token_cost": 8.33333333333333e-08, "cache_read_input_audio_token_cost": 1e-07, @@ -65519,7 +65540,7 @@ "supports_prompt_caching": true }, "openrouter/deepseek/deepseek-v4-flash-0731": { - "input_cost_per_token": 4e-08, + "input_cost_per_token": 3e-08, "output_cost_per_token": 3.2e-07, "cache_read_input_token_cost": 1.6e-08, "litellm_provider": "openrouter", @@ -67140,8 +67161,8 @@ "supports_web_search": false }, "openrouter/qwen/qwen3-30b-a3b-instruct-2507": { - "input_cost_per_token": 4.815e-08, - "output_cost_per_token": 1.9305e-07, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 32000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 0b606f5db4e..e19ddaf72a7 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41253,6 +41253,27 @@ "supports_vision": false, "supports_web_search": false }, + "openrouter/fireworks/ember-1": { + "input_cost_per_token": 3e-06, + "output_cost_per_token": 1.5e-05, + "cache_read_input_token_cost": 3e-07, + "litellm_provider": "openrouter", + "max_input_tokens": 1048576, + "max_output_tokens": 943718, + "max_tokens": 943718, + "mode": "chat", + "source": "https://openrouter.ai/api/v1/models", + "supports_audio_input": false, + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_parallel_function_calling": false, + "supports_pdf_input": false, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_web_search": false + }, "openrouter/google/gemini-2.5-flash": { "cache_creation_input_token_cost": 8.33333333333333e-08, "cache_read_input_audio_token_cost": 1e-07, @@ -65519,7 +65540,7 @@ "supports_prompt_caching": true }, "openrouter/deepseek/deepseek-v4-flash-0731": { - "input_cost_per_token": 4e-08, + "input_cost_per_token": 3e-08, "output_cost_per_token": 3.2e-07, "cache_read_input_token_cost": 1.6e-08, "litellm_provider": "openrouter", @@ -67140,8 +67161,8 @@ "supports_web_search": false }, "openrouter/qwen/qwen3-30b-a3b-instruct-2507": { - "input_cost_per_token": 4.815e-08, - "output_cost_per_token": 1.9305e-07, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 3e-07, "litellm_provider": "openrouter", "max_input_tokens": 262144, "max_output_tokens": 32000, From 42d8d08815236862963fac4d2f3302bb3a51bf49 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:25:24 -0700 Subject: [PATCH 44/96] fix(cost-map): source and chat completions endpoint for bedrock mantle gpt-5.4 and gpt-5.5 (#42890) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 8 ++++++-- model_prices_and_context_window.json | 8 ++++++-- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e19ddaf72a7..c01f7c2f396 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -57528,6 +57528,7 @@ "mode": "responses", "use_openai_responses_path": true, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -57545,7 +57546,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-55.html" }, "bedrock_mantle/openai.gpt-5.4": { "input_cost_per_token": 2.75e-06, @@ -57566,6 +57568,7 @@ "mode": "responses", "use_openai_responses_path": true, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -57583,7 +57586,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-54.html" }, "bedrock_mantle/google.gemma-4-31b": { "input_cost_per_token": 1.4e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e19ddaf72a7..c01f7c2f396 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -57528,6 +57528,7 @@ "mode": "responses", "use_openai_responses_path": true, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -57545,7 +57546,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-55.html" }, "bedrock_mantle/openai.gpt-5.4": { "input_cost_per_token": 2.75e-06, @@ -57566,6 +57568,7 @@ "mode": "responses", "use_openai_responses_path": true, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -57583,7 +57586,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-54.html" }, "bedrock_mantle/google.gemma-4-31b": { "input_cost_per_token": 1.4e-07, From a5b9b6da4dc7bd8d95dc0af4a920a77a4a1f85b4 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:41:10 -0700 Subject: [PATCH 45/96] feat(cost-map): add vertex_ai/gemini-3.8-live (#42891) * feat(cost-map): add vertex_ai/gemini-3.8-live Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost-map): price vertex_ai/gemini-3.8-live video tokens Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 36 +++++++++++++++++++ model_prices_and_context_window.json | 36 +++++++++++++++++++ 2 files changed, 72 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c01f7c2f396..b73f12bb673 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -27384,6 +27384,42 @@ "supports_web_search": false, "supports_native_streaming": true }, + "vertex_ai/gemini-3.8-live": { + "input_cost_per_audio_token": 3e-06, + "input_cost_per_image_token": 1e-06, + "input_cost_per_token": 7.5e-07, + "input_cost_per_video_per_second": 3.3333333333333335e-05, + "input_cost_per_video_token": 1e-06, + "litellm_provider": "vertex_ai", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "realtime", + "output_cost_per_audio_token": 1.2e-05, + "output_cost_per_token": 4.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/vertex_ai/live", + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_response_schema": false, + "supports_vision": true, + "supports_web_search": true, + "gemini_audio_only_live": true + }, "vertex_ai/gemini-3.1-pro-preview": { "prompt_cache_min_tokens": 4096, "cache_read_input_token_cost": 2e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c01f7c2f396..b73f12bb673 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -27384,6 +27384,42 @@ "supports_web_search": false, "supports_native_streaming": true }, + "vertex_ai/gemini-3.8-live": { + "input_cost_per_audio_token": 3e-06, + "input_cost_per_image_token": 1e-06, + "input_cost_per_token": 7.5e-07, + "input_cost_per_video_per_second": 3.3333333333333335e-05, + "input_cost_per_video_token": 1e-06, + "litellm_provider": "vertex_ai", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "realtime", + "output_cost_per_audio_token": 1.2e-05, + "output_cost_per_token": 4.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_endpoints": [ + "/vertex_ai/live", + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_response_schema": false, + "supports_vision": true, + "supports_web_search": true, + "gemini_audio_only_live": true + }, "vertex_ai/gemini-3.1-pro-preview": { "prompt_cache_min_tokens": 4096, "cache_read_input_token_cost": 2e-07, From 6dbd65b23098644c75d948ed2c26bd74b5e0e73c Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:42:24 -0700 Subject: [PATCH 46/96] fix(passthrough): log upstream 4xx/5xx error bodies and carry them into the failure hook (#42695) * test(integration): reproduce passthrough upstream error body missing from logs and spend row Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(passthrough): log upstream 4xx/5xx error bodies and carry them into the failure hook Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(error_normalization): let the passthrough prefix win over upstream body text Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(passthrough): honor message redaction for upstream error bodies Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(passthrough): bound the upstream error body read and sanitize it before logging Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(passthrough): use the Sequence import directly in the allowed-routes cast Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(passthrough): rechunk the upstream error stream so the preview read stays bounded Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): audit matrix for passthrough upstream error visibility Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(passthrough): drop the restating docstring on the upstream failure logger Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): drop the retired covers markers from the passthrough error tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(passthrough): keep the upstream status when the error body peek fails Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(passthrough): cover the relay aclose in the mid-read failure test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(passthrough): relay decoded partial body on mid-read failure Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/constants.py | 1 + .../litellm_core_utils/error_normalization.py | 2 +- .../pass_through_endpoints.py | 144 +++- tests/integration/_support/wire.py | 6 +- .../test_passthrough_upstream_error_chaos.py | 150 ++++ ...t_passthrough_upstream_error_visibility.py | 600 +++++++++++++++ .../test_error_normalization.py | 12 + .../test_pass_through_endpoints.py | 716 +++++++++++++++++- 8 files changed, 1595 insertions(+), 36 deletions(-) create mode 100644 tests/integration/observability/test_passthrough_upstream_error_chaos.py create mode 100644 tests/integration/observability/test_passthrough_upstream_error_visibility.py diff --git a/litellm/constants.py b/litellm/constants.py index e5b662bd515..dcc12ef2ba9 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1518,6 +1518,7 @@ PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES: Final = int( CLOUDZERO_EXPORT_INTERVAL_MINUTES: Final = int(os.getenv("CLOUDZERO_EXPORT_INTERVAL_MINUTES", 60)) MCP_TOOL_NAME_PREFIX: Final = "mcp_tool" MAXIMUM_TRACEBACK_LINES_TO_LOG: Final = int(os.getenv("MAXIMUM_TRACEBACK_LINES_TO_LOG", 100)) +PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS: Final = 4096 # Headers to control callbacks X_LITELLM_DISABLE_CALLBACKS: Final = "x-litellm-disable-callbacks" diff --git a/litellm/litellm_core_utils/error_normalization.py b/litellm/litellm_core_utils/error_normalization.py index 61eb47a9675..be0098ec34b 100644 --- a/litellm/litellm_core_utils/error_normalization.py +++ b/litellm/litellm_core_utils/error_normalization.py @@ -60,13 +60,13 @@ class _HasProxyErrorType(Protocol): _MESSAGE_PATTERNS: Final[tuple[tuple[re.Pattern[str], str], ...]] = ( + (re.compile(r"upstream passthrough request failed", re.IGNORECASE), UPSTREAM_PASSTHROUGH), ( re.compile(r"budget has been exceeded|max budget|crossed budget", re.IGNORECASE), BUDGET_EXCEEDED, ), (re.compile(r"no healthy deployments?|no deployments available", re.IGNORECASE), NO_HEALTHY_DEPLOYMENTS), (re.compile(r"not allowed to access model due to tags configuration", re.IGNORECASE), MODEL_ACCESS_DENIED), - (re.compile(r"upstream passthrough request failed", re.IGNORECASE), UPSTREAM_PASSTHROUGH), (re.compile(r"is not supported for provider|not implemented", re.IGNORECASE), UNSUPPORTED_OPERATION), ( re.compile(r"context window|context length|(prompt|input) is too long|tokens? ?> ?\d+ ?maximum", re.IGNORECASE), diff --git a/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py b/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py index f985c1d49d1..a119335ba46 100644 --- a/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py +++ b/litellm/proxy/pass_through_endpoints/pass_through_endpoints.py @@ -5,7 +5,7 @@ import json import posixpath import traceback from base64 import b64encode -from collections.abc import AsyncGenerator, Callable, Iterable, Mapping, Sequence +from collections.abc import AsyncGenerator, AsyncIterator, Callable, Iterable, Mapping, Sequence from dataclasses import dataclass from datetime import datetime from itertools import count, groupby @@ -41,6 +41,8 @@ from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid from litellm.constants import ( MAXIMUM_TRACEBACK_LINES_TO_LOG, + PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS, + REDACTED_BY_LITELLM, SESSION_ID_OMITTED_METADATA_KEY, WEBSOCKET_CLOSE_REASON_MAX_BYTES, ) @@ -56,6 +58,7 @@ from litellm.litellm_core_utils.internal_call_metadata import MODEL_ACCESS_GROUP from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.litellm_logging import _get_masked_values from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER +from litellm.litellm_core_utils.redact_messages import should_redact_message_logging from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.llms.base_llm.managed_resources.utils import ( resolve_passthrough_managed_id_provider, @@ -849,23 +852,106 @@ def _resolve_team_callback_wiring( ) +def _truncate_upstream_error_body(body: str) -> str: + if len(body) <= PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS: + return body + return ( + f"{body[:PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS]}... " + f"(truncated at {PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS} chars)" + ) + + +def _sanitize_upstream_error_body(body: str) -> str: + return " ".join("".join(char if char.isprintable() else " " for char in body).split()) + + +class _PrefixReplayStream(httpx.AsyncByteStream): + def __init__(self, prefix: bytes, rest: AsyncIterator[bytes], upstream: httpx.Response) -> None: + self._prefix: Final = prefix + self._rest: Final = rest + self._upstream: Final = upstream + + async def __aiter__(self) -> AsyncIterator[bytes]: + if self._prefix: + yield self._prefix + async for chunk in self._rest: + yield chunk + + async def aclose(self) -> None: + await self._upstream.aclose() + + +async def _no_more_chunks() -> AsyncIterator[bytes]: + return + yield b"" + + +async def _read_error_body_preview( + stream: AsyncIterator[bytes], +) -> tuple[bytes, AsyncIterator[bytes]]: + collected: Final[list[bytes]] = [] # mutable-ok: accumulated until the preview byte budget, then joined once + total = 0 # rebind-ok: running byte count against the preview budget + try: + async for chunk in stream: + collected.append(chunk) + total += len(chunk) + if total > PASSTHROUGH_UPSTREAM_ERROR_BODY_MAX_LOG_CHARS: + break + except httpx.HTTPError as err: + partial: Final = b"".join(collected) + verbose_proxy_logger.warning( + "pass_through_endpoint: upstream error body read failed after %d bytes: %s", + len(partial), + type(err).__name__, + ) + return partial, _no_more_chunks() + return b"".join(collected), stream + + +def _headers_without_body_framing(headers: httpx.Headers) -> httpx.Headers: + return httpx.Headers( + [(name, value) for name, value in headers.raw if name.lower() not in (b"content-encoding", b"content-length")] + ) + + +async def _error_body_preview_and_relay(response: httpx.Response) -> tuple[str, httpx.Response]: + if response.is_stream_consumed: + return response.text, response + body_iter: Final = response.aiter_bytes() + prefix, rest = await _read_error_body_preview(body_iter) + preview_text: Final = prefix.decode(response.encoding or "utf-8", errors="replace") + return preview_text, httpx.Response( + status_code=response.status_code, + headers=_headers_without_body_framing(response.headers), + stream=_PrefixReplayStream(prefix=prefix, rest=rest, upstream=response), + request=response.request, + extensions=response.extensions, + ) + + async def _log_passthrough_upstream_failure( response: httpx.Response, user_api_key_dict: UserAPIKeyAuth, request_payload: dict, -) -> None: - """Fire LiteLLM-side failure hooks (spend tracking, alerting callbacks) for - an upstream 4xx/5xx passthrough response. - - Passthrough must return the upstream status/body/headers to the client - unchanged, so this never raises or transforms the response - it only - mirrors the monitoring side effect that ``post_call_failure_hook`` would - have received had the error originated inside LiteLLM. - """ + logging_obj: LiteLLMLoggingObj, +) -> httpx.Response: if response.status_code < 400: - return + return response from litellm.proxy.proxy_server import proxy_logging_obj + preview_text, relay_response = await _error_body_preview_and_relay(response) + upstream_error_body: Final = ( + REDACTED_BY_LITELLM + if should_redact_message_logging(logging_obj.model_call_details) + else _truncate_upstream_error_body(_sanitize_upstream_error_body(preview_text)) + ) + verbose_proxy_logger.warning( + "pass_through_endpoint: upstream %s %s returned %s: %s", + response.request.method, + response.url.copy_with(query=None, fragment=None), + response.status_code, + upstream_error_body, + ) try: response.raise_for_status() except httpx.HTTPStatusError: @@ -878,7 +964,7 @@ async def _log_passthrough_upstream_failure( # rate-limit errors already are. synthetic_exception: Final = HTTPException( status_code=response.status_code, - detail=f"Upstream passthrough request failed with status {response.status_code}", + detail=f"Upstream passthrough request failed with status {response.status_code}: {upstream_error_body}", ) try: await proxy_logging_obj.post_call_failure_hook( @@ -892,6 +978,7 @@ async def _log_passthrough_upstream_failure( "pass_through_endpoint: post_call_failure_hook raised for upstream error", exc_info=True, ) + return relay_response async def _relay_reporting_failures( @@ -1321,7 +1408,7 @@ async def pass_through_request( headers=response.headers, ) - await _log_passthrough_upstream_failure( + relay_response: Final = await _log_passthrough_upstream_failure( response=response, user_api_key_dict=user_api_key_dict, request_payload=_build_passthrough_failure_request_payload( @@ -1331,17 +1418,18 @@ async def pass_through_request( custom_llm_provider=custom_llm_provider, upstream_usage=upstream_usage, ), + logging_obj=logging_obj, ) # Call response headers hook for streaming pass-through _response_headers = HttpPassThroughEndpointHelpers.get_response_headers( - headers=response.headers, + headers=relay_response.headers, litellm_call_id=litellm_call_id, ) callback_headers = await proxy_logging_obj.post_call_response_headers_hook( data=_parsed_body or {}, user_api_key_dict=user_api_key_dict, - response=response, + response=relay_response, request_headers=dict(request.headers), ) if callback_headers: @@ -1352,7 +1440,7 @@ async def pass_through_request( stream=_own_streamed_managed_ids( stream=_relay_reporting_failures( stream=PassThroughStreamingHandler.chunk_processor( - response=response, + response=relay_response, request_body=_parsed_body, litellm_logging_obj=logging_obj, endpoint_type=endpoint_type, @@ -1360,7 +1448,7 @@ async def pass_through_request( passthrough_success_handler_obj=pass_through_endpoint_logging, url_route=str(url), ), - upstream_status=response.status_code, + upstream_status=relay_response.status_code, user_api_key_dict=user_api_key_dict, request_payload=_build_passthrough_failure_request_payload( parsed_body=_parsed_body, @@ -1374,10 +1462,10 @@ async def pass_through_request( user_api_key_dict=user_api_key_dict, ), ping_interval_seconds=litellm.sse_keepalive_ping_interval_seconds, - upstream_headers=response.headers, + upstream_headers=relay_response.headers, ), headers=_response_headers, - status_code=response.status_code, + status_code=relay_response.status_code, ) if state_raw_body is not None: @@ -1412,7 +1500,7 @@ async def pass_through_request( logging_obj.stream = True logging_obj.model_call_details["stream"] = True - await _log_passthrough_upstream_failure( + detected_relay_response: Final = await _log_passthrough_upstream_failure( response=response, user_api_key_dict=user_api_key_dict, request_payload=_build_passthrough_failure_request_payload( @@ -1422,17 +1510,18 @@ async def pass_through_request( custom_llm_provider=custom_llm_provider, upstream_usage=upstream_usage, ), + logging_obj=logging_obj, ) # Call response headers hook for detected streaming pass-through _response_headers = HttpPassThroughEndpointHelpers.get_response_headers( - headers=response.headers, + headers=detected_relay_response.headers, litellm_call_id=litellm_call_id, ) callback_headers = await proxy_logging_obj.post_call_response_headers_hook( data=_parsed_body or {}, user_api_key_dict=user_api_key_dict, - response=response, + response=detected_relay_response, request_headers=dict(request.headers), ) if callback_headers: @@ -1443,7 +1532,7 @@ async def pass_through_request( stream=_own_streamed_managed_ids( stream=_relay_reporting_failures( stream=PassThroughStreamingHandler.chunk_processor( - response=response, + response=detected_relay_response, request_body=_parsed_body, litellm_logging_obj=logging_obj, endpoint_type=endpoint_type, @@ -1451,7 +1540,7 @@ async def pass_through_request( passthrough_success_handler_obj=pass_through_endpoint_logging, url_route=str(url), ), - upstream_status=response.status_code, + upstream_status=detected_relay_response.status_code, user_api_key_dict=user_api_key_dict, request_payload=_build_passthrough_failure_request_payload( parsed_body=_parsed_body, @@ -1465,10 +1554,10 @@ async def pass_through_request( user_api_key_dict=user_api_key_dict, ), ping_interval_seconds=litellm.sse_keepalive_ping_interval_seconds, - upstream_headers=response.headers, + upstream_headers=detected_relay_response.headers, ), headers=_response_headers, - status_code=response.status_code, + status_code=detected_relay_response.status_code, ) if not _should_buffer_passthrough_response(response): @@ -1526,6 +1615,7 @@ async def pass_through_request( response=response, user_api_key_dict=user_api_key_dict, request_payload=failure_request_payload, + logging_obj=logging_obj, ) if response.status_code < 400 and response_body is not None and guardrails_to_run: @@ -3435,7 +3525,7 @@ async def _filter_endpoints_by_team_allowed_routes( for endpoint in pass_through_endpoints if endpoint.path in cast( # cast-ok: guarded above; team metadata stores this key as a list of route paths - "Sequence[str]", team_metadata.get("allowed_passthrough_routes") + Sequence[str], team_metadata.get("allowed_passthrough_routes") ) ] diff --git a/tests/integration/_support/wire.py b/tests/integration/_support/wire.py index 5052c34e021..ed96d4e4e83 100644 --- a/tests/integration/_support/wire.py +++ b/tests/integration/_support/wire.py @@ -43,7 +43,9 @@ class Wire: @contextmanager -def wire_server(respond: Callable[[Request], Reply], tls: ssl.SSLContext | None = None) -> Generator[Wire, None, None]: +def wire_server( + respond: Callable[[Request], Reply], tls: ssl.SSLContext | None = None, port: int = 0 +) -> Generator[Wire, None, None]: """Owned TCP peer; requests traverse the real HTTP client and serialization.""" received: Final[SimpleQueue[Request]] = SimpleQueue() errors: Final[SimpleQueue[Exception]] = SimpleQueue() @@ -114,7 +116,7 @@ def wire_server(respond: Callable[[Request], Reply], tls: ssl.SSLContext | None if tls is not None: self.socket = tls.wrap_socket(self.socket, server_side=True) - with OwnedHTTPServer(("127.0.0.1", 0), Handler) as server: + with OwnedHTTPServer(("127.0.0.1", port), Handler) as server: thread: Final = threading.Thread(target=server.serve_forever, kwargs={"poll_interval": 0.05}) thread.start() try: diff --git a/tests/integration/observability/test_passthrough_upstream_error_chaos.py b/tests/integration/observability/test_passthrough_upstream_error_chaos.py new file mode 100644 index 00000000000..d94b3b24954 --- /dev/null +++ b/tests/integration/observability/test_passthrough_upstream_error_chaos.py @@ -0,0 +1,150 @@ +import asyncio +import json +import re +import signal +from pathlib import Path +from typing import Final + +import httpx +import psutil +import pytest +import yaml +from integration._support.client import Gateway, eventually, object_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy_process +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue + +_GENERATE_CONTENT: Final[dict[str, JsonValue]] = {"contents": [{"role": "user", "parts": [{"text": "hi"}]}]} +_NOT_FOUND_BODY: Final = json.dumps( + { + "error": { + "code": 404, + "message": "models/nope-9 is not found for this scripted upstream", + "status": "NOT_FOUND", + } + } +).encode() +_INTERNAL_BODY: Final = ( + '{"error":{"code":500,"message":"' + "chunked upstream failure body " * 200 + '","status":"INTERNAL"}}' +).encode() +_OK_CHUNKS: Final = tuple(f"data: ok-{index}\n\n".encode() for index in range(3)) +_STARTED_WORKER: Final = re.compile(r"Started server process \[(\d+)\]") + + +def _chaos_reply(request: Request) -> Reply: + if "streamGenerateContent" in request.target: + return Reply(status=500, chunks=tuple(_INTERNAL_BODY[i : i + 512] for i in range(0, len(_INTERNAL_BODY), 512))) + if "healthy-model" in request.target: + return Reply(status=200, chunks=_OK_CHUNKS, content_type="text/event-stream") + return Reply(status=404, body=_NOT_FOUND_BODY) + + +def _error_information(call_id: str) -> dict[str, JsonValue]: + rows: Final = eventually( + lambda: read_rows('SELECT metadata FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), + lambda values: len(values) == 1, + seconds=70, + ) + metadata: Final = rows[0]["metadata"] + parsed: Final = json.loads(metadata) if isinstance(metadata, str) else object_value(metadata) + return object_value(parsed["error_information"]) + + +def _single_spend_row(call_id: str) -> None: + rows: Final = eventually( + lambda: read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), + lambda values: len(values) == 1, + seconds=70, + ) + assert len(rows) == 1, call_id + + +async def _fire_burst( + base_url: str, key: str, count: int, *, tolerate_transport_errors: bool = False +) -> tuple[httpx.Response, ...]: + async def one(client: httpx.AsyncClient, index: int) -> httpx.Response: + if index % 3 == 0: + path: Final = "/gemini/v1beta/models/nope-9:generateContent" + elif index % 3 == 1: + path = "/gemini/v1beta/models/nope-9:streamGenerateContent?alt=sse" + else: + path = "/gemini/v1beta/models/healthy-model:streamGenerateContent?alt=sse" + return await client.post( + path, + json=_GENERATE_CONTENT, + headers={"Authorization": f"Bearer {key}", "x-goog-api-key": key}, + ) + + async with httpx.AsyncClient(base_url=base_url, timeout=30, trust_env=False) as client: + results: Final = await asyncio.gather( + *(one(client, index) for index in range(count)), return_exceptions=tolerate_transport_errors + ) + for result in results: + assert not isinstance(result, BaseException) or isinstance(result, httpx.TransportError), repr(result) + return tuple(result for result in results if isinstance(result, httpx.Response)) + + +async def test_passthrough_upstream_outage_mid_burst_still_logs_errors_once(gateway: Gateway, tmp_path: Path) -> None: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + path: Final = tmp_path / "chaos-outage.yaml" + with wire_server(_chaos_reply) as wire: + port: Final = int(wire.url.rsplit(":", 1)[1]) + config["environment_variables"] = {"GEMINI_API_BASE": wire.url, "GEMINI_API_KEY": "scripted"} + path.write_text(yaml.safe_dump(config)) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + burst: Final = asyncio.create_task(_fire_burst(str(candidate.client.base_url), candidate.key, 30)) + await asyncio.to_thread(eventually, lambda: wire.received.qsize(), lambda size: size >= 10, 30) + with wire_server(_chaos_reply, port=port): + responses: Final = await burst + assert len(responses) == 30 + for response in responses: + assert response.status_code in (200, 404, 500, 502), response.status_code + assert "x-litellm-call-id" in response.headers, response.status_code + assert len(_STARTED_WORKER.findall(owned.log.read_text())) >= 2 + for response in responses: + _single_spend_row(response.headers["x-litellm-call-id"]) + if response.status_code == 404: + error_information: Final = _error_information(response.headers["x-litellm-call-id"]) + assert "not found for this scripted upstream" in str(error_information["error_message"]), response.text + elif response.status_code == 500: + assert "chunked upstream failure body" in str( + _error_information(response.headers["x-litellm-call-id"])["error_message"] + ), response.text + + +async def test_passthrough_worker_sigkill_leaves_sibling_serving_and_logging(gateway: Gateway, tmp_path: Path) -> None: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + path: Final = tmp_path / "chaos-kill.yaml" + with wire_server(_chaos_reply) as wire: + config["environment_variables"] = {"GEMINI_API_BASE": wire.url, "GEMINI_API_KEY": "scripted"} + path.write_text(yaml.safe_dump(config)) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + workers: Final = eventually( + lambda: tuple(int(pid) for pid in _STARTED_WORKER.findall(owned.log.read_text())), + lambda pids: len(pids) == 2, + seconds=30, + ) + burst: Final = asyncio.create_task( + _fire_burst(str(candidate.client.base_url), candidate.key, 20, tolerate_transport_errors=True) + ) + await asyncio.to_thread(eventually, lambda: wire.received.qsize(), lambda size: size >= 5, 30) + psutil.Process(workers[0]).send_signal(signal.SIGKILL) + responses: Final = await burst + for response in responses: + assert response.status_code in (200, 404, 500, 502), response.status_code + follow_up: Final = candidate.request( + "POST", + "/gemini/v1beta/models/nope-9:generateContent", + _GENERATE_CONTENT, + headers={"x-goog-api-key": candidate.key}, + ) + assert follow_up.status_code == 404, follow_up.text + assert follow_up.json() == json.loads(_NOT_FOUND_BODY), follow_up.text + for response in responses: + if "x-litellm-call-id" in response.headers: + _single_spend_row(response.headers["x-litellm-call-id"]) + error_information: Final = _error_information(follow_up.headers["x-litellm-call-id"]) + assert "not found for this scripted upstream" in str(error_information["error_message"]), follow_up.text diff --git a/tests/integration/observability/test_passthrough_upstream_error_visibility.py b/tests/integration/observability/test_passthrough_upstream_error_visibility.py new file mode 100644 index 00000000000..bb18add2f2f --- /dev/null +++ b/tests/integration/observability/test_passthrough_upstream_error_visibility.py @@ -0,0 +1,600 @@ +import gzip +import json +from hashlib import sha256 +from pathlib import Path +from typing import Final + +import httpx +import pytest +import yaml +from integration._support.client import Gateway, eventually, object_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy_process +from integration._support.wire import Reply, Request, wire_server +from openai import AsyncOpenAI, NotFoundError, OpenAI +from pydantic import JsonValue + +_UPSTREAM_ERROR: Final[dict[str, JsonValue]] = { + "error": { + "code": 404, + "message": "Publisher Model `publishers/anthropic/models/claude-nope-9` was not found or your project does not have access to it. Please ensure you are using a valid model version.", + "status": "NOT_FOUND", + } +} + + +def test_gemini_passthrough_upstream_error_body_reaches_proxy_log_and_spend_row( + gateway: Gateway, tmp_path: Path +) -> None: + def respond(request: Request) -> Reply: + return Reply(status=404, body=json.dumps(_UPSTREAM_ERROR).encode()) + + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + path: Final = tmp_path / "gemini-passthrough.yaml" + with wire_server(respond) as wire: + config["environment_variables"] = {"GEMINI_API_BASE": wire.url, "GEMINI_API_KEY": "scripted"} + path.write_text(yaml.safe_dump(config)) + with owned_proxy_process(gateway, tmp_path, {}, config=path) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", + "/gemini/v1beta/models/claude-nope-9:generateContent", + {"contents": [{"role": "user", "parts": [{"text": "hi"}]}]}, + headers={"x-goog-api-key": candidate.key}, + ) + assert response.status_code == 404, response.text + assert response.json() == _UPSTREAM_ERROR, response.text + try: + eventually( + lambda: owned.log.read_text(), + lambda text: "was not found or your project" in text, + seconds=30, + ) + except AssertionError: + pytest.fail( + f"upstream 404 body never reached the proxy log after {response.status_code} passthrough; " + f"log tail: {owned.log.read_text()[-2000:]}" + ) + rows: Final = eventually( + lambda: read_rows( + 'SELECT metadata FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (response.headers["x-litellm-call-id"],), + ), + lambda values: len(values) == 1, + seconds=70, + ) + metadata: Final = rows[0]["metadata"] + parsed: Final = json.loads(metadata) if isinstance(metadata, str) else object_value(metadata) + error_information: Final = object_value(parsed["error_information"]) + assert "was not found or your project" in str(error_information["error_message"]), response.text + assert error_information["error_code"] == "404", response.text + + +_GEMINI_MODEL_PATH: Final = "/gemini/v1beta/models/claude-nope-9:generateContent" +_GEMINI_STREAM_PATH: Final = "/gemini/v1beta/models/claude-nope-9:streamGenerateContent" +_GENERATE_CONTENT: Final[dict[str, JsonValue]] = {"contents": [{"role": "user", "parts": [{"text": "hi"}]}]} +_UPSTREAM_500_BODY: Final = ( + '{"error":{"code":500,"message":"' + "chunked upstream failure body " * 200 + '","status":"INTERNAL"}}' +).encode() + + +def _gemini_config(path: Path, wire_url: str) -> None: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["environment_variables"] = {"GEMINI_API_BASE": wire_url, "GEMINI_API_KEY": "scripted"} + path.write_text(yaml.safe_dump(config)) + + +def _gemini_headers(candidate: Gateway) -> dict[str, str]: + return {"Authorization": f"Bearer {candidate.key}", "x-goog-api-key": candidate.key} + + +def _spend_error_information(call_id: str) -> dict[str, JsonValue]: + rows: Final = eventually( + lambda: read_rows('SELECT metadata FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), + lambda values: len(values) == 1, + seconds=70, + ) + metadata: Final = rows[0]["metadata"] + parsed: Final = json.loads(metadata) if isinstance(metadata, str) else object_value(metadata) + return object_value(parsed["error_information"]) + + +def _spend_status(call_id: str) -> str: + rows: Final = eventually( + lambda: read_rows('SELECT status FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (call_id,)), + lambda values: len(values) == 1, + seconds=70, + ) + return str(rows[0]["status"]) + + +def _upstream_warning(log: Path, needle: str = "pass_through_endpoint: upstream") -> str: + text: Final = eventually(lambda: log.read_text(), lambda content: needle in content, seconds=30) + return next(line for line in text.splitlines() if needle in line) + + +def _upstream_warnings(log: Path, needle: str = "pass_through_endpoint: upstream") -> tuple[str, ...]: + return tuple(line for line in log.read_text().splitlines() if needle in line) + + +async def test_gemini_passthrough_async_client_404_body_reaches_proxy_log_and_spend_row( + gateway: Gateway, tmp_path: Path +) -> None: + def respond(request: Request) -> Reply: + return Reply(status=404, body=json.dumps(_UPSTREAM_ERROR).encode()) + + path: Final = tmp_path / "gemini-async.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + async with httpx.AsyncClient( + base_url=str(candidate.client.base_url), timeout=15, trust_env=False + ) as async_client: + response: Final = await async_client.post( + _GEMINI_MODEL_PATH, json=_GENERATE_CONTENT, headers=_gemini_headers(candidate) + ) + assert response.status_code == 404, response.text + assert response.json() == _UPSTREAM_ERROR, response.text + warning: Final = _upstream_warning(owned.log) + assert "was not found or your project" in warning, warning + error_information: Final = _spend_error_information(response.headers["x-litellm-call-id"]) + assert "was not found or your project" in str(error_information["error_message"]), response.text + assert error_information["error_code"] == "404", response.text + + +def test_gemini_passthrough_streaming_500_relays_full_body_and_logs_bounded_preview( + gateway: Gateway, tmp_path: Path +) -> None: + body: Final = _UPSTREAM_500_BODY + assert len(body) == 6055 + chunks: Final = tuple(body[index * 512 : (index + 1) * 512] for index in range(11)) + (body[5632:],) + + def respond(request: Request) -> Reply: + return Reply(status=500, chunks=chunks) + + path: Final = tmp_path / "gemini-stream-500.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + with candidate.client.stream( + "POST", + _GEMINI_STREAM_PATH, + params={"alt": "sse"}, + json=_GENERATE_CONTENT, + headers=_gemini_headers(candidate), + ) as response: + assert response.status_code == 500, response.text + streamed: Final = response.read() + assert streamed == body + warning: Final = _upstream_warning(owned.log) + assert warning.endswith("... (truncated at 4096 chars)"), warning + error_information: Final = _spend_error_information(response.headers["x-litellm-call-id"]) + error_message: Final = str(error_information["error_message"]) + assert error_message.endswith("... (truncated at 4096 chars)"), error_message + assert error_information["error_code"] == "500", error_message + + +def test_gemini_passthrough_success_logs_nothing_and_spend_row_is_success(gateway: Gateway, tmp_path: Path) -> None: + upstream_ok: Final = { + "candidates": [{"content": {"parts": [{"text": "hello"}], "role": "model"}}], + "usageMetadata": {"promptTokenCount": 3, "candidatesTokenCount": 2, "totalTokenCount": 5}, + } + + def respond(request: Request) -> Reply: + return Reply(status=200, body=json.dumps(upstream_ok).encode()) + + path: Final = tmp_path / "gemini-200.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", _GEMINI_MODEL_PATH, _GENERATE_CONTENT, headers={"x-goog-api-key": candidate.key} + ) + assert response.status_code == 200, response.text + assert response.json() == upstream_ok, response.text + assert _spend_status(response.headers["x-litellm-call-id"]) == "success" + assert not _upstream_warnings(owned.log), owned.log.read_text()[-2000:] + + +def test_gemini_passthrough_streaming_200_relays_every_chunk(gateway: Gateway, tmp_path: Path) -> None: + chunks: Final = tuple(f"data: chunk-{index}\n\n".encode() for index in range(5)) + + def respond(request: Request) -> Reply: + return Reply(status=200, chunks=chunks, content_type="text/event-stream") + + path: Final = tmp_path / "gemini-stream-200.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + with candidate.client.stream( + "POST", + _GEMINI_STREAM_PATH, + params={"alt": "sse"}, + json=_GENERATE_CONTENT, + headers=_gemini_headers(candidate), + ) as response: + assert response.status_code == 200 + streamed: Final = response.read() + assert streamed == b"".join(chunks) + assert not _upstream_warnings(owned.log), owned.log.read_text()[-2000:] + + +def test_config_pass_through_route_logs_body_and_strips_query(gateway: Gateway, tmp_path: Path) -> None: + upstream_error: Final = {"error": {"message": "max budget reached for this deployment"}} + + def respond(request: Request) -> Reply: + return Reply(status=403, body=json.dumps(upstream_error).encode()) + + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + path: Final = tmp_path / "config-route.yaml" + with wire_server(respond) as wire: + config["general_settings"]["pass_through_endpoints"] = [ + { + "path": "/audit-pt", + "target": f"{wire.url}/upstream?trace=secret-q", + "include_subpath": True, + "headers": {"Authorization": "Bearer scripted"}, + } + ] + path.write_text(yaml.safe_dump(config)) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request("POST", "/audit-pt", _GENERATE_CONTENT) + assert response.status_code == 403, response.text + assert response.json() == upstream_error, response.text + warning: Final = _upstream_warning(owned.log) + assert "max budget reached for this deployment" in warning, warning + assert "?" not in warning and "secret-q" not in warning, warning + error_information: Final = _spend_error_information(response.headers["x-litellm-call-id"]) + assert error_information["normalized_error"] == "500_UPSTREAM_PASSTHROUGH", response.text + assert "max budget reached for this deployment" in str(error_information["error_message"]), response.text + + +_OPENAI_UPSTREAM_404: Final[dict[str, JsonValue]] = { + "error": {"message": "The model `nope-9` does not exist", "type": "invalid_request_error"} +} + + +def _openai_config(path: Path, wire_url: str) -> None: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["environment_variables"] = {"OPENAI_API_BASE": wire_url, "OPENAI_API_KEY": "scripted"} + path.write_text(yaml.safe_dump(config)) + + +def test_openai_passthrough_sdk_error_body_reaches_proxy_log_and_spend_row(gateway: Gateway, tmp_path: Path) -> None: + def respond(request: Request) -> Reply: + return Reply(status=404, body=json.dumps(_OPENAI_UPSTREAM_404).encode()) + + path: Final = tmp_path / "openai-404.yaml" + with wire_server(respond) as wire: + _openai_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + with OpenAI( + api_key=candidate.key, + base_url=f"{str(candidate.client.base_url).rstrip('/')}/openai", + max_retries=0, + http_client=httpx.Client(timeout=15, trust_env=False), + ) as sdk: + with pytest.raises(NotFoundError) as raised: + sdk.chat.completions.create(model="nope-9", messages=[{"role": "user", "content": "hi"}]) + assert "does not exist" in str(raised.value), raised.value + warning: Final = _upstream_warning(owned.log) + assert "does not exist" in warning, warning + error_information: Final = _spend_error_information(raised.value.response.headers["x-litellm-call-id"]) + assert "does not exist" in str(error_information["error_message"]) + + +async def test_openai_passthrough_async_sdk_error_body_reaches_proxy_log_and_spend_row( + gateway: Gateway, tmp_path: Path +) -> None: + def respond(request: Request) -> Reply: + return Reply(status=404, body=json.dumps(_OPENAI_UPSTREAM_404).encode()) + + path: Final = tmp_path / "openai-async-404.yaml" + with wire_server(respond) as wire: + _openai_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + async with AsyncOpenAI( + api_key=candidate.key, + base_url=f"{str(candidate.client.base_url).rstrip('/')}/openai", + max_retries=0, + http_client=httpx.AsyncClient(timeout=15, trust_env=False), + ) as sdk: + with pytest.raises(NotFoundError) as raised: + await sdk.chat.completions.create(model="nope-9", messages=[{"role": "user", "content": "hi"}]) + assert "does not exist" in str(raised.value), raised.value + warning: Final = _upstream_warning(owned.log) + assert "does not exist" in warning, warning + error_information: Final = _spend_error_information(raised.value.response.headers["x-litellm-call-id"]) + assert "does not exist" in str(error_information["error_message"]) + + +def test_gemini_passthrough_control_characters_cannot_forge_log_lines(gateway: Gateway, tmp_path: Path) -> None: + forged: Final = b'{"error": "line one"}\n2026-01-01 FAKE LOG LINE\x1b[31m\r' + b"x" * 4943 + b"\x00tail" + assert len(forged) == 5000 + + def respond(request: Request) -> Reply: + return Reply(status=502, body=forged, content_type="text/html") + + path: Final = tmp_path / "gemini-forged.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", _GEMINI_MODEL_PATH, _GENERATE_CONTENT, headers={"x-goog-api-key": candidate.key} + ) + assert response.status_code == 502, response.text + assert response.content == forged, response.text + warning: Final = _upstream_warning(owned.log) + assert "\n" not in warning and "\x1b" not in warning, warning + assert "line one" in warning and "FAKE LOG LINE" in warning, warning + assert warning.endswith("... (truncated at 4096 chars)"), warning + error_information: Final = _spend_error_information(response.headers["x-litellm-call-id"]) + assert error_information["error_code"] == "502", response.text + + +def test_gemini_passthrough_empty_error_body_still_logged_and_proxy_serves(gateway: Gateway, tmp_path: Path) -> None: + def respond(request: Request) -> Reply: + if "claude-nope-9" in request.target: + return Reply(status=404, body=b"") + return Reply(status=200, body=b'{"candidates": [{"content": {"parts": [{"text": "ok"}]}}]}') + + path: Final = tmp_path / "gemini-empty.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", _GEMINI_MODEL_PATH, _GENERATE_CONTENT, headers={"x-goog-api-key": candidate.key} + ) + assert response.status_code == 404, response.text + assert response.content == b"", response.text + warning: Final = _upstream_warning(owned.log) + assert "returned 404" in warning, warning + error_information: Final = _spend_error_information(response.headers["x-litellm-call-id"]) + assert error_information["error_code"] == "404", response.text + follow_up: Final = candidate.request( + "POST", + "/gemini/v1beta/models/healthy-model:generateContent", + _GENERATE_CONTENT, + headers={"x-goog-api-key": candidate.key}, + ) + assert follow_up.status_code == 200, follow_up.text + + +def test_gemini_passthrough_gzip_error_body_decoded_for_log_and_client(gateway: Gateway, tmp_path: Path) -> None: + upstream_error: Final = {"error": {"message": "gzipped upstream says the model is gone"}} + + def respond(request: Request) -> Reply: + return Reply( + status=400, + body=gzip.compress(json.dumps(upstream_error).encode()), + headers={"content-encoding": "gzip"}, + ) + + path: Final = tmp_path / "gemini-gzip.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", _GEMINI_MODEL_PATH, _GENERATE_CONTENT, headers={"x-goog-api-key": candidate.key} + ) + assert response.status_code == 400, response.text + assert response.json() == upstream_error, response.text + warning: Final = _upstream_warning(owned.log) + assert "gzipped upstream says the model is gone" in warning, warning + + +def test_gemini_passthrough_streaming_gzip_error_body_decoded_for_log_and_client( + gateway: Gateway, tmp_path: Path +) -> None: + upstream_error: Final = {"error": {"message": "streamed gzip upstream denies the deployment"}} + compressed: Final = gzip.compress(json.dumps(upstream_error).encode()) + third: Final = len(compressed) // 3 + + def respond(request: Request) -> Reply: + return Reply( + status=403, + chunks=(compressed[:third], compressed[third : 2 * third], compressed[2 * third :]), + headers={"content-encoding": "gzip"}, + ) + + path: Final = tmp_path / "gemini-stream-gzip.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + with candidate.client.stream( + "POST", + _GEMINI_STREAM_PATH, + params={"alt": "sse"}, + json=_GENERATE_CONTENT, + headers=_gemini_headers(candidate), + ) as response: + assert response.status_code == 403 + streamed: Final = response.read() + assert json.loads(streamed) == upstream_error, streamed + warning: Final = _upstream_warning(owned.log) + assert "streamed gzip upstream denies the deployment" in warning, warning + + +def test_gemini_passthrough_error_body_redacted_when_message_logging_off(gateway: Gateway, tmp_path: Path) -> None: + upstream_error: Final = {"error": {"message": "sensitive upstream explanation"}} + + def respond(request: Request) -> Reply: + return Reply(status=404, body=json.dumps(upstream_error).encode()) + + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + path: Final = tmp_path / "gemini-redacted.yaml" + with wire_server(respond) as wire: + config["environment_variables"] = {"GEMINI_API_BASE": wire.url, "GEMINI_API_KEY": "scripted"} + config["litellm_settings"]["turn_off_message_logging"] = True + path.write_text(yaml.safe_dump(config)) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", _GEMINI_MODEL_PATH, _GENERATE_CONTENT, headers={"x-goog-api-key": candidate.key} + ) + assert response.status_code == 404, response.text + assert response.json() == upstream_error, response.text + warning: Final = _upstream_warning(owned.log) + assert "redacted-by-litellm" in warning, warning + assert "sensitive upstream explanation" not in warning, warning + error_information: Final = _spend_error_information(response.headers["x-litellm-call-id"]) + error_message: Final = str(error_information["error_message"]) + assert "redacted-by-litellm" in error_message, error_message + assert "sensitive upstream explanation" not in error_message, error_message + + +def test_gemini_passthrough_exact_4096_byte_body_logged_without_marker(gateway: Gateway, tmp_path: Path) -> None: + body: Final = b'{"error": "' + b"y" * 4083 + b'"}' + assert len(body) == 4096 + + def respond(request: Request) -> Reply: + return Reply(status=404, body=body) + + path: Final = tmp_path / "gemini-exact.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", _GEMINI_MODEL_PATH, _GENERATE_CONTENT, headers={"x-goog-api-key": candidate.key} + ) + assert response.status_code == 404, response.text + warning: Final = _upstream_warning(owned.log) + assert body[:512].decode() in warning, warning + assert "(truncated at 4096 chars)" not in warning, warning + + +def test_gemini_passthrough_4097_byte_body_truncated_with_marker(gateway: Gateway, tmp_path: Path) -> None: + body: Final = b'{"error": "' + b"y" * 4084 + b'"}' + assert len(body) == 4097 + + def respond(request: Request) -> Reply: + return Reply(status=404, body=body) + + path: Final = tmp_path / "gemini-over.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", _GEMINI_MODEL_PATH, _GENERATE_CONTENT, headers={"x-goog-api-key": candidate.key} + ) + assert response.status_code == 404, response.text + warning: Final = _upstream_warning(owned.log) + assert body[:512].decode() in warning, warning + assert warning.endswith("... (truncated at 4096 chars)"), warning + + +def test_gemini_passthrough_one_byte_stream_chunks_reassembled_and_logged(gateway: Gateway, tmp_path: Path) -> None: + body: Final = json.dumps(_UPSTREAM_ERROR).encode() + + def respond(request: Request) -> Reply: + return Reply(status=404, chunks=tuple(bytes([byte]) for byte in body)) + + path: Final = tmp_path / "gemini-one-byte.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + with candidate.client.stream( + "POST", + _GEMINI_STREAM_PATH, + params={"alt": "sse"}, + json=_GENERATE_CONTENT, + headers=_gemini_headers(candidate), + ) as response: + assert response.status_code == 404 + streamed: Final = response.read() + assert streamed == body + warning: Final = _upstream_warning(owned.log) + assert "was not found or your project" in warning, warning + + +def test_gemini_passthrough_repeated_errors_each_get_row_and_log_line(gateway: Gateway, tmp_path: Path) -> None: + def respond(request: Request) -> Reply: + return Reply(status=404, body=json.dumps(_UPSTREAM_ERROR).encode()) + + path: Final = tmp_path / "gemini-twice.yaml" + with wire_server(respond) as wire: + _gemini_config(path, wire.url) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + responses: Final = tuple( + candidate.request( + "POST", _GEMINI_MODEL_PATH, _GENERATE_CONTENT, headers={"x-goog-api-key": candidate.key} + ) + for _ in range(2) + ) + call_ids: Final = tuple(response.headers["x-litellm-call-id"] for response in responses) + assert len(set(call_ids)) == 2 + for response in responses: + assert response.status_code == 404, response.text + error_information: Final = _spend_error_information(response.headers["x-litellm-call-id"]) + assert "was not found or your project" in str(error_information["error_message"]), response.text + eventually( + lambda: _upstream_warnings(owned.log, "returned 404"), + lambda lines: len(lines) == 2, + seconds=30, + ) + + +def test_budget_rejected_call_keeps_budget_normalized_error(gateway: Gateway, tmp_path: Path) -> None: + path: Final = tmp_path / "budget.yaml" + path.write_text(Path("tests/integration/proxy_config.yaml").read_text()) + with owned_proxy_process(gateway, tmp_path, {}, config=path, workers=2) as owned: + candidate: Final = owned.gateway + with candidate.scenario() as scenario: + model: Final = scenario.model() + key: Final = scenario.key(max_budget=0.000001) + first: Final = candidate.chat(model, key=key) + assert "id" in first, first + rejected: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": "over budget"}]}, + key=key, + ) + assert rejected.status_code == 422 and "budget_exceeded" in rejected.text, rejected.text + digest: Final = sha256(key.encode()).hexdigest() + rows: Final = eventually( + lambda: read_rows( + 'SELECT metadata FROM "LiteLLM_SpendLogs" WHERE api_key=%s', + (digest,), + ), + lambda values: any( + "BUDGET_EXCEEDED" + in str( + object_value( + json.loads(row["metadata"]) + if isinstance(row["metadata"], str) + else object_value(row["metadata"]) + )["error_information"] + ) + for row in values + ), + seconds=70, + ) + budget_rows: Final = tuple( + row + for row in rows + if "BUDGET_EXCEEDED" + in str( + object_value( + json.loads(row["metadata"]) + if isinstance(row["metadata"], str) + else object_value(row["metadata"]) + )["error_information"] + ) + ) + assert len(budget_rows) == 1, budget_rows diff --git a/tests/test_litellm/litellm_core_utils/test_error_normalization.py b/tests/test_litellm/litellm_core_utils/test_error_normalization.py index d65b6d316ac..87b9463cb01 100644 --- a/tests/test_litellm/litellm_core_utils/test_error_normalization.py +++ b/tests/test_litellm/litellm_core_utils/test_error_normalization.py @@ -165,6 +165,18 @@ def test_variants_of_one_failure_share_a_normalized_error(messages: tuple[Except assert normalized == {expected} +def test_normalize_error_passthrough_prefix_wins_over_upstream_body_text() -> None: + from fastapi import HTTPException + + for detail in ( + 'Upstream passthrough request failed with status 400: {"error": {"message": "no deployments available for this model"}}', + 'Upstream passthrough request failed with status 400: {"error": {"message": "max budget reached"}}', + ): + exc = HTTPException(status_code=400, detail=detail) + message = f"400: {detail}" + assert normalize_error(exc, "400", message) == "500_UPSTREAM_PASSTHROUGH", message + + def test_router_no_healthy_deployment_wording_clusters_as_no_healthy_deployments() -> None: for message in (RouterErrors.no_healthy_deployments.value, "No healthy deployments found."): exc = litellm.BadRequestError(message, llm_provider="openai", model="gpt-4o") diff --git a/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py b/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py index 6ad850866b7..7a64d5f2218 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/test_pass_through_endpoints.py @@ -1,8 +1,10 @@ import asyncio +import gzip import json import logging import os import sys +import zlib from collections.abc import Callable from contextlib import ExitStack, contextmanager from io import BytesIO @@ -12,12 +14,14 @@ from unittest.mock import AsyncMock, MagicMock, patch import httpx import pytest -from fastapi import Request, Response, UploadFile +from fastapi import HTTPException, Request, Response, UploadFile +from fastapi.responses import StreamingResponse from pydantic import ValidationError from starlette.datastructures import FormData, Headers, QueryParams from starlette.datastructures import UploadFile as StarletteUploadFile import litellm +from litellm._logging import verbose_proxy_logger from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.proxy._types import ProxyException, UserAPIKeyAuth @@ -27,6 +31,8 @@ from litellm.proxy.pass_through_endpoints.pass_through_endpoints import ( HttpPassThroughEndpointHelpers, InitPassThroughEndpointHelpers, _registered_pass_through_routes, + _truncate_upstream_error_body, + _with_trace_context, chat_completion_pass_through_endpoint, create_pass_through_route, initialize_pass_through_endpoints, @@ -34,7 +40,6 @@ from litellm.proxy.pass_through_endpoints.pass_through_endpoints import ( resolve_llm_passthrough_timeout, resolve_pass_through_request_timeout, websocket_passthrough_request, - _with_trace_context, ) from litellm.proxy.pass_through_endpoints.success_handler import ( PassThroughEndpointLogging, @@ -4126,6 +4131,705 @@ async def test_pass_through_request_streaming_upstream_error_returned_unchanged( assert failure_call_kwargs["original_exception"].status_code == 403 +class _UpstreamErrorBodyStream(httpx.AsyncByteStream): + def __init__(self, body: bytes) -> None: + self._body: Final = body + + async def __aiter__(self): + yield self._body + + +def _upstream_error_request() -> MagicMock: + mock_request: Final = MagicMock(spec=Request) + mock_request.method = "POST" + mock_request.url = "http://test-proxy.com/mock-upstream/v1beta/models/claude-nope-9:generateContent" + mock_request.body = AsyncMock(return_value=b'{"contents": []}') + mock_request.headers = Headers({"content-type": "application/json"}) + mock_request.query_params = QueryParams({}) + return mock_request + + +@pytest.mark.asyncio +async def test_pass_through_request_non_streaming_upstream_error_body_logged_and_in_failure_detail( + caplog: pytest.LogCaptureFixture, +): + upstream_body: Final = { + "error": { + "code": 404, + "message": "Publisher Model `publishers/anthropic/models/claude-nope-9` was not found or your project does not have access", + "status": "NOT_FOUND", + } + } + upstream_content: Final = json.dumps(upstream_body).encode("utf-8") + upstream_response: Final = httpx.Response( + status_code=404, + headers={"content-type": "application/json"}, + content=upstream_content, + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:generateContent"), + ) + + with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"): + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.ProxyBaseLLMRequestProcessing" + ) as mock_processing: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_processing.get_custom_headers.return_value = {} + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + response: Final = await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:generateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + ) + + warning_messages: Final = [record.getMessage() for record in caplog.records if record.levelno == logging.WARNING] + upstream_warnings: Final = [ + message for message in warning_messages if "upstream" in message and "returned 404" in message + ] + assert len(upstream_warnings) == 1, warning_messages + assert "was not found or your project" in upstream_warnings[0] + assert "/v1beta/models/claude-nope-9:generateContent" in upstream_warnings[0] + + assert response.status_code == 404 + assert response.body == upstream_content + + mock_proxy_logging.post_call_failure_hook.assert_called_once() + failure_call_kwargs: Final = mock_proxy_logging.post_call_failure_hook.call_args.kwargs + original_exception: Final = failure_call_kwargs["original_exception"] + assert isinstance(original_exception, HTTPException) + assert original_exception.status_code == 404 + assert "was not found or your project" in original_exception.detail + + +@pytest.mark.asyncio +async def test_pass_through_request_streaming_upstream_error_body_reaches_client_and_failure_detail(): + upstream_content: Final = ( + b'data: {"error": {"code": 403, "message": "stream access was not found or your project lacks"}}\n\n' + ) + upstream_response: Final = httpx.Response( + status_code=403, + headers={"content-type": "text/event-stream"}, + stream=_UpstreamErrorBodyStream(upstream_content), + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent"), + ) + + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.pass_through_endpoint_logging.pass_through_async_success_handler" + ) as mock_success_handler: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_success_handler.return_value = None + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + response: Final = await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + stream=True, + ) + + assert isinstance(response, StreamingResponse) + assert response.status_code == 403 + streamed_chunks: Final = [chunk async for chunk in response.body_iterator] + streamed_bytes: Final = b"".join( + chunk if isinstance(chunk, bytes) else chunk.encode("utf-8") for chunk in streamed_chunks + ) + assert streamed_bytes == upstream_content + + mock_proxy_logging.post_call_failure_hook.assert_called_once() + original_exception: Final = mock_proxy_logging.post_call_failure_hook.call_args.kwargs["original_exception"] + assert "was not found or your project" in original_exception.detail + + +@pytest.mark.asyncio +async def test_truncate_upstream_error_body_caps_at_log_limit(): + short_body: Final = "x" * 4096 + assert _truncate_upstream_error_body(short_body) == short_body + + long_body: Final = "a" * 5000 + truncated: Final = _truncate_upstream_error_body(long_body) + assert truncated == f"{'a' * 4096}... (truncated at 4096 chars)" + + upstream_response: Final = httpx.Response( + status_code=500, + headers={"content-type": "text/plain"}, + content=long_body.encode("utf-8"), + request=httpx.Request("POST", "http://target-api.com/api/big-error"), + ) + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.ProxyBaseLLMRequestProcessing" + ) as mock_processing: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_processing.get_custom_headers.return_value = {} + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/api/big-error", + custom_headers={}, + user_api_key_dict=MagicMock(), + ) + + detail: Final = mock_proxy_logging.post_call_failure_hook.call_args.kwargs["original_exception"].detail + assert detail == f"Upstream passthrough request failed with status 500: {'a' * 4096}... (truncated at 4096 chars)" + + +@pytest.mark.asyncio +async def test_pass_through_request_upstream_error_log_strips_provider_key_from_url(): + upstream_content: Final = b'{"error": "denied"}' + upstream_response: Final = httpx.Response( + status_code=404, + headers={"content-type": "application/json"}, + content=upstream_content, + request=httpx.Request( + "POST", + "http://target-api.com/v1beta/models/claude-nope-9:generateContent?key=AIzaSySecretProviderKey123", + ), + ) + + with patch.object(verbose_proxy_logger, "warning") as mock_warning: + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.ProxyBaseLLMRequestProcessing" + ) as mock_processing: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_processing.get_custom_headers.return_value = {} + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:generateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + ) + + upstream_warnings: Final = [ + call + for call in mock_warning.call_args_list + if call.args[0] == "pass_through_endpoint: upstream %s %s returned %s: %s" + ] + assert len(upstream_warnings) == 1, mock_warning.call_args_list + logged_url: Final = str(upstream_warnings[0].args[2]) + assert "/v1beta/models/claude-nope-9:generateContent" in logged_url + assert "AIzaSySecretProviderKey123" not in logged_url + assert "key=" not in logged_url + + +@pytest.mark.asyncio +@pytest.mark.parametrize("turn_off_message_logging", [True, False]) +async def test_passthrough_upstream_error_body_redacted_when_message_logging_off( + turn_off_message_logging: bool, +): + upstream_content: Final = b'{"error": {"message": "upstream body says the project was not found"}}' + upstream_response: Final = httpx.Response( + status_code=404, + headers={"content-type": "application/json"}, + content=upstream_content, + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:generateContent"), + ) + user_api_key_dict: Final = MagicMock() + user_api_key_dict.metadata = { + "logging": [ + { + "callback_name": "prometheus", + "callback_type": "success_and_failure", + "callback_vars": {"turn_off_message_logging": turn_off_message_logging}, + } + ] + } + user_api_key_dict.team_metadata = None + user_api_key_dict.team_id = None + + with patch.object(verbose_proxy_logger, "warning") as mock_warning: + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.ProxyBaseLLMRequestProcessing" + ) as mock_processing: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_processing.get_custom_headers.return_value = {} + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + response: Final = await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:generateContent", + custom_headers={}, + user_api_key_dict=user_api_key_dict, + ) + + assert response.status_code == 404 + assert response.body == upstream_content + + upstream_warnings: Final = [ + call + for call in mock_warning.call_args_list + if call.args[0] == "pass_through_endpoint: upstream %s %s returned %s: %s" + ] + assert len(upstream_warnings) == 1, mock_warning.call_args_list + logged_body: Final = str(upstream_warnings[0].args[4]) + + mock_proxy_logging.post_call_failure_hook.assert_called_once() + detail: Final = mock_proxy_logging.post_call_failure_hook.call_args.kwargs["original_exception"].detail + + if turn_off_message_logging: + assert logged_body == "redacted-by-litellm" + assert "upstream body says the project was not found" not in logged_body + assert detail == "Upstream passthrough request failed with status 404: redacted-by-litellm" + else: + assert "upstream body says the project was not found" in logged_body + assert detail == f"Upstream passthrough request failed with status 404: {upstream_content.decode()}" + + +class _ChunkedUpstreamErrorBodyStream(httpx.AsyncByteStream): + def __init__(self, chunks: tuple[bytes, ...]) -> None: + self._chunks: Final = chunks + self.served: int = 0 + + async def __aiter__(self): + for chunk in self._chunks: + self.served += 1 + yield chunk + + +@pytest.mark.asyncio +async def test_pass_through_request_streaming_upstream_error_reads_only_preview_and_relays_full_body(): + chunk_size: Final = 1024 + chunks: Final = tuple(b"x" * chunk_size for _ in range(10)) + upstream_content: Final = b"".join(chunks) + body_stream: Final = _ChunkedUpstreamErrorBodyStream(chunks) + upstream_response: Final = httpx.Response( + status_code=500, + headers={"content-type": "text/plain"}, + stream=body_stream, + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent"), + ) + + served_at_warning: list[int] = [] + real_warning: Final = verbose_proxy_logger.warning + + def _recording_warning(*args, **kwargs): + if args and args[0] == "pass_through_endpoint: upstream %s %s returned %s: %s": + served_at_warning.append(body_stream.served) + return real_warning(*args, **kwargs) + + with patch.object(verbose_proxy_logger, "warning", side_effect=_recording_warning): + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.pass_through_endpoint_logging.pass_through_async_success_handler" + ) as mock_success_handler: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_success_handler.return_value = None + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + response: Final = await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + stream=True, + ) + + assert isinstance(response, StreamingResponse) + assert response.status_code == 500 + streamed_chunks: Final = [chunk async for chunk in response.body_iterator] + streamed_bytes: Final = b"".join( + chunk if isinstance(chunk, bytes) else chunk.encode("utf-8") for chunk in streamed_chunks + ) + assert streamed_bytes == upstream_content + + assert served_at_warning == [5], ( + "each raw chunk is yielded as-is; five 1024-byte chunks are the first point the preview budget is exceeded" + ) + expected_body: Final = f"{'x' * 4096}... (truncated at 4096 chars)" + assert ( + mock_proxy_logging.post_call_failure_hook.call_args.kwargs["original_exception"].detail + == f"Upstream passthrough request failed with status 500: {expected_body}" + ) + + +@pytest.mark.asyncio +async def test_pass_through_request_streaming_upstream_error_single_large_chunk_stays_bounded(): + first_chunk: Final = b"x" * 65536 + second_chunk: Final = b'{"error": "tail"}' + upstream_content: Final = first_chunk + second_chunk + body_stream: Final = _ChunkedUpstreamErrorBodyStream((first_chunk, second_chunk)) + upstream_response: Final = httpx.Response( + status_code=500, + headers={"content-type": "text/plain"}, + stream=body_stream, + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent"), + ) + + served_at_warning: list[int] = [] + real_warning: Final = verbose_proxy_logger.warning + + def _recording_warning(*args, **kwargs): + if args and args[0] == "pass_through_endpoint: upstream %s %s returned %s: %s": + served_at_warning.append(body_stream.served) + return real_warning(*args, **kwargs) + + with patch.object(verbose_proxy_logger, "warning", side_effect=_recording_warning): + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.pass_through_endpoint_logging.pass_through_async_success_handler" + ) as mock_success_handler: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_success_handler.return_value = None + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + response: Final = await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + stream=True, + ) + + assert isinstance(response, StreamingResponse) + assert response.status_code == 500 + streamed_chunks: Final = [chunk async for chunk in response.body_iterator] + streamed_bytes: Final = b"".join( + chunk if isinstance(chunk, bytes) else chunk.encode("utf-8") for chunk in streamed_chunks + ) + assert streamed_bytes == upstream_content + + assert served_at_warning == [1], ( + "the rechunked preview is served from the first raw chunk; the second must not be pulled before the warning" + ) + expected_body: Final = f"{'x' * 4096}... (truncated at 4096 chars)" + assert ( + mock_proxy_logging.post_call_failure_hook.call_args.kwargs["original_exception"].detail + == f"Upstream passthrough request failed with status 500: {expected_body}" + ) + + +class _UpstreamErrorBodyStreamDropping(httpx.AsyncByteStream): + async def __aiter__(self): + yield b'{"error": "half' + raise httpx.ReadError("peer reset") + + +@pytest.mark.asyncio +async def test_pass_through_request_streaming_upstream_error_body_read_failure_keeps_status_and_partial_body(): + """ + Regression: a 502 whose upstream dies while the error preview is being read + must still reach the client with status 502 and the bytes already received; + the read failure must not escape as a ProxyException 500. + """ + upstream_response: Final = httpx.Response( + status_code=502, + headers={"content-type": "application/json"}, + stream=_UpstreamErrorBodyStreamDropping(), + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent"), + ) + + recorded_warnings: list[tuple] = [] + real_warning: Final = verbose_proxy_logger.warning + + def _recording_warning(*args, **kwargs): + if args and str(args[0]).startswith("pass_through_endpoint: upstream"): + recorded_warnings.append(args) + return real_warning(*args, **kwargs) + + with patch.object(verbose_proxy_logger, "warning", side_effect=_recording_warning): + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.pass_through_endpoint_logging.pass_through_async_success_handler" + ) as mock_success_handler: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_success_handler.return_value = None + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + response: Final = await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + stream=True, + ) + + assert isinstance(response, StreamingResponse) + assert response.status_code == 502 + streamed_chunks: Final = [chunk async for chunk in response.body_iterator] + streamed_bytes: Final = b"".join( + chunk if isinstance(chunk, bytes) else chunk.encode("utf-8") for chunk in streamed_chunks + ) + assert streamed_bytes == b'{"error": "half' + await upstream_response.aclose() + + rendered: Final = [str(args[0]) for args in recorded_warnings] + formats: Final = [args[0] for args in recorded_warnings] + assert any( + fmt == "pass_through_endpoint: upstream %s %s returned %s: %s" and '{"error": "half' in str(args[4]) + for args, fmt in zip(recorded_warnings, formats) + ), rendered + assert any( + fmt == "pass_through_endpoint: upstream error body read failed after %d bytes: %s" + and args[1] == 15 + and args[2] == "ReadError" + for args, fmt in zip(recorded_warnings, formats) + ), rendered + + +class _UpstreamErrorGzipStreamDropping(httpx.AsyncByteStream): + def __init__(self, flushed_prefix: bytes) -> None: + self._flushed_prefix: Final = flushed_prefix + + async def __aiter__(self): + yield self._flushed_prefix + raise httpx.ReadError("peer reset") + + +@pytest.mark.asyncio +async def test_pass_through_request_streaming_upstream_error_gzip_read_failure_relays_decoded_partial(): + """ + Regression: a mid-read failure on a gzip upstream must relay the decoded + plaintext, not the compressed bytes; the relay strips content-encoding so + raw compressed bytes would reach the client as garbage. + """ + plaintext: Final = b'{"error": "half' + compressor: Final = zlib.compressobj(level=6, wbits=31) + flushed_prefix: Final = compressor.compress(plaintext) + compressor.flush(zlib.Z_SYNC_FLUSH) + upstream_response: Final = httpx.Response( + status_code=502, + headers={"content-type": "application/json", "content-encoding": "gzip"}, + stream=_UpstreamErrorGzipStreamDropping(flushed_prefix), + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent"), + ) + + recorded_warnings: list[tuple] = [] + real_warning: Final = verbose_proxy_logger.warning + + def _recording_warning(*args, **kwargs): + if args and str(args[0]).startswith("pass_through_endpoint: upstream"): + recorded_warnings.append(args) + return real_warning(*args, **kwargs) + + with patch.object(verbose_proxy_logger, "warning", side_effect=_recording_warning): + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.pass_through_endpoint_logging.pass_through_async_success_handler" + ) as mock_success_handler: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_success_handler.return_value = None + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + response: Final = await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + stream=True, + ) + + assert isinstance(response, StreamingResponse) + assert response.status_code == 502 + assert "content-encoding" not in response.headers + streamed_chunks: Final = [chunk async for chunk in response.body_iterator] + streamed_bytes: Final = b"".join( + chunk if isinstance(chunk, bytes) else chunk.encode("utf-8") for chunk in streamed_chunks + ) + assert streamed_bytes == plaintext + await upstream_response.aclose() + + rendered: Final = [str(args[0]) for args in recorded_warnings] + assert any( + args[0] == "pass_through_endpoint: upstream %s %s returned %s: %s" and plaintext.decode() in str(args[4]) + for args in recorded_warnings + ), rendered + + +@pytest.mark.asyncio +async def test_pass_through_request_streaming_upstream_error_gzip_body_decoded_for_log_and_client(): + upstream_content: Final = b'{"error": {"message": "gzipped upstream says the project was not found"}}' + compressed: Final = gzip.compress(upstream_content) + upstream_response: Final = httpx.Response( + status_code=502, + headers={"content-type": "text/event-stream", "content-encoding": "gzip"}, + stream=_ChunkedUpstreamErrorBodyStream((compressed[:10], compressed[10:])), + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent"), + ) + + with patch.object(verbose_proxy_logger, "warning") as mock_warning: + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.pass_through_endpoint_logging.pass_through_async_success_handler" + ) as mock_success_handler: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_success_handler.return_value = None + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + response: Final = await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:streamGenerateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + stream=True, + ) + + assert isinstance(response, StreamingResponse) + streamed_chunks: Final = [chunk async for chunk in response.body_iterator] + streamed_bytes: Final = b"".join( + chunk if isinstance(chunk, bytes) else chunk.encode("utf-8") for chunk in streamed_chunks + ) + assert streamed_bytes == upstream_content + + upstream_warnings: Final = [ + call + for call in mock_warning.call_args_list + if call.args[0] == "pass_through_endpoint: upstream %s %s returned %s: %s" + ] + assert len(upstream_warnings) == 1, mock_warning.call_args_list + logged_body: Final = str(upstream_warnings[0].args[4]) + assert "gzipped upstream says the project was not found" in logged_body + + +@pytest.mark.asyncio +async def test_pass_through_request_upstream_error_body_sanitized_against_log_forging(): + upstream_content: Final = b'{"error": "line one"}\n2026-01-01 FAKE LOG LINE\x1b[31m' + upstream_response: Final = httpx.Response( + status_code=404, + headers={"content-type": "application/json"}, + content=upstream_content, + request=httpx.Request("POST", "http://target-api.com/v1beta/models/claude-nope-9:generateContent"), + ) + + with patch.object(verbose_proxy_logger, "warning") as mock_warning: + with patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.get_async_httpx_client" + ) as mock_get_client: + with patch( + "litellm.proxy.pass_through_endpoints.pass_through_endpoints.ProxyBaseLLMRequestProcessing" + ) as mock_processing: + mock_proxy_logging.pre_call_hook = AsyncMock(return_value={}) + mock_proxy_logging.post_call_failure_hook = AsyncMock() + mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) + mock_processing.get_custom_headers.return_value = {} + + async_client: Final = MagicMock() + async_client.build_request = MagicMock(return_value=MagicMock()) + async_client.send = AsyncMock(return_value=upstream_response) + mock_get_client.return_value = MagicMock(client=async_client) + + await pass_through_request( + request=_upstream_error_request(), + target="http://target-api.com/v1beta/models/claude-nope-9:generateContent", + custom_headers={}, + user_api_key_dict=MagicMock(), + ) + + upstream_warnings: Final = [ + call + for call in mock_warning.call_args_list + if call.args[0] == "pass_through_endpoint: upstream %s %s returned %s: %s" + ] + assert len(upstream_warnings) == 1, mock_warning.call_args_list + logged_body: Final = str(upstream_warnings[0].args[4]) + assert logged_body == '{"error": "line one"} 2026-01-01 FAKE LOG LINE [31m' + assert "\n" not in logged_body + assert "\x1b" not in logged_body + + detail: Final = mock_proxy_logging.post_call_failure_hook.call_args.kwargs["original_exception"].detail + assert ( + detail + == 'Upstream passthrough request failed with status 404: {"error": "line one"} 2026-01-01 FAKE LOG LINE [31m' + ) + + class _UpstreamDroppingMidStream(httpx.AsyncByteStream): async def __aiter__(self): yield b'data: {"id": "chatcmpl-1", "choices": [{"delta": {"content": "hi"}}]}\n\n' @@ -4287,7 +4991,9 @@ async def test_pass_through_request_claims_the_budget_reservation_only_when_its_ mock_proxy_logging.post_call_failure_hook = AsyncMock() mock_proxy_logging.post_call_response_headers_hook = AsyncMock(return_value=None) mock_processing.get_custom_headers.return_value = {} - mock_worker.ensure_initialized_and_enqueue = MagicMock(side_effect=lambda async_coroutine: async_coroutine.close()) + mock_worker.ensure_initialized_and_enqueue = MagicMock( + side_effect=lambda async_coroutine: async_coroutine.close() + ) async_client = MagicMock() async_client.build_request = MagicMock(return_value=MagicMock()) async_client.send = AsyncMock(return_value=upstream_response) @@ -5304,9 +6010,7 @@ async def test_websocket_passthrough_propagates_active_trace_context( mock_proxy_logging.post_call_success_hook = AsyncMock() mock_proxy_logging.post_call_failure_hook = AsyncMock() mock_worker = MagicMock() - mock_worker.ensure_initialized_and_enqueue = MagicMock( - side_effect=lambda async_coroutine: async_coroutine.close() - ) + mock_worker.ensure_initialized_and_enqueue = MagicMock(side_effect=lambda async_coroutine: async_coroutine.close()) monkeypatch.setattr("litellm.proxy.proxy_server.proxy_logging_obj", mock_proxy_logging) monkeypatch.setattr( "litellm.proxy.pass_through_endpoints.pass_through_endpoints.connect", From 3fb6f8740b109d638c55c19436cc9440595642f4 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 00:25:03 -0700 Subject: [PATCH 47/96] test(integration): add read-replica routing harness to the CircleCI integration suite (#42692) * test(integration): add read-replica routing harness * refactor(integration): hoist the maintenance url imports * fix(integration): keep per-test databases and the witness sequence readable under replica roles * fix(integration): opt bespoke database and pool tests out of the injected read replica * test(integration): commit recorded replica routing expectations * fix(integration): judge routing by role containment so shrinking role sets do not fail * fix(integration): run the pool-limit shutdown choreography on the superuser database url * ci(integration): add the mcp group to the replica matrix * fix(integration): judge routing by exact role sets with a named either-role allowlist * test(integration): drop containment-era routing expectations for re-recording * chore(integration): drop docstrings from the replica harness scripts * docs(integration): describe exact routing matching and the either-role list * test(integration): record exact replica routing expectations * test(integration): allow the SELECT 1 health probe on either role * test(integration): replace committed routing expectations with an on-demand base-vs-head parity run * test(integration): fix parity env scope, readme wording, and seed-deterministic serialization test * test(integration): make the sorted-role serialization test deterministic in-process * test(integration): swap all product code in parity runs and pin role gains * ci(integration): force tracked-file removal before parity checkout --------- Co-authored-by: yuneng --- .circleci/config.yml | 124 ++++- .circleci/scripts/prepare_replica_roles.py | 52 +++ .circleci/scripts/run_integration.sh | 31 +- tests/integration/README.md | 2 + tests/integration/_support/process.py | 18 +- tests/integration/_support/routing.py | 333 +++++++++++++ tests/integration/conftest.py | 3 + .../database/test_transaction_atomicity.py | 1 + ..._user_updates_wedged_coordination_redis.py | 1 + .../test_vector_store_config_ownership.py | 9 +- tests/integration/routing/either_role.json | 3 + .../routing/test_redis_recovery.py | 2 +- .../spend/test_daily_rollup_retry.py | 1 + .../integration/spend/test_shutdown_flush.py | 2 + tests/unit/integration_support/__init__.py | 0 .../unit/integration_support/test_routing.py | 438 ++++++++++++++++++ 16 files changed, 1008 insertions(+), 12 deletions(-) create mode 100644 .circleci/scripts/prepare_replica_roles.py create mode 100644 tests/integration/_support/routing.py create mode 100644 tests/integration/routing/either_role.json create mode 100644 tests/unit/integration_support/__init__.py create mode 100644 tests/unit/integration_support/test_routing.py diff --git a/.circleci/config.yml b/.circleci/config.yml index eb76244c1ab..370424dca86 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -12,6 +12,9 @@ parameters: migration_source_sha: type: string default: "" + routing_parity_base: + type: string + default: "" orbs: codecov: codecov/codecov@4.0.1 node: circleci/node@5.1.0 # Add this line to declare the node orb @@ -176,6 +179,9 @@ commands: image: type: string default: postgres:14@sha256:6a70deda415ec296f977890e11aba04a0db9f632a362e3fce45e845e3db74f26 + server_args: + type: string + default: "" steps: - run: name: Start PostgreSQL @@ -186,7 +192,7 @@ commands: -e POSTGRES_PASSWORD=postgres \ -e POSTGRES_DB=<< parameters.db_name >> \ -p 5432:5432 \ - << parameters.image >> + << parameters.image >> << parameters.server_args >> - wait_for_service: url: tcp://localhost:5432 timeout: "60" @@ -3108,6 +3114,10 @@ jobs: parameters: suite: type: string + mode: + type: enum + enum: [standard, replica] + default: standard machine: image: ubuntu-2204:2024.04.1 resource_class: large @@ -3142,18 +3152,19 @@ jobs: command: cd ui/litellm-dashboard && NEXT_TELEMETRY_DISABLED=1 npm run build - start_postgres: image: postgres:16@sha256:e17e86066e5ef83e0952a9347f5c792b7ece00972e2aa787a6986f471b3dd3d5 + server_args: "-c shared_preload_libraries=pg_stat_statements -c pg_stat_statements.track=all -c pg_stat_statements.max=20000" - start_redis - run: name: Run owned integration contracts - command: bash .circleci/scripts/run_integration.sh << parameters.suite >> + command: bash .circleci/scripts/run_integration.sh << parameters.suite >> << parameters.mode >> no_output_timeout: 15m - run: name: Stop owned database and Redis when: always command: | - mkdir -p test-results/integration-<< parameters.suite >> - docker logs postgres-db > test-results/integration-<< parameters.suite >>/postgres.log 2>&1 || true - docker logs redis-cache > test-results/integration-<< parameters.suite >>/redis.log 2>&1 || true + mkdir -p test-results/services-<< parameters.suite >>-<< parameters.mode >> + docker logs postgres-db > test-results/services-<< parameters.suite >>-<< parameters.mode >>/postgres.log 2>&1 || true + docker logs redis-cache > test-results/services-<< parameters.suite >>-<< parameters.mode >>/redis.log 2>&1 || true docker rm -f postgres-db redis-cache test -z "$(docker ps -aq --filter name=postgres-db --filter name=redis-cache)" - store_test_results: @@ -3161,6 +3172,76 @@ jobs: - store_artifacts: path: test-results + routing_parity: + parameters: + suite: + type: string + machine: + image: ubuntu-2204:2024.04.1 + resource_class: large + working_directory: ~/project + steps: + - setup_litellm_test_deps + - run: + name: Check out base product code + environment: + ROUTING_PARITY_BASE: << pipeline.parameters.routing_parity_base >> + command: | + [[ "$ROUTING_PARITY_BASE" =~ ^[0-9a-f]{40}$ ]] || exit 1 + git fetch --depth 1 origin "$ROUTING_PARITY_BASE" + git rm -r -f --quiet litellm enterprise litellm-proxy-extras + git checkout "$ROUTING_PARITY_BASE" -- litellm enterprise litellm-proxy-extras + git reset --quiet + test -f litellm/rust_bridge/_native.abi3.so + - start_postgres: + image: postgres:16@sha256:e17e86066e5ef83e0952a9347f5c792b7ece00972e2aa787a6986f471b3dd3d5 + server_args: "-c shared_preload_libraries=pg_stat_statements -c pg_stat_statements.track=all -c pg_stat_statements.max=20000" + - start_redis + - run: + name: Run base side + command: bash .circleci/scripts/run_integration.sh << parameters.suite >> parity base + no_output_timeout: 15m + - run: + name: Stop base database and Redis + when: always + command: | + mkdir -p test-results/services-<< parameters.suite >>-parity-base + docker logs postgres-db > test-results/services-<< parameters.suite >>-parity-base/postgres.log 2>&1 || true + docker logs redis-cache > test-results/services-<< parameters.suite >>-parity-base/redis.log 2>&1 || true + docker rm -f postgres-db redis-cache + test -z "$(docker ps -aq --filter name=postgres-db --filter name=redis-cache)" + - run: + name: Check out head product code + command: | + git rm -r -f --quiet litellm enterprise litellm-proxy-extras + git checkout "$CIRCLE_SHA1" -- litellm enterprise litellm-proxy-extras + git reset --quiet + test -f litellm/rust_bridge/_native.abi3.so + - start_postgres: + image: postgres:16@sha256:e17e86066e5ef83e0952a9347f5c792b7ece00972e2aa787a6986f471b3dd3d5 + server_args: "-c shared_preload_libraries=pg_stat_statements -c pg_stat_statements.track=all -c pg_stat_statements.max=20000" + - start_redis + - run: + name: Run head side + command: bash .circleci/scripts/run_integration.sh << parameters.suite >> parity head + no_output_timeout: 15m + - run: + name: Stop head database and Redis + when: always + command: | + mkdir -p test-results/services-<< parameters.suite >>-parity-head + docker logs postgres-db > test-results/services-<< parameters.suite >>-parity-head/postgres.log 2>&1 || true + docker logs redis-cache > test-results/services-<< parameters.suite >>-parity-head/redis.log 2>&1 || true + docker rm -f postgres-db redis-cache + test -z "$(docker ps -aq --filter name=postgres-db --filter name=redis-cache)" + - run: + name: Compare routing parity + command: PYTHONPATH="$PWD/tests" .venv/bin/python -m integration._support.routing check test-results/parity-<< parameters.suite >>/base test-results/parity-<< parameters.suite >>/head + - store_test_results: + path: test-results + - store_artifacts: + path: test-results + unit: machine: image: ubuntu-2204:2024.04.1 @@ -3224,8 +3305,22 @@ workflows: branches: only: main jobs: *migration_jobs + routing_parity: + when: + not: + equal: ["", << pipeline.parameters.routing_parity_base >>] + jobs: + - routing_parity: + name: routing-parity-<< matrix.suite >> + matrix: + parameters: + suite: [management, accounting, database, providers, extensions, cost, mcp] integration: - unless: << pipeline.parameters.run_migration_tests >> + unless: + or: + - << pipeline.parameters.run_migration_tests >> + - not: + equal: ["", << pipeline.parameters.routing_parity_base >>] jobs: - integration_contracts: name: integration-<< matrix.suite >> @@ -3237,8 +3332,23 @@ workflows: only: - main - /litellm_.*/ + - integration_contracts: + name: integration-<< matrix.suite >>-replica + matrix: + parameters: + suite: [management, database] + mode: [replica] + filters: + branches: + only: + - main + - /litellm_.*/ build_and_test: - unless: << pipeline.parameters.run_migration_tests >> + unless: + or: + - << pipeline.parameters.run_migration_tests >> + - not: + equal: ["", << pipeline.parameters.routing_parity_base >>] jobs: - using_litellm_on_windows: filters: &main_branches diff --git a/.circleci/scripts/prepare_replica_roles.py b/.circleci/scripts/prepare_replica_roles.py new file mode 100644 index 00000000000..fb4b7fcae97 --- /dev/null +++ b/.circleci/scripts/prepare_replica_roles.py @@ -0,0 +1,52 @@ +from __future__ import annotations + +import os +from typing import Final +from urllib.parse import urlsplit, urlunsplit + +import psycopg + +DATABASE_URL: Final = os.environ["DATABASE_URL"] + + +def postgres_url() -> str: + parsed: Final = urlsplit(DATABASE_URL) + return urlunsplit(parsed._replace(path="/postgres")) + + +def main() -> None: + with psycopg.connect(postgres_url(), autocommit=True) as admin: + admin.execute("CREATE EXTENSION IF NOT EXISTS pg_stat_statements") + admin.execute("CREATE ROLE litellm_writer LOGIN PASSWORD 'litellm-writer' NOSUPERUSER") + admin.execute("CREATE ROLE litellm_reader LOGIN PASSWORD 'litellm-reader' NOSUPERUSER NOINHERIT") + admin.execute("ALTER ROLE litellm_reader SET default_transaction_read_only = on") + admin.execute("ALTER DATABASE circle_test OWNER TO litellm_writer") + admin.execute("GRANT CONNECT ON DATABASE circle_test TO litellm_reader") + with psycopg.connect(DATABASE_URL, autocommit=True) as admin: + admin.execute("GRANT USAGE ON SCHEMA public TO litellm_reader") + admin.execute( + "ALTER DEFAULT PRIVILEGES FOR ROLE litellm_writer IN SCHEMA public GRANT SELECT ON TABLES TO litellm_reader" + ) + admin.execute("GRANT SELECT ON ALL TABLES IN SCHEMA public TO litellm_reader") + + parsed: Final = urlsplit(DATABASE_URL) + reader_url: Final = urlunsplit( + parsed._replace(netloc=f"litellm_reader:litellm-reader@{parsed.hostname}:{parsed.port}") + ) + writer_url: Final = urlunsplit( + parsed._replace(netloc=f"litellm_writer:litellm-writer@{parsed.hostname}:{parsed.port}") + ) + with psycopg.connect(reader_url, autocommit=True) as reader: + assert reader.execute("SHOW transaction_read_only").fetchone() == ("on",) + try: + reader.execute("CREATE TABLE integration_readonly_probe (id int)") + except psycopg.errors.ReadOnlySqlTransaction: + pass + else: + raise AssertionError("litellm_reader executed a write statement") + with psycopg.connect(writer_url, autocommit=True) as writer: + assert writer.execute("SELECT current_user").fetchone() == ("litellm_writer",) + + +if __name__ == "__main__": + main() diff --git a/.circleci/scripts/run_integration.sh b/.circleci/scripts/run_integration.sh index d16ac9cd124..b617a79946c 100644 --- a/.circleci/scripts/run_integration.sh +++ b/.circleci/scripts/run_integration.sh @@ -7,7 +7,15 @@ if [ "${GITHUB_ACTIONS:-}" = true ]; then fi suite="${1:?integration suite required}" -results="test-results/integration-${suite}" +mode="${2:-standard}" +side="${3:-}" +if [ "$mode" = replica ]; then + results="test-results/integration-${suite}-replica" +elif [ "$mode" = parity ]; then + results="test-results/parity-${suite}/${side:?parity side required}" +else + results="test-results/integration-${suite}" +fi mkdir -p "$results" integration_identity="$(.venv/bin/python -c 'import uuid; print(uuid.uuid4().hex)')" upstream_pid="" @@ -80,6 +88,18 @@ export INTEGRATION_ORDER_SEED="$INTEGRATION_SEED" uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma > "$results/prisma-generate.log" 2>&1 +export INTEGRATION_PROXY_DATABASE_URL="" +export INTEGRATION_PROXY_READ_REPLICA_URL="" +export INTEGRATION_ROUTING="" +if [ "$mode" = replica ] || [ "$mode" = parity ]; then + .venv/bin/python .circleci/scripts/prepare_replica_roles.py > "$results/prepare-replica-roles.log" 2>&1 + export INTEGRATION_PROXY_DATABASE_URL="postgresql://litellm_writer:litellm-writer@127.0.0.1:5432/circle_test" + export INTEGRATION_PROXY_READ_REPLICA_URL="postgresql://litellm_reader:litellm-reader@127.0.0.1:5432/circle_test" +fi +if [ "$mode" = parity ]; then + export INTEGRATION_ROUTING=capture +fi + sudo iptables -N integration_only guard_created=true sudo iptables -A integration_only -o lo -j ACCEPT @@ -137,8 +157,12 @@ start_proxy() { else cost_map_env=("LITELLM_LOCAL_MODEL_COST_MAP=True") fi + local -a database_env=("DATABASE_URL=${INTEGRATION_PROXY_DATABASE_URL:-$DATABASE_URL}") + if [ -n "$INTEGRATION_PROXY_READ_REPLICA_URL" ]; then + database_env+=("DATABASE_URL_READ_REPLICA=$INTEGRATION_PROXY_READ_REPLICA_URL") + fi setsid env -i PATH="$PATH" HOME="$HOME" PYTHONPATH="$PYTHONPATH" INTEGRATION_RUN_ID="$integration_identity" \ - DATABASE_URL="$DATABASE_URL" REDIS_HOST="$REDIS_HOST" REDIS_PORT="$REDIS_PORT" \ + "${database_env[@]}" REDIS_HOST="$REDIS_HOST" REDIS_PORT="$REDIS_PORT" \ INTEGRATION_UPSTREAM_URL="$INTEGRATION_UPSTREAM_URL" \ LITELLM_MASTER_KEY="$LITELLM_MASTER_KEY" LITELLM_SALT_KEY="$LITELLM_SALT_KEY" LITELLM_UI_PATH="$LITELLM_UI_PATH" PROXY_BASE_URL="http://127.0.0.1:$port" \ LITELLM_MODE=PRODUCTION STORE_MODEL_IN_DB=True "${cost_map_env[@]}" \ @@ -195,6 +219,9 @@ env -i PATH="$PATH" HOME="$HOME" PYTHONPATH="$PYTHONPATH" \ INTEGRATION_SEED="$INTEGRATION_SEED" \ INTEGRATION_ORDER_SEED="$INTEGRATION_ORDER_SEED" \ LITELLM_LOCAL_MODEL_COST_MAP=True AWS_EC2_METADATA_DISABLED=true DO_NOT_TRACK=1 \ + INTEGRATION_PROXY_DATABASE_URL="$INTEGRATION_PROXY_DATABASE_URL" \ + INTEGRATION_PROXY_READ_REPLICA_URL="$INTEGRATION_PROXY_READ_REPLICA_URL" \ + INTEGRATION_ROUTING="$INTEGRATION_ROUTING" \ .venv/bin/python tests/integration/run.py "$suite" --results "$results" if [ "${INTEGRATION_COVERAGE:-0}" = 1 ]; then diff --git a/tests/integration/README.md b/tests/integration/README.md index f7c1305ad2e..ac9b01786b9 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -35,3 +35,5 @@ The extensions shard uses the built-in generic callback and guardrail transports The mcp shard runs the MCP gateway against SDK peers owned by each test (`_support/mcp.py`): streamable HTTP, SSE and stdio peers, an OpenAPI-spec app, and an OAuth 2.1 authorization-server double. Every peer records the requests it receives so a test can assert what reached the peer, not only what the proxy answered. The shard runs with `INTEGRATION_WORKERS` set and with `INTEGRATION_COVERAGE=1`, which starts the proxy under `coverage run --parallel-mode` limited to the MCP modules and stores `coverage.txt` plus an HTML report with the job artifacts. A test that fails because the product is wrong is skipped with `pytest.skip("BUG: ")` so the skip list in `execution.json` is the open MCP bug list Browser contracts live in `tests/e2e/ui/tests/integrationCritical` and run only through `tests/e2e/ui/integration.config.ts`. The expected browser results are listed in `expected.json` in that directory and checked by `.circleci/scripts/verify_integration_browser.py`. The CircleCI browser shard builds the checked-out dashboard, starts the owned proxy with that build, and verifies one exact browser result without retries or skips. The default Playwright selection excludes this directory. The focused project flow asserts the submitted create and clear values, fresh SQL state and actual blocked/restored serving while preserving model restrictions + +Two always-on `-replica` CircleCI jobs (management, database) run their groups in replica mode, where every proxy connects through a real `litellm_writer` role and a real read-only `litellm_reader` role against the same PostgreSQL. Nothing is captured there: the job passes when the tests pass, and a write routed to the read-only reader fails the test that issued it. A deeper check runs on demand as the `routing_parity` workflow, triggered through the CircleCI API v2 pipeline endpoint on the PR branch with `{"parameters": {"routing_parity_base": "<40-hex merge-base sha>"}}`. The workflow fans out over the seven groups, and each `routing-parity-` job runs its own group twice against the same test harness, once with `litellm/`, `enterprise/`, and `litellm-proxy-extras/` checked out from the base revision and once from the head, with a pytest plugin snapshotting `pg_stat_statements` into `routing-observed.json` per side. The `check` step then compares the two observations and writes `routing-diff.txt`: a statement seen on both sides fails when its role set changed, globally or for the same test (per-test capture is skipped under xdist), unless it is listed in `tests/integration/routing/either_role.json`, where each entry names the statement and a one-line reason it legitimately runs on whichever role asks for it, printed under `== either role ==`. Queries seen on only one side are listed, never failed, `pg_stat_statements` evictions and a role that never ran a statement are failures diff --git a/tests/integration/_support/process.py b/tests/integration/_support/process.py index 5c44beaa570..fcbaf7c8d8c 100644 --- a/tests/integration/_support/process.py +++ b/tests/integration/_support/process.py @@ -9,6 +9,7 @@ from collections.abc import Iterator, Mapping from contextlib import contextmanager from dataclasses import dataclass from pathlib import Path +from types import MappingProxyType from typing import Final import httpx @@ -16,6 +17,17 @@ import psutil from integration._support.client import Gateway +def proxy_database_environment() -> Mapping[str, str]: + writer: Final = os.environ.get("INTEGRATION_PROXY_DATABASE_URL", "") + reader: Final = os.environ.get("INTEGRATION_PROXY_READ_REPLICA_URL", "") + return MappingProxyType( + { + **({"DATABASE_URL": writer} if writer else {}), + **({"DATABASE_URL_READ_REPLICA": reader} if reader else {}), + } + ) + + def in_group(process: psutil.Process, group: int) -> bool: try: return os.getpgid(process.pid) == group @@ -83,7 +95,11 @@ def owned_proxy_process( port: Final = reserve.getsockname()[1] root: Final = Path(os.environ.get("INTEGRATION_PROXY_ROOT") or Path(__file__).resolve().parents[3]) environment: Final = { - **{name: value for name, value in os.environ.items() if name not in remove_environment}, + **{ + name: value + for name, value in {**os.environ, **proxy_database_environment()}.items() + if name not in remove_environment + }, "LITELLM_MASTER_KEY": gateway.key, "LITELLM_SALT_KEY": os.environ.get("LITELLM_SALT_KEY", "sk-integration-salt"), "STORE_MODEL_IN_DB": "True", diff --git a/tests/integration/_support/routing.py b/tests/integration/_support/routing.py new file mode 100644 index 00000000000..14f4741367a --- /dev/null +++ b/tests/integration/_support/routing.py @@ -0,0 +1,333 @@ +from __future__ import annotations + +import argparse +import itertools +import json +import os +import re +import sys +from collections.abc import Iterator, Mapping +from dataclasses import dataclass +from pathlib import Path +from types import MappingProxyType +from typing import Final +from urllib.parse import urlsplit, urlunsplit + +import psycopg +import pytest +from pydantic import TypeAdapter + +WRITER_ROLE: Final = "litellm_writer" +READER_ROLE: Final = "litellm_reader" +ROLES: Final = (READER_ROLE, WRITER_ROLE) +DATABASE_NAME: Final = "circle_test" +OBSERVED_FILE: Final = "routing-observed.json" +DIFF_FILE: Final = "routing-diff.txt" +EITHER_ROLE_FILE: Final = Path(__file__).resolve().parents[1] / "routing" / "either_role.json" + +RoleSet = frozenset[str] +RoutingMap = Mapping[str, frozenset[str]] +Snapshot = Mapping[tuple[str, str], int] + +_PLACEHOLDERS: Final = re.compile(r"\$\d+(?:\s*,\s*\$\d+)*") +_QUERIES: Final = TypeAdapter(dict[str, tuple[str, ...]]) +_OBSERVED: Final = TypeAdapter(dict[str, object]) + + +def normalize(query: str) -> str: + return _PLACEHOLDERS.sub("$n", " ".join(query.split())) + + +@dataclass(frozen=True, slots=True) +class Observation: + queries: RoutingMap + tests: Mapping[str, RoutingMap] + calls: Mapping[str, int] + dealloc: int + + +@dataclass(frozen=True, slots=True) +class Mismatch: + test: str | None + query: str + base: tuple[str, ...] + head: tuple[str, ...] + + +@dataclass(frozen=True, slots=True) +class Report: + mismatches: tuple[Mismatch, ...] + only_base: tuple[str, ...] + only_head: tuple[str, ...] + calls: Mapping[str, Mapping[str, int]] + dealloc: Mapping[str, int] + either_role: tuple[str, ...] = () + + def failures(self) -> tuple[str, ...]: + mismatch_failures: Final = tuple( + f"{mismatch.test if mismatch.test is not None else 'global'}: {mismatch.query}: " + f"base [{', '.join(mismatch.base)}] head [{', '.join(mismatch.head)}]" + for mismatch in self.mismatches + ) + side_failures: Final = tuple( + failure + for side in ("base", "head") + for failure in ( + *( + (f"{side}: pg_stat_statements evicted {self.dealloc[side]} entries (dealloc > 0)",) + if self.dealloc[side] > 0 + else () + ), + *(f"{side}: no {role} calls observed" for role in ROLES if self.calls[side].get(role, 0) == 0), + ) + ) + return (*mismatch_failures, *side_failures) + + +def _sorted_map(value: RoutingMap) -> RoutingMap: + return MappingProxyType(dict(sorted(value.items()))) + + +def compare(base: Observation, head: Observation, either_role: frozenset[str] = frozenset()) -> Report: + mismatches: Final = ( + *( + Mismatch( + None, + query, + tuple(sorted(base_roles)), + tuple(sorted(head.queries[query])), + ) + for query, base_roles in base.queries.items() + if query in head.queries and head.queries[query] != base_roles and query not in either_role + ), + *( + Mismatch( + test, + query, + tuple(sorted(base_roles)), + tuple(sorted(head.tests[test][query])), + ) + for test, queries in base.tests.items() + if test in head.tests + for query, base_roles in queries.items() + if query in head.tests[test] and head.tests[test][query] != base_roles and query not in either_role + ), + ) + varying: Final = frozenset( + query + for query in either_role + if (query in base.queries and query in head.queries and head.queries[query] != base.queries[query]) + or any( + query in base.tests[test] + and query in head.tests[test] + and head.tests[test][query] != base.tests[test][query] + for test in frozenset(base.tests) & frozenset(head.tests) + ) + ) + return Report( + mismatches, + tuple(sorted(query for query in base.queries if query not in head.queries)), + tuple(sorted(query for query in head.queries if query not in base.queries)), + MappingProxyType({"base": base.calls, "head": head.calls}), + MappingProxyType({"base": base.dealloc, "head": head.dealloc}), + tuple(sorted(varying)), + ) + + +def render(report: Report) -> str: + failures: Final = report.failures() + lines: Final = ( + "== failures ==", + *(failures or ("none",)), + "", + "== either role ==", + *(report.either_role or ("none",)), + "", + "== queries only in base ==", + *(report.only_base or ("none",)), + "", + "== queries only in head ==", + *(report.only_head or ("none",)), + "", + "== calls ==", + *( + line + for side in ("base", "head") + for line in ( + *(f"{side} {role}: {report.calls[side].get(role, 0)}" for role in ROLES), + f"{side} dealloc: {report.dealloc[side]}", + ) + ), + ) + return "\n".join(lines) + "\n" + + +def _roles(document: Mapping[str, tuple[str, ...]]) -> RoutingMap: + return _sorted_map({query: frozenset(roles) for query, roles in document.items()}) + + +def _tests(document: Mapping[str, Mapping[str, tuple[str, ...]]]) -> Mapping[str, RoutingMap]: + return MappingProxyType({node: _roles(queries) for node, queries in document.items()}) + + +def load_observation(path: Path) -> Observation: + document: Final = _OBSERVED.validate_python(json.loads(path.read_text())) + queries: Final = _QUERIES.validate_python(document.get("queries", {})) + tests: Final = TypeAdapter(dict[str, dict[str, tuple[str, ...]]]).validate_python(document.get("tests", {})) + calls: Final = TypeAdapter(dict[str, int]).validate_python(document.get("calls", {})) + dealloc: Final = TypeAdapter(int).validate_python(document.get("dealloc", 0)) + return Observation(_roles(queries), _tests(tests), MappingProxyType(calls), dealloc) + + +def load_either_role(path: Path) -> frozenset[str]: + if not path.exists(): + return frozenset() + document: Final = TypeAdapter(dict[str, str]).validate_python(json.loads(path.read_text())) + return frozenset(document) + + +def _serializable(queries: RoutingMap, tests: Mapping[str, RoutingMap]) -> dict[str, object]: + return { + "queries": {query: sorted(roles) for query, roles in queries.items()}, + "tests": {node: {query: sorted(roles) for query, roles in mapping.items()} for node, mapping in tests.items()}, + } + + +def dump_observation(observation: Observation) -> str: + document: Final = _serializable(observation.queries, observation.tests) + return ( + json.dumps( + {**document, "calls": dict(observation.calls), "dealloc": observation.dealloc}, + indent=2, + sort_keys=True, + ) + + "\n" + ) + + +def _maintenance_url() -> str: + parsed: Final = urlsplit(os.environ["DATABASE_URL"]) + return urlunsplit(parsed._replace(path="/postgres")) + + +def snapshot(connection: psycopg.Connection[object]) -> Mapping[tuple[str, str], int]: + rows: Final = connection.execute( + """ + SELECT r.rolname, s.query, s.calls + FROM pg_stat_statements s + JOIN pg_roles r ON r.oid = s.userid + WHERE s.dbid = (SELECT oid FROM pg_database WHERE datname = %s) + AND r.rolname = ANY(%s) + """, + (DATABASE_NAME, list(ROLES)), + ).fetchall() + return MappingProxyType( + { + key: sum(calls for _, _, calls in grouped) + for key, grouped in itertools.groupby( + sorted((str(role), normalize(str(query)), int(calls)) for role, query, calls in rows), + key=lambda row: (row[0], row[1]), + ) + } + ) + + +def delta(before: Snapshot, after: Snapshot) -> RoutingMap: + pairs: Final = {key: after.get(key, 0) - before.get(key, 0) for key in frozenset(before) | frozenset(after)} + queries: Final = frozenset(query for (_, query), change in pairs.items() if change > 0) + return MappingProxyType( + {query: frozenset(role for role in ROLES if pairs.get((role, query), 0) > 0) for query in sorted(queries)} + ) + + +def role_calls(before: Snapshot, after: Snapshot) -> Mapping[str, int]: + return MappingProxyType( + { + role: sum( + max(after.get((role, query), 0) - before.get((role, query), 0), 0) + for query in frozenset(q for _, q in before) | frozenset(q for _, q in after) + ) + for role in ROLES + } + ) + + +def dealloc(connection: psycopg.Connection[object]) -> int: + return int(connection.execute("SELECT dealloc FROM pg_stat_statements_info").fetchone()[0]) + + +class RoutingPlugin: + def __init__(self, config: pytest.Config) -> None: + self.config = config + self._session_start: Snapshot | None = None + self._tests: tuple[tuple[str, RoutingMap], ...] = () + + def _snapshot(self) -> Snapshot: + with psycopg.connect(_maintenance_url(), autocommit=True) as connection: + return snapshot(connection) + + def pytest_sessionstart(self, session: pytest.Session) -> None: + if hasattr(self.config, "workerinput"): + return + self._session_start = self._snapshot() + + @pytest.hookimpl(hookwrapper=True) + def pytest_runtest_protocol(self, item: pytest.Item, nextitem: pytest.Item | None) -> Iterator[None]: + if self.config.getoption("numprocesses", default=None) or hasattr(self.config, "workerinput"): + yield + return + before: Final = self._snapshot() + yield + after: Final = self._snapshot() + self._tests = (*self._tests, (item.nodeid, delta(before, after))) + + def pytest_sessionfinish(self, session: pytest.Session, exitstatus: int) -> None: + if hasattr(self.config, "workerinput"): + return + end: Final = self._snapshot() + with psycopg.connect(_maintenance_url(), autocommit=True) as connection: + evictions: Final = dealloc(connection) + start: Final = self._session_start or {} + tests: Final = MappingProxyType({node: mapping for node, mapping in self._tests}) + destination: Final = Path(os.environ["INTEGRATION_RESULTS_DIR"]) + destination.mkdir(parents=True, exist_ok=True) + (destination / OBSERVED_FILE).write_text( + dump_observation( + Observation( + delta(start, end), + tests, + role_calls(start, end), + evictions, + ) + ) + ) + + +def main(argv: tuple[str, ...] | list[str]) -> int: + parser: Final = argparse.ArgumentParser() + parser.add_argument("command", choices=("check",)) + parser.add_argument("base_dir", type=Path) + parser.add_argument("head_dir", type=Path) + parser.add_argument("--either-role", type=Path, default=EITHER_ROLE_FILE) + parser.add_argument("--diff", type=Path, default=None) + options: Final = parser.parse_args(argv) + base_path: Final = options.base_dir / OBSERVED_FILE + head_path: Final = options.head_dir / OBSERVED_FILE + for path in (base_path, head_path): + if not path.exists(): + sys.stderr.write(f"observed routing file missing: {path}\n") + if not base_path.exists() or not head_path.exists(): + return 1 + report: Final = compare( + load_observation(base_path), + load_observation(head_path), + load_either_role(options.either_role), + ) + diff: Final = render(report) + (options.diff or options.head_dir.parent / DIFF_FILE).write_text(diff) + sys.stdout.write(diff) + return 1 if report.failures() else 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py index 4986f5ddcc0..368b3ebee75 100644 --- a/tests/integration/conftest.py +++ b/tests/integration/conftest.py @@ -15,6 +15,7 @@ from redis import Redis from tests.integration._support.client import Gateway, eventually, gateway_from_environment from tests.integration._support.generation import LIFECYCLE_SETTINGS from tests.integration._support.manifest import OWNED_DIRECTORIES +from tests.integration._support.routing import RoutingPlugin COLLECTED: Final = pytest.StashKey[tuple[str, ...]]() REPORTS: Final = pytest.StashKey[list[pytest.TestReport]]() @@ -29,6 +30,8 @@ def pytest_configure(config: pytest.Config) -> None: config.addinivalue_line("markers", "covers(*ids): legacy contract IDs kept for existing tests, not enforced") config.stash[REPORTS] = [] config.pluginmanager.register(IntegrationReportPlugin(config)) + if os.environ.get("INTEGRATION_ROUTING"): + config.pluginmanager.register(RoutingPlugin(config)) class IntegrationReportPlugin: diff --git a/tests/integration/database/test_transaction_atomicity.py b/tests/integration/database/test_transaction_atomicity.py index c150354d9a6..030f0f3e445 100644 --- a/tests/integration/database/test_transaction_atomicity.py +++ b/tests/integration/database/test_transaction_atomicity.py @@ -43,6 +43,7 @@ def test_access_group_second_key_constraint_failure_rolls_back_all_writes(gatewa ) with psycopg.connect(os.environ["DATABASE_URL"], autocommit=True) as connection, ExitStack() as cleanup: connection.execute(sql.SQL("CREATE SEQUENCE {}").format(sql.Identifier(witness))) + connection.execute(sql.SQL("GRANT USAGE ON SEQUENCE {} TO PUBLIC").format(sql.Identifier(witness))) cleanup.callback(connection.execute, sql.SQL("DROP SEQUENCE {}").format(sql.Identifier(witness))) connection.execute( sql.SQL( diff --git a/tests/integration/management/test_user_updates_wedged_coordination_redis.py b/tests/integration/management/test_user_updates_wedged_coordination_redis.py index d8c84a70778..0715059744e 100644 --- a/tests/integration/management/test_user_updates_wedged_coordination_redis.py +++ b/tests/integration/management/test_user_updates_wedged_coordination_redis.py @@ -121,6 +121,7 @@ def test_user_budget_updates_return_promptly_while_coordination_redis_is_wedged( "REDIS_PORT": str(coordination.port), }, config=Path("tests/integration/coordination_redis_proxy_config.yaml"), + remove_environment=("DATABASE_URL_READ_REPLICA",), workers=2, ) as candidate, Redis(host=coordination.host, port=coordination.port, socket_timeout=1) as subscriber_client, diff --git a/tests/integration/management/test_vector_store_config_ownership.py b/tests/integration/management/test_vector_store_config_ownership.py index e1e7ac42472..e4ea4e324ff 100644 --- a/tests/integration/management/test_vector_store_config_ownership.py +++ b/tests/integration/management/test_vector_store_config_ownership.py @@ -318,7 +318,14 @@ def test_redis_outage_keeps_config_store_served_and_recovers( "REDIS_PORT": str(cache.port), "REDIS_CIRCUIT_BREAKER_RECOVERY_TIMEOUT": "1", } - with owned_proxy(gateway, tmp_path, overrides, config=PROXY_CONFIG, workers=2) as candidate: + with owned_proxy( + gateway, + tmp_path, + overrides, + config=PROXY_CONFIG, + workers=2, + remove_environment=("DATABASE_URL_READ_REPLICA",), + ) as candidate: db_store_id: Final = f"vs_db_{uuid.uuid4().hex}" for phase in ("before", "during", "after"): if phase == "during": diff --git a/tests/integration/routing/either_role.json b/tests/integration/routing/either_role.json new file mode 100644 index 00000000000..68824b50e11 --- /dev/null +++ b/tests/integration/routing/either_role.json @@ -0,0 +1,3 @@ +{ + "SELECT $n": "SELECT 1 health probe: the database watchdog and probe target query whichever pool reader_unavailable selects and the reconnect smoke test always uses the writer, all timer driven, so it lands under whichever test is in flight" +} diff --git a/tests/integration/routing/test_redis_recovery.py b/tests/integration/routing/test_redis_recovery.py index 81d27a190b0..9b41d44926f 100644 --- a/tests/integration/routing/test_redis_recovery.py +++ b/tests/integration/routing/test_redis_recovery.py @@ -26,7 +26,7 @@ def test_owned_redis_outage_recovers_requests_and_real_response_cache(gateway: G try: with owned_redis(tmp_path) as cache, monkeypatch.context() as environment: environment.setenv("DATABASE_URL", database_url) - with owned_proxy(gateway, tmp_path, {"DATABASE_URL": database_url, "REDIS_HOST": cache.host, "REDIS_PORT": str(cache.port), "REDIS_CIRCUIT_BREAKER_RECOVERY_TIMEOUT": "1"}) as candidate, candidate.scenario() as scenario, httpx.Client(base_url=gateway.upstream_url, timeout=5, trust_env=False) as upstream: + with owned_proxy(gateway, tmp_path, {"DATABASE_URL": database_url, "REDIS_HOST": cache.host, "REDIS_PORT": str(cache.port), "REDIS_CIRCUIT_BREAKER_RECOVERY_TIMEOUT": "1"}, remove_environment=("DATABASE_URL_READ_REPLICA",)) as candidate, candidate.scenario() as scenario, httpx.Client(base_url=gateway.upstream_url, timeout=5, trust_env=False) as upstream: model: Final = scenario.model() key: Final = scenario.key(models=[model]) for generation in ("before", "after"): diff --git a/tests/integration/spend/test_daily_rollup_retry.py b/tests/integration/spend/test_daily_rollup_retry.py index cf1b989639a..5b48d841f75 100644 --- a/tests/integration/spend/test_daily_rollup_retry.py +++ b/tests/integration/spend/test_daily_rollup_retry.py @@ -26,6 +26,7 @@ def _install_daily_user_rollup_fault(user_id: str) -> str: _execute( ( sql.SQL("CREATE SEQUENCE {}").format(sequence), + sql.SQL("GRANT USAGE ON SEQUENCE {} TO PUBLIC").format(sequence), sql.SQL( "CREATE FUNCTION {}() RETURNS trigger LANGUAGE plpgsql AS $fault$ " "BEGIN PERFORM nextval({}); " diff --git a/tests/integration/spend/test_shutdown_flush.py b/tests/integration/spend/test_shutdown_flush.py index b744c1f4b5f..5441f70d59e 100644 --- a/tests/integration/spend/test_shutdown_flush.py +++ b/tests/integration/spend/test_shutdown_flush.py @@ -196,12 +196,14 @@ def _proxy_with_one_seeded_row( gateway, tmp_path, { + "DATABASE_URL": os.environ["DATABASE_URL"], "LITELLM_LOG": "DEBUG", "GRACEFUL_SHUTDOWN_TIMEOUT": "1", "SCHEDULED_JOB_SHUTDOWN_FINISH_TIMEOUT_SECONDS": "1", "SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS": str(cancel_timeout_seconds), }, config=_config_with_pool_limit(tmp_path, pool_limit), + remove_environment=("DATABASE_URL_READ_REPLICA",), workers=workers, ) as owned: key: Final = string_value( diff --git a/tests/unit/integration_support/__init__.py b/tests/unit/integration_support/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/unit/integration_support/test_routing.py b/tests/unit/integration_support/test_routing.py new file mode 100644 index 00000000000..22a74423eb7 --- /dev/null +++ b/tests/unit/integration_support/test_routing.py @@ -0,0 +1,438 @@ +from __future__ import annotations + +import json +from collections.abc import Iterator +from pathlib import Path +from types import MappingProxyType +from typing import Final + +import pytest + +from tests.integration._support.routing import ( + DIFF_FILE, + OBSERVED_FILE, + READER_ROLE, + WRITER_ROLE, + Mismatch, + Observation, + compare, + delta, + dump_observation, + load_either_role, + load_observation, + main, + normalize, + render, + role_calls, +) + +TOKEN_QUERY: Final = 'UPDATE "LiteLLM_VerificationToken" SET token = $n WHERE token = $n' +NODE_ID: Final = "tests/integration/management/test_keys.py::test_generate" + + +def _routing(entries: dict[str, tuple[str, ...]]) -> MappingProxyType[str, frozenset[str]]: + return MappingProxyType({query: frozenset(roles) for query, roles in entries.items()}) + + +def _observation( + queries: dict[str, tuple[str, ...]], + tests: dict[str, dict[str, tuple[str, ...]]] | None = None, + calls: dict[str, int] | None = None, + dealloc: int = 0, +) -> Observation: + return Observation( + _routing(queries), + MappingProxyType({node: _routing(mapping) for node, mapping in (tests or {}).items()}), + MappingProxyType(calls if calls is not None else {"litellm_reader": 3, "litellm_writer": 7}), + dealloc, + ) + + +@pytest.mark.parametrize( + ("raw", "expected"), + [ + ("SELECT a\n FROM t", "SELECT a FROM t"), + ("SELECT * FROM t WHERE id IN ($1, $2, $3)", "SELECT * FROM t WHERE id IN ($n)"), + ("SELECT * FROM t WHERE id IN ($1,$2)", "SELECT * FROM t WHERE id IN ($n)"), + ("SELECT * FROM t WHERE id IN ($4)", "SELECT * FROM t WHERE id IN ($n)"), + ( + "INSERT INTO t VALUES ($1, $2) ON CONFLICT ($3, $4, $5) DO NOTHING", + "INSERT INTO t VALUES ($n) ON CONFLICT ($n) DO NOTHING", + ), + ], +) +def test_normalize_collapses_whitespace_and_placeholders(raw: str, expected: str) -> None: + assert normalize(raw) == expected + + +def test_compare_reports_global_role_mismatch() -> None: + base: Final = _observation({TOKEN_QUERY: ("litellm_reader",), "SELECT 1": ("litellm_writer",)}) + head: Final = _observation({TOKEN_QUERY: ("litellm_writer",), "SELECT 1": ("litellm_writer",)}) + report: Final = compare(base, head) + assert report.mismatches == (Mismatch(None, TOKEN_QUERY, ("litellm_reader",), ("litellm_writer",)),) + assert report.failures() == (f"global: {TOKEN_QUERY}: base [litellm_reader] head [litellm_writer]",) + + +def test_compare_reports_global_shrink_mismatch() -> None: + base: Final = _observation({TOKEN_QUERY: ("litellm_reader", "litellm_writer")}) + head: Final = _observation({TOKEN_QUERY: ("litellm_writer",)}) + report: Final = compare(base, head) + assert report.mismatches == ( + Mismatch(None, TOKEN_QUERY, ("litellm_reader", "litellm_writer"), ("litellm_writer",)), + ) + assert report.failures() == (f"global: {TOKEN_QUERY}: base [litellm_reader, litellm_writer] head [litellm_writer]",) + + +def test_compare_reports_per_test_mismatch_with_nodeid() -> None: + base: Final = _observation( + {TOKEN_QUERY: ("litellm_reader",)}, + {NODE_ID: {TOKEN_QUERY: ("litellm_reader",)}}, + ) + head: Final = _observation( + {TOKEN_QUERY: ("litellm_reader",)}, + {NODE_ID: {TOKEN_QUERY: ("litellm_writer",)}}, + ) + report: Final = compare(base, head) + assert report.mismatches == (Mismatch(NODE_ID, TOKEN_QUERY, ("litellm_reader",), ("litellm_writer",)),) + assert report.failures() == (f"{NODE_ID}: {TOKEN_QUERY}: base [litellm_reader] head [litellm_writer]",) + + +def test_compare_per_test_mismatch_ignores_global_observation() -> None: + base: Final = _observation( + {TOKEN_QUERY: ("litellm_reader", "litellm_writer")}, + {NODE_ID: {TOKEN_QUERY: ("litellm_reader",)}}, + ) + head: Final = _observation( + {TOKEN_QUERY: ("litellm_reader", "litellm_writer")}, + {NODE_ID: {TOKEN_QUERY: ("litellm_writer",)}}, + ) + report: Final = compare(base, head) + assert report.mismatches == (Mismatch(NODE_ID, TOKEN_QUERY, ("litellm_reader",), ("litellm_writer",)),) + assert report.failures() == (f"{NODE_ID}: {TOKEN_QUERY}: base [litellm_reader] head [litellm_writer]",) + + +def test_compare_reports_per_test_shrink_mismatch() -> None: + base: Final = _observation( + {TOKEN_QUERY: ("litellm_reader", "litellm_writer")}, + {NODE_ID: {TOKEN_QUERY: ("litellm_reader", "litellm_writer")}}, + ) + head: Final = _observation( + {TOKEN_QUERY: ("litellm_reader", "litellm_writer")}, + {NODE_ID: {TOKEN_QUERY: ("litellm_reader",)}}, + ) + report: Final = compare(base, head) + assert report.mismatches == ( + Mismatch(NODE_ID, TOKEN_QUERY, ("litellm_reader", "litellm_writer"), ("litellm_reader",)), + ) + assert report.failures() == ( + f"{NODE_ID}: {TOKEN_QUERY}: base [litellm_reader, litellm_writer] head [litellm_reader]", + ) + + +def test_compare_reports_global_gain_mismatch() -> None: + base: Final = _observation({TOKEN_QUERY: ("litellm_reader",)}) + head: Final = _observation({TOKEN_QUERY: ("litellm_reader", "litellm_writer")}) + report: Final = compare(base, head) + assert report.mismatches == ( + Mismatch(None, TOKEN_QUERY, ("litellm_reader",), ("litellm_reader", "litellm_writer")), + ) + assert report.failures() == (f"global: {TOKEN_QUERY}: base [litellm_reader] head [litellm_reader, litellm_writer]",) + + +def test_compare_reports_per_test_gain_mismatch() -> None: + base: Final = _observation( + {TOKEN_QUERY: ("litellm_reader",)}, + {NODE_ID: {TOKEN_QUERY: ("litellm_reader",)}}, + ) + head: Final = _observation( + {TOKEN_QUERY: ("litellm_reader",)}, + {NODE_ID: {TOKEN_QUERY: ("litellm_reader", "litellm_writer")}}, + ) + report: Final = compare(base, head) + assert report.mismatches == ( + Mismatch(NODE_ID, TOKEN_QUERY, ("litellm_reader",), ("litellm_reader", "litellm_writer")), + ) + assert report.failures() == ( + f"{NODE_ID}: {TOKEN_QUERY}: base [litellm_reader] head [litellm_reader, litellm_writer]", + ) + + +def test_compare_either_role_suppresses_and_reports_variance() -> None: + base: Final = _observation( + {TOKEN_QUERY: ("litellm_reader", "litellm_writer"), "SELECT quiet": ("litellm_reader",)}, + {NODE_ID: {TOKEN_QUERY: ("litellm_reader",)}}, + ) + head: Final = _observation( + {TOKEN_QUERY: ("litellm_writer",), "SELECT quiet": ("litellm_reader",)}, + {NODE_ID: {TOKEN_QUERY: ("litellm_writer",)}}, + ) + report: Final = compare(base, head, either_role=frozenset({TOKEN_QUERY, "SELECT quiet"})) + assert report.mismatches == () + assert report.failures() == () + assert report.either_role == (TOKEN_QUERY,) + assert "== either role ==\n" + TOKEN_QUERY + "\n" in render(report) + + +def test_compare_either_role_matches_exact_keys_only() -> None: + base: Final = _observation( + { + "SELECT $n": ("litellm_reader",), + "SELECT $n FROM x": ("litellm_reader",), + "SELECT $n FROM x WHERE y = $n": ("litellm_reader",), + } + ) + head: Final = _observation( + { + "SELECT $n": ("litellm_writer",), + "SELECT $n FROM x": ("litellm_writer",), + "SELECT $n FROM x WHERE y = $n": ("litellm_writer",), + } + ) + report: Final = compare(base, head, either_role=frozenset({"SELECT $n FROM x"})) + assert frozenset(mismatch.query for mismatch in report.mismatches) == frozenset( + {"SELECT $n", "SELECT $n FROM x WHERE y = $n"} + ) + other: Final = compare(base, head, either_role=frozenset({"SELECT $n"})) + assert frozenset(mismatch.query for mismatch in other.mismatches) == frozenset( + {"SELECT $n FROM x", "SELECT $n FROM x WHERE y = $n"} + ) + + +def test_compare_one_sided_queries_are_listed_not_failed() -> None: + base: Final = _observation({"SELECT a": ("litellm_reader",), "SELECT gone": ("litellm_writer",)}) + head: Final = _observation({"SELECT a": ("litellm_reader",), "SELECT new": ("litellm_writer",)}) + report: Final = compare(base, head) + assert report.only_base == ("SELECT gone",) + assert report.only_head == ("SELECT new",) + assert report.mismatches == () + assert report.failures() == () + + +def test_failures_flags_dealloc_evictions_on_base() -> None: + report: Final = compare(_observation({}, dealloc=1), _observation({})) + assert report.failures() == ("base: pg_stat_statements evicted 1 entries (dealloc > 0)",) + + +def test_failures_flags_dealloc_evictions_on_head() -> None: + report: Final = compare(_observation({}), _observation({}, dealloc=1)) + assert report.failures() == ("head: pg_stat_statements evicted 1 entries (dealloc > 0)",) + + +def test_failures_flags_silent_reader_on_base() -> None: + report: Final = compare( + _observation({}, calls={"litellm_reader": 0, "litellm_writer": 5}), + _observation({}), + ) + assert report.failures() == ("base: no litellm_reader calls observed",) + + +def test_failures_flags_silent_reader_on_head() -> None: + report: Final = compare( + _observation({}), + _observation({}, calls={"litellm_reader": 0, "litellm_writer": 5}), + ) + assert report.failures() == ("head: no litellm_reader calls observed",) + + +def test_failures_flags_silent_writer() -> None: + report: Final = compare( + _observation({}, calls={"litellm_reader": 5, "litellm_writer": 0}), + _observation({}), + ) + assert report.failures() == ("base: no litellm_writer calls observed",) + + +def test_failures_counts_missing_role_as_silent() -> None: + report: Final = compare(_observation({}), _observation({}, calls={"litellm_writer": 5})) + assert report.failures() == ("head: no litellm_reader calls observed",) + + +def test_compare_skips_per_test_mismatches_for_xdist_shape() -> None: + base: Final = _observation( + {TOKEN_QUERY: ("litellm_reader",)}, + {NODE_ID: {TOKEN_QUERY: ("litellm_reader",)}}, + ) + head: Final = _observation({TOKEN_QUERY: ("litellm_reader",)}) + assert head.tests == {} + report: Final = compare(base, head) + assert report.mismatches == () + assert report.failures() == () + + +class _WriterFirst(frozenset[str]): + def __iter__(self) -> Iterator[str]: + return iter((WRITER_ROLE, READER_ROLE)) + + +def test_dump_observation_sorts_role_lists_and_round_trips(tmp_path: Path) -> None: + queries: Final = [f"SELECT {index}" for index in range(4)] + observation: Final = Observation( + MappingProxyType({query: _WriterFirst({WRITER_ROLE, READER_ROLE}) for query in queries}), + MappingProxyType( + {NODE_ID: MappingProxyType({query: _WriterFirst({WRITER_ROLE, READER_ROLE}) for query in queries})} + ), + MappingProxyType({READER_ROLE: 1, WRITER_ROLE: 2}), + 0, + ) + expected: Final = ( + json.dumps( + { + "queries": {query: ["litellm_reader", "litellm_writer"] for query in queries}, + "tests": {NODE_ID: {query: ["litellm_reader", "litellm_writer"] for query in queries}}, + "calls": {"litellm_reader": 1, "litellm_writer": 2}, + "dealloc": 0, + }, + sort_keys=True, + indent=2, + ) + + "\n" + ) + dumped: Final = dump_observation(observation) + assert dumped == expected + path: Final = tmp_path / OBSERVED_FILE + path.write_text(dumped) + loaded: Final = load_observation(path) + assert loaded.queries == _routing({query: (WRITER_ROLE, READER_ROLE) for query in queries}) + assert loaded.tests == {NODE_ID: loaded.queries} + + +def _write_observed(results: Path, observation: Observation) -> None: + results.mkdir(parents=True, exist_ok=True) + (results / OBSERVED_FILE).write_text(dump_observation(observation)) + + +def test_main_check_returns_zero_for_matching_routes(tmp_path: Path) -> None: + base_dir: Final = tmp_path / "base" + head_dir: Final = tmp_path / "head" + observation: Final = _observation({TOKEN_QUERY: ("litellm_reader",)}) + _write_observed(base_dir, observation) + _write_observed(head_dir, observation) + assert main(["check", str(base_dir), str(head_dir)]) == 0 + diff: Final = (tmp_path / DIFF_FILE).read_text() + assert "== failures ==\nnone\n" in diff + + +def test_main_check_returns_one_and_writes_exact_diff(tmp_path: Path) -> None: + base_dir: Final = tmp_path / "parity" / "base" + head_dir: Final = tmp_path / "parity" / "head" + _write_observed( + base_dir, + _observation({TOKEN_QUERY: ("litellm_reader",), "SELECT absent": ("litellm_writer",)}), + ) + _write_observed( + head_dir, + _observation( + {TOKEN_QUERY: ("litellm_writer",)}, + calls={"litellm_reader": 0, "litellm_writer": 5}, + dealloc=2, + ), + ) + assert main(["check", str(base_dir), str(head_dir)]) == 1 + assert (head_dir.parent / DIFF_FILE).read_text() == ( + "== failures ==\n" + f"global: {TOKEN_QUERY}: base [litellm_reader] head [litellm_writer]\n" + "head: pg_stat_statements evicted 2 entries (dealloc > 0)\n" + "head: no litellm_reader calls observed\n" + "\n" + "== either role ==\n" + "none\n" + "\n" + "== queries only in base ==\n" + "SELECT absent\n" + "\n" + "== queries only in head ==\n" + "none\n" + "\n" + "== calls ==\n" + "base litellm_reader: 3\n" + "base litellm_writer: 7\n" + "base dealloc: 0\n" + "head litellm_reader: 0\n" + "head litellm_writer: 5\n" + "head dealloc: 2\n" + ) + + +def test_main_check_missing_observed_returns_one(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None: + base_dir: Final = tmp_path / "base" + head_dir: Final = tmp_path / "head" + _write_observed(base_dir, _observation({})) + head_dir.mkdir() + assert main(["check", str(base_dir), str(head_dir)]) == 1 + assert "observed routing file missing" in capsys.readouterr().err + + +def test_main_check_either_role_suppresses_shrink(tmp_path: Path) -> None: + base_dir: Final = tmp_path / "base" + head_dir: Final = tmp_path / "head" + _write_observed(base_dir, _observation({TOKEN_QUERY: ("litellm_reader", "litellm_writer")})) + _write_observed(head_dir, _observation({TOKEN_QUERY: ("litellm_writer",)})) + argv: Final = ["check", str(base_dir), str(head_dir)] + allowlist: Final = tmp_path / "either.json" + allowlist.write_text(json.dumps({TOKEN_QUERY: "timer probe may use either pool"})) + assert main([*argv, "--either-role", str(allowlist)]) == 0 + assert "== either role ==\n" + TOKEN_QUERY + "\n" in (tmp_path / DIFF_FILE).read_text() + assert main(argv) == 1 + + +def test_delta_maps_positive_increases_per_role() -> None: + before: Final = MappingProxyType( + { + ("litellm_reader", "SELECT both"): 1, + ("litellm_writer", "SELECT both"): 2, + ("litellm_reader", "SELECT reader"): 3, + ("litellm_writer", "SELECT gone"): 4, + ("litellm_reader", "SELECT same"): 5, + } + ) + after: Final = MappingProxyType( + { + ("litellm_reader", "SELECT both"): 2, + ("litellm_writer", "SELECT both"): 5, + ("litellm_reader", "SELECT reader"): 6, + ("litellm_reader", "SELECT same"): 5, + ("litellm_writer", "SELECT writer"): 7, + } + ) + assert delta(before, after) == { + "SELECT both": frozenset({"litellm_reader", "litellm_writer"}), + "SELECT reader": frozenset({"litellm_reader"}), + "SELECT writer": frozenset({"litellm_writer"}), + } + + +def test_role_calls_sums_positive_increases_per_role() -> None: + before: Final = MappingProxyType( + { + ("litellm_reader", "SELECT a"): 10, + ("litellm_reader", "SELECT b"): 4, + ("litellm_writer", "SELECT a"): 1, + } + ) + after: Final = MappingProxyType( + { + ("litellm_reader", "SELECT a"): 11, + ("litellm_reader", "SELECT b"): 2, + ("litellm_writer", "SELECT a"): 1, + ("litellm_writer", "SELECT c"): 6, + } + ) + assert role_calls(before, after) == {"litellm_reader": 1, "litellm_writer": 6} + + +def test_load_either_role_missing_path_returns_empty(tmp_path: Path) -> None: + assert load_either_role(tmp_path / "absent.json") == frozenset() + + +def test_load_either_role_reads_query_keys(tmp_path: Path) -> None: + path: Final = tmp_path / "either.json" + path.write_text(json.dumps({"SELECT $n": "probe", "SELECT now()": "clock"})) + assert load_either_role(path) == frozenset({"SELECT $n", "SELECT now()"}) + + +def test_load_observation_reads_calls_and_dealloc(tmp_path: Path) -> None: + observation: Final = _observation({TOKEN_QUERY: ("litellm_reader",)}, dealloc=0) + path: Final = tmp_path / OBSERVED_FILE + path.write_text(dump_observation(observation)) + loaded: Final = load_observation(path) + assert loaded == observation From 09ebb28473e6e9e09c80ce2f822b88d8bc24f2e4 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 07:32:00 +0000 Subject: [PATCH 48/96] fix(s3_v2): bound concurrent S3 uploads per flush and add opt-in JSONL batch files (#41258) * fix(s3_v2): bound concurrent S3 uploads per flush and add opt-in JSONL batch files Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(s3_v2): keep failed uploads queued, parse env-backed flags, add integration coverage Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(s3_v2): type test helpers and honor constructor bound when config value is null Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(s3): annotate required casts for the type-discipline gate Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(s3_v2): keep tenant prefixes, stable retries and cold storage safety in batch file mode Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(s3_v2): cover root-level batch file keys for codecov patch target Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(s3_v2): audit matrix across chat, messages and responses surfaces with sink faults Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(s3_v2): read sink objects under the lock in the audit cells Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(s3_v2): rebind the retry queue instead of slicing in place Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(s3_v2): ignore stray non-POST requests in the surface upstream Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(s3_v2): keep fake upload state on the fake client instead of nonlocal counters Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: yucheng --- litellm/constants.py | 1 + litellm/integrations/s3.py | 35 +- litellm/integrations/s3_v2.py | 126 +++- litellm/types/integrations/s3_v2.py | 2 + .../observability/_s3_v2_support.py | 344 ++++++++++ .../test_s3_v2_flush_surfaces.py | 97 +++ .../observability/test_s3_v2_upload_fanout.py | 630 ++++++++++++++++++ tests/test_litellm/integrations/test_s3_v2.py | 450 +++++++++++++ 8 files changed, 1665 insertions(+), 20 deletions(-) create mode 100644 tests/integration/observability/_s3_v2_support.py create mode 100644 tests/integration/observability/test_s3_v2_flush_surfaces.py create mode 100644 tests/integration/observability/test_s3_v2_upload_fanout.py diff --git a/litellm/constants.py b/litellm/constants.py index dcc12ef2ba9..3ff80d8b7dd 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -48,6 +48,7 @@ DEFAULT_BATCH_SIZE: Final = int(os.getenv("DEFAULT_BATCH_SIZE", 512)) DEFAULT_FLUSH_INTERVAL_SECONDS: Final = int(os.getenv("DEFAULT_FLUSH_INTERVAL_SECONDS", 5)) DEFAULT_S3_FLUSH_INTERVAL_SECONDS: Final = int(os.getenv("DEFAULT_S3_FLUSH_INTERVAL_SECONDS", 10)) DEFAULT_S3_BATCH_SIZE: Final = int(os.getenv("DEFAULT_S3_BATCH_SIZE", 512)) +DEFAULT_S3_MAX_CONCURRENT_UPLOADS: Final = int(os.getenv("DEFAULT_S3_MAX_CONCURRENT_UPLOADS", "16")) # https://docs.aws.amazon.com/AmazonS3/latest/userguide/object-keys.html MAX_S3_OBJECT_KEY_BYTES: Final = 1024 S3_BOUNDED_OBJECT_KEY_HEAD_BYTES: Final = 64 diff --git a/litellm/integrations/s3.py b/litellm/integrations/s3.py index 54ec876fd0d..3c8619e82b2 100644 --- a/litellm/integrations/s3.py +++ b/litellm/integrations/s3.py @@ -20,7 +20,8 @@ from litellm.constants import ( ) from litellm.types.utils import StandardLoggingPayload -_S3_LOG_PROMPTS_ONLY: Final = TypeAdapter(bool) +_S3_BOOL: Final = TypeAdapter(bool) +_UPLOAD_BOUND: Final = TypeAdapter(int) def resolve_s3_log_prompts_only(configured: object, environ: Mapping[str, str] | None = None) -> bool: @@ -29,12 +30,42 @@ def resolve_s3_log_prompts_only(configured: object, environ: Mapping[str, str] | if raw is None or raw == "": return False try: - return _S3_LOG_PROMPTS_ONLY.validate_python(raw.strip() if isinstance(raw, str) else raw) + return _S3_BOOL.validate_python(raw.strip() if isinstance(raw, str) else raw) except ValidationError: verbose_logger.warning("s3 logging: s3_log_prompts_only=%r is not a boolean, logging prompts only", raw) return True +def resolve_s3_max_concurrent_uploads(configured: object, fallback: int) -> int: + if configured is None or configured == "": + return fallback + try: + bound: Final = _UPLOAD_BOUND.validate_python(configured.strip() if isinstance(configured, str) else configured) + except ValidationError: + verbose_logger.warning( + "s3 logging: s3_max_concurrent_uploads=%r is not an integer, using %s", configured, fallback + ) + return fallback + if bound < 1: + verbose_logger.warning( + "s3 logging: s3_max_concurrent_uploads=%r must be at least 1, using %s", configured, fallback + ) + return fallback + return bound + + +def resolve_s3_batch_file_upload(configured: object) -> bool: + if configured is None or configured == "": + return False + try: + return _S3_BOOL.validate_python(configured.strip() if isinstance(configured, str) else configured) + except ValidationError: + verbose_logger.warning( + "s3 logging: s3_batch_file_upload=%r is not a boolean, keeping per-request objects", configured + ) + return False + + def prompts_only_payload(payload: StandardLoggingPayload) -> StandardLoggingPayload: return {**payload, "response": None} diff --git a/litellm/integrations/s3_v2.py b/litellm/integrations/s3_v2.py index 826f55cc798..dc33fe6c2bd 100644 --- a/litellm/integrations/s3_v2.py +++ b/litellm/integrations/s3_v2.py @@ -3,26 +3,33 @@ s3 Bucket Logging Integration async_log_success_event: Processes the event, stores it in memory for DEFAULT_S3_FLUSH_INTERVAL_SECONDS seconds or until DEFAULT_S3_BATCH_SIZE and then flushes to s3 async_log_failure_event: Processes the event, stores it in memory for DEFAULT_S3_FLUSH_INTERVAL_SECONDS seconds or until DEFAULT_S3_BATCH_SIZE and then flushes to s3 -NOTE 1: S3 does not provide a BATCH PUT API endpoint, so we create tasks to upload each element individually +NOTE 1: S3 does not provide a BATCH PUT API endpoint; by default each element is uploaded concurrently (bounded by s3_max_concurrent_uploads), or with s3_batch_file_upload the whole flush is written as one .jsonl file """ import asyncio import time from collections.abc import Mapping -from datetime import datetime +from datetime import datetime, timezone from typing import TYPE_CHECKING, Final, cast from urllib.parse import quote +from uuid import uuid4 import httpx import litellm from litellm._logging import print_verbose, verbose_logger -from litellm.constants import DEFAULT_S3_BATCH_SIZE, DEFAULT_S3_FLUSH_INTERVAL_SECONDS +from litellm.constants import ( + DEFAULT_S3_BATCH_SIZE, + DEFAULT_S3_FLUSH_INTERVAL_SECONDS, + DEFAULT_S3_MAX_CONCURRENT_UPLOADS, +) from litellm.integrations.s3 import ( get_s3_object_download_filename, get_s3_object_key, prompts_only_payload, + resolve_s3_batch_file_upload, resolve_s3_log_prompts_only, + resolve_s3_max_concurrent_uploads, resolve_sse_params, ) from litellm.litellm_core_utils.aws_partition import get_aws_dns_suffix @@ -43,7 +50,20 @@ if TYPE_CHECKING: from botocore.credentials import Credentials +def _s3_key_parent(s3_object_key: str) -> str: + return s3_object_key.rsplit("/", 1)[0] if "/" in s3_object_key else "" + + +class S3BatchUploadError(Exception): + def __init__(self, failed: int, total: int) -> None: + self.failed = failed + self.total = total + super().__init__(f"{failed} of {total} S3 uploads failed; events kept in queue for the next flush") + + class S3Logger(CustomBatchLogger, BaseAWSLLM): + preserve_events_added_during_flush = True + def __init__( self, s3_bucket_name: str | None = None, @@ -71,6 +91,8 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): s3_server_side_encryption: str | None = None, s3_sse_kms_key_id: str | None = None, s3_log_prompts_only: bool | None = None, + s3_max_concurrent_uploads: int = DEFAULT_S3_MAX_CONCURRENT_UPLOADS, + s3_batch_file_upload: bool = False, s3_callback_params_override: dict | None = None, **kwargs, ): @@ -112,7 +134,10 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): s3_server_side_encryption=s3_server_side_encryption, s3_sse_kms_key_id=s3_sse_kms_key_id, s3_log_prompts_only=s3_log_prompts_only, + s3_max_concurrent_uploads=s3_max_concurrent_uploads, + s3_batch_file_upload=s3_batch_file_upload, ) + self._upload_semaphore = asyncio.Semaphore(self.s3_max_concurrent_uploads) verbose_logger.debug("s3 logger using endpoint url %s", s3_endpoint_url) # IMPORTANT @@ -168,6 +193,8 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): s3_server_side_encryption: str | None = None, s3_sse_kms_key_id: str | None = None, s3_log_prompts_only: bool | None = None, + s3_max_concurrent_uploads: int = DEFAULT_S3_MAX_CONCURRENT_UPLOADS, + s3_batch_file_upload: bool = False, params_source: dict | None = None, ): """ @@ -226,6 +253,16 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): params.get("s3_sse_kms_key_id") or s3_sse_kms_key_id, ) + configured_bound: Final = params.get("s3_max_concurrent_uploads") + self.s3_max_concurrent_uploads = resolve_s3_max_concurrent_uploads( + s3_max_concurrent_uploads if configured_bound is None or configured_bound == "" else configured_bound, + DEFAULT_S3_MAX_CONCURRENT_UPLOADS, + ) + + self.s3_batch_file_upload = s3_batch_file_upload or resolve_s3_batch_file_upload( + params.get("s3_batch_file_upload") + ) + def _build_object_url(self, s3_object_key: str) -> str: """ Build the exact URL that is both signed and sent, with the key percent-encoded once. @@ -347,7 +384,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): verbose_logger.exception("s3 Layer Error - %s", e) self.handle_callback_failure(callback_name="S3Logger") - async def async_upload_data_to_s3(self, batch_logging_element: s3BatchLoggingElement): + async def async_upload_data_to_s3(self, batch_logging_element: s3BatchLoggingElement) -> bool: try: import base64 import hashlib @@ -364,7 +401,11 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): url: Final = self._build_object_url(batch_logging_element.s3_object_key) # Convert JSON to string - json_string: Final = safe_dumps(batch_logging_element.payload) + json_string: Final = ( + batch_logging_element.body + if batch_logging_element.body is not None + else safe_dumps(batch_logging_element.payload) + ) # Calculate SHA256 hash of the content content_hash: Final = hashlib.sha256(json_string.encode("utf-8")).hexdigest() @@ -374,7 +415,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): # Prepare the request headers: Final = { - "Content-Type": "application/json", + "Content-Type": batch_logging_element.content_type, "Content-MD5": content_md5, "x-amz-content-sha256": content_hash, "Content-Language": "en", @@ -421,27 +462,72 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): except Exception as e: verbose_logger.exception("Error uploading to s3: %s", e) self.handle_callback_failure(callback_name="S3Logger") + return False + return True - async def async_send_batch(self): + async def async_send_batch(self) -> None: """ + Sends runs from self.log_queue. - Sends runs from self.log_queue - - Returns: None - - Raises: Does not raise an exception, will only verbose_logger.exception() + Raises S3BatchUploadError when any upload failed; CustomBatchLogger.flush_queue + keeps the surviving queue entries for the next flush. """ - verbose_logger.debug("s3_v2 logger - sending batch of %s", len(self.log_queue)) - if not self.log_queue: + batch: Final = tuple(self.log_queue) + if not batch: return + verbose_logger.debug("s3_v2 logger - sending batch of %s", len(batch)) ######################################################### # Flush the log queue to s3 # the log queue can be bounded by DEFAULT_S3_BATCH_SIZE # see custom_batch_logger.py which triggers the flush ######################################################### - for payload in self.log_queue: - asyncio.create_task(self.async_upload_data_to_s3(payload)) + uploads: Final = self._batch_file_elements(batch) if self._batch_file_mode_active() else batch + results: Final = await asyncio.gather(*(self._upload_bounded(element) for element in uploads)) + failed: Final = tuple(element for element, ok in zip(uploads, results, strict=True) if not ok) + if not failed: + return + self.log_queue = [*failed, *self.log_queue[len(batch) :]] + raise S3BatchUploadError(failed=len(failed), total=len(uploads)) + + def _batch_file_mode_active(self) -> bool: + if not self.s3_batch_file_upload: + return False + if litellm.cold_storage_custom_logger == "s3_v2": + verbose_logger.warning( + "s3 logging: s3_batch_file_upload is ignored because s3_v2 is the cold storage logger; " + "per-request objects are required for spend log lookups" + ) + return False + return True + + async def _upload_bounded(self, element: s3BatchLoggingElement) -> bool: + async with self._upload_semaphore: + return await self.async_upload_data_to_s3(element) + + def _batch_file_elements(self, batch: tuple[s3BatchLoggingElement, ...]) -> tuple[s3BatchLoggingElement, ...]: + now: Final = datetime.now(timezone.utc) + groups: Final = { + parent: tuple( + element for element in batch if element.body is None and _s3_key_parent(element.s3_object_key) == parent + ) + for parent in sorted({_s3_key_parent(element.s3_object_key) for element in batch if element.body is None}) + } + return tuple(element for element in batch if element.body is not None) + tuple( + self._build_batch_file_element(elements, parent, now) for parent, elements in groups.items() + ) + + def _build_batch_file_element( + self, elements: tuple[s3BatchLoggingElement, ...], parent: str, now: datetime + ) -> s3BatchLoggingElement: + batch_name: Final = f"batch_{now.strftime('%H-%M-%S')}_{uuid4().hex}" + return s3BatchLoggingElement( + payload={}, + body="\n".join(safe_dumps(element.payload) for element in elements), + content_type="application/x-ndjson", + s3_object_key=f"{parent}/{batch_name}.jsonl" if parent else f"{batch_name}.jsonl", + s3_object_download_filename=f"{batch_name}.jsonl", + ) def create_s3_batch_logging_element( self, @@ -521,7 +607,11 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): url: Final = self._build_object_url(batch_logging_element.s3_object_key) # Convert JSON to string - json_string: Final = safe_dumps(batch_logging_element.payload) + json_string: Final = ( + batch_logging_element.body + if batch_logging_element.body is not None + else safe_dumps(batch_logging_element.payload) + ) # Calculate SHA256 hash of the content content_hash: Final = hashlib.sha256(json_string.encode("utf-8")).hexdigest() @@ -531,7 +621,7 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): # Prepare the request headers: Final = { - "Content-Type": "application/json", + "Content-Type": batch_logging_element.content_type, "Content-MD5": content_md5, "x-amz-content-sha256": content_hash, "Content-Language": "en", diff --git a/litellm/types/integrations/s3_v2.py b/litellm/types/integrations/s3_v2.py index 32864bf5b8c..555b16dc141 100644 --- a/litellm/types/integrations/s3_v2.py +++ b/litellm/types/integrations/s3_v2.py @@ -9,3 +9,5 @@ class s3BatchLoggingElement(BaseModel): payload: dict s3_object_key: str s3_object_download_filename: str + body: str | None = None + content_type: str = "application/json" diff --git a/tests/integration/observability/_s3_v2_support.py b/tests/integration/observability/_s3_v2_support.py new file mode 100644 index 00000000000..104c0eda863 --- /dev/null +++ b/tests/integration/observability/_s3_v2_support.py @@ -0,0 +1,344 @@ +import asyncio +import json +import threading +import time +from collections.abc import Mapping +from concurrent.futures import ThreadPoolExecutor +from dataclasses import dataclass, field +from pathlib import Path +from types import MappingProxyType +from typing import Final + +import anthropic +import openai +import yaml +from integration._support.client import Gateway, JsonValue, eventually, object_value +from integration._support.wire import Reply, Request + +BUCKET: Final = "integration-bucket" +PREFIX: Final = "integration-logs" + + +@dataclass(slots=True) +class RecordingS3Sink: + """Records every accepted PUT body by target, tracks peak concurrency, and can reject a leading + run of PUT attempts with a chosen status before accepting. Serves stored bodies back on GET.""" + + fail_attempts: int = 0 + fail_until: float = 0.0 + fail_status: int = 503 + delay_seconds: float = 0.5 + lock: threading.Lock = field(default_factory=threading.Lock) + in_flight: int = 0 + peak: int = 0 + attempts: int = 0 + store: dict[str, bytes] = field(default_factory=dict) # mutable-ok: GET reads must see writes from earlier PUTs + + def respond(self, request: Request) -> Reply: + if request.method == "GET": + body: Final = self.store.get(request.target) + if body is None: + return Reply(status=404) + return Reply(body=body) + assert request.method == "PUT", request.method + assert request.target.startswith(f"/{BUCKET}/{PREFIX}/"), request.target + with self.lock: + self.attempts += 1 + if self.attempts <= self.fail_attempts or time.time() < self.fail_until: + return Reply( + status=self.fail_status, + body=b"SinkFailure", + content_type="application/xml", + ) + self.in_flight += 1 + self.peak = max(self.peak, self.in_flight) + self.store[request.target] = request.body + time.sleep(self.delay_seconds) + with self.lock: + self.in_flight -= 1 + return Reply() + + def objects(self) -> Mapping[str, bytes]: + with self.lock: + return MappingProxyType(dict(self.store)) + + def payloads(self) -> tuple[dict[str, JsonValue], ...]: + return tuple(object_value(json.loads(line)) for body in self.objects().values() for line in body.splitlines()) + + +def s3_config( + path: Path, sink_url: str, extra: Mapping[str, JsonValue], settings: Mapping[str, JsonValue] | None = None +) -> Path: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["litellm_settings"].update( + { + "callbacks": ["s3_v2"], + "s3_callback_params": { + "s3_bucket_name": BUCKET, + "s3_region_name": "us-east-1", + "s3_endpoint_url": sink_url, + "s3_path": PREFIX, + "s3_aws_access_key_id": "AKIAIOSFODNN7EXAMPLE", + "s3_aws_secret_access_key": "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY", + **extra, + }, + **(settings or {}), + } + ) + target: Final = path / "s3_v2.yaml" + target.write_text(yaml.safe_dump(config)) + return target + + +def _chat_completion(identity: str) -> dict[str, JsonValue]: + return { + "id": identity, + "object": "chat.completion", + "created": 1, + "model": "gpt-4o-mini", + "choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 11, "completion_tokens": 4, "total_tokens": 15}, + } + + +def _chat_stream_frames(identity: str) -> tuple[bytes, ...]: + chunks: Final = ( + { + "id": identity, + "object": "chat.completion.chunk", + "created": 1, + "model": "gpt-4o-mini", + "choices": [{"index": 0, "delta": {"role": "assistant", "content": "ok"}, "finish_reason": None}], + }, + { + "id": identity, + "object": "chat.completion.chunk", + "created": 1, + "model": "gpt-4o-mini", + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 11, "completion_tokens": 4, "total_tokens": 15}, + }, + ) + return tuple(f"data: {json.dumps(chunk)}\n\n".encode() for chunk in chunks) + (b"data: [DONE]\n\n",) + + +def _messages_completion(identity: str) -> dict[str, JsonValue]: + return { + "id": identity, + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [{"type": "text", "text": "ok"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 11, "output_tokens": 4}, + } + + +def _messages_stream_frames(identity: str) -> tuple[bytes, ...]: + events: Final = ( + ( + "message_start", + { + "type": "message_start", + "message": { + "id": identity, + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5-20250929", + "content": [], + "stop_reason": None, + "usage": {"input_tokens": 11, "output_tokens": 1}, + }, + }, + ), + ( + "content_block_start", + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + ), + ( + "content_block_delta", + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "ok"}}, + ), + ("content_block_stop", {"type": "content_block_stop", "index": 0}), + ( + "message_delta", + {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {"output_tokens": 4}}, + ), + ("message_stop", {"type": "message_stop"}), + ) + return tuple(f"event: {name}\ndata: {json.dumps(payload)}\n\n".encode() for name, payload in events) + + +def _responses_completion(identity: str) -> dict[str, JsonValue]: + return { + "id": identity, + "object": "response", + "created_at": 1, + "status": "completed", + "model": "gpt-4o-mini", + "output": [ + { + "type": "message", + "id": f"msg_{identity}", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "ok", "annotations": []}], + } + ], + "usage": {"input_tokens": 11, "output_tokens": 4, "total_tokens": 15}, + } + + +def _responses_stream_frames(identity: str) -> tuple[bytes, ...]: + events: Final = ( + ( + "response.created", + { + "type": "response.created", + "response": {**_responses_completion(identity), "status": "in_progress", "output": []}, + }, + ), + ( + "response.output_text.delta", + { + "type": "response.output_text.delta", + "item_id": f"msg_{identity}", + "output_index": 0, + "content_index": 0, + "delta": "ok", + }, + ), + ("response.completed", {"type": "response.completed", "response": _responses_completion(identity)}), + ) + return tuple(f"event: {name}\ndata: {json.dumps(payload)}\n\n".encode() for name, payload in events) + + +def surface_reply(request: Request) -> Reply: + """Scripted upstream that echoes the caller's marker string back as the response id.""" + if request.method != "POST" or not request.body: + return Reply(status=404) + body: Final = json.loads(request.body) + if request.target.endswith("/chat/completions"): + identity: Final = body["messages"][0]["content"] + if body.get("stream"): + return Reply(content_type="text/event-stream", chunks=_chat_stream_frames(identity)) + return Reply(body=json.dumps(_chat_completion(identity)).encode()) + if request.target.endswith("/messages"): + identity_messages: Final = body["messages"][0]["content"] + if body.get("stream"): + return Reply(content_type="text/event-stream", chunks=_messages_stream_frames(identity_messages)) + return Reply(body=json.dumps(_messages_completion(identity_messages)).encode()) + assert request.target.endswith("/responses"), request.target + identity_responses: Final = body["input"] + if body.get("stream"): + return Reply(content_type="text/event-stream", chunks=_responses_stream_frames(identity_responses)) + return Reply(body=json.dumps(_responses_completion(identity_responses)).encode()) + + +SURFACES: Final = ("chat", "chat_stream", "messages", "messages_stream", "responses", "responses_stream") + + +def call_surface( + candidate: Gateway, surface: str, openai_model: str, anthropic_model: str, key: str, marker: str +) -> tuple[str, str | None]: + """Drive one request through the given surface; return (client-visible response id, x-litellm-call-id).""" + base: Final = str(candidate.client.base_url).rstrip("/") + headers: Final = {"Authorization": f"Bearer {key}"} + if surface == "chat": + reply: Final = openai.OpenAI(base_url=f"{base}/v1", api_key=key).chat.completions.create( + model=openai_model, + messages=[{"role": "user", "content": marker}], + extra_body={"cache": {"no-cache": True}}, + ) + return reply.id, None + + async def chat_stream() -> str: + stream = await openai.AsyncOpenAI(base_url=f"{base}/v1", api_key=key).chat.completions.create( + model=openai_model, + messages=[{"role": "user", "content": marker}], + stream=True, + extra_body={"cache": {"no-cache": True}}, + ) + seen = "" + async for chunk in stream: + seen = chunk.id # rebind-ok: the stream yields one chunk at a time + return seen + + if surface == "chat_stream": + return asyncio.run(chat_stream()), None + if surface in ("messages", "messages_stream"): + client: Final = anthropic.Anthropic(base_url=base, api_key="anthropic-placeholder", default_headers=headers) + if surface == "messages": + reply_messages: Final = client.messages.create( + model=anthropic_model, max_tokens=16, messages=[{"role": "user", "content": marker}] + ) + return reply_messages.id, None + with client.messages.stream( + model=anthropic_model, max_tokens=16, messages=[{"role": "user", "content": marker}] + ) as stream: + final: Final = stream.get_final_message() + return final.id, None + if surface == "responses": + response: Final = candidate.request( + "POST", + "/v1/responses", + {"model": openai_model, "input": marker, "cache": {"no-cache": True}}, + key=key, + ) + assert response.status_code == 200, response.text + return str(response.json()["id"]), response.headers.get("x-litellm-call-id") + assert surface == "responses_stream", surface + with candidate.client.stream( + "POST", + "/v1/responses", + json={"model": openai_model, "input": marker, "stream": True}, + headers=headers, + ) as response: + text: Final = response.read().decode() + assert response.status_code == 200, text + call_id: Final = response.headers.get("x-litellm-call-id") + assert marker in text, text + return marker, call_id + + +def collect_payloads(sink: RecordingS3Sink, count: int, seconds: float = 60) -> tuple[dict[str, JsonValue], ...]: + """Wait until `count` stored payload lines exist, then return every stored payload object.""" + + def delivered() -> int: + return sum(len(body.splitlines()) for body in sink.objects().values()) + + eventually(delivered, lambda total: total >= count, seconds=seconds) + return sink.payloads() + + +def mixed_burst( + candidate: Gateway, openai_model: str, anthropic_model: str, key: str, marker: str, per_surface: int = 8 +) -> tuple[tuple[str, str | None], ...]: + """Fire `per_surface` requests on every surface; returns (response id, x-litellm-call-id) per request.""" + jobs: Final = tuple( + (surface, f"{marker}-{surface}-{index}") for surface in SURFACES for index in range(per_surface) + ) + + def call(job: tuple[str, str]) -> tuple[str, str | None]: + surface, identity = job + return call_surface(candidate, surface, openai_model, anthropic_model, key, identity) + + with ThreadPoolExecutor(max_workers=48) as pool: + return tuple(pool.map(call, jobs)) + + +def matched_ids( + payloads: tuple[dict[str, JsonValue], ...], answered: tuple[tuple[str, str | None], ...] +) -> frozenset[str]: + """Every payload must be accountable to an answered request by response id or litellm_call_id.""" + response_ids: Final = frozenset(observed for observed, _ in answered) + call_ids: Final = frozenset(call_id for _, call_id in answered if call_id is not None) + landed: Final = [] + for payload in payloads: + if payload["id"] in response_ids: + landed.append(payload["id"]) + continue + assert payload["litellm_call_id"] in call_ids, f"unmatched payload {payload['id']!r}" + landed.append(str(payload["id"])) + return frozenset(landed) diff --git a/tests/integration/observability/test_s3_v2_flush_surfaces.py b/tests/integration/observability/test_s3_v2_flush_surfaces.py new file mode 100644 index 00000000000..2e0b7260a13 --- /dev/null +++ b/tests/integration/observability/test_s3_v2_flush_surfaces.py @@ -0,0 +1,97 @@ +import re +import uuid +from pathlib import Path +from typing import Final + +import pytest +from _s3_v2_support import ( + BUCKET, + PREFIX, + RecordingS3Sink, + collect_payloads, + matched_ids, + mixed_burst, + s3_config, + surface_reply, +) +from integration._support.client import Gateway +from integration._support.process import owned_proxy +from integration._support.wire import wire_server + +PER_REQUEST_KEY: Final = re.compile(rf"^/{BUCKET}/{PREFIX}/\d{{4}}-\d{{2}}-\d{{2}}/.+\.json$") +BATCH_KEY: Final = re.compile( + rf"^/{BUCKET}/{PREFIX}/\d{{4}}-\d{{2}}-\d{{2}}/batch_\d{{2}}-\d{{2}}-\d{{2}}_[0-9a-f]{{32}}\.jsonl$" +) + + +@pytest.mark.covers("other.observability.s3_v2.mixed_surface_burst_bounds_puts_one_object_per_response_id") +def test_s3_v2_mixed_surface_burst_bounds_puts_one_object_per_response_id(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3mix" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(surface_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + openai_model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + anthropic_model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=provider.url, api_key="synthetic-provider-key" + ) + key: Final = scenario.key(models=[openai_model, anthropic_model]) + answered: Final = mixed_burst(candidate, openai_model, anthropic_model, key, marker) + payloads: Final = collect_payloads(sink, len(answered)) + targets: Final = tuple(sink.objects()) + assert sum(1 for r in provider.drain() if r.method == "POST") == 48 + assert sink.peak <= 16, f"peak concurrent PUTs {sink.peak} exceeded the default bound" + assert all(PER_REQUEST_KEY.match(target) for target in targets), list(targets) + assert len(targets) == 48 + assert matched_ids(payloads, answered) + + +@pytest.mark.covers("other.observability.s3_v2.mixed_surface_batch_writes_ndjson_lines_per_response_id") +def test_s3_v2_mixed_surface_batch_writes_ndjson_lines_per_response_id(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3mixb" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(surface_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = s3_config(tmp_path, bucket.url, {"s3_batch_file_upload": True}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + openai_model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + anthropic_model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=provider.url, api_key="synthetic-provider-key" + ) + key: Final = scenario.key(models=[openai_model, anthropic_model]) + answered: Final = mixed_burst(candidate, openai_model, anthropic_model, key, marker) + payloads: Final = collect_payloads(sink, len(answered)) + targets: Final = tuple(sink.objects()) + puts: Final = bucket.drain() + assert sum(1 for r in provider.drain() if r.method == "POST") == 48 + assert all(BATCH_KEY.match(target) for target in targets), list(targets) + assert all(put.headers["content-type"] == "application/x-ndjson" for put in puts), [put.headers for put in puts] + assert matched_ids(payloads, answered) + assert len(payloads) == 48 + + +@pytest.mark.covers("other.observability.s3_v2.sink_outage_mid_mixed_burst_recovers_every_response_id") +def test_s3_v2_sink_outage_mid_mixed_burst_recovers_every_response_id(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3mixo" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink(fail_attempts=30, fail_status=503, delay_seconds=0.2) + with wire_server(surface_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + openai_model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + anthropic_model: Final = scenario.model( + model="anthropic/claude-sonnet-4-5-20250929", api_base=provider.url, api_key="synthetic-provider-key" + ) + key: Final = scenario.key(models=[openai_model, anthropic_model]) + answered: Final = mixed_burst(candidate, openai_model, anthropic_model, key, marker) + payloads: Final = collect_payloads(sink, len(answered), seconds=90) + assert sum(1 for r in provider.drain() if r.method == "POST") == 48 + assert matched_ids(payloads, answered) + assert len(payloads) == 48, "a stored id was overwritten or duplicated" diff --git a/tests/integration/observability/test_s3_v2_upload_fanout.py b/tests/integration/observability/test_s3_v2_upload_fanout.py new file mode 100644 index 00000000000..3ebca152327 --- /dev/null +++ b/tests/integration/observability/test_s3_v2_upload_fanout.py @@ -0,0 +1,630 @@ +import json +import re +import threading +import time +import uuid +from collections.abc import Mapping +from concurrent.futures import ThreadPoolExecutor +from dataclasses import dataclass, field +from pathlib import Path +from typing import Final + +import httpx +import pytest +import yaml +from _s3_v2_support import RecordingS3Sink, collect_payloads +from _s3_v2_support import s3_config as _recording_s3_config +from integration._support.client import Gateway, JsonValue, eventually +from integration._support.process import group_members, owned_proxy, owned_proxy_process +from integration._support.wire import Reply, Request, Wire, wire_server + +BUCKET: Final = "integration-bucket" +PREFIX: Final = "integration-logs" +REQUESTS: Final = 64 +PUT_DELAY_SECONDS: Final = 0.5 + + +@dataclass(slots=True) +class S3Sink: + """Accepts every PUT after a fixed delay and records the peak number of PUTs in flight.""" + + lock: threading.Lock = field(default_factory=threading.Lock) + in_flight: int = 0 + peak: int = 0 + + def respond(self, request: Request) -> Reply: + assert request.method == "PUT", request.method + assert request.target.startswith(f"/{BUCKET}/{PREFIX}/"), request.target + with self.lock: + self.in_flight += 1 + self.peak = max(self.peak, self.in_flight) + time.sleep(PUT_DELAY_SECONDS) + with self.lock: + self.in_flight -= 1 + return Reply() + + +def _chat_reply(request: Request) -> Reply: + if request.method != "POST" or not request.body: + return Reply(status=404) + text: Final = json.loads(request.body)["messages"][0]["content"] + return Reply( + body=json.dumps( + { + "id": text, + "object": "chat.completion", + "created": 1, + "model": "gpt-4o-mini", + "choices": [{"index": 0, "message": {"role": "assistant", "content": text}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 11, "completion_tokens": 4, "total_tokens": 15}, + } + ).encode() + ) + + +def _s3_config(path: Path, sink_url: str, extra: Mapping[str, JsonValue]) -> Path: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["litellm_settings"].update( + { + "callbacks": ["s3_v2"], + "s3_callback_params": { + "s3_bucket_name": BUCKET, + "s3_region_name": "us-east-1", + "s3_endpoint_url": sink_url, + "s3_path": PREFIX, + "s3_aws_access_key_id": "AKIAIOSFODNN7EXAMPLE", + "s3_aws_secret_access_key": "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY", + **extra, + }, + } + ) + target: Final = path / "s3_v2.yaml" + target.write_text(yaml.safe_dump(config)) + return target + + +def _burst(candidate: Gateway, model: str, key: str, marker: str) -> frozenset[str]: + ids: Final = tuple(f"{marker}-{index}" for index in range(REQUESTS)) + + def request(identity: str) -> str: + response: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": identity}], "cache": {"no-cache": True}}, + key=key, + ) + assert response.status_code == 200, response.text + return response.json()["id"] + + with ThreadPoolExecutor(max_workers=32) as pool: + returned: Final = frozenset(pool.map(request, ids)) + assert returned == frozenset(ids) + return returned + + +def _collect(bucket: Wire, count_lines: bool, expected: int) -> tuple[Request, ...]: + puts: Final[list[Request]] = [] # mutable-ok: drain() consumes the queue, later polls must keep earlier PUTs + + def delivered() -> int: + puts.extend(bucket.drain()) + return sum(len(put.body.splitlines()) if count_lines else 1 for put in puts) + + eventually(delivered, lambda total: total >= expected, seconds=30) + return tuple(puts) + + +PER_REQUEST_KEY: Final = re.compile(rf"^/{BUCKET}/{PREFIX}/\d{{4}}-\d{{2}}-\d{{2}}/.+\.json$") +BATCH_KEY: Final = re.compile( + rf"^/{BUCKET}/{PREFIX}/\d{{4}}-\d{{2}}-\d{{2}}/batch_\d{{2}}-\d{{2}}-\d{{2}}_[0-9a-f]{{32}}\.jsonl$" +) + + +@pytest.mark.covers("other.observability.s3_v2.flush_bounds_concurrent_puts_to_default_and_keeps_every_log") +def test_s3_v2_flush_bounds_concurrent_puts_to_the_default_of_sixteen(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3fan" + uuid.uuid4().hex[:8] + sink: Final = S3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + ids: Final = _burst(candidate, model, key, marker) + puts: Final = _collect(bucket, count_lines=False, expected=REQUESTS) + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert sink.peak <= 16, f"peak concurrent PUTs {sink.peak} exceeded the default bound for {REQUESTS} queued logs" + assert all(PER_REQUEST_KEY.match(put.target) for put in puts), [put.target for put in puts] + assert frozenset(json.loads(put.body)["id"] for put in puts) == ids + assert len({put.target for put in puts}) == REQUESTS + + +@pytest.mark.covers("other.observability.s3_v2.configured_bound_and_env_backed_false_keeps_per_request_objects") +def test_s3_v2_honors_configured_bound_and_env_backed_false_batch_flag(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3cap" + uuid.uuid4().hex[:8] + sink: Final = S3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config( + tmp_path, + bucket.url, + {"s3_max_concurrent_uploads": 4, "s3_batch_file_upload": "os.environ/INTEGRATION_S3_BATCH_FILE_UPLOAD"}, + ) + with ( + owned_proxy( + gateway, + tmp_path, + {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3", "INTEGRATION_S3_BATCH_FILE_UPLOAD": "false"}, + config=config, + ) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + ids: Final = _burst(candidate, model, key, marker) + puts: Final = _collect(bucket, count_lines=False, expected=REQUESTS) + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert sink.peak <= 4, f"peak concurrent PUTs {sink.peak} exceeded s3_max_concurrent_uploads=4" + assert all(PER_REQUEST_KEY.match(put.target) for put in puts), [put.target for put in puts] + assert frozenset(json.loads(put.body)["id"] for put in puts) == ids + + +@pytest.mark.covers("other.observability.s3_v2.batch_file_upload_writes_one_ndjson_object_per_flush") +def test_s3_v2_batch_file_upload_writes_one_jsonl_object_per_flush(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3jsonl" + uuid.uuid4().hex[:8] + sink: Final = S3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {"s3_batch_file_upload": True}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + ids: Final = _burst(candidate, model, key, marker) + puts: Final = _collect(bucket, count_lines=True, expected=REQUESTS) + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert len(puts) <= 2, f"{len(puts)} PUTs for {REQUESTS} logs; batch mode must write one object per flush" + assert all(BATCH_KEY.match(put.target) for put in puts), [put.target for put in puts] + assert all(put.headers["content-type"] == "application/x-ndjson" for put in puts), [put.headers for put in puts] + lines: Final = tuple(line for put in puts for line in put.body.decode().splitlines()) + assert frozenset(json.loads(line)["id"] for line in lines) == ids + assert len(lines) == REQUESTS + + +@pytest.mark.covers("other.observability.s3_v2.batch_file_upload_keeps_team_prefix_in_object_key") +def test_s3_v2_batch_file_upload_keeps_team_alias_prefix(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3team" + uuid.uuid4().hex[:8] + team_alias: Final = f"alpha-{uuid.uuid4().hex[:8]}" + team_batch_key: Final = re.compile( + rf"^/{BUCKET}/{PREFIX}/{team_alias}/\d{{4}}-\d{{2}}-\d{{2}}/batch_\d{{2}}-\d{{2}}-\d{{2}}_[0-9a-f]{{32}}\.jsonl$" + ) + sink: Final = S3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {"s3_batch_file_upload": True, "s3_use_team_prefix": True}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + team: Final = scenario.team(team_alias=team_alias, models=[model]) + key: Final = scenario.key(team_id=team, models=[model]) + ids: Final = _burst(candidate, model, key, marker) + puts: Final = _collect(bucket, count_lines=True, expected=REQUESTS) + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert len(puts) >= 1 + assert all(team_batch_key.match(put.target) for put in puts), [put.target for put in puts] + lines: Final = tuple(line for put in puts for line in put.body.decode().splitlines()) + assert frozenset(json.loads(line)["id"] for line in lines) == ids + assert len(lines) == REQUESTS + + +@pytest.mark.covers("other.observability.s3_v2.upstream_failure_events_land_alongside_successes") +def test_s3_v2_upstream_failure_events_land_alongside_successes(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3fail" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + + def provider(request: Request) -> Reply: + text: Final = json.loads(request.body)["messages"][0]["content"] + if text.endswith("-fail"): + return Reply( + status=401, + body=b'{"error": {"message": "synthetic upstream rejection", "code": "synthetic_401"}}', + ) + return _chat_reply(request) + + with wire_server(provider) as upstream, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=upstream.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + success_ids: Final = tuple(f"{marker}-{index}" for index in range(8)) + failure_ids: Final = tuple(f"{marker}-{index}-fail" for index in range(4)) + + def send(identity: str) -> httpx.Response: + return candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": identity}], "cache": {"no-cache": True}}, + key=key, + ) + + with ThreadPoolExecutor(max_workers=12) as pool: + responses: Final = tuple(pool.map(send, (*success_ids, *failure_ids))) + ok: Final = responses[:8] + rejected: Final = responses[8:] + assert all(response.status_code == 200 for response in ok), [r.text for r in ok] + assert tuple(response.json()["id"] for response in ok) == success_ids + for response in rejected: + assert response.status_code in (400, 401), response.status_code + assert "synthetic upstream rejection" in response.text, response.text + failure_call_ids: Final = frozenset(response.headers["x-litellm-call-id"] for response in rejected) + payloads: Final = collect_payloads(sink, len(success_ids) + len(failure_ids)) + assert len(upstream.drain()) == len(success_ids) + len(failure_ids) + delivered: Final = frozenset(payload["id"] for payload in payloads if payload["status"] == "success") + assert delivered == frozenset(success_ids) + failures: Final = tuple(payload for payload in payloads if payload["status"] == "failure") + assert len(failures) == len(failure_ids) + assert frozenset(payload["litellm_call_id"] for payload in failures) == failure_call_ids + assert all("synthetic upstream rejection" in json.dumps(payload["error_information"]) for payload in failures) + + +@pytest.mark.covers("other.observability.s3_v2.invalid_or_empty_bound_falls_back_to_sixteen") +@pytest.mark.parametrize( + ("bad", "warns"), + [ + pytest.param("abc", True, id="non_integer"), + pytest.param(0, True, id="below_one"), + pytest.param("", False, id="empty"), + ], +) +def test_s3_v2_invalid_or_empty_bound_falls_back_to_sixteen( + gateway: Gateway, tmp_path: Path, bad: JsonValue, warns: bool +) -> None: + marker: Final = "s3bound" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {"s3_max_concurrent_uploads": bad}) + with ( + owned_proxy_process(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as owned, + owned.gateway.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + ids: Final = _burst(owned.gateway, model, key, marker) + payloads: Final = collect_payloads(sink, REQUESTS) + if warns: + eventually( + lambda: owned.log.read_text(), + lambda text: "s3_max_concurrent_uploads" in text, + seconds=15, + ) + else: + assert "s3_max_concurrent_uploads" not in owned.log.read_text() + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert sink.peak <= 16, f"peak concurrent PUTs {sink.peak} exceeded the fallback bound" + assert frozenset(payload["id"] for payload in payloads) == ids + + +@pytest.mark.covers("other.observability.s3_v2.sink_rejection_requeues_and_delivers_every_id_once") +def test_s3_v2_sink_rejection_requeues_and_delivers_every_id_once(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3deny" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink(fail_status=403, delay_seconds=0.2) + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy_process(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as owned, + owned.gateway.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + sink.fail_until = time.time() + 10 + ids: Final = _burst(owned.gateway, model, key, marker) + payloads: Final = collect_payloads(sink, REQUESTS, seconds=90) + eventually( + lambda: owned.log.read_text(), + lambda text: "S3BatchUploadError" in text, + seconds=15, + ) + readiness: Final = owned.gateway.client.get("/health/readiness") + assert readiness.status_code == 200, readiness.text + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert len(sink.objects()) == REQUESTS + assert frozenset(payload["id"] for payload in payloads) == ids + + +@pytest.mark.covers("other.observability.s3_v2.batch_retry_resends_identical_key_and_body") +def test_s3_v2_batch_retry_resends_identical_key_and_body(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3retry" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink(fail_status=500, delay_seconds=0.2) + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {"s3_batch_file_upload": True}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + sink.fail_until = time.time() + 8 + ids: Final = _burst(candidate, model, key, marker) + payloads: Final = collect_payloads(sink, REQUESTS, seconds=90) + puts: Final = bucket.drain() + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + by_target: Final = {} + for put in puts: + by_target.setdefault(put.target, set()).add(put.body) # mutable-ok: grouping attempts seen so far per target + assert all(len(bodies) == 1 for bodies in by_target.values()), "a retried batch PUT changed key or body" + assert max(sum(1 for put in puts if put.target == target) for target in by_target) >= 2, "no retried PUT observed" + assert frozenset(payload["id"] for payload in payloads) == ids + assert len(payloads) == REQUESTS + + +@pytest.mark.covers("other.observability.s3_v2.unknown_model_rejection_keeps_other_requests_logging") +def test_s3_v2_unknown_model_rejection_keeps_other_requests_logging(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3ghost" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + ghost: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": f"ghost-{uuid.uuid4().hex}", "messages": [{"role": "user", "content": "hi"}]}, + key=key, + ) + assert ghost.status_code in (400, 403, 404), ghost.text + ids: Final = _burst(candidate, model, key, marker) + eventually( + lambda: frozenset(payload["id"] for payload in sink.payloads()), + lambda landed: ids <= landed, + seconds=90, + ) + payloads: Final = sink.payloads() + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert ids <= frozenset(payload["id"] for payload in payloads) + extras: Final = tuple(payload for payload in payloads if payload["id"] not in ids) + assert all(payload["status"] == "failure" for payload in extras), extras + + +@pytest.mark.covers("other.observability.s3_v2.batch_flag_ignored_when_s3_v2_is_cold_storage_logger") +def test_s3_v2_batch_flag_ignored_when_s3_v2_is_cold_storage_logger(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3cold" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _recording_s3_config( + tmp_path, + bucket.url, + {"s3_batch_file_upload": True}, + {"cold_storage_custom_logger": "s3_v2"}, + ) + with ( + owned_proxy_process(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as owned, + owned.gateway.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + response: Final = owned.gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": marker}], "cache": {"no-cache": True}}, + key=key, + ) + assert response.status_code == 200, response.text + request_id: Final = str(response.json()["id"]) + payloads: Final = collect_payloads(sink, 1) + assert all(PER_REQUEST_KEY.match(target) for target in sink.objects()), list(sink.objects()) + eventually( + lambda: owned.log.read_text(), + lambda text: "s3_batch_file_upload is ignored because s3_v2 is the cold storage logger" in text, + seconds=15, + ) + spend: Final = eventually( + lambda: owned.gateway.request("GET", f"/spend/logs/ui/{request_id}"), + lambda reply: reply.status_code == 200 and bool((reply.json() or {}).get("messages")), + seconds=60, + ) + assert spend.status_code == 200, spend.text + body: Final = spend.json() + assert body["messages"], spend.text + assert body["response"], spend.text + assert payloads[0]["id"] == request_id + + +@pytest.mark.covers("other.observability.s3_v2.identical_requests_land_distinct_objects") +def test_s3_v2_identical_requests_land_distinct_objects(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3same" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + + def send(_: int) -> str: + response: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": marker}], "cache": {"no-cache": True}}, + key=key, + ) + assert response.status_code == 200, response.text + return str(response.json()["id"]) + + with ThreadPoolExecutor(max_workers=16) as pool: + returned: Final = frozenset(pool.map(send, range(16))) + payloads: Final = collect_payloads(sink, 16) + assert sum(1 for r in provider.drain() if r.method == "POST") == 16 + assert returned == {marker}, "the upstream echo keeps the same id for identical requests" + assert len(sink.objects()) == 16, "identical requests must still land as distinct objects" + assert all(payload["id"] == marker for payload in payloads) + + +@pytest.mark.covers("other.observability.s3_v2.two_workers_bound_and_deliver_every_id") +def test_s3_v2_two_workers_bound_and_deliver_every_id(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3work" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy( + gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config, workers=2 + ) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + ids: Final = _burst(candidate, model, key, marker) + payloads: Final = collect_payloads(sink, REQUESTS) + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert sink.peak <= 32, f"peak concurrent PUTs {sink.peak} exceeded two workers at the default bound" + assert len(sink.objects()) == REQUESTS + assert frozenset(payload["id"] for payload in payloads) == ids + + +@pytest.mark.covers("other.observability.s3_v2.slow_sink_never_duplicates_or_stalls_readiness") +def test_s3_v2_slow_sink_never_duplicates_or_stalls_readiness(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3slow" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink(delay_seconds=1.5) + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {"s3_batch_file_upload": True}) + with ( + owned_proxy(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "1"}, config=config) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + ids: Final = _burst(candidate, model, key, marker) + + def delivered() -> int: + readiness: Final = candidate.client.get("/health/readiness") + assert readiness.status_code == 200, readiness.text + return sum(len(body.splitlines()) for body in sink.objects().values()) + + eventually(delivered, lambda total: total >= REQUESTS, seconds=90) + payloads: Final = sink.payloads() + puts: Final = bucket.drain() + targets: Final = tuple(put.target for put in puts) + assert sum(1 for r in provider.drain() if r.method == "POST") == REQUESTS + assert len(set(targets)) == len(targets), "the same object was PUT more than once" + assert frozenset(payload["id"] for payload in payloads) == ids + assert len(payloads) == REQUESTS + + +@pytest.mark.covers("other.observability.s3_v2.worker_kill_mid_burst_keeps_surviving_deliveries") +def test_s3_v2_worker_kill_mid_burst_keeps_surviving_deliveries(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3kill" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {}) + with ( + owned_proxy_process( + gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config, workers=2 + ) as owned, + owned.gateway.scenario() as scenario, + ): + model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key") + key: Final = scenario.key(models=[model]) + sent: Final = tuple(f"{marker}-{index}" for index in range(REQUESTS)) + + def send(identity: str) -> tuple[str, bool]: + try: + response: Final = owned.gateway.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": identity}], + "cache": {"no-cache": True}, + }, + key=key, + ) + except Exception: + return identity, False + return identity, response.status_code == 200 + + with ThreadPoolExecutor(max_workers=32) as pool: + futures: Final = tuple(pool.submit(send, identity) for identity in sent) + time.sleep(0.5) + children: Final = tuple( + process for process in group_members(owned.process.pid) if process.pid != owned.process.pid + ) + assert children, "no worker children found to kill" + children[0].kill() + results: Final = tuple(future.result() for future in futures) + survivors: Final = frozenset(identity for identity, ok in results if ok) + assert survivors, "no request survived the worker kill" + readiness: Final = owned.gateway.client.get("/health/readiness") + assert readiness.status_code == 200, readiness.text + payloads: Final = collect_payloads(sink, len(survivors), seconds=90) + landed: Final = frozenset(payload["id"] for payload in payloads) + assert survivors <= landed, "an id whose response succeeded never landed" + assert landed <= frozenset(sent), "an id that was never sent landed" + + +@pytest.mark.covers("other.observability.s3_v2.sigterm_mid_burst_loses_only_inflight_without_duplicates") +def test_s3_v2_sigterm_mid_burst_loses_only_inflight_without_duplicates(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = "s3term" + uuid.uuid4().hex[:8] + sink: Final = RecordingS3Sink() + with wire_server(_chat_reply) as provider, wire_server(sink.respond) as bucket: + config: Final = _s3_config(tmp_path, bucket.url, {}) + owned: Final = owned_proxy_process(gateway, tmp_path, {"DEFAULT_S3_FLUSH_INTERVAL_SECONDS": "3"}, config=config) + candidate_owned: Final = owned.__enter__() + try: + created: Final = candidate_owned.gateway.post( + "/model/new", + { + "model_name": f"integration-{marker}", + "litellm_params": { + "model": "openai/gpt-4o-mini", + "api_key": "synthetic-provider-key", + "api_base": provider.url + "/v1", + }, + "model_info": {}, + }, + ) + model: Final = str(created["model_name"]) + key: Final = str(candidate_owned.gateway.post("/key/generate", {"models": [model]})["key"]) + sent: Final = tuple(f"{marker}-{index}" for index in range(REQUESTS)) + + def send(identity: str) -> tuple[str, bool]: + try: + response: Final = candidate_owned.gateway.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": identity}], + "cache": {"no-cache": True}, + }, + key=key, + ) + except Exception: + return identity, False + return identity, response.status_code == 200 + + with ThreadPoolExecutor(max_workers=32) as pool: + futures: Final = tuple(pool.submit(send, identity) for identity in sent) + time.sleep(0.5) + candidate_owned.process.terminate() + results: Final = tuple(future.result() for future in futures) + candidate_owned.process.wait(timeout=30) + finally: + owned.__exit__(None, None, None) + answered: Final = frozenset(identity for identity, ok in results if ok) + landed: Final = frozenset(payload["id"] for payload in sink.payloads()) + assert landed <= answered, ( + "a delivered object has no matching answered request; lost in-flight ids are expected, extras are not" + ) + targets: Final = tuple(sink.objects()) + assert len(set(targets)) == len(targets), "the same object was PUT more than once" diff --git a/tests/test_litellm/integrations/test_s3_v2.py b/tests/test_litellm/integrations/test_s3_v2.py index a9d13038180..c67eaa45112 100644 --- a/tests/test_litellm/integrations/test_s3_v2.py +++ b/tests/test_litellm/integrations/test_s3_v2.py @@ -2468,3 +2468,453 @@ def test_prompts_only_toggle_is_exposed_to_admin_ui_for_both_s3_callbacks(callba from litellm.integrations.custom_logger import CustomLogger assert "S3_LOG_PROMPTS_ONLY" in CustomLogger.get_callback_env_vars(callback_name) + + +def _element(payload: dict[str, object], key_suffix: str) -> s3BatchLoggingElement: + return s3BatchLoggingElement( + s3_object_key=f"2025-09-14/test-{key_suffix}.json", + payload=payload, + s3_object_download_filename=f"test-{key_suffix}.json", + ) + + +def _ok_response() -> MagicMock: + response = MagicMock() + response.status_code = 200 + response.raise_for_status = MagicMock() + return response + + +class _CountingPut: + def __init__(self) -> None: + self.in_flight = 0 + self.peak = 0 + self.calls = 0 + + async def __call__(self, url: str, data: str | None = None, headers: dict[str, str] | None = None) -> MagicMock: + self.in_flight += 1 + self.peak = max(self.peak, self.in_flight) + self.calls += 1 + await asyncio.sleep(0.01) + self.in_flight -= 1 + return _ok_response() + + +class _RecordingPut: + def __init__(self) -> None: + self.calls: tuple[tuple[str, str | None, dict[str, str] | None], ...] = () + + async def __call__(self, url: str, data: str | None = None, headers: dict[str, str] | None = None) -> MagicMock: + self.calls = (*self.calls, (url, data, headers)) + return _ok_response() + + +class _LateAppendingPut: + def __init__(self, logger: S3Logger, element: s3BatchLoggingElement, fail_first: bool = False) -> None: + self.logger = logger + self.element = element + self.fail_first = fail_first + self.appended = False + + async def __call__(self, url: str, data: str | None = None, headers: dict[str, str] | None = None) -> MagicMock: + if not self.appended: + self.appended = True + self.logger.log_queue.append(self.element) + if self.fail_first: + return _failure_response() + return _ok_response() + + +class _FailOnSuffixPut: + def __init__(self, suffixes: tuple[str, ...]) -> None: + self.failing = True + self.suffixes = suffixes + + async def __call__(self, url: str, data: str | None = None, headers: dict[str, str] | None = None) -> MagicMock: + if self.failing and url.endswith(self.suffixes): + return _failure_response() + return _ok_response() + + +class _FailUntilClearedPut: + def __init__(self) -> None: + self.failing = True + self.calls: tuple[tuple[str, str | None], ...] = () + + async def __call__(self, url: str, data: str | None = None, headers: dict[str, str] | None = None) -> MagicMock: + self.calls = (*self.calls, (url, data)) + if self.failing: + return _failure_response() + return _ok_response() + + +@pytest.mark.asyncio +async def test_async_send_batch_bounds_concurrent_uploads() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_max_concurrent_uploads=4, + ) + + put = _CountingPut() + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + + logger.log_queue = [_element({"i": i}, f"{i}") for i in range(40)] + + await logger.async_send_batch() + + assert put.peak == 4 + assert put.calls == 40 + + +@pytest.mark.asyncio +async def test_async_send_batch_uploads_single_jsonl_file() -> None: + import json + + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_batch_file_upload=True, + ) + + put = _RecordingPut() + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + + payloads = [{"id": "req-1"}, {"id": "req-2"}, {"id": "req-3"}] + logger.log_queue = [_element(payload, f"{i}") for i, payload in enumerate(payloads)] + + await logger.async_send_batch() + + assert len(put.calls) == 1 + url, data, headers = put.calls[0] + assert url.endswith(".jsonl") + assert data is not None + assert headers is not None + assert [json.loads(line) for line in data.splitlines()] == payloads + assert headers["Content-Type"] == "application/x-ndjson" + + +@pytest.mark.asyncio +async def test_flush_queue_preserves_events_added_during_upload() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + ) + + late_element = _element({"id": "late"}, "late") + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = _LateAppendingPut(logger, late_element) + + logger.log_queue = [_element({"id": "first"}, "first")] + + await logger.flush_queue() + + assert logger.log_queue == [late_element] + + +def _override_logger(**overrides: object) -> S3Logger: + return S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_callback_params_override=overrides, + ) + + +def test_env_backed_false_string_keeps_per_request_uploads() -> None: + assert _override_logger(s3_batch_file_upload="false").s3_batch_file_upload is False + assert _override_logger(s3_batch_file_upload="true").s3_batch_file_upload is True + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_batch_file_upload=True, + s3_callback_params_override={"s3_batch_file_upload": "false"}, + ) + assert logger.s3_batch_file_upload is True + + +@pytest.mark.parametrize("bad", [0, -3, "0", "abc", ""]) +def test_invalid_concurrency_falls_back_to_default(bad: object) -> None: + from litellm.constants import DEFAULT_S3_MAX_CONCURRENT_UPLOADS + + logger = _override_logger(s3_max_concurrent_uploads=bad) + + assert logger.s3_max_concurrent_uploads == DEFAULT_S3_MAX_CONCURRENT_UPLOADS + assert logger._upload_semaphore._value == DEFAULT_S3_MAX_CONCURRENT_UPLOADS + + +def test_env_backed_concurrency_string_is_parsed() -> None: + logger = _override_logger(s3_max_concurrent_uploads="4") + + assert logger.s3_max_concurrent_uploads == 4 + assert logger._upload_semaphore._value == 4 + + +@pytest.mark.parametrize("empty", [None, ""]) +def test_empty_config_concurrency_falls_back_to_constructor_value(empty: object) -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_max_concurrent_uploads=4, + s3_callback_params_override={"s3_max_concurrent_uploads": empty}, + ) + + assert logger.s3_max_concurrent_uploads == 4 + assert logger._upload_semaphore._value == 4 + + +def _failure_response() -> MagicMock: + response = MagicMock() + response.status_code = 400 + response.raise_for_status = MagicMock(side_effect=Exception("s3 rejected the object")) + return response + + +@pytest.mark.asyncio +async def test_failed_uploads_stay_queued_for_next_flush() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + ) + + elements = [_element({"i": i}, f"{i}") for i in range(5)] + put = _FailOnSuffixPut(("test-2.json", "test-4.json")) + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + logger.log_queue = list(elements) + + await logger.flush_queue() + + assert logger.log_queue == [elements[2], elements[4]] + + put.failing = False + await logger.flush_queue() + + assert logger.log_queue == [] + + +@pytest.mark.asyncio +async def test_batch_file_upload_failure_keeps_whole_batch() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_batch_file_upload=True, + ) + + put = _FailUntilClearedPut() + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + + elements = [_element({"i": i}, f"{i}") for i in range(3)] + logger.log_queue = list(elements) + + await logger.flush_queue() + + assert len(put.calls) == 1 + assert len(logger.log_queue) == 1 + assert logger.log_queue[0].body == "\n".join(json.dumps(element.payload) for element in elements) + + +@pytest.mark.asyncio +async def test_events_appended_during_failed_flush_survive() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + ) + + late = _element({"id": "late"}, "late") + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = _LateAppendingPut(logger, late, fail_first=True) + + first = _element({"id": "first"}, "first") + logger.log_queue = [first] + + await logger.flush_queue() + + assert logger.log_queue == [first, late] + + +@pytest.mark.asyncio +async def test_batch_file_key_shape() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_path="logs", + s3_batch_file_upload=True, + ) + + put = _RecordingPut() + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + logger.log_queue = [_element({"id": "req-1"}, "0")] + + await logger.async_send_batch() + + ((url, _data, headers),) = put.calls + assert headers is not None + assert re.search(r".*/2025-09-14/batch_\d{2}-\d{2}-\d{2}_[0-9a-f]{32}\.jsonl$", url) + assert headers["Content-Disposition"].endswith('.jsonl"') + + +@pytest.mark.asyncio +async def test_batch_file_groups_raw_elements_by_key_parent() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_batch_file_upload=True, + ) + + put = _RecordingPut() + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + + alpha = s3BatchLoggingElement( + s3_object_key="logs/alpha/2026-01-01/a.json", payload={"id": "a"}, s3_object_download_filename="a.json" + ) + beta = s3BatchLoggingElement( + s3_object_key="logs/beta/2026-01-01/b.json", payload={"id": "b"}, s3_object_download_filename="b.json" + ) + plain = s3BatchLoggingElement( + s3_object_key="logs/2026-01-01/c.json", payload={"id": "c"}, s3_object_download_filename="c.json" + ) + root = s3BatchLoggingElement( + s3_object_key="solo.json", payload={"id": "d"}, s3_object_download_filename="solo.json" + ) + logger.log_queue = [alpha, beta, plain, root] + + await logger.async_send_batch() + + assert len(put.calls) == 4 + by_parent = { + re.sub(r"(^|/)batch_\d{2}-\d{2}-\d{2}_[0-9a-f]{32}\.jsonl$", "", url.split(".com/", 1)[-1]): (url, data) + for url, data, _headers in put.calls + } + assert sorted(by_parent) == ["", "logs/2026-01-01", "logs/alpha/2026-01-01", "logs/beta/2026-01-01"] + assert [line for line in by_parent[""][1].splitlines()] == [json.dumps({"id": "d"})] + assert [line for line in by_parent["logs/alpha/2026-01-01"][1].splitlines()] == [json.dumps({"id": "a"})] + assert [line for line in by_parent["logs/beta/2026-01-01"][1].splitlines()] == [json.dumps({"id": "b"})] + assert [line for line in by_parent["logs/2026-01-01"][1].splitlines()] == [json.dumps({"id": "c"})] + + +@pytest.mark.asyncio +async def test_failed_batch_file_is_requeued_and_resent_unchanged() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_batch_file_upload=True, + ) + + put = _FailUntilClearedPut() + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + logger.log_queue = [_element({"i": i}, f"{i}") for i in range(3)] + + await logger.flush_queue() + + assert len(logger.log_queue) == 1 + assert logger.log_queue[0].body is not None + assert logger.log_queue[0].s3_object_key.endswith(".jsonl") + + put.failing = False + await logger.flush_queue() + + assert logger.log_queue == [] + assert len(put.calls) == 2 + assert put.calls[0] == put.calls[1] + + +@pytest.mark.asyncio +async def test_elements_appended_after_failed_batch_file_get_their_own_file() -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_batch_file_upload=True, + ) + + put = _FailUntilClearedPut() + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + logger.log_queue = [_element({"id": "first"}, "first")] + + await logger.flush_queue() + + late = _element({"id": "late"}, "late") + logger.log_queue.append(late) + + put.failing = False + await logger.flush_queue() + + assert logger.log_queue == [] + assert len(put.calls) == 3 + assert put.calls[0] == put.calls[1] + assert put.calls[2][0] != put.calls[0][0] + assert put.calls[2][1] == json.dumps({"id": "late"}) + + +@pytest.mark.asyncio +async def test_batch_file_mode_disabled_when_s3_v2_is_cold_storage_logger(monkeypatch: pytest.MonkeyPatch) -> None: + logger = S3Logger( + s3_bucket_name="test-bucket", + s3_aws_access_key_id="test-key", + s3_aws_secret_access_key="test-secret", + s3_region_name="us-east-1", + s3_batch_file_upload=True, + ) + + put = _RecordingPut() + + logger.async_httpx_client = AsyncMock() + logger.async_httpx_client.put = put + + import litellm + + monkeypatch.setattr(litellm, "cold_storage_custom_logger", "s3_v2") + logger.log_queue = [_element({"id": "req-1"}, "0")] + + await logger.async_send_batch() + + assert len(put.calls) == 1 + assert put.calls[0][0].endswith("test-0.json") + + monkeypatch.setattr(litellm, "cold_storage_custom_logger", None) + logger.log_queue = [_element({"id": "req-2"}, "1")] + + await logger.async_send_batch() + + assert len(put.calls) == 2 + assert put.calls[1][0].endswith(".jsonl") From d11705a24d4aabacefc40b2f8c85a2a21c23bf5c Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:04:31 -0700 Subject: [PATCH 49/96] fix(cost-map): source for bedrock mantle gpt-5.6 luna, sol, terra and grok-4.6 (#42898) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 12 ++++++++---- model_prices_and_context_window.json | 12 ++++++++---- 2 files changed, 16 insertions(+), 8 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index b73f12bb673..36866bca3ff 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -56802,7 +56802,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-sol.html" }, "bedrock_mantle/openai.gpt-5.6-terra": { "input_cost_per_token": 2.2e-06, @@ -56844,7 +56845,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-terra.html" }, "bedrock_mantle/openai.gpt-5.6-cyber": { "input_cost_per_token": 1.375e-05, @@ -56951,7 +56953,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html" }, "us.openai.gpt-5.6-sol": { "input_cost_per_token": 4.4e-06, @@ -57722,7 +57725,8 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "supports_vision": true + "supports_vision": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-xai-grok-4-6.html" }, "bedrock_mantle/anthropic.claude-haiku-4-5": { "cache_creation_input_token_cost": 1.25e-06, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index b73f12bb673..36866bca3ff 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -56802,7 +56802,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-sol.html" }, "bedrock_mantle/openai.gpt-5.6-terra": { "input_cost_per_token": 2.2e-06, @@ -56844,7 +56845,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-terra.html" }, "bedrock_mantle/openai.gpt-5.6-cyber": { "input_cost_per_token": 1.375e-05, @@ -56951,7 +56953,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_web_search": true + "supports_web_search": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html" }, "us.openai.gpt-5.6-sol": { "input_cost_per_token": 4.4e-06, @@ -57722,7 +57725,8 @@ "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "supports_vision": true + "supports_vision": true, + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-xai-grok-4-6.html" }, "bedrock_mantle/anthropic.claude-haiku-4-5": { "cache_creation_input_token_cost": 1.25e-06, From 265f874eaf3f1de13a77f819546555e11e378212 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 24 Sep 2026 01:05:55 -0700 Subject: [PATCH 50/96] test(integration): edge-case matrices for malformed token limits and callback_settings shapes (#42895) * test(integration): edge-case matrices for malformed token limits and callback_settings shapes Extends the integration suite so two classes of issues found by gauntlet reviews are caught end to end against a real proxy: - non-numeric or odd model_info token limits (from /model/new and from config YAML) must be listed as absent on /v1/models, /models, /v1/models/{id} and /model/info, keep sibling models listed, and still serve chat - every callback_settings shape (top level and per consumer) must let the proxy boot, register the configured callbacks and serve chat Four product bugs on main surfaced by the matrices are recorded as BUG skips per the suite convention: chat 500 and /model_group/info 500 on non-numeric token limits, a startup crash on a non-object callback_settings, and otel silently dropped on a non-object callback_settings.otel * test(integration): pin the exact coerced value for numeric-edge token limits Addresses review feedback: the numeric-edge matrix only asserted 'int or absent'. It now asserts the listed value for each case on /v1/models, /models and /v1/models/{id}, which also lets the listing helper drop its optional-expectation branch. --- .../test_callback_settings_boot.py | 154 +++++++++++++ .../test_model_listing_token_limits.py | 215 ++++++++++++++++++ 2 files changed, 369 insertions(+) create mode 100644 tests/integration/configuration/test_callback_settings_boot.py diff --git a/tests/integration/configuration/test_callback_settings_boot.py b/tests/integration/configuration/test_callback_settings_boot.py new file mode 100644 index 00000000000..53e81b74572 --- /dev/null +++ b/tests/integration/configuration/test_callback_settings_boot.py @@ -0,0 +1,154 @@ +import json +import uuid +from collections.abc import Mapping +from pathlib import Path +from typing import Final + +import pytest +from pydantic import JsonValue + +from tests.integration._support.client import Gateway +from tests.integration._support.process import owned_proxy + +SERVING_CONSUMERS: Final = { + "compression_interception": "CompressionInterceptionLogger", + "code_interpreter_interception": "CodeInterpreterInterceptionLogger", + "websearch_interception": "WebSearchInterceptionLogger", +} +OTEL_CONSUMER: Final = {"otel": "OpenTelemetry"} +GUARDRAIL_CONSUMERS: Final = { + "presidio": "_OPTIONAL_PresidioPIIMasking", + "lakera_prompt_injection": "lakeraAI_Moderation", +} + +TOP_LEVEL_SHAPES: Final = ( + pytest.param({}, id="empty-object"), + pytest.param(None, id="null"), + pytest.param("otel", id="string"), + pytest.param(["otel"], id="list"), + pytest.param(True, id="bool"), + pytest.param(0, id="zero"), +) + +CONSUMER_SHAPES: Final = ( + pytest.param({}, id="empty-object"), + pytest.param(None, id="null"), + pytest.param("on", id="string"), + pytest.param(True, id="bool"), + pytest.param([], id="empty-list"), + pytest.param(["on"], id="list"), + pytest.param(7, id="int"), +) + +TOP_LEVEL_BOOT_CRASH: Final = ( + "BUG: a non-object callback_settings is stored verbatim and proxy startup crashes calling .get on it" +) +TOP_LEVEL_BOOT_CRASH_IDS: Final = frozenset({"string", "list", "bool"}) +OTEL_DROPPED: Final = ( + "BUG: a non-object callback_settings.otel fails dict() and the otel callback is silently not registered" +) +OTEL_DROPPED_IDS: Final = frozenset({"null", "string", "bool", "int"}) + + +def _write_config( + directory: Path, upstream_url: str, model: str, callbacks: tuple[str, ...], callback_settings: JsonValue +) -> Path: + config: Final = directory / f"callback_settings_{uuid.uuid4().hex}.yaml" + config.write_text( + json.dumps( + { + "model_list": [ + { + "model_name": model, + "litellm_params": { + "model": f"openai/{model}", + "api_base": f"{upstream_url}/v1", + "api_key": "integration-provider-key", + }, + } + ], + "litellm_settings": {"callbacks": list(callbacks)}, + "callback_settings": callback_settings, + "general_settings": { + "master_key": "os.environ/LITELLM_MASTER_KEY", + "database_url": "os.environ/DATABASE_URL", + }, + } + ) + ) + return config + + +def _assert_registered(candidate: Gateway, consumers: Mapping[str, str]) -> None: + response: Final = candidate.request("GET", "/active/callbacks") + assert response.status_code == 200, response.text + missing: Final = sorted(name for name, class_name in consumers.items() if class_name not in response.text) + assert missing == [], response.text + + +@pytest.mark.parametrize("callback_settings", TOP_LEVEL_SHAPES) +def test_top_level_callback_settings_shape_boots_registers_and_serves_chat( + gateway: Gateway, tmp_path: Path, callback_settings: JsonValue, request: pytest.FixtureRequest +) -> None: + if request.node.callspec.id in TOP_LEVEL_BOOT_CRASH_IDS: + pytest.skip(TOP_LEVEL_BOOT_CRASH) + consumers: Final = {**SERVING_CONSUMERS, **OTEL_CONSUMER} + model: Final = f"integration-callback-settings-{uuid.uuid4().hex}" + config: Final = _write_config(tmp_path, gateway.upstream_url, model, tuple(consumers), callback_settings) + with owned_proxy(gateway, tmp_path, {"STORE_MODEL_IN_DB": "False"}, config=config) as candidate: + _assert_registered(candidate, consumers) + reply: Final = candidate.chat(model, text=f"callback settings {uuid.uuid4().hex}") + assert reply["model"] == model, reply + + +@pytest.mark.parametrize("value", CONSUMER_SHAPES) +def test_serving_consumer_settings_shape_boots_registers_and_serves_chat( + gateway: Gateway, tmp_path: Path, value: JsonValue +) -> None: + model: Final = f"integration-callback-settings-{uuid.uuid4().hex}" + config: Final = _write_config( + tmp_path, + gateway.upstream_url, + model, + tuple(SERVING_CONSUMERS), + {consumer: value for consumer in SERVING_CONSUMERS}, + ) + with owned_proxy(gateway, tmp_path, {"STORE_MODEL_IN_DB": "False"}, config=config) as candidate: + _assert_registered(candidate, SERVING_CONSUMERS) + reply: Final = candidate.chat(model, text=f"callback settings {uuid.uuid4().hex}") + assert reply["model"] == model, reply + + +@pytest.mark.parametrize("value", CONSUMER_SHAPES) +def test_otel_settings_shape_boots_registers_and_serves_chat( + gateway: Gateway, tmp_path: Path, value: JsonValue, request: pytest.FixtureRequest +) -> None: + if request.node.callspec.id in OTEL_DROPPED_IDS: + pytest.skip(OTEL_DROPPED) + model: Final = f"integration-callback-settings-{uuid.uuid4().hex}" + config: Final = _write_config(tmp_path, gateway.upstream_url, model, tuple(OTEL_CONSUMER), {"otel": value}) + with owned_proxy(gateway, tmp_path, {"STORE_MODEL_IN_DB": "False"}, config=config) as candidate: + _assert_registered(candidate, OTEL_CONSUMER) + reply: Final = candidate.chat(model, text=f"callback settings {uuid.uuid4().hex}") + assert reply["model"] == model, reply + + +@pytest.mark.parametrize("value", CONSUMER_SHAPES) +def test_guardrail_consumer_settings_shape_boots_and_registers( + gateway: Gateway, tmp_path: Path, value: JsonValue +) -> None: + model: Final = f"integration-callback-settings-{uuid.uuid4().hex}" + config: Final = _write_config( + tmp_path, + gateway.upstream_url, + model, + tuple(GUARDRAIL_CONSUMERS), + {consumer: value for consumer in GUARDRAIL_CONSUMERS}, + ) + environment: Final = { + "STORE_MODEL_IN_DB": "False", + "PRESIDIO_ANALYZER_API_BASE": gateway.upstream_url, + "PRESIDIO_ANONYMIZER_API_BASE": gateway.upstream_url, + } + with owned_proxy(gateway, tmp_path, environment, config=config) as candidate: + _assert_registered(candidate, GUARDRAIL_CONSUMERS) diff --git a/tests/integration/pricing/test_model_listing_token_limits.py b/tests/integration/pricing/test_model_listing_token_limits.py index 13702748ec5..5f75b5bb38a 100644 --- a/tests/integration/pricing/test_model_listing_token_limits.py +++ b/tests/integration/pricing/test_model_listing_token_limits.py @@ -1,9 +1,93 @@ +import json import uuid +from collections.abc import Mapping +from pathlib import Path from typing import Final +import pytest from integration._support.client import Gateway, object_value +from integration._support.process import owned_proxy from pydantic import JsonValue +SIBLING_LIMITS: Final = {"max_input_tokens": 4321, "max_output_tokens": 987} + +NON_NUMERIC_LIMITS: Final = ( + pytest.param("", id="empty-string"), + pytest.param(" ", id="blank-string"), + pytest.param("128,000", id="thousands-separator"), + pytest.param("unlimited", id="word"), + pytest.param("NaN", id="nan-string"), + pytest.param("inf", id="inf-string"), + pytest.param([], id="empty-list"), + pytest.param([4096], id="list"), + pytest.param({}, id="empty-object"), + pytest.param({"tokens": 4096}, id="object"), + pytest.param(True, id="bool"), + pytest.param(None, id="null"), +) + +NUMERIC_EDGE_LIMITS: Final = ( + pytest.param(0, id="zero"), + pytest.param(-1, id="negative"), + pytest.param(1.5, id="float"), + pytest.param("1.5", id="float-string"), + pytest.param("1e9", id="exponent-string"), + pytest.param(10**12, id="huge"), +) +NUMERIC_EDGE_EXPECTED: Final = { + "zero": 0, + "negative": -1, + "float": 1, + "float-string": 1, + "exponent-string": 1_000_000_000, + "huge": 10**12, +} + +MODEL_GROUP_INFO_500: Final = ( + "BUG: /model_group/info returns 500 for every caller when one deployment's token limit is non-numeric" +) +CHAT_500: Final = ( + "BUG: chat completions return 500 from ModelGroupInfo validation when the deployment's token limit is non-numeric" +) +MODEL_GROUP_INFO_500_IDS: Final = frozenset( + {"empty-string", "blank-string", "thousands-separator", "word", "nan-string", "inf-string"} + | {"empty-list", "list", "empty-object", "object"} +) +CHAT_500_IDS: Final = frozenset( + {"empty-string", "blank-string", "thousands-separator", "word", "empty-list", "list", "empty-object", "object"} +) + + +def _listed(gateway: Gateway, path: str) -> dict[str, dict[str, JsonValue]]: + entries: Final = gateway.get(path)["data"] + assert isinstance(entries, list) + return {str(object_value(entry)["id"]): object_value(entry) for entry in entries} + + +def _limits(entry: Mapping[str, JsonValue]) -> tuple[JsonValue, JsonValue]: + return entry.get("max_input_tokens"), entry.get("max_output_tokens") + + +def _assert_listing_spares_the_sibling( + gateway: Gateway, broken: str, sibling: str, broken_limits: tuple[JsonValue, JsonValue] +) -> None: + for path in ("/v1/models", "/models"): + listed: Final = _listed(gateway, path) + assert _limits(listed[sibling]) == (4321, 987), (path, listed[sibling]) + assert _limits(listed[broken]) == broken_limits, (path, listed[broken]) + single: Final = gateway.get(f"/v1/models/{broken}") + assert single["id"] == broken, single + assert _limits(single) == broken_limits, single + registered: Final = gateway.get("/model/info")["data"] + assert isinstance(registered, list) + assert {broken, sibling} <= {str(object_value(entry)["model_name"]) for entry in registered} + + +def _assert_serves_chat(gateway: Gateway, *models: str) -> None: + for model in models: + reply: Final = gateway.chat(model, text=f"token limit edge {uuid.uuid4().hex}") + assert reply["model"] == model, reply + def _listed_model(gateway: Gateway, model: str) -> dict[str, JsonValue]: entries: Final = gateway.get("/v1/models")["data"] @@ -27,3 +111,134 @@ def test_v1_models_carries_deployment_model_info_limits_for_an_unknown_model(gat listed: Final = _listed_model(gateway, model) assert listed["max_input_tokens"] == 4321, listed assert listed["max_output_tokens"] == 987, listed + + +def test_numeric_string_token_limit_is_coerced_to_an_int(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"openai/custom-{uuid.uuid4().hex}", + model_info={"max_input_tokens": "4096", "max_output_tokens": "512"}, + ) + assert _limits(_listed_model(gateway, model)) == (4096, 512) + + +@pytest.mark.parametrize("value", NON_NUMERIC_LIMITS) +def test_non_numeric_token_limit_is_listed_as_absent_without_breaking_the_listing( + gateway: Gateway, value: JsonValue +) -> None: + with gateway.scenario() as scenario: + sibling: Final = scenario.model(model=f"openai/custom-{uuid.uuid4().hex}", model_info=SIBLING_LIMITS) + broken: Final = scenario.model( + model=f"openai/custom-{uuid.uuid4().hex}", + model_info={"max_input_tokens": value, "max_output_tokens": value}, + ) + _assert_listing_spares_the_sibling(gateway, broken, sibling, (None, None)) + + +@pytest.mark.parametrize("value", NON_NUMERIC_LIMITS) +def test_non_numeric_token_limit_still_serves_chat( + gateway: Gateway, value: JsonValue, request: pytest.FixtureRequest +) -> None: + if request.node.callspec.id in CHAT_500_IDS: + pytest.skip(CHAT_500) + with gateway.scenario() as scenario: + sibling: Final = scenario.model(model=f"openai/custom-{uuid.uuid4().hex}", model_info=SIBLING_LIMITS) + broken: Final = scenario.model( + model=f"openai/custom-{uuid.uuid4().hex}", + model_info={"max_input_tokens": value, "max_output_tokens": value}, + ) + _assert_serves_chat(gateway, broken, sibling) + + +@pytest.mark.parametrize("value", NUMERIC_EDGE_LIMITS) +def test_numeric_edge_token_limit_is_listed_as_its_integer_without_breaking_the_listing( + gateway: Gateway, value: JsonValue, request: pytest.FixtureRequest +) -> None: + expected: Final = NUMERIC_EDGE_EXPECTED[request.node.callspec.id] + with gateway.scenario() as scenario: + sibling: Final = scenario.model(model=f"openai/custom-{uuid.uuid4().hex}", model_info=SIBLING_LIMITS) + broken: Final = scenario.model( + model=f"openai/custom-{uuid.uuid4().hex}", + model_info={"max_input_tokens": value, "max_output_tokens": value}, + ) + _assert_listing_spares_the_sibling(gateway, broken, sibling, (expected, expected)) + _assert_serves_chat(gateway, broken, sibling) + + +@pytest.mark.parametrize("field", ("max_input_tokens", "max_output_tokens")) +def test_one_malformed_limit_does_not_disturb_the_other(gateway: Gateway, field: str) -> None: + other: Final = "max_output_tokens" if field == "max_input_tokens" else "max_input_tokens" + with gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"openai/custom-{uuid.uuid4().hex}", model_info={field: "128,000", other: 2048} + ) + listed: Final = _listed_model(gateway, model) + assert listed.get(field) is None, listed + assert listed[other] == 2048, listed + + +@pytest.mark.parametrize("value", NON_NUMERIC_LIMITS + NUMERIC_EDGE_LIMITS) +def test_malformed_token_limit_keeps_model_group_info_serving( + gateway: Gateway, value: JsonValue, request: pytest.FixtureRequest +) -> None: + if request.node.callspec.id in MODEL_GROUP_INFO_500_IDS: + pytest.skip(MODEL_GROUP_INFO_500) + with gateway.scenario() as scenario: + sibling: Final = scenario.model(model=f"openai/custom-{uuid.uuid4().hex}", model_info=SIBLING_LIMITS) + broken: Final = scenario.model( + model=f"openai/custom-{uuid.uuid4().hex}", + model_info={"max_input_tokens": value, "max_output_tokens": value}, + ) + groups: Final = gateway.get("/model_group/info")["data"] + assert isinstance(groups, list) + assert {broken, sibling} <= {str(object_value(group)["model_group"]) for group in groups} + single: Final = gateway.get("/model_group/info", {"model_group": broken})["data"] + assert isinstance(single, list) + assert [object_value(group)["model_group"] for group in single] == [broken] + + +def _yaml_deployment(name: str, upstream_url: str, model_info: Mapping[str, JsonValue]) -> dict[str, JsonValue]: + return { + "model_name": name, + "litellm_params": { + "model": f"openai/custom-{uuid.uuid4().hex}", + "api_base": f"{upstream_url}/v1", + "api_key": "integration-provider-key", + }, + "model_info": dict(model_info), + } + + +def test_non_numeric_token_limits_in_config_yaml_are_listed_as_absent(gateway: Gateway, tmp_path: Path) -> None: + run: Final = uuid.uuid4().hex + sibling: Final = f"integration-yaml-sibling-{run}" + broken: Final = {f"integration-yaml-{parameter.id}-{run}": parameter.values[0] for parameter in NON_NUMERIC_LIMITS} + serving: Final = tuple( + f"integration-yaml-{parameter.id}-{run}" for parameter in NON_NUMERIC_LIMITS if parameter.id not in CHAT_500_IDS + ) + config: Final = tmp_path / "malformed_token_limits.yaml" + config.write_text( + json.dumps( + { + "model_list": [ + _yaml_deployment(sibling, gateway.upstream_url, SIBLING_LIMITS), + *( + _yaml_deployment( + name, gateway.upstream_url, {"max_input_tokens": value, "max_output_tokens": value} + ) + for name, value in broken.items() + ), + ], + "general_settings": { + "master_key": "os.environ/LITELLM_MASTER_KEY", + "database_url": "os.environ/DATABASE_URL", + "store_model_in_db": True, + }, + "router_settings": {"disable_cooldowns": True}, + } + ) + ) + with owned_proxy(gateway, tmp_path, {}, config=config) as candidate: + for name in broken: + _assert_listing_spares_the_sibling(candidate, name, sibling, (None, None)) + _assert_serves_chat(candidate, sibling, *serving) From 1c8a0ff6023b3eba6b01e955c49edcfa97bf0d22 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:06:54 -0700 Subject: [PATCH 51/96] fix(cost-map): add azure deprecation dates for gpt-6 and gpt-realtime-whisper (#42897) * fix(cost-map): add azure deprecation dates for gpt-6 and gpt-realtime-whisper Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost-map): add azure deprecation dates to dated gpt-6 keys Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 7 +++++++ model_prices_and_context_window.json | 7 +++++++ 2 files changed, 14 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 36866bca3ff..d2aec118d31 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -6159,6 +6159,7 @@ ] }, "azure/gpt-realtime-whisper": { + "deprecation_date": "2027-05-06", "input_cost_per_second": 0.0002833333333333333, "litellm_provider": "azure", "mode": "audio_transcription", @@ -8060,6 +8061,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 2.5e-05, "cache_read_input_token_cost": 1e-06, "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "deprecation_date": "2028-01-11", "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "litellm_provider": "azure", @@ -8108,6 +8110,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 2.5e-05, "cache_read_input_token_cost": 1e-06, "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "deprecation_date": "2028-01-11", "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "litellm_provider": "azure", @@ -8156,6 +8159,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 2.5e-07, "cache_read_input_token_cost": 1e-08, "cache_read_input_token_cost_above_272k_tokens": 2e-08, + "deprecation_date": "2028-03-11", "input_cost_per_token": 1e-07, "input_cost_per_token_above_272k_tokens": 2e-07, "litellm_provider": "azure", @@ -8204,6 +8208,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 2.5e-07, "cache_read_input_token_cost": 1e-08, "cache_read_input_token_cost_above_272k_tokens": 2e-08, + "deprecation_date": "2028-03-11", "input_cost_per_token": 1e-07, "input_cost_per_token_above_272k_tokens": 2e-07, "litellm_provider": "azure", @@ -8252,6 +8257,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 5e-06, "cache_read_input_token_cost": 2e-07, "cache_read_input_token_cost_above_272k_tokens": 4e-07, + "deprecation_date": "2028-03-11", "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "litellm_provider": "azure", @@ -8300,6 +8306,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 5e-06, "cache_read_input_token_cost": 2e-07, "cache_read_input_token_cost_above_272k_tokens": 4e-07, + "deprecation_date": "2028-03-11", "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "litellm_provider": "azure", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 36866bca3ff..d2aec118d31 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -6159,6 +6159,7 @@ ] }, "azure/gpt-realtime-whisper": { + "deprecation_date": "2027-05-06", "input_cost_per_second": 0.0002833333333333333, "litellm_provider": "azure", "mode": "audio_transcription", @@ -8060,6 +8061,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 2.5e-05, "cache_read_input_token_cost": 1e-06, "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "deprecation_date": "2028-01-11", "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "litellm_provider": "azure", @@ -8108,6 +8110,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 2.5e-05, "cache_read_input_token_cost": 1e-06, "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "deprecation_date": "2028-01-11", "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "litellm_provider": "azure", @@ -8156,6 +8159,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 2.5e-07, "cache_read_input_token_cost": 1e-08, "cache_read_input_token_cost_above_272k_tokens": 2e-08, + "deprecation_date": "2028-03-11", "input_cost_per_token": 1e-07, "input_cost_per_token_above_272k_tokens": 2e-07, "litellm_provider": "azure", @@ -8204,6 +8208,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 2.5e-07, "cache_read_input_token_cost": 1e-08, "cache_read_input_token_cost_above_272k_tokens": 2e-08, + "deprecation_date": "2028-03-11", "input_cost_per_token": 1e-07, "input_cost_per_token_above_272k_tokens": 2e-07, "litellm_provider": "azure", @@ -8252,6 +8257,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 5e-06, "cache_read_input_token_cost": 2e-07, "cache_read_input_token_cost_above_272k_tokens": 4e-07, + "deprecation_date": "2028-03-11", "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "litellm_provider": "azure", @@ -8300,6 +8306,7 @@ "cache_creation_input_token_cost_above_272k_tokens": 5e-06, "cache_read_input_token_cost": 2e-07, "cache_read_input_token_cost_above_272k_tokens": 4e-07, + "deprecation_date": "2028-03-11", "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "litellm_provider": "azure", From 04f3ade124476fa1eba61e99a7c8e34ae0b6cc0e Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 02:07:12 -0700 Subject: [PATCH 52/96] feat(cost-map): add wandb DeepSeek-V4.1-Flash and gemma-4-26B-A4B-it (#42924) * feat(cost-map): add wandb DeepSeek-V4.1-Flash and gemma-4-26B-A4B-it Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost-map): mark wandb gemma-4-26B-A4B-it as reasoning capable Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 28 +++++++++++++++++++ model_prices_and_context_window.json | 28 +++++++++++++++++++ 2 files changed, 56 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d2aec118d31..5ebc0416aff 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -70438,6 +70438,34 @@ "output_cost_per_token": 0.0, "source": "https://docs.typesafe.ai/models" }, + "wandb/deepseek-ai/DeepSeek-V4.1-Flash": { + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 2e-07, + "litellm_provider": "wandb", + "max_input_tokens": 1049000, + "max_tokens": 1048576, + "mode": "chat", + "output_cost_per_token": 6.5e-07, + "source": "https://wandb.ai/site/pricing/tokens/", + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_vision": true + }, + "wandb/google/gemma-4-26B-A4B-it": { + "cache_read_input_token_cost": 5e-08, + "input_cost_per_token": 1e-07, + "litellm_provider": "wandb", + "max_input_tokens": 262000, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://wandb.ai/site/pricing/tokens/", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_vision": true + }, "wandb/zai-org/GLM-5.3-Flash": { "cache_read_input_token_cost": 5e-08, "input_cost_per_token": 1.5e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d2aec118d31..5ebc0416aff 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -70438,6 +70438,34 @@ "output_cost_per_token": 0.0, "source": "https://docs.typesafe.ai/models" }, + "wandb/deepseek-ai/DeepSeek-V4.1-Flash": { + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 2e-07, + "litellm_provider": "wandb", + "max_input_tokens": 1049000, + "max_tokens": 1048576, + "mode": "chat", + "output_cost_per_token": 6.5e-07, + "source": "https://wandb.ai/site/pricing/tokens/", + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_vision": true + }, + "wandb/google/gemma-4-26B-A4B-it": { + "cache_read_input_token_cost": 5e-08, + "input_cost_per_token": 1e-07, + "litellm_provider": "wandb", + "max_input_tokens": 262000, + "max_tokens": 262144, + "mode": "chat", + "output_cost_per_token": 3e-07, + "source": "https://wandb.ai/site/pricing/tokens/", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_vision": true + }, "wandb/zai-org/GLM-5.3-Flash": { "cache_read_input_token_cost": 5e-08, "input_cost_per_token": 1.5e-07, From b550db1b4fcfa954326da18c1126e37d0c3a7bd7 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 02:58:44 -0700 Subject: [PATCH 53/96] fix(cost-map): update openrouter kimi-k2.7-code input price (#42932) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 2 +- model_prices_and_context_window.json | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 5ebc0416aff..af66504c341 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -65843,7 +65843,7 @@ "supports_web_search": false }, "openrouter/moonshotai/kimi-k2.7-code": { - "input_cost_per_token": 7.062e-07, + "input_cost_per_token": 6.562e-07, "output_cost_per_token": 3.3e-06, "cache_read_input_token_cost": 1.8e-07, "litellm_provider": "openrouter", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 5ebc0416aff..af66504c341 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -65843,7 +65843,7 @@ "supports_web_search": false }, "openrouter/moonshotai/kimi-k2.7-code": { - "input_cost_per_token": 7.062e-07, + "input_cost_per_token": 6.562e-07, "output_cost_per_token": 3.3e-06, "cache_read_input_token_cost": 1.8e-07, "litellm_provider": "openrouter", From 8bbe7edb711ee894deb5ad7e3cc10391a9837de9 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 03:03:53 -0700 Subject: [PATCH 54/96] fix(cost-map): add azure deprecation dates for regional gpt-6 rows (#42933) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 6 ++++++ model_prices_and_context_window.json | 6 ++++++ 2 files changed, 12 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index af66504c341..54475f38364 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -8647,6 +8647,7 @@ "supports_minimal_reasoning_effort": false }, "azure/us/gpt-6-astra": { + "deprecation_date": "2028-01-11", "cache_creation_input_token_cost": 1.375e-05, "cache_creation_input_token_cost_above_272k_tokens": 2.75e-05, "cache_read_input_token_cost": 1.1e-06, @@ -8695,6 +8696,7 @@ "supports_xhigh_reasoning_effort": true }, "azure/us/gpt-6-luna": { + "deprecation_date": "2028-03-11", "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_272k_tokens": 2.75e-07, "cache_read_input_token_cost": 1.1e-08, @@ -8743,6 +8745,7 @@ "supports_xhigh_reasoning_effort": true }, "azure/us/gpt-6-sol": { + "deprecation_date": "2028-03-11", "cache_creation_input_token_cost": 2.75e-06, "cache_creation_input_token_cost_above_272k_tokens": 5.5e-06, "cache_read_input_token_cost": 2.2e-07, @@ -68811,6 +68814,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-6-astra": { + "deprecation_date": "2028-01-11", "cache_creation_input_token_cost": 1.375e-05, "cache_creation_input_token_cost_above_272k_tokens": 2.75e-05, "cache_read_input_token_cost": 1.1e-06, @@ -68824,6 +68828,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-6-luna": { + "deprecation_date": "2028-03-11", "cache_creation_input_token_cost": 1.5e-07, "cache_creation_input_token_cost_above_272k_tokens": 3e-07, "cache_read_input_token_cost": 1.2e-08, @@ -68872,6 +68877,7 @@ "supports_xhigh_reasoning_effort": true }, "azure/eu/gpt-6-sol": { + "deprecation_date": "2028-03-11", "cache_creation_input_token_cost": 3e-06, "cache_creation_input_token_cost_above_272k_tokens": 6e-06, "cache_read_input_token_cost": 2.4e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index af66504c341..54475f38364 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -8647,6 +8647,7 @@ "supports_minimal_reasoning_effort": false }, "azure/us/gpt-6-astra": { + "deprecation_date": "2028-01-11", "cache_creation_input_token_cost": 1.375e-05, "cache_creation_input_token_cost_above_272k_tokens": 2.75e-05, "cache_read_input_token_cost": 1.1e-06, @@ -8695,6 +8696,7 @@ "supports_xhigh_reasoning_effort": true }, "azure/us/gpt-6-luna": { + "deprecation_date": "2028-03-11", "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_272k_tokens": 2.75e-07, "cache_read_input_token_cost": 1.1e-08, @@ -8743,6 +8745,7 @@ "supports_xhigh_reasoning_effort": true }, "azure/us/gpt-6-sol": { + "deprecation_date": "2028-03-11", "cache_creation_input_token_cost": 2.75e-06, "cache_creation_input_token_cost_above_272k_tokens": 5.5e-06, "cache_read_input_token_cost": 2.2e-07, @@ -68811,6 +68814,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-6-astra": { + "deprecation_date": "2028-01-11", "cache_creation_input_token_cost": 1.375e-05, "cache_creation_input_token_cost_above_272k_tokens": 2.75e-05, "cache_read_input_token_cost": 1.1e-06, @@ -68824,6 +68828,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-6-luna": { + "deprecation_date": "2028-03-11", "cache_creation_input_token_cost": 1.5e-07, "cache_creation_input_token_cost_above_272k_tokens": 3e-07, "cache_read_input_token_cost": 1.2e-08, @@ -68872,6 +68877,7 @@ "supports_xhigh_reasoning_effort": true }, "azure/eu/gpt-6-sol": { + "deprecation_date": "2028-03-11", "cache_creation_input_token_cost": 3e-06, "cache_creation_input_token_cost_above_272k_tokens": 6e-06, "cache_read_input_token_cost": 2.4e-07, From d17c0d77240e24f89de1003c5cf68ea331f9e258 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 04:17:23 -0700 Subject: [PATCH 55/96] refactor: daily fresh tech debt cleanup, rolling PR (#42710) * refactor: clear fresh tech debt from the last 24 hours (2026-09-05, 2026-09-06) Drop the TID251 cast import and both cast-ok casts from the refusal message_delta rebuild by narrowing the TypedDict union on its type literal, drop the redundant Mapping cast after the isinstance check in _mapping_field, and type the Lyria predict read-only helpers as Mapping[str, object] instead of a bare dict with mutable-ok. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: clear fresh tech debt from the last 24 hours (2026-09-09) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: clear fresh tech debt from the last 24 hours (2026-09-10) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: keep the pre-existing cost-estimate comment and usage cost read out of the cleanup Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: clear fresh tech debt from the last 24 hours (2026-09-13) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: keep the model info pricing helper out of the cleanup Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: drop suppressions that no longer suppress anything (2026-09-16) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: keep the rebind-ok reason inside the line limit Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(anthropic): rebuild the refusal message_delta by spreading the chunk Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: clear fresh tech debt from the last 24 hours (2026-09-17) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: clear fresh tech debt from the last 24 hours (2026-09-18) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: clear fresh tech debt from the last 24 hours (2026-09-19) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: keep the pre-existing protected-resource return type out of the cleanup Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: clear fresh tech debt from the last 24 hours (2026-09-20) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: type fresh getattr, Any, and bare dict debt from 2026-09-22 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(mcp): keep the string guard on tools/list next_cursor Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(vercel_ai_gateway): type the embedding error headers dict Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(techdebt): fix inert suppressions and missing Final in 2026-09-22 changes Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(techdebt): shorten suppression reason to fit line length Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(techdebt): format provider spread so its suppression sits on the literal Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(techdebt): drop the logger extras suppression that LIT013 now flags as inert Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(techdebt): clear fresh suppressions, Any aliases and slop from the 24h window Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(vercel): take a read-only headers mapping in get_error_class Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: mateo Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/cost_calculator.py | 8 ++++--- litellm/experimental_mcp_client/tools.py | 2 +- litellm/integrations/arize/arize.py | 2 +- .../mavvrik_focus/mavvrik_focus_logger.py | 24 ++++--------------- .../websearch_interception/handler.py | 2 +- litellm/litellm_core_utils/core_helpers.py | 2 +- litellm/litellm_core_utils/litellm_logging.py | 4 ++-- .../llm_cost_calc/zero_cost_diagnostic.py | 11 ++++++--- .../prompt_templates/common_utils.py | 4 ++-- .../litellm_core_utils/streaming_handler.py | 2 +- litellm/llms/a2a/chat/transformation.py | 2 +- .../chat/guardrail_translation/handler.py | 2 +- .../adapters/streaming_iterator.py | 24 ++++++++----------- .../messages/utils.py | 2 +- .../responses_adapters/transformation.py | 4 ++-- .../guardrail_translation/handler.py | 2 +- .../embedding/transformation.py | 7 ++++-- .../vertex_and_google_ai_studio_gemini.py | 2 +- litellm/main.py | 4 ++-- litellm/passthrough/main.py | 4 +++- .../mcp_server/bridge_token_flow.py | 4 ++-- .../mcp_server/discoverable_endpoints.py | 4 +++- .../mcp_server/mcp_server_manager.py | 2 +- .../_experimental/mcp_server/tool_search.py | 5 ++-- litellm/proxy/auth/auth_checks.py | 2 +- litellm/proxy/common_request_processing.py | 2 +- litellm/proxy/db/baseline_accounting.py | 4 ++-- .../guardrails/guardrail_initializers.py | 2 +- .../proxy/hooks/proxy_track_cost_callback.py | 2 +- .../management_endpoints/common_utils.py | 8 +++---- .../management_helpers/bulk_user_creation.py | 4 ++-- .../management_helpers/bulk_user_deletion.py | 2 +- .../llm_passthrough_endpoints.py | 2 +- .../vertex_passthrough_logging_handler.py | 4 ++-- .../proxy/response_api_endpoints/endpoints.py | 4 ++-- .../daily_global_spend_rollup.py | 4 ---- litellm/proxy/utils.py | 1 - .../transformation.py | 2 +- litellm/responses/streaming_iterator.py | 2 +- .../complexity_router/jev_classifier.py | 4 ++-- .../encrypted_content_affinity_check.py | 6 ++--- litellm/rust_bridge/dispatch.py | 10 ++++---- litellm/rust_bridge/response_metadata.py | 4 ++-- litellm/types/llms/bedrock.py | 1 - litellm/types/utils.py | 4 ++-- 45 files changed, 96 insertions(+), 107 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 7990832dc48..6cc0d9444cd 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -2017,8 +2017,8 @@ def _deployment_model_info( return cast(ModelInfo, registered_deployment_info) # cast-ok: router registers deployment prices under its id if litellm_logging_obj is None: return None - litellm_params: Final = getattr(litellm_logging_obj, "litellm_params", None) - if litellm_params is None: + litellm_params: Final = litellm_logging_obj.litellm_params + if not litellm_params: return None return next( ( @@ -2036,7 +2036,9 @@ def _ocr_model_info( router_model_id: str | None, ) -> OCRPricing | None: deployment_info: Final = _deployment_model_info(litellm_logging_obj, custom_pricing, router_model_id) - litellm_params: Final = getattr(litellm_logging_obj, "litellm_params", None) if custom_pricing else None + litellm_params: Final = ( + litellm_logging_obj.litellm_params if custom_pricing and litellm_logging_obj is not None else None + ) if litellm_params is None: return deployment_info return _layered_ocr_pricing(litellm_params, deployment_info) diff --git a/litellm/experimental_mcp_client/tools.py b/litellm/experimental_mcp_client/tools.py index a9ee851d529..df644fd7f4a 100644 --- a/litellm/experimental_mcp_client/tools.py +++ b/litellm/experimental_mcp_client/tools.py @@ -129,7 +129,7 @@ async def list_tools_with_pagination( ) tools.extend(result.tools) - next_cursor = getattr(result, "next_cursor", None) + next_cursor = result.next_cursor if not isinstance(next_cursor, str) or not next_cursor: return tools if next_cursor in seen_cursors: diff --git a/litellm/integrations/arize/arize.py b/litellm/integrations/arize/arize.py index 4ab8d9796b2..57e60fea759 100644 --- a/litellm/integrations/arize/arize.py +++ b/litellm/integrations/arize/arize.py @@ -112,7 +112,7 @@ class ArizeLogger(OpenTelemetry): if value is None or value in ("", "None"): return None try: - rate = float(value) + rate: Final = float(value) except (TypeError, ValueError): verbose_logger.warning( "ArizeLogger: %s value %r is not a number; exporting the request", diff --git a/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py b/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py index 7e3c4cc3ce8..4c2f75bb4d7 100644 --- a/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py +++ b/litellm/integrations/mavvrik_focus/mavvrik_focus_logger.py @@ -21,7 +21,7 @@ from __future__ import annotations import os from datetime import datetime, timedelta, timezone -from typing import TYPE_CHECKING, Any, Final, Protocol +from typing import TYPE_CHECKING, Any, Final import litellm from litellm._logging import verbose_proxy_logger @@ -35,17 +35,6 @@ else: AsyncIOScheduler = Any -class _PodLockManager(Protocol): - """The subset of PodLockManager this logger drives to serialize the export across pods.""" - - @property - def redis_cache(self) -> object: ... - - async def acquire_lock(self, cronjob_id: str) -> bool | None: ... - - async def release_lock(self, cronjob_id: str) -> None: ... - - def _parse_metrics_marker( marker: object | None, ) -> datetime | None: @@ -237,13 +226,10 @@ class MavvrikFocusLogger(FocusLogger): """Scheduler entry point — uses Mavvrik-specific pod-lock key.""" from litellm.proxy.proxy_server import proxy_logging_obj # noqa: PLC0415 - pod_lock_manager: _PodLockManager | None = None - if proxy_logging_obj is not None: - writer: Final[object] = getattr(proxy_logging_obj, "db_spend_update_writer", None) - if writer is not None: - pod_lock_manager = getattr(writer, "pod_lock_manager", None) - - if pod_lock_manager and pod_lock_manager.redis_cache: + pod_lock_manager: Final = ( + proxy_logging_obj.db_spend_update_writer.pod_lock_manager if proxy_logging_obj is not None else None + ) + if pod_lock_manager is not None and pod_lock_manager.redis_cache: acquired: Final = await pod_lock_manager.acquire_lock(cronjob_id=MAVVRIK_FOCUS_EXPORT_JOB_NAME) if not acquired: verbose_proxy_logger.debug("Mavvrik FOCUS export: unable to acquire pod lock") diff --git a/litellm/integrations/websearch_interception/handler.py b/litellm/integrations/websearch_interception/handler.py index 54578323fa4..90bbd5a00d8 100644 --- a/litellm/integrations/websearch_interception/handler.py +++ b/litellm/integrations/websearch_interception/handler.py @@ -1849,7 +1849,7 @@ class WebSearchInterceptionLogger(CustomLogger): for tool_call in tool_calls: # Handle both Anthropic-style input and OpenAI-style function.arguments query = None - tool_args: dict | None = None # mutable-ok: the tool call's own arguments dict + tool_args: dict[str, object] | None = None # mutable-ok: the tool call's own arguments dict if "input" in tool_call and isinstance(tool_call["input"], dict): tool_args = tool_call["input"] query = tool_args.get("query") diff --git a/litellm/litellm_core_utils/core_helpers.py b/litellm/litellm_core_utils/core_helpers.py index a2d40279c49..3afa6a913b5 100644 --- a/litellm/litellm_core_utils/core_helpers.py +++ b/litellm/litellm_core_utils/core_helpers.py @@ -365,7 +365,7 @@ def _budget_reservation_on_auth_object(user_api_key_auth: object) -> object: return getattr(user_api_key_auth, "budget_reservation", None) -def budget_reservation_from_metadata(metadata: Mapping[str, object]) -> dict | None: +def budget_reservation_from_metadata(metadata: Mapping[str, object]) -> dict[str, object] | None: stamped: Final = metadata.get("user_api_key_budget_reservation") if isinstance(stamped, dict): return stamped diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 0603414cabd..2cecec729c2 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -5191,7 +5191,7 @@ def _maybe_construct_otel_v2(callback_name: str, _in_memory_loggers: list[Custom for callback in _in_memory_loggers: if ( isinstance(callback, OpenTelemetryV2) - and getattr(callback, "callback_name", None) == callback_name + and callback.callback_name == callback_name and (serves_a_destination or not _exports_nowhere(callback.config)) ): return callback @@ -6663,7 +6663,7 @@ def get_standard_logging_object_payload( cost_breakdown=request_cost_breakdown, autorouter_savings=autorouter_savings, autorouter_savings_estimate=( - { + { # mutable-ok: spend-log JSON serialization requires plain mappings "version": 3, "status": "unknown", "reason": "pending_projection", diff --git a/litellm/litellm_core_utils/llm_cost_calc/zero_cost_diagnostic.py b/litellm/litellm_core_utils/llm_cost_calc/zero_cost_diagnostic.py index 6331d815bdc..9688b511ea8 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/zero_cost_diagnostic.py +++ b/litellm/litellm_core_utils/llm_cost_calc/zero_cost_diagnostic.py @@ -5,7 +5,12 @@ from typing import Final from pydantic import TypeAdapter, ValidationError from typing_extensions import assert_never -from litellm.types.utils import StandardLoggingZeroCostDiagnostic, Usage +from litellm.types.utils import ( + CompletionTokensDetailsWrapper, + PromptTokensDetailsWrapper, + StandardLoggingZeroCostDiagnostic, + Usage, +) ZERO_COST_COUNTER_NAME: Final = "litellm_zero_cost_requests_total" @@ -18,8 +23,8 @@ _NESTED_PRICING: Final = TypeAdapter(Mapping[str, object] | tuple[object, ...]) _MAX_PRICING_DEPTH: Final = 4 -def _audio_tokens(details: object) -> int: - audio_tokens: Final = getattr(details, "audio_tokens", None) +def _audio_tokens(details: PromptTokensDetailsWrapper | CompletionTokensDetailsWrapper | None) -> int: + audio_tokens: Final = details.audio_tokens if details is not None else None return audio_tokens if isinstance(audio_tokens, int) and audio_tokens > 0 else 0 diff --git a/litellm/litellm_core_utils/prompt_templates/common_utils.py b/litellm/litellm_core_utils/prompt_templates/common_utils.py index 378295e1b7a..8e5d2cd0a17 100644 --- a/litellm/litellm_core_utils/prompt_templates/common_utils.py +++ b/litellm/litellm_core_utils/prompt_templates/common_utils.py @@ -2003,11 +2003,11 @@ def strip_encrypted_reasoning_from_messages(messages: object) -> None: """ if not isinstance(messages, list): return - for content in _anthropic_content_lists(cast(list[object], messages)): # cast-ok: untyped client json + for content in anthropic_content_lists(cast(list[object], messages)): # cast-ok: untyped client json _strip_encrypted_reasoning_from_blocks(content) -def _anthropic_content_lists(messages: Sequence[object]) -> Iterator[object]: +def anthropic_content_lists(messages: Sequence[object]) -> Iterator[object]: return ( cast(list[object], content) # cast-ok: narrowed by isinstance for message in messages diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index f97a274708f..fa687b585f5 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -1329,7 +1329,7 @@ class CustomStreamWrapper: "is_finished": chunk_finish_reason is not None, "finish_reason": chunk_finish_reason, "original_chunk": cached_chunk, - "tool_calls": (getattr(cached_choice.delta, "tool_calls", None) if cached_choice is not None else None), + "tool_calls": cached_choice.delta.tool_calls if cached_choice is not None else None, } completion_obj["content"] = response_obj["text"] diff --git a/litellm/llms/a2a/chat/transformation.py b/litellm/llms/a2a/chat/transformation.py index c5a71daaba5..0813e0827d2 100644 --- a/litellm/llms/a2a/chat/transformation.py +++ b/litellm/llms/a2a/chat/transformation.py @@ -48,7 +48,7 @@ def _registry_api_key(agent_litellm_params: Mapping[str, object]) -> str | None: return configured_api_key if isinstance(configured_api_key, str) else None -def _registry_headers(agent_litellm_params: Mapping[str, object]) -> dict[str, Any] | None: +def _registry_headers(agent_litellm_params: Mapping[str, object]) -> dict[str, object] | None: stored_headers: Final = agent_litellm_params.get("headers") if not isinstance(stored_headers, Mapping): return None diff --git a/litellm/llms/anthropic/chat/guardrail_translation/handler.py b/litellm/llms/anthropic/chat/guardrail_translation/handler.py index a78f633f5d7..e1c727ad235 100644 --- a/litellm/llms/anthropic/chat/guardrail_translation/handler.py +++ b/litellm/llms/anthropic/chat/guardrail_translation/handler.py @@ -685,7 +685,7 @@ class AnthropicMessagesHandler(BaseTranslation): return data - def _hoisted_top_level_system_message(self, data: dict) -> AllMessageValues | None: + def _hoisted_top_level_system_message(self, data: Mapping[str, object]) -> AllMessageValues | None: """Return the system message produced by translating the top-level prompt.""" system: Final = data.get("system") if not system: diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py index 4486eb0985a..20753afee5c 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py @@ -11,7 +11,6 @@ from typing import ( Final, Literal, Protocol, - cast, # noqa: TID251 # rebuilt message_delta dict spans the ContentBlockDelta/MessageBlockDelta union get_args, ) @@ -27,6 +26,7 @@ from litellm.types.llms.anthropic import ( ContentBlockDelta, ContextManagementResponse, MessageBlockDelta, + MessageDelta, StreamingContentBlockDeltaType, UsageDelta, UsageIteration, @@ -1028,26 +1028,22 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): self, processed_chunk: ContentBlockDelta | MessageBlockDelta, ) -> ContentBlockDelta | MessageBlockDelta: - if processed_chunk.get("type") != "message_delta" or not self._refusal_text: + if processed_chunk["type"] != "message_delta" or not self._refusal_text: return processed_chunk - delta: Final = cast(Mapping[str, object], processed_chunk["delta"]) # cast-ok: keys checked before use + delta: Final = processed_chunk["delta"] if delta.get("stop_reason") == "max_tokens": return processed_chunk from litellm.llms.anthropic.experimental_pass_through.messages.utils import ( refusal_stop_details, ) - return cast( # cast-ok: rebuilt dict matches the message_delta TypedDict shape for this branch - ContentBlockDelta | MessageBlockDelta, - { # mutable-ok: fresh translation payload; never mutated after construction - **processed_chunk, - "delta": { # mutable-ok: fresh message_delta payload; never mutated after construction - **delta, - "stop_reason": "refusal", - "stop_details": refusal_stop_details(self._refusal_text), - }, - }, - ) + refusal_delta: Final[MessageDelta] = { + **delta, + "stop_reason": "refusal", + "stop_details": refusal_stop_details(self._refusal_text), + } + refusal_chunk: Final[MessageBlockDelta] = {**processed_chunk, "delta": refusal_delta} + return refusal_chunk @staticmethod def _delta_has_content(processed_chunk: Mapping[str, object]) -> bool: diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/utils.py b/litellm/llms/anthropic/experimental_pass_through/messages/utils.py index 7545dff1408..89105c00428 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/utils.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/utils.py @@ -37,7 +37,7 @@ def _mapping_field(container: object, key: str) -> object | None: """One key of a raw provider payload, or None when the payload is not a mapping.""" if not isinstance(container, Mapping): return None - return cast(Mapping[str, object], container).get(key) # cast-ok: raw payload, callers re-check every value + return container.get(key) def _mapping_str_field(container: object, key: str) -> str | None: diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py index e3d3425f8a6..6a31173a9c6 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py @@ -169,7 +169,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter: cls, summary: Iterable[object], encrypted_content: object, - ) -> dict[str, Any] | None: # mutable-ok: API message payload + ) -> dict[str, object] | None: # mutable-ok: API message payload """The one Anthropic block for a Responses reasoning item. The item's encrypted reasoning rides the block's opaque field (`signature`, or @@ -198,7 +198,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter: @classmethod def _assistant_group_to_input_items( cls, group: tuple[Mapping[str, object], ...] - ) -> tuple[dict[str, Any], ...]: # mutable-ok: API message payload + ) -> tuple[dict[str, object], ...]: # mutable-ok: API message payload first: Final = group[0] btype: Final = first.get("type") if btype in ("thinking", "redacted_thinking"): diff --git a/litellm/llms/openai/responses/guardrail_translation/handler.py b/litellm/llms/openai/responses/guardrail_translation/handler.py index ad8e29ad4ac..66cebe0175d 100644 --- a/litellm/llms/openai/responses/guardrail_translation/handler.py +++ b/litellm/llms/openai/responses/guardrail_translation/handler.py @@ -994,7 +994,7 @@ class OpenAIResponsesHandler(BaseTranslation): def _spread_text_rewrite_over_stream_events( self, - stream_events: Sequence[Any], + stream_events: Sequence[object], rewritten_text: str, guardrail_name: str, ) -> None: diff --git a/litellm/llms/vercel_ai_gateway/embedding/transformation.py b/litellm/llms/vercel_ai_gateway/embedding/transformation.py index fc9c6bcc19f..9cfaaab89f0 100644 --- a/litellm/llms/vercel_ai_gateway/embedding/transformation.py +++ b/litellm/llms/vercel_ai_gateway/embedding/transformation.py @@ -7,6 +7,7 @@ Vercel AI Gateway is OpenAI-compatible and supports embeddings via the /v1/embed Docs: https://vercel.com/docs/ai-gateway/openai-compat/embeddings """ +from collections.abc import Mapping from typing import TYPE_CHECKING, Any, Final import httpx @@ -161,12 +162,14 @@ class VercelAIGatewayEmbeddingConfig(BaseEmbeddingConfig): optional_params[param] = value return optional_params - def get_error_class(self, error_message: str, status_code: int, headers: Any) -> BaseLLMException: + def get_error_class( + self, error_message: str, status_code: int, headers: Mapping[str, str] | httpx.Headers + ) -> BaseLLMException: """ Get the error class for Vercel AI Gateway errors. """ return VercelAIGatewayException( message=error_message, status_code=status_code, - headers=headers, + headers=headers if isinstance(headers, httpx.Headers) else httpx.Headers(headers), ) diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index a50665c71df..941ec4ad419 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -286,7 +286,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): """ Check if the model is Gemini 3 or newer. """ - model_name = model.split("/")[-1].lower() + model_name: Final = model.split("/")[-1].lower() is_vertex_fine_tuned_model: Final = model_name.isdigit() or ( model.startswith("gemini/") and not model_name.startswith("gemini-") ) diff --git a/litellm/main.py b/litellm/main.py index 4c40d864169..ceac729d3f0 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1182,7 +1182,7 @@ def _is_claude_tool_target(custom_llm_provider: str | None, model: str) -> bool: return False -def _without_anthropic_only_tool_keys(tool: dict) -> dict: +def _without_anthropic_only_tool_keys(tool: dict[str, object]) -> dict[str, object]: kept: Final = {key: value for key, value in tool.items() if key not in _ANTHROPIC_ONLY_TOOL_KEYS} function: Final = tool.get("function") if not isinstance(function, dict): @@ -1193,7 +1193,7 @@ def _without_anthropic_only_tool_keys(tool: dict) -> dict: } -def _drop_anthropic_only_tool_keys(tools: list[dict] | None) -> list[dict] | None: +def _drop_anthropic_only_tool_keys(tools: list[dict[str, object]] | None) -> list[dict[str, object]] | None: if tools is None: return None return [_without_anthropic_only_tool_keys(tool) if isinstance(tool, dict) else tool for tool in tools] diff --git a/litellm/passthrough/main.py b/litellm/passthrough/main.py index ef931827d85..45ab52690bf 100644 --- a/litellm/passthrough/main.py +++ b/litellm/passthrough/main.py @@ -540,7 +540,9 @@ def llm_passthrough_route( ) ## IS STREAMING REQUEST - _streaming_request_data: dict = data if isinstance(data, dict) else (json if isinstance(json, dict) else {}) + _streaming_request_data: Final[dict[str, object]] = ( + data if isinstance(data, dict) else (json if isinstance(json, dict) else {}) + ) is_streaming_request: Final = provider_config.is_streaming_request( endpoint=endpoint, request_data=_streaming_request_data, diff --git a/litellm/proxy/_experimental/mcp_server/bridge_token_flow.py b/litellm/proxy/_experimental/mcp_server/bridge_token_flow.py index e5d11271c67..77bdbd26b35 100644 --- a/litellm/proxy/_experimental/mcp_server/bridge_token_flow.py +++ b/litellm/proxy/_experimental/mcp_server/bridge_token_flow.py @@ -86,8 +86,8 @@ async def oauth_authorization_uses_gateway_credential(request: Request) -> bool: async def _opaque_bearer_is_gateway_credential(token: str) -> bool: - from litellm.proxy._experimental.mcp_server.outbound_credentials.envelope import ( - is_envelope, # noqa: PLC0415 # envelope imports bridge types + from litellm.proxy._experimental.mcp_server.outbound_credentials.envelope import ( # noqa: PLC0415 # envelope imports bridge types + is_envelope, is_refresh_envelope, ) from litellm.proxy._types import hash_token # noqa: PLC0415 # proxy import cycle diff --git a/litellm/proxy/_experimental/mcp_server/discoverable_endpoints.py b/litellm/proxy/_experimental/mcp_server/discoverable_endpoints.py index ade829a1b67..d42c1c6b879 100644 --- a/litellm/proxy/_experimental/mcp_server/discoverable_endpoints.py +++ b/litellm/proxy/_experimental/mcp_server/discoverable_endpoints.py @@ -2634,7 +2634,9 @@ def _build_aggregate_protected_resource_response(request: Request) -> dict: } -def _build_aggregate_authorization_server_response(request: Request, token_exchange_available: bool) -> dict: +def _build_aggregate_authorization_server_response( + request: Request, token_exchange_available: bool +) -> dict[str, object]: """RFC 8414 metadata for the gateway as the aggregate authorization server. The issuer is ``{base}/mcp`` and must stay equal to the value the diff --git a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py index 0c520142fb3..312dcb27d89 100644 --- a/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py +++ b/litellm/proxy/_experimental/mcp_server/mcp_server_manager.py @@ -3490,7 +3490,7 @@ class MCPServerManager: passthrough_server_ids: Final = [ server.server_id for server in self.get_registry().values() - if getattr(server, "auth_type", None) == MCPAuth.true_passthrough + if server.auth_type == MCPAuth.true_passthrough ] combined_servers.update(passthrough_server_ids) diff --git a/litellm/proxy/_experimental/mcp_server/tool_search.py b/litellm/proxy/_experimental/mcp_server/tool_search.py index 3c060752934..71e46f8df25 100644 --- a/litellm/proxy/_experimental/mcp_server/tool_search.py +++ b/litellm/proxy/_experimental/mcp_server/tool_search.py @@ -128,14 +128,15 @@ def with_mcp_proxy_identity(tool: Tool, server_id: str) -> Tool: def _mcp_proxy_identity(tool: Tool) -> MCPProxyToolIdentity: - identity: Final = (tool.meta or {}).get(_MCP_PROXY_IDENTITY_META_KEY) # mutable-ok: absent metadata default + identity: Final = None if tool.meta is None else tool.meta.get(_MCP_PROXY_IDENTITY_META_KEY) if not isinstance(identity, Mapping): raise TypeError("MCP proxy tool identity is missing") server_id: Final = identity.get("server_id") tool_name: Final = identity.get("tool_name") if not isinstance(server_id, str) or not isinstance(tool_name, str): raise TypeError("MCP proxy tool identity is invalid") - return {"server_id": server_id, "tool_name": tool_name} # mutable-ok: TypedDict identity payload + resolved: Final[MCPProxyToolIdentity] = {"server_id": server_id, "tool_name": tool_name} + return resolved def mcp_proxy_tool_id(tool: Tool) -> str: diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index d2fed1eb421..f5e98b40ca7 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -4073,7 +4073,7 @@ async def get_org_object_for_request( ) except OrganizationNotFoundError: return None - except Exception as e: # noqa: BLE001 # only a DB outage may fail auth here, anything else degrades to no org limits + except Exception as e: if not PrismaDBExceptionHandler.is_database_service_unavailable_error_in_chain(e): verbose_proxy_logger.debug("org lookup failed, continuing without org limits", exc_info=True) return None diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 2e43c6b0d22..904070cfadd 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -2323,7 +2323,7 @@ class ProxyBaseLLMRequestProcessing: return fallbacks if isinstance(fallbacks, list) and fallbacks else None @staticmethod - def _resolve_fallback_models(model: str, fallbacks: list) -> list | None: + def _resolve_fallback_models(model: str, fallbacks: list) -> list[str] | None: from litellm.router_utils.fallback_event_handlers import get_fallback_model_group fallback_model_group, generic_fallback_idx = get_fallback_model_group( diff --git a/litellm/proxy/db/baseline_accounting.py b/litellm/proxy/db/baseline_accounting.py index 78036237993..05a4a989152 100644 --- a/litellm/proxy/db/baseline_accounting.py +++ b/litellm/proxy/db/baseline_accounting.py @@ -486,7 +486,7 @@ class BaselineAccountingStore: async def _pages( self, db: SupportsRawQueries, scope: str, after_revision: int, withdraw_from: float | None = None ) -> AsyncIterator[tuple[_StoredRecord, ...]]: - cursor: float | None = None + cursor: float | None = None # rebind-ok: keyset pagination advances after each complete timestamp group while page := _RECORDS.validate_python( tuple(await db.query_raw(_READ_PAGE, scope, after_revision, cursor, _PAGE_TIMESTAMPS, withdraw_from)) ): @@ -627,7 +627,7 @@ async def flush_baseline_accounting(client: PrismaClient) -> None: more_queued: Final = bool(client.baseline_accounting_transactions) try: remaining: Final = await asyncio.wait_for(_flush_records(store, batch), timeout=5) - except (Exception, asyncio.CancelledError) as error: # noqa: BLE001 # unknown acknowledgements can be replayed safely + except (Exception, asyncio.CancelledError) as error: async with client.baseline_accounting_lock: client.baseline_accounting_transactions.extend(batch) if isinstance(error, asyncio.CancelledError): diff --git a/litellm/proxy/guardrails/guardrail_initializers.py b/litellm/proxy/guardrails/guardrail_initializers.py index 7bafad26569..c422902d30d 100644 --- a/litellm/proxy/guardrails/guardrail_initializers.py +++ b/litellm/proxy/guardrails/guardrail_initializers.py @@ -120,7 +120,7 @@ def initialize_presidio(litellm_params: LitellmParams, guardrail: Guardrail) -> _OPTIONAL_PresidioPIIMasking, ) - explicit_filter_scope: Final = getattr(litellm_params, "presidio_filter_scope", None) + explicit_filter_scope: Final = litellm_params.presidio_filter_scope filter_scope: Final = explicit_filter_scope or ("input" if _is_mcp_only_mode(litellm_params.mode) else "both") run_input: Final = filter_scope in ("input", "both") run_output: Final = filter_scope in ("output", "both") diff --git a/litellm/proxy/hooks/proxy_track_cost_callback.py b/litellm/proxy/hooks/proxy_track_cost_callback.py index 81894a5ff12..5dae9e8bb10 100644 --- a/litellm/proxy/hooks/proxy_track_cost_callback.py +++ b/litellm/proxy/hooks/proxy_track_cost_callback.py @@ -770,7 +770,7 @@ async def _reconcile_budget_reservation_before_db_update( "Failed to invalidate budget reservation counters after pre-persist reconcile failed" ) finally: - budget_reservation["finalized"] = True # rebind-ok: the counter update reads the stamp off the shared dict + budget_reservation["finalized"] = True # rebind-ok: stamps the caller's shared dict for the counter update async def _release_budget_reservation(budget_reservation: dict | None) -> None: diff --git a/litellm/proxy/management_endpoints/common_utils.py b/litellm/proxy/management_endpoints/common_utils.py index 0bd4eb5a5d8..59c06a3f888 100644 --- a/litellm/proxy/management_endpoints/common_utils.py +++ b/litellm/proxy/management_endpoints/common_utils.py @@ -528,7 +528,7 @@ def _prisma_value(value: object) -> object: return list(value) if isinstance(value, tuple) else value -def member_budget_patch(source: BaseModel) -> dict[str, Any]: +def member_budget_patch(source: BaseModel) -> Mapping[str, object]: """Map the per-member limit fields a request actually set to their budget-table columns (merge-patch: a sent value updates, an explicit null clears, an absent field is left untouched).""" @@ -561,7 +561,7 @@ async def _upsert_budget_and_membership( user_id: str, existing_budget_id: str | None, user_api_key_dict: UserAPIKeyAuth, - budget_patch: dict[str, Any], + budget_patch: Mapping[str, object], team_default_budget_id: str | None = None, shared_budget_ids: frozenset[str] | None = None, ): @@ -624,9 +624,9 @@ async def _upsert_budget_and_membership( if is_shared_default and not temp_only else None ) - source: Final[Mapping[str, Any]] = source_row.model_dump() if source_row is not None else MappingProxyType({}) + source: Final[Mapping[str, object]] = source_row.model_dump() if source_row is not None else MappingProxyType({}) - create_data: Final[dict[str, Any]] = { # mutable-ok: Prisma create payloads are dict-shaped + create_data: Final[dict[str, object]] = { # mutable-ok: Prisma create payloads are dict-shaped "created_by": user_api_key_dict.user_id or "", "updated_by": user_api_key_dict.user_id or "", **MappingProxyType( diff --git a/litellm/proxy/management_helpers/bulk_user_creation.py b/litellm/proxy/management_helpers/bulk_user_creation.py index dd3f4ff1b12..ec8fd312766 100644 --- a/litellm/proxy/management_helpers/bulk_user_creation.py +++ b/litellm/proxy/management_helpers/bulk_user_creation.py @@ -348,7 +348,7 @@ async def _prepare_user(user: _PendingUser, prisma_client: PrismaClient) -> _Pre data: Final = {**dumped, "user_id": user.user_id} # mutable-ok: /user/new defaults helper mutates in place data_json: Final = _JSON_OBJECT.validate_python(_update_internal_new_user_params(data, user.request)) with_permission: Final = _JSON_OBJECT.validate_python( - await _set_object_permission(data_json=data_json, prisma_client=prisma_client) # pyright: ignore[reportUnknownArgumentType] # validated by the adapter + await _set_object_permission(data_json=data_json, prisma_client=prisma_client) ) return _PreparedUser(user, _USER_ROW.validate_python(with_permission)) except Exception as exc: # noqa: BLE001 # any preparation failure is reported on this row only @@ -509,7 +509,7 @@ class _TeamsData(TypedDict): def _default_member_budget_id(team: LiteLLM_TeamTable) -> str | None: metadata: Final = ( _JSON_OBJECT.validate_python( - team.metadata # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # LiteLLM_TeamTable.metadata is a bare dict; validated by the adapter + team.metadata # pyright: ignore[reportUnknownMemberType] # LiteLLM_TeamTable.metadata is a bare dict; validated by the adapter ) if team.metadata # pyright: ignore[reportUnknownMemberType] # same bare dict else None diff --git a/litellm/proxy/management_helpers/bulk_user_deletion.py b/litellm/proxy/management_helpers/bulk_user_deletion.py index 8b5b601fe8a..c7b89a6dd6c 100644 --- a/litellm/proxy/management_helpers/bulk_user_deletion.py +++ b/litellm/proxy/management_helpers/bulk_user_deletion.py @@ -204,7 +204,7 @@ def _error_message(exc: BaseException) -> str: if isinstance(exc, HTTPException) and isinstance(exc.detail, dict): return str(exc.detail.get("error", exc.detail)) # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # HTTPException.detail is untyped if isinstance(exc, HTTPException): - return str(exc.detail) # pyright: ignore[reportUnknownArgumentType] # HTTPException.detail is untyped + return str(exc.detail) return str(exc) or type(exc).__name__ diff --git a/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py b/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py index 4e60c318f03..eaa03b67b40 100644 --- a/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py +++ b/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py @@ -3902,7 +3902,7 @@ async def handle_gigachat_passthrough_router_model( is_streaming: Final = request_body.get("stream", False) # pyright: ignore[reportUnknownVariableType] # request_body is dict[Unknown, Unknown] - data: dict[str, Any] = await _read_request_body(request=request) # Any needed for proxy pipeline + data: Final[dict[str, object]] = await _read_request_body(request=request) if user_api_key_dict is not None: auth_metadata: Final = { metadata_key: value diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py index 2cdeddbea30..040250637ea 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py @@ -458,7 +458,7 @@ class VertexPassthroughLoggingHandler: @staticmethod def _is_audio_predict_response( model: str, - json_response: dict, # mutable-ok: predicate inspects the decoded provider response dictionary without mutation + json_response: Mapping[str, object], ) -> bool: return ( VertexPassthroughLoggingHandler._get_audio_prediction_count(json_response=json_response) > 0 @@ -467,7 +467,7 @@ class VertexPassthroughLoggingHandler: @staticmethod def _get_audio_prediction_count( - json_response: dict, # mutable-ok: counter inspects the decoded provider response dictionary without mutation + json_response: Mapping[str, object], ) -> int: predictions: Final = json_response.get("predictions") if not isinstance(predictions, list): diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index ca4008dba3d..75eefb2e73b 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -6,7 +6,7 @@ from collections.abc import AsyncIterator, Awaitable, Mapping, Sequence from enum import Enum from functools import partial from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, NamedTuple, Protocol, cast, get_args +from typing import TYPE_CHECKING, Any, Final, NamedTuple, Protocol, TypeAlias, cast, get_args from uuid import uuid4 import fastapi @@ -49,7 +49,7 @@ if TYPE_CHECKING: router: Final = APIRouter() -_ResponseDocSchemas = dict[int | str, dict[str, Any]] # pyright: ignore[reportExplicitAny] # fastapi's responses kwarg +_ResponseDocSchemas: TypeAlias = dict[int | str, dict[str, object]] # fastapi's responses kwarg RESPONSES_API_RESPONSE_SCHEMAS: Final[_ResponseDocSchemas] = {200: {"model": ResponsesAPIResponse}} RESPONSES_API_CREATE_RESPONSE_SCHEMAS: Final[_ResponseDocSchemas] = { diff --git a/litellm/proxy/spend_tracking/daily_global_spend_rollup.py b/litellm/proxy/spend_tracking/daily_global_spend_rollup.py index 376b113ed02..b81b6c1943e 100644 --- a/litellm/proxy/spend_tracking/daily_global_spend_rollup.py +++ b/litellm/proxy/spend_tracking/daily_global_spend_rollup.py @@ -198,10 +198,6 @@ async def _scan_pending(prisma_client: "PrismaClient") -> _PendingScan: return _PendingScan(marker, db_now.now, tuple(_DateRow.model_validate(row).date for row in rows)) -async def pending_days(prisma_client: "PrismaClient") -> tuple[str, ...]: - return (await _scan_pending(prisma_client)).days - - async def reconcile_day(prisma_client: "PrismaClient", day: str) -> None: """Rewrite one day of the global table from the per-key sums. Idempotent: a rerun overwrites every group with the same totals.""" diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index bc64293c9b3..fe161f5d50b 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -2389,7 +2389,6 @@ class ProxyLogging: ) try: - # Execute guardrail pipelines before the normal callback loop if not skip_guardrails: data, _ = await self._maybe_execute_pipelines( # rebind-ok: pipeline edits feed the callback loop below data=data, diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index e421cae0724..bd239922fd3 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -2254,7 +2254,7 @@ class LiteLLMCompletionResponsesConfig: ) -> Mapping[str, ResponseFunctionWebSearch]: calls: Final[dict[str, ResponseFunctionWebSearch]] = {} # mutable-ok: indexes provider-built calls for choice in chat_completion_response.choices: - provider_fields = getattr(choice.message, "provider_specific_fields", None) + provider_fields = choice.message.provider_specific_fields if not isinstance(provider_fields, Mapping): continue web_search_calls = provider_fields.get("web_search_calls") diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 64989c4cf1c..70f2a7db6da 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -1360,7 +1360,7 @@ def _billed_terminal_response( return None usage: Final[object] = response_obj.get("usage") # pyright: ignore[reportUnknownMemberType, reportUnknownVariableType] # a model_constructed terminal event leaves response as an untyped dict return ResponsesAPIResponse.model_construct( - **{**response_obj, "usage": usage if usage is not None or estimate is None else estimate()} # pyright: ignore[reportUnknownArgumentType, reportArgumentType] # same untyped dict spread + **{**response_obj, "usage": usage if usage is not None or estimate is None else estimate()} # pyright: ignore[reportArgumentType] # same untyped dict spread ) diff --git a/litellm/router_strategy/complexity_router/jev_classifier.py b/litellm/router_strategy/complexity_router/jev_classifier.py index c0d0d1de8e3..02e57975626 100644 --- a/litellm/router_strategy/complexity_router/jev_classifier.py +++ b/litellm/router_strategy/complexity_router/jev_classifier.py @@ -1,7 +1,7 @@ from collections.abc import Mapping from datetime import datetime, timezone from types import MappingProxyType -from typing import Annotated, Final, Literal, NamedTuple, Protocol +from typing import Annotated, Final, Literal, NamedTuple, Protocol, TypeAlias from uuid import uuid4 import httpx @@ -24,7 +24,7 @@ from litellm.proxy.pass_through_endpoints.llm_provider_handlers.typesafe_passthr from litellm.router_strategy.complexity_router.config import DEFAULT_JEV_INSTRUCTIONS as _DEFAULT_JEV_INSTRUCTIONS from litellm.types.utils import AUTOROUTER_CLASSIFIER_CALL_ORIGIN -JevProbability = Annotated[float, Field(ge=0.0, le=1.0)] +JevProbability: TypeAlias = Annotated[float, Field(ge=0.0, le=1.0)] DEFAULT_JEV_INSTRUCTIONS: Final = _DEFAULT_JEV_INSTRUCTIONS diff --git a/litellm/router_utils/pre_call_checks/encrypted_content_affinity_check.py b/litellm/router_utils/pre_call_checks/encrypted_content_affinity_check.py index cb5c3089685..e5e40d7d6f5 100644 --- a/litellm/router_utils/pre_call_checks/encrypted_content_affinity_check.py +++ b/litellm/router_utils/pre_call_checks/encrypted_content_affinity_check.py @@ -50,6 +50,7 @@ from litellm.exceptions import ( from litellm.integrations.custom_logger import CustomLogger, Span from litellm.litellm_core_utils.credential_accessor import CredentialAccessor from litellm.litellm_core_utils.prompt_templates.common_utils import ( + anthropic_content_lists, encrypted_content_of_block, strip_encrypted_reasoning_from_messages, ) @@ -155,10 +156,7 @@ class EncryptedContentAffinityCheck(CustomLogger): return iter(()) return ( cast(Mapping[str, object], block) # cast-ok: narrowed by isinstance - for message in cast(list[object], messages) # cast-ok: narrowed by isinstance - if isinstance(message, Mapping) - for content in (cast(Mapping[str, object], message).get("content"),) # cast-ok: narrowed by isinstance - if isinstance(content, list) + for content in anthropic_content_lists(cast(list[object], messages)) # cast-ok: narrowed by isinstance for block in cast(list[object], content) # cast-ok: narrowed by isinstance if isinstance(block, Mapping) ) diff --git a/litellm/rust_bridge/dispatch.py b/litellm/rust_bridge/dispatch.py index 076b7759c6d..94ccddc92b7 100644 --- a/litellm/rust_bridge/dispatch.py +++ b/litellm/rust_bridge/dispatch.py @@ -2,7 +2,7 @@ from __future__ import annotations from collections.abc import Awaitable, Callable, Mapping from dataclasses import dataclass -from typing import Final, Generic, TypeVar +from typing import Final, Generic, TypeAlias, TypeVar from litellm.rust_bridge import catalog, runtime from litellm.rust_bridge.bindings import NativeBinding @@ -10,11 +10,11 @@ from litellm.rust_bridge.catalog import Route, RouteContext, RouteRule, Rules from litellm.rust_bridge.configuration import Decision from litellm.rust_bridge.configuration import decision as rollout_decision -RequestT = TypeVar("RequestT") -NativeT = TypeVar("NativeT") -ResultT = TypeVar("ResultT") +RequestT: Final = TypeVar("RequestT") +NativeT: Final = TypeVar("NativeT") +ResultT: Final = TypeVar("ResultT") -NativeHook = Callable[[RequestT, tuple[object, ...], Mapping[str, object]], ResultT] +NativeHook: TypeAlias = Callable[[RequestT, tuple[object, ...], Mapping[str, object]], ResultT] def call_hook( diff --git a/litellm/rust_bridge/response_metadata.py b/litellm/rust_bridge/response_metadata.py index 1c03515720e..ae459710b34 100644 --- a/litellm/rust_bridge/response_metadata.py +++ b/litellm/rust_bridge/response_metadata.py @@ -1,10 +1,10 @@ -from typing import TypeVar +from typing import Final, TypeVar from litellm.router_utils.add_retry_fallback_headers import ( _add_headers_to_response, # pyright: ignore[reportPrivateUsage] # reuse the proxy's identity-preserving response metadata writer ) -ResultT = TypeVar("ResultT") +ResultT: Final = TypeVar("ResultT") def mark_rust_response(response: ResultT) -> ResultT: diff --git a/litellm/types/llms/bedrock.py b/litellm/types/llms/bedrock.py index 3674bb670d5..e4c41c3ee5b 100644 --- a/litellm/types/llms/bedrock.py +++ b/litellm/types/llms/bedrock.py @@ -558,7 +558,6 @@ class AmazonTitanMultimodalEmbeddingResponse(TypedDict): message: str # Specifies any errors that occur during generation. -# TwelveLabs Marengo Embed types TWELVELABS_EMBEDDING_INPUT_TYPES = Literal["text", "image", "video", "audio"] TWELVELABS_EMBEDDING_OPTIONS = Literal["visual-text", "visual-image", "audio"] diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 064e3040054..caf88e5d517 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3860,7 +3860,7 @@ def without_server_derived_pricing(model_info: Mapping[str, Any]) -> Mapping[str ) -def echoed_cost_map_pricing_fields(model_info: Mapping[str, Any]) -> tuple[str, ...]: +def echoed_cost_map_pricing_fields(model_info: Mapping[str, object]) -> tuple[str, ...]: """Pricing fields a stored ``model_info`` blob copied from a ``/model/info`` response. Only ``litellm.get_model_info`` emits ``key`` (the resolved cost-map entry), so a stored @@ -3891,7 +3891,7 @@ def echoed_cost_map_fields( ) -def pricing_override_fields(*sources: Mapping[str, Any]) -> tuple[str, ...]: +def pricing_override_fields(*sources: Mapping[str, object]) -> tuple[str, ...]: return tuple( sorted( frozenset( From 2eeb16266b906494d3a961d0d03da20810b49924 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 04:39:09 -0700 Subject: [PATCH 56/96] fix(cost-map): sync vertex-ai rows (gemma 4 maas cache price, chirp_2) (#42942) * fix(cost-map): sync vertex-ai rows (gemma 4 maas cache price, chirp_2) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(vertex-ai): expect chirp_2 as speech-to-text model now that catalog row exists Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../model_prices_and_context_window_backup.json | 15 +++++++++++++++ model_prices_and_context_window.json | 15 +++++++++++++++ .../test_vertex_ai_realtime_transformation.py | 2 +- 3 files changed, 31 insertions(+), 1 deletion(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 54475f38364..298ec16fc55 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -48131,6 +48131,20 @@ "/v1/realtime" ] }, + "vertex_ai/chirp_2": { + "input_cost_per_second": 0.00026667, + "litellm_provider": "vertex_ai", + "metadata": { + "calculation": "$0.016/60 seconds = $0.00026667 per second", + "original_pricing_per_minute": 0.016 + }, + "mode": "audio_transcription", + "source": "https://cloud.google.com/speech-to-text/pricing", + "supported_endpoints": [ + "/v1/audio/transcriptions", + "/v1/realtime" + ] + }, "vertex_ai/claude-3-5-haiku": { "deprecation_date": "2026-07-05", "input_cost_per_token": 1e-06, @@ -50155,6 +50169,7 @@ ] }, "vertex_ai/google/gemma-4-26b-a4b-it-maas": { + "cache_read_input_token_cost": 1.5e-08, "input_cost_per_token": 1.5e-07, "litellm_provider": "vertex_ai-openai_models", "max_input_tokens": 262144, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 54475f38364..298ec16fc55 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -48131,6 +48131,20 @@ "/v1/realtime" ] }, + "vertex_ai/chirp_2": { + "input_cost_per_second": 0.00026667, + "litellm_provider": "vertex_ai", + "metadata": { + "calculation": "$0.016/60 seconds = $0.00026667 per second", + "original_pricing_per_minute": 0.016 + }, + "mode": "audio_transcription", + "source": "https://cloud.google.com/speech-to-text/pricing", + "supported_endpoints": [ + "/v1/audio/transcriptions", + "/v1/realtime" + ] + }, "vertex_ai/claude-3-5-haiku": { "deprecation_date": "2026-07-05", "input_cost_per_token": 1e-06, @@ -50155,6 +50169,7 @@ ] }, "vertex_ai/google/gemma-4-26b-a4b-it-maas": { + "cache_read_input_token_cost": 1.5e-08, "input_cost_per_token": 1.5e-07, "litellm_provider": "vertex_ai-openai_models", "max_input_tokens": 262144, diff --git a/tests/test_litellm/llms/vertex_ai/audio_transcription/test_vertex_ai_realtime_transformation.py b/tests/test_litellm/llms/vertex_ai/audio_transcription/test_vertex_ai_realtime_transformation.py index 84c3a4e244a..b50d6e19458 100644 --- a/tests/test_litellm/llms/vertex_ai/audio_transcription/test_vertex_ai_realtime_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/audio_transcription/test_vertex_ai_realtime_transformation.py @@ -115,7 +115,7 @@ def _commands(config: VertexChirpRealtimeConfig, payload: str) -> list[object]: [ ("vertex_ai/chirp_3", True), ("chirp_3", True), - ("chirp_2", False), + ("chirp_2", True), ("gemini-live-2.5-flash", False), ("vertex_ai/gemini-2.0-flash-live-preview-04-09", False), ("vertex_ai/gemini-3.5-transcribe-live-preview", False), From ddc7ee6838726f2c329202bc4bef9abca58e1b35 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 04:54:10 -0700 Subject: [PATCH 57/96] feat(bedrock): add gpt-5.4 and gpt-5.5 us and global inference profile pricing (#42941) * feat(bedrock): add gpt-5.4 and gpt-5.5 us and global inference profile pricing Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(bedrock): drop supports_max_reasoning_effort from gpt-5.4 and gpt-5.5 rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 124 ++++++++++++++++++ model_prices_and_context_window.json | 124 ++++++++++++++++++ 2 files changed, 248 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 298ec16fc55..0f2782efffa 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -57151,6 +57151,130 @@ "/v1/responses" ] }, + "us.openai.gpt-5.4": { + "input_cost_per_token": 2.75e-06, + "input_cost_per_token_above_272k_tokens": 5.5e-06, + "cache_read_input_token_cost": 2.75e-07, + "cache_read_input_token_cost_above_272k_tokens": 5.5e-07, + "output_cost_per_token": 1.65e-05, + "output_cost_per_token_above_272k_tokens": 2.475e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-54.html", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, + "supports_vision": true, + "supports_sampling_params": false, + "supported_endpoints": [ + "/v1/responses" + ] + }, + "global.openai.gpt-5.4": { + "input_cost_per_token": 2.75e-06, + "input_cost_per_token_above_272k_tokens": 5.5e-06, + "cache_read_input_token_cost": 2.75e-07, + "cache_read_input_token_cost_above_272k_tokens": 5.5e-07, + "output_cost_per_token": 1.65e-05, + "output_cost_per_token_above_272k_tokens": 2.475e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-54.html", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, + "supports_vision": true, + "supports_sampling_params": false, + "supported_endpoints": [ + "/v1/responses" + ] + }, + "us.openai.gpt-5.5": { + "input_cost_per_token": 5.5e-06, + "input_cost_per_token_above_272k_tokens": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1.1e-06, + "output_cost_per_token": 3.3e-05, + "output_cost_per_token_above_272k_tokens": 4.95e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-55.html", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, + "supports_vision": true, + "supports_sampling_params": false, + "supported_endpoints": [ + "/v1/responses" + ] + }, + "global.openai.gpt-5.5": { + "input_cost_per_token": 5.5e-06, + "input_cost_per_token_above_272k_tokens": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1.1e-06, + "output_cost_per_token": 3.3e-05, + "output_cost_per_token_above_272k_tokens": 4.95e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-55.html", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, + "supports_vision": true, + "supports_sampling_params": false, + "supported_endpoints": [ + "/v1/responses" + ] + }, "global.openai.gpt-5.6-luna": { "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 298ec16fc55..0f2782efffa 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -57151,6 +57151,130 @@ "/v1/responses" ] }, + "us.openai.gpt-5.4": { + "input_cost_per_token": 2.75e-06, + "input_cost_per_token_above_272k_tokens": 5.5e-06, + "cache_read_input_token_cost": 2.75e-07, + "cache_read_input_token_cost_above_272k_tokens": 5.5e-07, + "output_cost_per_token": 1.65e-05, + "output_cost_per_token_above_272k_tokens": 2.475e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-54.html", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, + "supports_vision": true, + "supports_sampling_params": false, + "supported_endpoints": [ + "/v1/responses" + ] + }, + "global.openai.gpt-5.4": { + "input_cost_per_token": 2.75e-06, + "input_cost_per_token_above_272k_tokens": 5.5e-06, + "cache_read_input_token_cost": 2.75e-07, + "cache_read_input_token_cost_above_272k_tokens": 5.5e-07, + "output_cost_per_token": 1.65e-05, + "output_cost_per_token_above_272k_tokens": 2.475e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-54.html", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, + "supports_vision": true, + "supports_sampling_params": false, + "supported_endpoints": [ + "/v1/responses" + ] + }, + "us.openai.gpt-5.5": { + "input_cost_per_token": 5.5e-06, + "input_cost_per_token_above_272k_tokens": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1.1e-06, + "output_cost_per_token": 3.3e-05, + "output_cost_per_token_above_272k_tokens": 4.95e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-55.html", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, + "supports_vision": true, + "supports_sampling_params": false, + "supported_endpoints": [ + "/v1/responses" + ] + }, + "global.openai.gpt-5.5": { + "input_cost_per_token": 5.5e-06, + "input_cost_per_token_above_272k_tokens": 1.1e-05, + "cache_read_input_token_cost": 5.5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1.1e-06, + "output_cost_per_token": 3.3e-05, + "output_cost_per_token_above_272k_tokens": 4.95e-05, + "litellm_provider": "bedrock_converse", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-55.html", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_xhigh_reasoning_effort": true, + "supports_vision": true, + "supports_sampling_params": false, + "supported_endpoints": [ + "/v1/responses" + ] + }, "global.openai.gpt-5.6-luna": { "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, From c4b56b6adad3c2fc8f9af691b5c0e325c5702a9e Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 06:15:55 -0700 Subject: [PATCH 58/96] fix(cost-map): update azure gpt-4.1-nano and gpt-4o-2024-05-13 retirement dates (#42947) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../model_prices_and_context_window_backup.json | 16 ++++++++-------- model_prices_and_context_window.json | 16 ++++++++-------- 2 files changed, 16 insertions(+), 16 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 0f2782efffa..bba80677a80 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -5442,7 +5442,7 @@ "supports_web_search": false }, "azure/gpt-4.1-nano": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -5476,7 +5476,7 @@ "supports_vision": true }, "azure/gpt-4.1-nano-2025-04-14": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -5546,7 +5546,7 @@ "supports_vision": true }, "azure/gpt-4o-2024-05-13": { - "deprecation_date": "2026-10-01", + "deprecation_date": "2026-12-09", "input_cost_per_token": 5e-06, "input_cost_per_token_batches": 2.5e-06, "litellm_provider": "azure", @@ -10660,7 +10660,7 @@ "supports_web_search": false }, "azure/us/gpt-4.1-nano-2025-04-14": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -68776,7 +68776,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-4.1-nano": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -68787,7 +68787,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-4o-2024-05-13": { - "deprecation_date": "2026-10-01", + "deprecation_date": "2026-12-09", "input_cost_per_token": 5.5e-06, "input_cost_per_token_batches": 2.75e-06, "litellm_provider": "azure", @@ -69211,7 +69211,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/us/gpt-4.1-nano": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -69222,7 +69222,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/us/gpt-4o-2024-05-13": { - "deprecation_date": "2026-10-01", + "deprecation_date": "2026-12-09", "input_cost_per_token": 5.5e-06, "input_cost_per_token_batches": 2.75e-06, "litellm_provider": "azure", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 0f2782efffa..bba80677a80 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -5442,7 +5442,7 @@ "supports_web_search": false }, "azure/gpt-4.1-nano": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -5476,7 +5476,7 @@ "supports_vision": true }, "azure/gpt-4.1-nano-2025-04-14": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -5546,7 +5546,7 @@ "supports_vision": true }, "azure/gpt-4o-2024-05-13": { - "deprecation_date": "2026-10-01", + "deprecation_date": "2026-12-09", "input_cost_per_token": 5e-06, "input_cost_per_token_batches": 2.5e-06, "litellm_provider": "azure", @@ -10660,7 +10660,7 @@ "supports_web_search": false }, "azure/us/gpt-4.1-nano-2025-04-14": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -68776,7 +68776,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-4.1-nano": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -68787,7 +68787,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-4o-2024-05-13": { - "deprecation_date": "2026-10-01", + "deprecation_date": "2026-12-09", "input_cost_per_token": 5.5e-06, "input_cost_per_token_batches": 2.75e-06, "litellm_provider": "azure", @@ -69211,7 +69211,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/us/gpt-4.1-nano": { - "deprecation_date": "2027-04-14", + "deprecation_date": "2026-10-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -69222,7 +69222,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/us/gpt-4o-2024-05-13": { - "deprecation_date": "2026-10-01", + "deprecation_date": "2026-12-09", "input_cost_per_token": 5.5e-06, "input_cost_per_token_batches": 2.75e-06, "litellm_provider": "azure", From 8477fe4108b742acbf7e40aabb79e3147fb9d329 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 06:41:13 -0700 Subject: [PATCH 59/96] fix(proxy): list key and team model aliases in GET /v1/models (#42908) * fix(proxy): list key and team model aliases in GET /v1/models Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): keep alias listing helpers within the type discipline budget Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(proxy): cover alias rows on GET /v1/models and /v1/models/{id} Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): apply team then key aliases like chat completions and keep the alias as the retrieved id Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): apply key aliases twice like chat completions and skip only malformed alias entries Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): apply the global model_alias_map between the key alias passes like chat completions Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): list only the caller's own aliases and never rewrite a listed model id on retrieval Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * style(proxy): ruff format model_info alias lookup Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): hide undiscoverable names from model retrieval so an alias named like one resolves to its target Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): keep undiscoverable models retrievable by id while excluding them from the alias guard Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(proxy): pass an immutable name sequence into the model_info alias guard Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(proxy): type the model list alias test helpers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(proxy): annotate the new alias listing test fixtures and helpers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../proxy/common_utils/model_listing_utils.py | 80 +++++++- litellm/proxy/proxy_server.py | 39 +++- tests/e2e/coverage_registry/other.yaml | 1 + tests/e2e/models.py | 1 + tests/e2e/other/other_client.py | 16 +- tests/e2e/other/test_jwt_auth_e2e.py | 47 ++++- .../common_utils/test_model_listing_utils.py | 177 +++++++++++++++--- .../proxy/test_model_list_aliases.py | 151 +++++++++++++++ 8 files changed, 477 insertions(+), 35 deletions(-) create mode 100644 tests/test_litellm/proxy/test_model_list_aliases.py diff --git a/litellm/proxy/common_utils/model_listing_utils.py b/litellm/proxy/common_utils/model_listing_utils.py index 4b3ff2711b3..3c6555662e6 100644 --- a/litellm/proxy/common_utils/model_listing_utils.py +++ b/litellm/proxy/common_utils/model_listing_utils.py @@ -13,9 +13,12 @@ from __future__ import annotations import re from collections.abc import Container, Mapping, Sequence from dataclasses import dataclass +from functools import reduce from types import MappingProxyType from typing import TYPE_CHECKING, Final, cast +from pydantic import TypeAdapter, ValidationError + import litellm if TYPE_CHECKING: @@ -28,6 +31,8 @@ CLAUDE_CODE_CLIENT: Final = "claude-code" _CLAUDE_CODE_ALIAS_PREFIX: Final = "claude-router-" _ONE_MILLION_SUFFIX: Final = "[1m]" _ONE_MILLION_TOKENS: Final = 1_000_000 +_ALIAS_ENTRIES: Final = TypeAdapter(Mapping[object, object]) +_NO_ALIASES: Final[Mapping[str, str]] = MappingProxyType({}) def configured_display_names( @@ -152,6 +157,77 @@ class ClaudeCodeRoutingNames: ) +@dataclass(frozen=True, slots=True) +class CallerAliases: + """`own` are the caller's key and team alias maps, the names `/v1/models` lists for it. + `rewrite` are the maps `/chat/completions` rewrites its model through, in the order it + applies them: the team's, the key's in `add_litellm_data_to_request`, then the global + `model_alias_map` and the key's again in `common_processing_pre_call_logic`.""" + + own: tuple[object, ...] + rewrite: tuple[object, ...] + + +def caller_alias_maps( + key_aliases: object, + team_aliases: object, + key_team_id: str | None, + listed_team_id: str | None, +) -> CallerAliases: + """Team aliases count only when listing the team the key authenticated as.""" + if listed_team_id is not None and listed_team_id != key_team_id: + return CallerAliases((key_aliases,), (key_aliases, litellm.model_alias_map, key_aliases)) + return CallerAliases((team_aliases, key_aliases), (team_aliases, key_aliases, litellm.model_alias_map, key_aliases)) + + +def _alias_map(aliases: object) -> Mapping[str, str]: + try: + entries: Final = _ALIAS_ENTRIES.validate_python(aliases, strict=True) + except ValidationError: + return _NO_ALIASES + return MappingProxyType( + {alias: target for alias, target in entries.items() if isinstance(alias, str) and isinstance(target, str)} + ) + + +def _alias_names(alias_maps: Sequence[Mapping[str, str]]) -> tuple[str, ...]: + return tuple(dict.fromkeys(alias for aliases in alias_maps for alias in aliases)) + + +def _rewrite(model_id: str, alias_maps: Sequence[Mapping[str, str]]) -> str | None: + target: Final = reduce(lambda name, aliases: aliases.get(name, name), alias_maps, model_id) + return None if target == model_id else target + + +def alias_target(model_id: str, aliases: CallerAliases, listed: Container[str] = frozenset()) -> str | None: + """The model group `/chat/completions` rewrites `model_id` to, else None. A `model_id` + already `listed` keeps its own row, so it is never rewritten.""" + if model_id in listed: + return None + return _rewrite(model_id, tuple(_alias_map(alias_map) for alias_map in aliases.rewrite)) + + +def alias_listing_entries( + entries: Sequence[tuple[str, str]], + aliases: CallerAliases, +) -> tuple[tuple[str, str], ...]: + """`entries` plus one `(alias, lookup_id)` row per key or team alias whose target is + listed. An alias colliding with a listed id keeps the listed entry.""" + maps: Final = tuple(_alias_map(alias_map) for alias_map in aliases.rewrite) + own: Final = tuple(_alias_map(alias_map) for alias_map in aliases.own) + lookup_by_response: Final = MappingProxyType(dict(entries)) + lookup_ids: Final = frozenset(lookup_by_response.values()) + targets: Final = MappingProxyType( + {alias: _rewrite(alias, maps) for alias in _alias_names(own) if alias not in lookup_by_response} + ) + added: Final = tuple( + (alias, lookup_by_response.get(target, target)) + for alias, target in targets.items() + if target is not None and (target in lookup_by_response or target in lookup_ids) + ) + return (*entries, *added) + + def claude_code_requested_group( requested: str, llm_router: Router, @@ -218,7 +294,7 @@ class TeamModelNameTranslator: @staticmethod def _response_to_lookup_map( - model_names: list[str], + model_names: Sequence[str], internal_to_public: dict[str, str], ) -> dict[str, str]: """Map each public response id to the first internal lookup id seen in @@ -235,7 +311,7 @@ class TeamModelNameTranslator: @staticmethod def listing_entries( - model_names: list[str], + model_names: Sequence[str], llm_router: Router | None, general_settings: Mapping[str, object], ) -> list[tuple[str, str]]: diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 7e18db742cd..9fb72d2d1b5 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -422,6 +422,9 @@ from litellm.proxy.common_utils.model_deprecation import collect_model_deprecati from litellm.proxy.common_utils.model_listing_utils import ( ClaudeCodeRoutingNames, TeamModelNameTranslator, + alias_listing_entries, + alias_target, + caller_alias_maps, claude_code_view_ids, configured_display_names, is_claude_code_client, @@ -11172,14 +11175,13 @@ async def model_list( view_aliases: Final = ( view_router_settings.get("model_group_alias") if isinstance(view_router_settings, Mapping) else None ) + caller_aliases: Final = caller_alias_maps( + user_api_key_dict.aliases, user_api_key_dict.team_model_aliases, user_api_key_dict.team_id, team_id + ) routing_names: Final = ClaudeCodeRoutingNames( llm_router, team_id or user_api_key_dict.team_id, - ( - user_api_key_dict.aliases, - user_api_key_dict.team_model_aliases, - view_aliases, - ), + (*caller_aliases.rewrite, view_aliases), ) # Validate scope parameter if provided @@ -11307,7 +11309,9 @@ async def model_list( # The internal routing key drives the metadata/fallback lookup, while the # public name is what the client sees as the model id. model_data = [] - entries: Final = TeamModelNameTranslator.listing_entries(all_models, llm_router, settings) + entries: Final = alias_listing_entries( + TeamModelNameTranslator.listing_entries(all_models, llm_router, settings), caller_aliases + ) for response_id, lookup_id in entries: model_info = create_model_info_response( model_id=lookup_id, @@ -11391,7 +11395,8 @@ async def model_info( ) # Mirror /v1/models' visibility filter so first-occurrence resolution - # cannot land on a deployment the listing had hidden. + # cannot land on a deployment the listing had hidden. Undiscoverable + # models stay retrievable by id, they only drop out of the alias guard. blocked_names: Final = llm_router.get_fully_blocked_model_names() if llm_router is not None else set() unhealthy_names: Final = await get_hidden_unhealthy_model_names( healthy_only=healthy_only, @@ -11401,10 +11406,25 @@ async def model_info( hidden_names: Final = blocked_names | unhealthy_names if hidden_names: all_models = [m for m in all_models if m not in hidden_names] + undiscoverable_names: Final = undiscoverable_model_names( + all_models, llm_router, user_api_key_dict, team_id or user_api_key_dict.team_id + ) internal_to_public: Final = TeamModelNameTranslator.build_internal_to_public_map(llm_router, settings) + aliased_model_id: Final = alias_target( + model_id, + caller_alias_maps( + user_api_key_dict.aliases, user_api_key_dict.team_model_aliases, user_api_key_dict.team_id, team_id + ), + frozenset( + response_id + for response_id, _ in TeamModelNameTranslator.listing_entries( + tuple(m for m in all_models if m not in undiscoverable_names), llm_router, settings + ) + ), + ) resolved_model_id: Final = TeamModelNameTranslator.resolve_public_name( - model_id=model_id, + model_id=aliased_model_id or model_id, available_models=all_models, llm_router=llm_router, general_settings=settings, @@ -11434,7 +11454,8 @@ async def model_info( fallback_type=None, llm_router=llm_router, ) - return {**response, "id": internal_to_public.get(resolved_model_id, model_id)} # mutable-ok: response id differs + response_id: Final = model_id if aliased_model_id else internal_to_public.get(resolved_model_id, model_id) + return {**response, "id": response_id} # mutable-ok: response id differs def _blocked_response_usage(original_response: object | None) -> "litellm.Usage": diff --git a/tests/e2e/coverage_registry/other.yaml b/tests/e2e/coverage_registry/other.yaml index f695af3cb11..3bd98ff5b0b 100644 --- a/tests/e2e/coverage_registry/other.yaml +++ b/tests/e2e/coverage_registry/other.yaml @@ -17,6 +17,7 @@ - {id: other.auth.jwt.virtual_key_unaffected, module: other, tier: P0, area: auth, assertions: [virtual_key_unaffected], source: "handle_jwt.py:213 is_jwt / user_api_key_auth.py:1332-1333", rationale: "enable_jwt_auth only routes three-segment bearer tokens into the JWT branch, so sk- virtual keys keep working on the same proxy"} - {id: other.auth.jwt.team_header_alias_binds_team, module: other, tier: P0, area: auth, assertions: [team_header_alias_binds_team], source: "handle_jwt.py JWTAuthManager.resolve_team_from_header / LIT-7181", fail_before_fix: proven, rationale: "x-litellm-team-id carrying the team alias binds and attributes the same team as the team id, so a managed client can pin a stable alias instead of a uuid"} - {id: other.auth.jwt.team_header_non_member_alias_denied, module: other, tier: P0, area: auth, assertions: [team_header_non_member_alias_denied], source: "handle_jwt.py JWTAuthManager.resolve_team_from_header / LIT-7181", rationale: "x-litellm-team-id naming the alias of a team the JWT does not grant is denied 403 with the same body as an unknown value, so the response does not reveal whether that team exists"} +- {id: other.auth.jwt.team_model_alias_listed_and_routes, module: other, tier: P1, area: auth, assertions: [team_model_alias_listed_and_routes], source: "proxy_server.py model_list / common_utils/model_listing_utils.py alias_listing_entries / LIT-8515", fail_before_fix: proven, rationale: "A team model_aliases name the JWT caller can complete on is also listed by GET /v1/models for that caller, in the OpenAI and the Anthropic (Claude Code) shapes, next to its target, so a managed client can discover the alias it is meant to send"} - {id: other.auth.model_access_group.wildcard_bare_name_allowed, module: other, tier: P0, area: auth, assertions: [wildcard_bare_name_allowed], source: "auth_checks.py:3232 / LIT-5813", fail_before_fix: proven, rationale: "A grant of a group holding a wildcard deployment covers the bare model names callers actually send, not only the provider-prefixed spelling"} - {id: other.auth.model_access_group.member_allowed, module: other, tier: P0, area: auth, assertions: [member_allowed], source: "auth_checks.py:3232", rationale: "A key whose allow-list is a model access group can call the deployments in that group"} - {id: other.auth.model_access_group.non_member_denied, module: other, tier: P0, area: auth, assertions: [non_member_denied], source: "auth_checks.py:3232", rationale: "That same grant reaches nothing outside the group, including provider models the group's wildcard does not cover"} diff --git a/tests/e2e/models.py b/tests/e2e/models.py index bd7f5171172..c96f4b0bef1 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -1426,6 +1426,7 @@ class TeamNewBody(BaseModel): team_id: str | None = None organization_id: str | None = None metadata: TeamMetadata | None = None + model_aliases: dict[str, str] | None = None class TeamNewResponse(BaseModel): diff --git a/tests/e2e/other/other_client.py b/tests/e2e/other/other_client.py index 057a735e5e5..93c198586f6 100644 --- a/tests/e2e/other/other_client.py +++ b/tests/e2e/other/other_client.py @@ -14,12 +14,15 @@ own endpoints, so no test ever holds a signing key. from __future__ import annotations from dataclasses import dataclass +from typing import Final -from e2e_http import AuthHeaders, NoBody, ProbeResult, Result +from e2e_http import AnthropicHeaders, AuthHeaders, NoBody, ProbeResult, Result from idp import Keycloak, keycloak_from_env from models import ( ChatBody, ChatResponse, + ModelsListParams, + ModelsListResponse, ReadinessDetailsResponse, ReadinessResponse, UserListParams, @@ -88,6 +91,17 @@ class OtherClient: response_type=ChatResponse, ) + def list_models_as(self, token: str, *, anthropic: bool = False) -> Result[ModelsListResponse]: + """GET /v1/models under `token`, in the OpenAI shape or, with `anthropic`, the + Anthropic Models API shape Claude Code reads. Both carry `data[].id`.""" + bearer: Final = self.proxy.transport.bearer(token) + return self.proxy.transport.get( + "/v1/models", + headers=AnthropicHeaders(authorization=bearer.authorization) if anthropic else bearer, + params=ModelsListParams(return_wildcard_routes=False), + response_type=ModelsListResponse, + ) + def list_users_as(self, key: str) -> Result[UserListResponse]: """GET /user/list under `key`. Admin-only, so it doubles as the master key's authorization proof: the master key (proxy admin) reads it, a diff --git a/tests/e2e/other/test_jwt_auth_e2e.py b/tests/e2e/other/test_jwt_auth_e2e.py index de06c376f76..a8aa474f04f 100644 --- a/tests/e2e/other/test_jwt_auth_e2e.py +++ b/tests/e2e/other/test_jwt_auth_e2e.py @@ -78,9 +78,35 @@ def bound_team(client: OtherClient, resources: ResourceManager) -> BoundTeam: return BoundTeam(identity=provisioned, team_id=provisioned.group, team_alias=team_alias) -def _ping() -> ChatBody: +@dataclass(frozen=True, slots=True) +class AliasedTeam: + identity: Identity + alias: str + target: str + + +@pytest.fixture +def aliased_team(client: OtherClient, resources: ResourceManager) -> AliasedTeam: + """An identity whose team carries a model_aliases entry, the name a managed + client such as Claude Code sends and the team rewrites to a real model group.""" + marker: Final = unique_marker() + provisioned: Final = _provision(client, resources, marker=marker) + alias: Final = f"e2e-jwt-model-alias-{marker}" + team_id: Final = client.proxy.create_team( + TeamNewBody( + team_alias=f"e2e-jwt-aliased-{marker}", + team_id=provisioned.group, + models=[CHEAP_OPENAI_MODEL], + model_aliases={alias: CHEAP_OPENAI_MODEL}, + ) + ) + resources.defer(lambda: client.proxy.delete_team(team_id)) + return AliasedTeam(identity=provisioned, alias=alias, target=CHEAP_OPENAI_MODEL) + + +def _ping(model: str = CHEAP_OPENAI_MODEL) -> ChatBody: return ChatBody( - model=CHEAP_OPENAI_MODEL, + model=model, messages=[ChatMessage(role="user", content=f"Reply with the single word pong. {unique_marker()}")], max_tokens=16, ) @@ -221,6 +247,23 @@ class TestJwtTeamHeader: f"{bound_team.team_id!r}, got {by_alias!r}" ) + @pytest.mark.covers("other.auth.jwt.team_model_alias_listed_and_routes") + @pytest.mark.parametrize("anthropic", [False, True], ids=["openai_shape", "anthropic_shape"]) + def test_team_model_alias_is_listed_by_v1_models_under_the_same_token_that_routes_it( + self, client: OtherClient, aliased_team: AliasedTeam, anthropic: bool + ) -> None: + token: Final = client.idp.access_token(aliased_team.identity) + + routed: Final = unwrap(client.proxy.chat(token, _ping(model=aliased_team.alias))) + assert routed.choices, f"precondition: /chat/completions must route the team alias, got {routed}" + + listed: Final = tuple(entry.id for entry in unwrap(client.list_models_as(token, anthropic=anthropic)).data) + assert aliased_team.alias in listed, ( + f"/v1/models must list team alias {aliased_team.alias!r} that the same token routes on " + f"/chat/completions, got {listed}" + ) + assert aliased_team.target in listed, f"the alias target {aliased_team.target!r} must stay listed, got {listed}" + @pytest.mark.covers("other.auth.jwt.team_header_non_member_alias_denied") def test_team_header_with_the_alias_of_a_team_the_caller_is_not_in_is_rejected_like_an_unknown_value( self, client: OtherClient, resources: ResourceManager, bound_team: BoundTeam diff --git a/tests/test_litellm/proxy/common_utils/test_model_listing_utils.py b/tests/test_litellm/proxy/common_utils/test_model_listing_utils.py index 7ef03140093..97e38d5e17a 100644 --- a/tests/test_litellm/proxy/common_utils/test_model_listing_utils.py +++ b/tests/test_litellm/proxy/common_utils/test_model_listing_utils.py @@ -4,9 +4,14 @@ from itertools import combinations import pytest +import litellm from litellm import Router from litellm.proxy.common_utils.model_listing_utils import ( + CallerAliases, ClaudeCodeRoutingNames, + alias_listing_entries, + alias_target, + caller_alias_maps, claude_code_group_name, claude_code_model_id, claude_code_requested_group, @@ -22,6 +27,10 @@ def _marked(name): return f"{_encoded(name)}[1m]" +def _caller(*maps: object) -> CallerAliases: + return CallerAliases(maps, maps) + + def _row(name, limit=1000000): return {"id": name, "object": "model", "created": 0, "owned_by": "openai", "max_input_tokens": limit} @@ -29,15 +38,16 @@ def _row(name, limit=1000000): def _router(*names, aliases=None): return Router( model_list=[ - {"model_name": name, "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}} - for name in names + {"model_name": name, "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}} for name in names ], model_group_alias=aliases, ) @pytest.mark.parametrize("limit", [None, 999999, 1000000]) -@pytest.mark.parametrize("name", ["foo", "foo[1m]", "foo[1M]", "a/b: 世界", "claude-router-foo", "claude-opus-5", "claude-opus-5[1m]"]) +@pytest.mark.parametrize( + "name", ["foo", "foo[1m]", "foo[1M]", "a/b: 世界", "claude-router-foo", "claude-opus-5", "claude-opus-5[1m]"] +) def test_listing_round_trips_entire_source_name(name, limit): names = frozenset({name}) view = claude_code_model_id(name, limit, names) @@ -48,7 +58,15 @@ def test_listing_round_trips_entire_source_name(name, limit): def test_collision_matrix_round_trips_without_duplicate_ids(): - universe = ("foo", "foo[1m]", "claude-router-foo", _encoded("foo"), _encoded("foo") + "[1m]", "claude-opus-5", "claude-opus-5[1m]") + universe = ( + "foo", + "foo[1m]", + "claude-router-foo", + _encoded("foo"), + _encoded("foo") + "[1m]", + "claude-opus-5", + "claude-opus-5[1m]", + ) for pair in combinations(universe, 2): for visible in (pair, pair[:1], pair[1:]): names = frozenset(pair) @@ -57,18 +75,31 @@ def test_collision_matrix_round_trips_without_duplicate_ids(): assert all((claude_code_group_name(shown, names) or shown) == source for source, shown in view.items()) -@pytest.mark.parametrize("spelling", ["claude-router-foo", "claude-router-ff", "claude-router-66 6f6f", "claude-router-666F6F", "claude-router-", _encoded("missing")]) +@pytest.mark.parametrize( + "spelling", + [ + "claude-router-foo", + "claude-router-ff", + "claude-router-66 6f6f", + "claude-router-666F6F", + "claude-router-", + _encoded("missing"), + ], +) def test_unknown_or_noncanonical_ids_are_never_guessed(spelling): assert claude_code_group_name(spelling, frozenset({"foo"})) is None -@pytest.mark.parametrize("headers,enabled", [ - ({"user-agent": "claude-code/2.1.267"}, True), - ({"user-agent": "claude-cli/2.1.267 (external, sdk-cli)"}, True), - ({"x-gateway-client": "Claude-Code"}, True), - ({"user-agent": "anthropic-sdk-python/0.40"}, False), - ({}, False), -]) +@pytest.mark.parametrize( + "headers,enabled", + [ + ({"user-agent": "claude-code/2.1.267"}, True), + ({"user-agent": "claude-cli/2.1.267 (external, sdk-cli)"}, True), + ({"x-gateway-client": "Claude-Code"}, True), + ({"user-agent": "anthropic-sdk-python/0.40"}, False), + ({}, False), + ], +) def test_only_claude_code_gets_the_view(headers, enabled): rows = (_row("foo"), _row("claude-opus-5")) view = claude_code_view_ids(rows, headers, frozenset(row["id"] for row in rows)) @@ -77,12 +108,15 @@ def test_only_claude_code_gets_the_view(headers, enabled): @pytest.mark.parametrize("layer", ["literal", "global", "router", "key", "team", "wildcard"]) def test_configured_names_outrank_generated_ids_even_when_hidden_from_listing(monkeypatch, layer): - import litellm - encoded = _encoded("foo") alias = {encoded: "other"} monkeypatch.setattr(litellm, "model_alias_map", alias if layer == "global" else {}) - router = _router("foo", "other", *( (encoded,) if layer == "literal" else ("*",) if layer == "wildcard" else ()), aliases=alias if layer == "router" else None) + router = _router( + "foo", + "other", + *((encoded,) if layer == "literal" else ("*",) if layer == "wildcard" else ()), + aliases=alias if layer == "router" else None, + ) maps = (alias,) if layer in ("key", "team") else () names = ClaudeCodeRoutingNames(router, None, maps) assert claude_code_requested_group(encoded, router, None, maps) is None @@ -98,12 +132,113 @@ def test_mutation_breaking_the_hex_name_cannot_route_to_the_source(source): assert claude_code_requested_group(_marked(source), router, None) == source +def test_team_alias_is_listed_under_its_target_metadata_and_only_when_the_target_is_accessible() -> None: + entries = [("gpt-4.1-mini", "gpt-4.1-mini"), ("team-public", "model_name_team_1_abc")] + aliases = ( + {"gpt-4.1-mini": "team-public"}, + None, + {"claude-sonnet-4-5": "gpt-4.1-mini", "via-public": "team-public", "not-granted": "gpt-4.1"}, + ) + assert alias_listing_entries(entries, _caller(*aliases)) == ( + *entries, + ("claude-sonnet-4-5", "gpt-4.1-mini"), + ("via-public", "model_name_team_1_abc"), + ) + assert alias_listing_entries(entries, _caller(None, {})) == tuple(entries) + + +def test_alias_target_resolves_the_requested_alias_across_key_and_team_maps() -> None: + maps = _caller({"o": "gpt-4.1"}, {"claude-sonnet-4-5": "gpt-4.1-mini"}) + assert alias_target("claude-sonnet-4-5", maps) == "gpt-4.1-mini" + assert alias_target("gpt-4.1-mini", _caller(None, {"claude-sonnet-4-5": "gpt-4.1-mini"})) is None + + +def test_alias_colliding_with_a_listed_id_keeps_the_listed_model_at_list_and_retrieval() -> None: + entries = [("fast", "fast"), ("gpt-4.1-mini", "gpt-4.1-mini")] + maps = _caller({"fast": "gpt-4.1-mini"}) + listed = frozenset(response_id for response_id, _ in entries) + + assert alias_listing_entries(entries, maps) == tuple(entries) + assert alias_target("fast", maps, listed) is None + assert alias_target("fast", maps) == "gpt-4.1-mini" + + +def test_alias_maps_apply_in_the_order_chat_completions_applies_them() -> None: + team_then_key = _caller({"fast": "gpt-4.1-mini", "hop": "mid"}, {"fast": "gpt-4.1", "mid": "gpt-4.1"}) + entries = [("gpt-4.1-mini", "gpt-4.1-mini"), ("gpt-4.1", "gpt-4.1")] + + assert alias_target("fast", team_then_key) == "gpt-4.1-mini" + assert alias_target("hop", team_then_key) == "gpt-4.1" + assert alias_listing_entries(entries, team_then_key) == ( + *entries, + ("fast", "gpt-4.1-mini"), + ("hop", "gpt-4.1"), + ("mid", "gpt-4.1"), + ) + + +def test_one_bad_alias_entry_hides_only_itself() -> None: + aliases = {"fast": "gpt-4.1-mini", "broken": 5, 7: "gpt-4.1-mini"} + entries = [("gpt-4.1-mini", "gpt-4.1-mini")] + + assert alias_listing_entries(entries, _caller(aliases)) == (*entries, ("fast", "gpt-4.1-mini")) + assert alias_target("fast", _caller(aliases)) == "gpt-4.1-mini" + + +def test_chained_key_alias_is_listed_only_when_its_final_target_is_listable() -> None: + key_aliases = {"a": "b", "b": "hidden"} + entries = [("b", "b")] + + assert alias_listing_entries(entries, caller_alias_maps(key_aliases, None, "team-a", None)) == (*entries,) + assert alias_target("a", caller_alias_maps(key_aliases, None, "team-a", None)) == "hidden" + + +def test_team_aliases_only_apply_when_listing_the_team_the_key_authenticated_as( + monkeypatch: pytest.MonkeyPatch, +) -> None: + key_aliases, team_aliases, global_aliases = {"k": "gpt-4.1"}, {"t": "gpt-4.1-mini"}, {"g": "gpt-4.1"} + monkeypatch.setattr(litellm, "model_alias_map", global_aliases) + own_team = CallerAliases((team_aliases, key_aliases), (team_aliases, key_aliases, global_aliases, key_aliases)) + assert caller_alias_maps(key_aliases, team_aliases, "team-a", None) == own_team + assert caller_alias_maps(key_aliases, team_aliases, "team-a", "team-a") == own_team + assert caller_alias_maps(key_aliases, team_aliases, "team-a", "team-b") == CallerAliases( + (key_aliases,), (key_aliases, global_aliases, key_aliases) + ) + + +def test_global_alias_rewrites_between_the_two_key_passes_like_chat_completions( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(litellm, "model_alias_map", {"b": "d"}) + key_aliases = {"a": "b", "b": "c"} + entries = [("c", "c"), ("d", "d")] + maps = caller_alias_maps(key_aliases, None, "team-a", None) + + assert alias_target("a", maps) == "d" + assert alias_listing_entries(entries, maps) == (*entries, ("a", "d"), ("b", "c")) + + +def test_global_aliases_rewrite_but_are_not_listed_as_caller_rows(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(litellm, "model_alias_map", {"g": "gpt-4.1-mini"}) + entries = [("gpt-4.1-mini", "gpt-4.1-mini")] + maps = caller_alias_maps({"k": "g"}, None, "team-a", None) + + assert alias_listing_entries(entries, maps) == (*entries, ("k", "gpt-4.1-mini")) + assert alias_target("g", maps) == "gpt-4.1-mini" + + def test_team_public_name_uses_the_same_scope_at_list_and_request(): - router = Router(model_list=[{ - "model_name": "model_name_team-a_id", - "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}, - "model_info": {"team_id": "team-a", "team_public_model_name": "shared"}, - }]) - shown = claude_code_view_ids((_row("shared"),), {"user-agent": "claude-code/2.1.267"}, ClaudeCodeRoutingNames(router, "team-a"))["shared"] + router = Router( + model_list=[ + { + "model_name": "model_name_team-a_id", + "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}, + "model_info": {"team_id": "team-a", "team_public_model_name": "shared"}, + } + ] + ) + shown = claude_code_view_ids( + (_row("shared"),), {"user-agent": "claude-code/2.1.267"}, ClaudeCodeRoutingNames(router, "team-a") + )["shared"] assert claude_code_requested_group(shown, router, "team-a") == "shared" assert claude_code_requested_group(shown, router, "team-b") is None diff --git a/tests/test_litellm/proxy/test_model_list_aliases.py b/tests/test_litellm/proxy/test_model_list_aliases.py new file mode 100644 index 00000000000..25a941bf2ae --- /dev/null +++ b/tests/test_litellm/proxy/test_model_list_aliases.py @@ -0,0 +1,151 @@ +""" +Tests for key and team `model_aliases` on the model listing endpoints: GET /v1/models +(`model_list`, OpenAI and Anthropic shapes) and GET /v1/models/{id} (`model_info`). +An alias the caller can complete on is listed next to its target and resolves by name. +""" + +import pytest +from starlette.requests import Request + +from litellm import Router +from litellm.proxy import proxy_server +from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth + + +def _deployment(model_name: str, model: str = "openai/gpt-4.1-mini", **model_info: str | bool) -> dict[str, object]: + return { + "model_name": model_name, + "litellm_params": {"model": model, "api_key": "sk-fake"}, + "model_info": {"id": f"{model_name}-id", **model_info}, + } + + +@pytest.fixture +def router(monkeypatch: pytest.MonkeyPatch) -> Router: + router = Router( + model_list=[ + _deployment("gpt-4.1-mini"), + _deployment("gpt-4.1", model="openai/gpt-4.1"), + _deployment("model_name_team1_abc", team_id="team1", team_public_model_name="team-chat"), + _deployment("hidden", model="anthropic/claude-sonnet-4-5", discoverable=False), + ] + ) + monkeypatch.setattr(proxy_server, "llm_router", router) + monkeypatch.setattr(proxy_server, "llm_model_list", router.model_list) + monkeypatch.setattr(proxy_server, "prisma_client", None) + monkeypatch.setattr(proxy_server, "general_settings", {}) + monkeypatch.setattr(proxy_server, "user_model", None) + return router + + +def _team_member( + team_id: str = "team1", models: list[str] | None = None, **aliases: dict[str, str] | None +) -> UserAPIKeyAuth: + return UserAPIKeyAuth( + api_key="sk-test", + user_id="u", + user_role=LitellmUserRoles.INTERNAL_USER, + team_id=team_id, + team_models=["gpt-4.1-mini", "model_name_team1_abc"], + models=models or ["gpt-4.1-mini", "model_name_team1_abc"], + **aliases, + ) + + +def _anthropic_request(*extra_headers: tuple[bytes, bytes]) -> Request: + return Request( + scope={ + "type": "http", + "method": "GET", + "path": "/v1/models", + "query_string": b"", + "headers": [(b"anthropic-version", b"2023-06-01"), *extra_headers], + } + ) + + +def _claude_code_request() -> Request: + return _anthropic_request((b"user-agent", b"claude-cli/2.1.267 (external, cli)")) + + +async def _v1_models(user_api_key_dict: UserAPIKeyAuth, request: Request | None = None) -> list[str]: + response = await proxy_server.model_list(user_api_key_dict=user_api_key_dict, request=request) + return [m["id"] for m in response["data"]] + + +@pytest.mark.asyncio +async def test_v1_models_lists_team_alias_next_to_its_target_in_both_shapes(router: Router) -> None: + caller = _team_member(team_model_aliases={"claude-sonnet-4-5": "gpt-4.1-mini"}) + + assert await _v1_models(caller) == ["gpt-4.1-mini", "team-chat", "claude-sonnet-4-5"] + assert await _v1_models(caller, request=_anthropic_request()) == ["gpt-4.1-mini", "team-chat", "claude-sonnet-4-5"] + + +@pytest.mark.asyncio +async def test_claude_code_picker_lists_the_alias_under_its_own_name(router: Router) -> None: + caller = _team_member(team_model_aliases={"claude-sonnet-4-5": "gpt-4.1-mini"}) + + picker_ids = await _v1_models(caller, request=_claude_code_request()) + assert any(picker_id.startswith("claude-sonnet-4-5") for picker_id in picker_ids), picker_ids + + +@pytest.mark.asyncio +async def test_v1_models_lists_key_alias_and_hides_alias_to_a_model_the_caller_cannot_list(router: Router) -> None: + caller = _team_member(aliases={"mini": "gpt-4.1-mini", "big": "gpt-4.1"}) + + assert await _v1_models(caller) == ["gpt-4.1-mini", "team-chat", "mini"] + + +@pytest.mark.asyncio +async def test_v1_models_resolves_a_team_alias_through_the_key_alias_like_chat_completions_does(router: Router) -> None: + caller = _team_member(team_model_aliases={"fast": "mid"}, aliases={"fast": "gpt-4.1", "mid": "gpt-4.1-mini"}) + + assert await _v1_models(caller) == ["gpt-4.1-mini", "team-chat", "fast", "mid"] + response = await proxy_server.model_info(model_id="fast", user_api_key_dict=caller) + assert response["id"] == "fast" + + +@pytest.mark.asyncio +async def test_v1_models_skips_only_the_malformed_alias_entries(router: Router) -> None: + caller = _team_member(team_model_aliases={"claude-sonnet-4-5": 5, "fast": "gpt-4.1-mini"}) + + assert await _v1_models(caller) == ["gpt-4.1-mini", "team-chat", "fast"] + + +@pytest.mark.asyncio +async def test_v1_models_by_id_resolves_a_team_alias_to_its_target_metadata(router: Router) -> None: + caller = _team_member(team_model_aliases={"claude-sonnet-4-5": "gpt-4.1-mini"}) + + response = await proxy_server.model_info(model_id="claude-sonnet-4-5", user_api_key_dict=caller) + assert response["id"] == "claude-sonnet-4-5" + assert response["owned_by"] == "openai" + + +@pytest.mark.asyncio +async def test_v1_models_by_id_retrieves_the_listed_model_when_an_alias_collides_with_its_id(router: Router) -> None: + caller = _team_member(aliases={"team-chat": "gpt-4.1"}) + + assert await _v1_models(caller) == ["gpt-4.1-mini", "team-chat"] + response = await proxy_server.model_info(model_id="team-chat", user_api_key_dict=caller) + assert response["id"] == "team-chat" + + +@pytest.mark.asyncio +async def test_v1_models_by_id_resolves_an_alias_named_like_an_undiscoverable_model_to_the_alias_target( + router: Router, +) -> None: + caller = _team_member(aliases={"hidden": "gpt-4.1-mini"}, models=["gpt-4.1-mini", "model_name_team1_abc", "hidden"]) + + assert await _v1_models(caller) == ["gpt-4.1-mini", "team-chat", "hidden"] + target = await proxy_server.model_info(model_id="gpt-4.1-mini", user_api_key_dict=caller) + response = await proxy_server.model_info(model_id="hidden", user_api_key_dict=caller) + assert response == {**target, "id": "hidden"} + + +@pytest.mark.asyncio +async def test_v1_models_by_id_keeps_the_alias_as_id_when_it_targets_a_team_scoped_model(router: Router) -> None: + caller = _team_member(team_model_aliases={"chat": "team-chat"}) + + assert "chat" in await _v1_models(caller) + response = await proxy_server.model_info(model_id="chat", user_api_key_dict=caller) + assert response["id"] == "chat" From 785d4974ba177f90f4368530751b50193c85cdfd Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 06:49:17 -0700 Subject: [PATCH 60/96] fix(cost-map): sync openrouter prices for deepseek v4 and glm-5.3 (#42952) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 22 +++++++++---------- model_prices_and_context_window.json | 22 +++++++++---------- 2 files changed, 22 insertions(+), 22 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index bba80677a80..a1878c00678 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41236,21 +41236,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.396e-07, + "input_cost_per_token": 9.31074e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.8792e-06, + "output_cost_per_token": 1.862148e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.83e-08, + "cache_read_input_token_cost": 7.75895e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, @@ -65596,13 +65596,13 @@ "supports_web_search": false }, "openrouter/z-ai/glm-5.3": { - "input_cost_per_token": 8.4e-07, - "output_cost_per_token": 2.64e-06, - "cache_read_input_token_cost": 1.56e-07, + "input_cost_per_token": 1.4e-06, + "output_cost_per_token": 4.4e-06, + "cache_read_input_token_cost": 2.6e-07, "litellm_provider": "openrouter", "max_input_tokens": 1310720, - "max_output_tokens": 131072, - "max_tokens": 131072, + "max_output_tokens": 943717, + "max_tokens": 943717, "mode": "chat", "source": "https://openrouter.ai/api/v1/models", "supports_audio_input": false, @@ -66307,9 +66307,9 @@ "supports_web_search": true }, "openrouter/deepseek/deepseek-v4-flash": { - "input_cost_per_token": 8.8606e-08, - "output_cost_per_token": 1.77212e-07, - "cache_read_input_token_cost": 1.77212e-08, + "input_cost_per_token": 8.554e-08, + "output_cost_per_token": 1.7108e-07, + "cache_read_input_token_cost": 1.7108e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index bba80677a80..a1878c00678 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41236,21 +41236,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.396e-07, + "input_cost_per_token": 9.31074e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.8792e-06, + "output_cost_per_token": 1.862148e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.83e-08, + "cache_read_input_token_cost": 7.75895e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, @@ -65596,13 +65596,13 @@ "supports_web_search": false }, "openrouter/z-ai/glm-5.3": { - "input_cost_per_token": 8.4e-07, - "output_cost_per_token": 2.64e-06, - "cache_read_input_token_cost": 1.56e-07, + "input_cost_per_token": 1.4e-06, + "output_cost_per_token": 4.4e-06, + "cache_read_input_token_cost": 2.6e-07, "litellm_provider": "openrouter", "max_input_tokens": 1310720, - "max_output_tokens": 131072, - "max_tokens": 131072, + "max_output_tokens": 943717, + "max_tokens": 943717, "mode": "chat", "source": "https://openrouter.ai/api/v1/models", "supports_audio_input": false, @@ -66307,9 +66307,9 @@ "supports_web_search": true }, "openrouter/deepseek/deepseek-v4-flash": { - "input_cost_per_token": 8.8606e-08, - "output_cost_per_token": 1.77212e-07, - "cache_read_input_token_cost": 1.77212e-08, + "input_cost_per_token": 8.554e-08, + "output_cost_per_token": 1.7108e-07, + "cache_read_input_token_cost": 1.7108e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, From 3acdfda19c37b88d2011982d3fb6d5850ce43d49 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 06:52:06 -0700 Subject: [PATCH 61/96] fix(cost-map): add batch prices for vertex gemini-3.8-flash-cyber (#42953) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 6 ++++++ model_prices_and_context_window.json | 6 ++++++ 2 files changed, 12 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a1878c00678..b5282f12539 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -27350,9 +27350,11 @@ }, "vertex_ai/gemini-3.8-flash-cyber": { "cache_read_input_token_cost": 1.5e-07, + "cache_read_input_token_cost_batches": 7.5e-08, "cache_read_input_token_cost_flex": 7.5e-08, "cache_read_input_token_cost_priority": 2.7e-07, "input_cost_per_token": 1.5e-06, + "input_cost_per_token_batches": 7.5e-07, "input_cost_per_token_flex": 7.5e-07, "input_cost_per_token_priority": 2.7e-06, "litellm_provider": "vertex_ai", @@ -27362,6 +27364,7 @@ "mode": "chat", "output_cost_per_reasoning_token": 7.5e-06, "output_cost_per_token": 7.5e-06, + "output_cost_per_token_batches": 3.75e-06, "output_cost_per_token_flex": 3.75e-06, "output_cost_per_token_priority": 1.35e-05, "regional_endpoint_uplift_multiplier": 1.1, @@ -29604,9 +29607,11 @@ }, "gemini-3.8-flash-cyber": { "cache_read_input_token_cost": 1.5e-07, + "cache_read_input_token_cost_batches": 7.5e-08, "cache_read_input_token_cost_flex": 7.5e-08, "cache_read_input_token_cost_priority": 2.7e-07, "input_cost_per_token": 1.5e-06, + "input_cost_per_token_batches": 7.5e-07, "input_cost_per_token_flex": 7.5e-07, "input_cost_per_token_priority": 2.7e-06, "litellm_provider": "vertex_ai-language-models", @@ -29616,6 +29621,7 @@ "mode": "chat", "output_cost_per_reasoning_token": 7.5e-06, "output_cost_per_token": 7.5e-06, + "output_cost_per_token_batches": 3.75e-06, "output_cost_per_token_flex": 3.75e-06, "output_cost_per_token_priority": 1.35e-05, "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a1878c00678..b5282f12539 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -27350,9 +27350,11 @@ }, "vertex_ai/gemini-3.8-flash-cyber": { "cache_read_input_token_cost": 1.5e-07, + "cache_read_input_token_cost_batches": 7.5e-08, "cache_read_input_token_cost_flex": 7.5e-08, "cache_read_input_token_cost_priority": 2.7e-07, "input_cost_per_token": 1.5e-06, + "input_cost_per_token_batches": 7.5e-07, "input_cost_per_token_flex": 7.5e-07, "input_cost_per_token_priority": 2.7e-06, "litellm_provider": "vertex_ai", @@ -27362,6 +27364,7 @@ "mode": "chat", "output_cost_per_reasoning_token": 7.5e-06, "output_cost_per_token": 7.5e-06, + "output_cost_per_token_batches": 3.75e-06, "output_cost_per_token_flex": 3.75e-06, "output_cost_per_token_priority": 1.35e-05, "regional_endpoint_uplift_multiplier": 1.1, @@ -29604,9 +29607,11 @@ }, "gemini-3.8-flash-cyber": { "cache_read_input_token_cost": 1.5e-07, + "cache_read_input_token_cost_batches": 7.5e-08, "cache_read_input_token_cost_flex": 7.5e-08, "cache_read_input_token_cost_priority": 2.7e-07, "input_cost_per_token": 1.5e-06, + "input_cost_per_token_batches": 7.5e-07, "input_cost_per_token_flex": 7.5e-07, "input_cost_per_token_priority": 2.7e-06, "litellm_provider": "vertex_ai-language-models", @@ -29616,6 +29621,7 @@ "mode": "chat", "output_cost_per_reasoning_token": 7.5e-06, "output_cost_per_token": 7.5e-06, + "output_cost_per_token_batches": 3.75e-06, "output_cost_per_token_flex": 3.75e-06, "output_cost_per_token_priority": 1.35e-05, "source": "https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing", From 77c7a870d542131ac6dd0fdea603d99bb88cdc84 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 06:55:14 -0700 Subject: [PATCH 62/96] fix(cost-map): add azure realtime, audio and partner model rows (#42954) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 217 ++++++++++++++++++ model_prices_and_context_window.json | 217 ++++++++++++++++++ 2 files changed, 434 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index b5282f12539..d87a21e6d1f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -63384,6 +63384,115 @@ "supports_tool_choice": true, "supports_vision": true }, + "azure_ai/FW-DeepSeek-V4.1-Flash": { + "cache_read_input_token_cost": 8e-09, + "input_cost_per_token": 3.75e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 1000000, + "max_output_tokens": 384000, + "max_tokens": 384000, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/FW-DeepSeek-V4-Flash": { + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 1000000, + "max_output_tokens": 384000, + "max_tokens": 384000, + "mode": "chat", + "output_cost_per_token": 3.1e-07, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/FW-GLM-5.3": { + "cache_read_input_token_cost": 3.25e-07, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "azure_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 5.5e-06, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/FW-GLM-5.3-Flash": { + "cache_read_input_token_cost": 3.8e-08, + "input_cost_per_token": 1.88e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 6.25e-07, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/FW-GPT-OSS-120B": { + "cache_read_input_token_cost": 8.2e-08, + "input_cost_per_token": 1.65e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 6.6e-07, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure_ai/Cohere-command-a-plus-05-2026": { + "input_cost_per_token": 8e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 128000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 3.2e-06, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/mistral-medium-3-5": { + "input_cost_per_token": 1.5e-06, + "litellm_provider": "azure_ai", + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 7.5e-06, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_response_schema": true, + "supports_vision": true + }, "bedrock/us-gov-west-1/nvidia.nemotron-nano-3-30b": { "input_cost_per_token": 7.2e-08, "litellm_provider": "bedrock", @@ -69433,6 +69542,114 @@ "mode": "embedding", "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, + "azure/gpt-realtime-2": { + "cache_creation_input_audio_token_cost": 4e-07, + "cache_read_input_audio_token_cost": 4e-07, + "cache_read_input_token_cost": 4e-07, + "input_cost_per_audio_token": 3.2e-05, + "input_cost_per_image_token": 5e-06, + "input_cost_per_token": 4e-06, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "realtime", + "output_cost_per_audio_token": 6.4e-05, + "output_cost_per_token": 2.4e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "azure/gpt-live-1": { + "input_cost_per_second": 0.000833333333333, + "litellm_provider": "azure", + "mode": "realtime", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true + }, + "azure/gpt-live-transcribe": { + "input_cost_per_second": 0.000283333333333, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "audio_transcription", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/realtime", + "/v1/realtime/transcription_sessions" + ], + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_input": true + }, + "azure/gpt-transcribe": { + "input_cost_per_second": 7.5e-05, + "litellm_provider": "azure", + "mode": "audio_transcription", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/audio/transcriptions", + "/v1/realtime/transcription_sessions" + ], + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_input": true + }, + "azure/gpt-realtime-translate": { + "input_cost_per_second": 0.000566666666667, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "realtime", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_modalities": [ + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true + }, "aihubmix/agnes-2.5-flash": { "input_cost_per_token": 3e-08, "litellm_provider": "aihubmix", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index b5282f12539..d87a21e6d1f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -63384,6 +63384,115 @@ "supports_tool_choice": true, "supports_vision": true }, + "azure_ai/FW-DeepSeek-V4.1-Flash": { + "cache_read_input_token_cost": 8e-09, + "input_cost_per_token": 3.75e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 1000000, + "max_output_tokens": 384000, + "max_tokens": 384000, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/FW-DeepSeek-V4-Flash": { + "cache_read_input_token_cost": 3e-08, + "input_cost_per_token": 1.5e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 1000000, + "max_output_tokens": 384000, + "max_tokens": 384000, + "mode": "chat", + "output_cost_per_token": 3.1e-07, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/FW-GLM-5.3": { + "cache_read_input_token_cost": 3.25e-07, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "azure_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 5.5e-06, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/FW-GLM-5.3-Flash": { + "cache_read_input_token_cost": 3.8e-08, + "input_cost_per_token": 1.88e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 6.25e-07, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/FW-GPT-OSS-120B": { + "cache_read_input_token_cost": 8.2e-08, + "input_cost_per_token": 1.65e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 6.6e-07, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "azure_ai/Cohere-command-a-plus-05-2026": { + "input_cost_per_token": 8e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 128000, + "max_output_tokens": 64000, + "max_tokens": 64000, + "mode": "chat", + "output_cost_per_token": 3.2e-06, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_function_calling": true, + "supports_reasoning": true, + "supports_tool_choice": true + }, + "azure_ai/mistral-medium-3-5": { + "input_cost_per_token": 1.5e-06, + "litellm_provider": "azure_ai", + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 7.5e-06, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_response_schema": true, + "supports_vision": true + }, "bedrock/us-gov-west-1/nvidia.nemotron-nano-3-30b": { "input_cost_per_token": 7.2e-08, "litellm_provider": "bedrock", @@ -69433,6 +69542,114 @@ "mode": "embedding", "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, + "azure/gpt-realtime-2": { + "cache_creation_input_audio_token_cost": 4e-07, + "cache_read_input_audio_token_cost": 4e-07, + "cache_read_input_token_cost": 4e-07, + "input_cost_per_audio_token": 3.2e-05, + "input_cost_per_image_token": 5e-06, + "input_cost_per_token": 4e-06, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "realtime", + "output_cost_per_audio_token": 6.4e-05, + "output_cost_per_token": 2.4e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "azure/gpt-live-1": { + "input_cost_per_second": 0.000833333333333, + "litellm_provider": "azure", + "mode": "realtime", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true + }, + "azure/gpt-live-transcribe": { + "input_cost_per_second": 0.000283333333333, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "audio_transcription", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/realtime", + "/v1/realtime/transcription_sessions" + ], + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_input": true + }, + "azure/gpt-transcribe": { + "input_cost_per_second": 7.5e-05, + "litellm_provider": "azure", + "mode": "audio_transcription", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/audio/transcriptions", + "/v1/realtime/transcription_sessions" + ], + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text" + ], + "supports_audio_input": true + }, + "azure/gpt-realtime-translate": { + "input_cost_per_second": 0.000566666666667, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "realtime", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_modalities": [ + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true + }, "aihubmix/agnes-2.5-flash": { "input_cost_per_token": 3e-08, "litellm_provider": "aihubmix", From 11ed3335c8be943da40815bc9bd3ae5990176c80 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 07:18:08 -0700 Subject: [PATCH 63/96] fix(cost-map): sync openrouter deepseek-v4-pro prices (#42956) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 6 +++--- model_prices_and_context_window.json | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d87a21e6d1f..6257405ed7f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41242,21 +41242,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.31074e-07, + "input_cost_per_token": 9.27768e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.862148e-06, + "output_cost_per_token": 1.855536e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.75895e-08, + "cache_read_input_token_cost": 7.7314e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d87a21e6d1f..6257405ed7f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41242,21 +41242,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.31074e-07, + "input_cost_per_token": 9.27768e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.862148e-06, + "output_cost_per_token": 1.855536e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.75895e-08, + "cache_read_input_token_cost": 7.7314e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, From b8f3ba03b30b7cb19e21820b9818f32935401ccd Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 07:57:26 -0700 Subject: [PATCH 64/96] fix(cost-map): sync openrouter deepseek-v4-pro prices (#42964) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 6 +++--- model_prices_and_context_window.json | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6257405ed7f..69a64de6e04 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41242,21 +41242,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.27768e-07, + "input_cost_per_token": 9.24462e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.855536e-06, + "output_cost_per_token": 1.848924e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.7314e-08, + "cache_read_input_token_cost": 7.70385e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6257405ed7f..69a64de6e04 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41242,21 +41242,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.27768e-07, + "input_cost_per_token": 9.24462e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.855536e-06, + "output_cost_per_token": 1.848924e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.7314e-08, + "cache_read_input_token_cost": 7.70385e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, From e135a199ad840336f60b61426967444bbd71cea5 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:09:49 -0500 Subject: [PATCH 65/96] fix(proxy): fail parked DB lookups at a deadline and flip readiness while they stall (#42654) * fix(proxy): fail parked DB lookups at a deadline and flip readiness while they stall Under a load burst with a slow authentication database every request parked inside the pod with no deadline while /health/readiness kept answering 200 (its own ping gets a fresh connection), so the load balancer kept sending traffic until the pod hit its memory limit, and the parked requests completed against the provider minutes after every client had hung up Every pre-request read (key, team, user, end user, budget, membership, organization, object permission, jwt mapping, project, proxy budget, spend counter reseed) now runs under one deadline, PROXY_DB_LOOKUP_DEADLINE_SECONDS (default 10 s). A lookup that hits it fails the request with the existing 503 "authentication database is temporarily unreachable" answer, honours allow_requests_on_db_unavailable, and never triggers the transport reconnect (the transport is fine, the query is slow), which is what turned the repro's stall into "too many clients". Writes stay unbounded A deadline hit marks the pod stalled for PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS (default 30 s, 0 disables), during which /health/readiness answers 503 with "db": "stalled" behind the same fail-open gate, so the pod leaves rotation before it fills its memory. The existing litellm_in_flight_requests gauge already exposes the parked set on /metrics The deadline is enforced on the wall clock: bounded_db_lookup waits on the lookup task with asyncio.wait and raises DBLookupDeadlineExceeded when the deadline passes even if the lookup absorbs its cancellation, where asyncio.wait_for on 3.12+ would sit on the cancelled task for as long as it takes The failure spend-log row no longer re-runs the key and team lookups when the failure itself is a database connection or deadline error, so a request that hit the deadline is answered after one deadline instead of two * fix(proxy): bound the spend counter gate wait and narrow the stalled lookup shortcut Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(proxy): keep the global spend lookup on the prisma client handle Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Co-authored-by: yassin Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/constants.py | 2 + litellm/proxy/auth/auth_checks.py | 119 ++++- litellm/proxy/auth/user_api_key_auth.py | 6 +- litellm/proxy/db/db_lookup_gate.py | 71 ++- litellm/proxy/db/exception_handler.py | 3 +- litellm/proxy/db/spend_counter_reseed.py | 71 +-- .../health_endpoints/_health_endpoints.py | 33 +- .../proxy/hooks/proxy_track_cost_callback.py | 11 +- .../proxy/auth/test_auth_checks.py | 163 +++++- .../proxy/auth/test_user_api_key_auth.py | 486 +++++++++--------- .../proxy/db/test_db_lookup_gate.py | 111 ++++ .../proxy/db/test_exception_handler.py | 15 + .../proxy/db/test_spend_counter_reseed.py | 27 + .../health_endpoints/test_health_endpoints.py | 128 +++++ .../hooks/test_proxy_track_cost_callback.py | 100 +++- 15 files changed, 1016 insertions(+), 330 deletions(-) create mode 100644 tests/test_litellm/proxy/db/test_db_lookup_gate.py diff --git a/litellm/constants.py b/litellm/constants.py index 3ff80d8b7dd..7b40f432446 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1772,6 +1772,8 @@ RESPONSES_SESSION_LOOKUP_MAX_ATTEMPTS: Final = max(1, int(os.getenv("RESPONSES_S RESPONSES_SESSION_LOOKUP_RETRY_INTERVAL: Final = float(os.getenv("RESPONSES_SESSION_LOOKUP_RETRY_INTERVAL", "0.2")) SPEND_COUNTER_RESEED_LOCKS_MAX_SIZE: Final = int(os.getenv("SPEND_COUNTER_RESEED_LOCKS_MAX_SIZE", 10000)) PROXY_DB_LOOKUP_MAX_CONCURRENCY: Final = max(1, int(os.getenv("PROXY_DB_LOOKUP_MAX_CONCURRENCY", "25"))) +PROXY_DB_LOOKUP_DEADLINE_SECONDS: Final = max(0.1, float(os.getenv("PROXY_DB_LOOKUP_DEADLINE_SECONDS", "10"))) +PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS: Final = max(0.0, float(os.getenv("PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS", "30"))) DEFAULT_CRON_JOB_LOCK_TTL_SECONDS: Final = int(os.getenv("DEFAULT_CRON_JOB_LOCK_TTL_SECONDS", 60)) # 1 minute PROXY_BUDGET_RESCHEDULER_MIN_TIME: Final = int(os.getenv("PROXY_BUDGET_RESCHEDULER_MIN_TIME", 597)) RESET_BUDGET_JOB_BATCH_SIZE: Final = max(1, int(os.getenv("RESET_BUDGET_JOB_BATCH_SIZE", "500"))) diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index f5e98b40ca7..67950e603c0 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -15,11 +15,11 @@ import re import time from collections.abc import Awaitable, Callable, Iterator, Mapping, Sequence from types import MappingProxyType -from typing import TYPE_CHECKING, Any, Final, Literal, Optional, Protocol, TypeAlias +from typing import TYPE_CHECKING, Any, Final, Generic, Literal, Optional, Protocol, TypeAlias from fastapi import HTTPException, Request, status from pydantic import BaseModel, TypeAdapter -from typing_extensions import ReadOnly, TypedDict +from typing_extensions import NotRequired, ReadOnly, Required, TypedDict, Unpack import litellm from litellm._logging import verbose_proxy_logger @@ -110,7 +110,7 @@ from litellm.proxy.common_utils.user_api_key_cache import ( team_membership_auth_cache_key, team_membership_reservation_cache_key, ) -from litellm.proxy.db.db_lookup_gate import db_lookup_gate +from litellm.proxy.db.db_lookup_gate import bounded_db_lookup, db_lookup_gate from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler from litellm.proxy.guardrails.tool_name_extraction import ( TOOL_CAPABLE_CALL_TYPES, @@ -223,30 +223,79 @@ class _PrismaTableHolder(Protocol[RowT_co]): def table(self) -> _PrismaAuthTable[RowT_co]: ... -def _dictable_table(repo: _PrismaTableHolder[_PrismaDictableRow]) -> _PrismaAuthTable[_PrismaDictableRow]: - return repo.table +class _FindOneKwargs(TypedDict): + where: ReadOnly[Required[Mapping[str, object]]] + include: ReadOnly[NotRequired[Mapping[str, object] | None]] + + +class _FindManyKwargs(TypedDict): + where: ReadOnly[NotRequired[Mapping[str, object] | None]] + include: ReadOnly[NotRequired[Mapping[str, object] | None]] + take: ReadOnly[NotRequired[int | None]] + + +class _DeadlineBoundedTable(Generic[RowT_co]): + """Every read on the wrapped table fails with ``DBLookupDeadlineExceeded`` once + ``PROXY_DB_LOOKUP_DEADLINE_SECONDS`` passes, so a stalled database fails the + request fast instead of parking it in the pod until it fills its memory.""" + + __slots__ = ("_lookup", "_table") + + def __init__(self, table: _PrismaAuthTable[RowT_co], lookup: str) -> None: + self._table: Final = table + self._lookup: Final = lookup + + async def find_unique( + self, + **kwargs: Unpack[_FindOneKwargs], # kwargs-ok: typed pass-through that forwards exactly what the caller passed + ) -> RowT_co | None: + return await bounded_db_lookup(self._table.find_unique(**kwargs), name=self._lookup) + + async def find_first( + self, + **kwargs: Unpack[_FindOneKwargs], # kwargs-ok: typed pass-through that forwards exactly what the caller passed + ) -> RowT_co | None: + return await bounded_db_lookup(self._table.find_first(**kwargs), name=self._lookup) + + async def find_many( + self, + **kwargs: Unpack[_FindManyKwargs], # kwargs-ok: typed pass-through that forwards exactly what the caller passed + ) -> Sequence[RowT_co]: + return await bounded_db_lookup(self._table.find_many(**kwargs), name=self._lookup) + + async def update(self, *, where: Mapping[str, object], data: Mapping[str, object]) -> RowT_co | None: + return await self._table.update(where=where, data=data) + + async def create(self, *, data: Mapping[str, object], include: Mapping[str, object] | None = None) -> RowT_co: + return await self._table.create(data=data, include=include) + + +def _dictable_table(repo: _PrismaTableHolder[_PrismaDictableRow], lookup: str) -> _PrismaAuthTable[_PrismaDictableRow]: + return _DeadlineBoundedTable(repo.table, lookup) def _jwt_key_mapping_table( repo: _PrismaTableHolder[_PrismaJWTKeyMappingRow], ) -> _PrismaAuthTable[_PrismaJWTKeyMappingRow]: - return repo.table + return _DeadlineBoundedTable(repo.table, "jwt_key_mapping") -def _model_dump_table(repo: _PrismaTableHolder[_PrismaModelDumpRow]) -> _PrismaAuthTable[_PrismaModelDumpRow]: - return repo.table +def _model_dump_table( + repo: _PrismaTableHolder[_PrismaModelDumpRow], lookup: str +) -> _PrismaAuthTable[_PrismaModelDumpRow]: + return _DeadlineBoundedTable(repo.table, lookup) def _team_table(repo: _PrismaTableHolder[_PrismaTeamRow]) -> _PrismaAuthTable[_PrismaTeamRow]: - return repo.table + return _DeadlineBoundedTable(repo.table, "team") def _vector_store_table(repo: _PrismaTableHolder[_PrismaVectorStoreRow]) -> _PrismaAuthTable[_PrismaVectorStoreRow]: - return repo.table + return _DeadlineBoundedTable(repo.table, "vector_store") def _user_table(repo: _PrismaTableHolder[_PrismaUserRow]) -> _PrismaAuthTable[_PrismaUserRow]: - return repo.table + return _DeadlineBoundedTable(repo.table, "user") class _VectorStorePermissionsRow(Protocol): @@ -257,7 +306,7 @@ class _VectorStorePermissionsRow(Protocol): def _object_permission_table( repo: _PrismaTableHolder[_VectorStorePermissionsRow], ) -> _PrismaAuthTable[_VectorStorePermissionsRow]: - return repo.table + return _DeadlineBoundedTable(repo.table, "object_permission") class _PrismaTagRow(Protocol): @@ -1422,7 +1471,7 @@ async def get_default_end_user_budget( # Fetch from database try: - budget_record: Final = await _dictable_table(BudgetRepository(prisma_client)).find_unique( + budget_record: Final = await _dictable_table(BudgetRepository(prisma_client), "budget").find_unique( where={"budget_id": default_budget_id} # mutable-ok: prisma where clause ) @@ -1483,7 +1532,7 @@ async def get_team_member_default_budget( return cached_budget try: - budget_record: Final = await _dictable_table(BudgetRepository(prisma_client)).find_unique( + budget_record: Final = await _dictable_table(BudgetRepository(prisma_client), "budget").find_unique( where={"budget_id": budget_id} ) except Exception: @@ -1877,7 +1926,7 @@ async def get_end_user_object( # Fetch from database try: - response: Final = await _dictable_table(EndUserRepository(prisma_client)).find_unique( + response: Final = await _dictable_table(EndUserRepository(prisma_client), "end_user").find_unique( where={"user_id": end_user_id}, include={"litellm_budget_table": True, "object_permission": True}, ) @@ -2286,7 +2335,7 @@ async def _fetch_team_membership_from_db( proxy_logging_obj: ProxyLogging | None = None, ) -> LiteLLM_TeamMembership | None: _ = parent_otel_span, proxy_logging_obj - response: Final = await _dictable_table(TeamMembershipRepository(prisma_client)).find_unique( + response: Final = await _dictable_table(TeamMembershipRepository(prisma_client), "team_membership").find_unique( where={"user_id_team_id": {"user_id": user_id, "team_id": team_id}}, include={"litellm_budget_table": True}, ) @@ -3290,7 +3339,7 @@ async def get_access_object( # Not in cache - fetch from DB try: - response: Final = await _dictable_table(AccessGroupRepository(prisma_client)).find_unique( + response: Final = await _dictable_table(AccessGroupRepository(prisma_client), "access_group").find_unique( where={"access_group_id": access_group_id} ) @@ -3472,7 +3521,7 @@ async def get_org_object_by_alias( # Query database by organization_alias try: - orgs = await _model_dump_table(OrganizationRepository(prisma_client)).find_many( + orgs = await _model_dump_table(OrganizationRepository(prisma_client), "organization").find_many( where={"organization_alias": org_alias} ) @@ -3650,10 +3699,32 @@ async def _fetch_key_object_from_db_with_reconnect( prisma_client: PrismaClient, parent_otel_span: Span | None, proxy_logging_obj: ProxyLogging | None, + deadline_seconds: float | None = None, ) -> BaseModel | None: """ Fetch key object from DB and retry once if a DB connection error can be healed. + The gate wait, the query, the reconnect, and the retry share one deadline, so a + stalled database fails the request with ``DBLookupDeadlineExceeded`` instead of + parking it. """ + return await bounded_db_lookup( + _fetch_key_object_from_db_unbounded( + hashed_token=hashed_token, + prisma_client=prisma_client, + parent_otel_span=parent_otel_span, + proxy_logging_obj=proxy_logging_obj, + ), + name="key", + deadline_seconds=deadline_seconds, + ) + + +async def _fetch_key_object_from_db_unbounded( + hashed_token: str, + prisma_client: PrismaClient, + parent_otel_span: Span | None, + proxy_logging_obj: ProxyLogging | None, +) -> BaseModel | None: async with db_lookup_gate.current(): try: return await prisma_client.get_data( @@ -3874,9 +3945,9 @@ async def get_object_permission( # else, check db try: - response: Final = await _dictable_table(ObjectPermissionRepository(prisma_client)).find_unique( - where={"object_permission_id": object_permission_id} - ) + response: Final = await _dictable_table( + ObjectPermissionRepository(prisma_client), "object_permission" + ).find_unique(where={"object_permission_id": object_permission_id}) if response is None: return None @@ -4008,7 +4079,9 @@ async def get_org_object( if include_budget_table: query_kwargs["include"] = {"litellm_budget_table": True} - response: Final = await _model_dump_table(OrganizationRepository(prisma_client)).find_unique(**query_kwargs) + response: Final = await _model_dump_table(OrganizationRepository(prisma_client), "organization").find_unique( + **query_kwargs + ) except Exception: # An operational failure (DB down, timeout, cache fault) is NOT the same fact as a confirmed # missing row, and relabelling it as "doesn't exist" made every caller unable to tell them @@ -5948,7 +6021,7 @@ async def get_project_object( return deserialized_project # Fetch from DB - project_row: Final = await _model_dump_table(ProjectRepository(prisma_client)).find_unique( + project_row: Final = await _model_dump_table(ProjectRepository(prisma_client), "project").find_unique( where={"project_id": project_id}, include={"litellm_budget_table": True}, ) diff --git a/litellm/proxy/auth/user_api_key_auth.py b/litellm/proxy/auth/user_api_key_auth.py index a6c0792a86f..ae95e94dd2d 100644 --- a/litellm/proxy/auth/user_api_key_auth.py +++ b/litellm/proxy/auth/user_api_key_auth.py @@ -120,6 +120,7 @@ from litellm.proxy.common_utils.user_api_key_cache import ( UserApiKeyCache, team_membership_auth_cache_key, ) +from litellm.proxy.db.db_lookup_gate import bounded_db_lookup from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup from litellm.proxy.spend_tracking.carried_budget_state import carry_team_and_user_budget_state @@ -735,8 +736,9 @@ async def _fetch_global_spend_with_event_coordination( """ async def _load_global_spend() -> float | None: - proxy_budget_row: Final = await prisma_client.db.litellm_usertable.find_unique( - where={"user_id": LITELLM_PROXY_BUDGET_NAME} + proxy_budget_row: Final = await bounded_db_lookup( + prisma_client.db.litellm_usertable.find_unique(where={"user_id": LITELLM_PROXY_BUDGET_NAME}), + name="proxy_budget", ) return float(proxy_budget_row.spend) if proxy_budget_row is not None else None diff --git a/litellm/proxy/db/db_lookup_gate.py b/litellm/proxy/db/db_lookup_gate.py index 2fd427687bd..d2aefde929f 100644 --- a/litellm/proxy/db/db_lookup_gate.py +++ b/litellm/proxy/db/db_lookup_gate.py @@ -1,7 +1,14 @@ import asyncio -from typing import Final +import time +from collections.abc import Awaitable, Callable +from typing import Final, TypeVar -from litellm.constants import PROXY_DB_LOOKUP_MAX_CONCURRENCY +from litellm.constants import ( + PROXY_DB_LOOKUP_DEADLINE_SECONDS, + PROXY_DB_LOOKUP_MAX_CONCURRENCY, +) + +LookupT = TypeVar("LookupT") class LoopBoundSemaphore: @@ -20,4 +27,64 @@ class LoopBoundSemaphore: return self._semaphore +class DBLookupDeadlineExceeded(asyncio.TimeoutError): + def __init__(self, lookup: str, deadline_seconds: float) -> None: + super().__init__(f"{lookup} lookup did not answer within {deadline_seconds:g}s") + self.lookup: Final = lookup + self.deadline_seconds: Final = deadline_seconds + + +class DBLookupStallTracker: + __slots__ = ("_clock", "_last_hit") + + def __init__(self, clock: Callable[[], float] = time.monotonic) -> None: + self._clock: Final = clock + self._last_hit: float | None = None + + def record_hit(self) -> None: + self._last_hit = self._clock() + + def clear(self) -> None: + self._last_hit = None + + def stalled_within(self, window_seconds: float) -> bool: + if self._last_hit is None: + return False + return self._clock() - self._last_hit < window_seconds + + db_lookup_gate: Final = LoopBoundSemaphore(PROXY_DB_LOOKUP_MAX_CONCURRENCY) +db_lookup_stall_tracker: Final = DBLookupStallTracker() + + +def _consume_abandoned_lookup(task: asyncio.Future[LookupT]) -> None: + if not task.cancelled(): + task.exception() + + +async def bounded_db_lookup( + lookup: Awaitable[LookupT], + *, + name: str, + deadline_seconds: float | None = None, + tracker: DBLookupStallTracker = db_lookup_stall_tracker, +) -> LookupT: + timeout: Final = PROXY_DB_LOOKUP_DEADLINE_SECONDS if deadline_seconds is None else deadline_seconds + task: Final = asyncio.ensure_future(lookup) + try: + done, _ = await asyncio.wait({task}, timeout=timeout) + except asyncio.CancelledError: + task.cancel() + raise + if task not in done: + task.cancel() + task.add_done_callback(_consume_abandoned_lookup) + tracker.record_hit() + raise DBLookupDeadlineExceeded(name, timeout) + try: + return task.result() + except DBLookupDeadlineExceeded: + raise + except asyncio.TimeoutError as e: + tracker.record_hit() + raise DBLookupDeadlineExceeded(name, timeout) from e diff --git a/litellm/proxy/db/exception_handler.py b/litellm/proxy/db/exception_handler.py index 38d6fb9b99d..022ca2efc9e 100644 --- a/litellm/proxy/db/exception_handler.py +++ b/litellm/proxy/db/exception_handler.py @@ -11,6 +11,7 @@ from litellm.proxy._types import ( ProxyErrorTypes, ProxyException, ) +from litellm.proxy.db.db_lookup_gate import DBLookupDeadlineExceeded from litellm.secret_managers.main import str_to_bool # Bounds the __cause__/__context__ walk in find_database_service_unavailable_error_in_chain. @@ -104,7 +105,7 @@ class PrismaDBExceptionHandler: """ import prisma.engine.errors - if isinstance(e, DB_CONNECTION_ERROR_TYPES): + if isinstance(e, (*DB_CONNECTION_ERROR_TYPES, DBLookupDeadlineExceeded)): return True if isinstance(e, _exception_types(prisma.engine.errors.EngineConnectionError)): return True diff --git a/litellm/proxy/db/spend_counter_reseed.py b/litellm/proxy/db/spend_counter_reseed.py index 2dd028454d6..f8e102d2682 100644 --- a/litellm/proxy/db/spend_counter_reseed.py +++ b/litellm/proxy/db/spend_counter_reseed.py @@ -23,7 +23,7 @@ from litellm._logging import verbose_proxy_logger from litellm.constants import SPEND_COUNTER_RESEED_LOCKS_MAX_SIZE from litellm.litellm_core_utils.duration_parser import duration_in_seconds from litellm.proxy._types import Litellm_EntityType -from litellm.proxy.db.db_lookup_gate import db_lookup_gate +from litellm.proxy.db.db_lookup_gate import bounded_db_lookup, db_lookup_gate from litellm.proxy.spend_tracking.spend_counter_batch import read_batched_spend_counter, record_spend_counter_value from litellm.repositories.organization_repository import OrganizationRepository from litellm.repositories.project_repository import ProjectRepository @@ -134,36 +134,9 @@ class SpendCounterReseed: if SpendCounterReseed._is_key_or_team_window_counter(counter_key): return None try: - async with db_lookup_gate.current(): - if counter_key.startswith("spend:key:"): - token: Final = counter_key[len("spend:key:") :] - row = await VerificationTokenRepository(prisma_client).table.find_unique(where={"token": token}) - elif counter_key.startswith("spend:team_member:"): - suffix: Final = counter_key[len("spend:team_member:") :] - if ":" not in suffix: - return None - user_id, team_id = suffix.rsplit(":", 1) - row = await TeamMembershipRepository(prisma_client).table.find_unique( - where={"user_id_team_id": {"user_id": user_id, "team_id": team_id}} - ) - elif counter_key.startswith("spend:team:"): - team_id = counter_key[len("spend:team:") :] - row = await TeamRepository(prisma_client).table.find_unique(where={"team_id": team_id}) - elif counter_key.startswith("spend:user:"): - user_id = counter_key[len("spend:user:") :] - row = await UserRepository(prisma_client).table.find_unique(where={"user_id": user_id}) - elif counter_key.startswith(END_USER_COUNTER_PREFIX) or counter_key.startswith("spend:tag:"): - return None - elif counter_key.startswith("spend:org:"): - org_id: Final = counter_key[len("spend:org:") :] - row = await OrganizationRepository(prisma_client).table.find_unique( - where={"organization_id": org_id} - ) - elif counter_key.startswith("spend:project:"): - project_id: Final = counter_key[len("spend:project:") :] - row = await ProjectRepository(prisma_client).table.find_unique(where={"project_id": project_id}) - else: - return None + row: Final = await bounded_db_lookup( + SpendCounterReseed._counter_row(prisma_client, counter_key), name="spend_counter" + ) except Exception: verbose_proxy_logger.exception("SpendCounterReseed.from_db: failed for %s", counter_key) return None @@ -171,13 +144,47 @@ class SpendCounterReseed: return None return float(getattr(row, "spend", 0.0) or 0.0) + @staticmethod + async def _counter_row(prisma_client: "PrismaClient", counter_key: str) -> object | None: + async with db_lookup_gate.current(): + if counter_key.startswith("spend:key:"): + token: Final = counter_key[len("spend:key:") :] + return await VerificationTokenRepository(prisma_client).table.find_unique(where={"token": token}) + if counter_key.startswith("spend:team_member:"): + suffix: Final = counter_key[len("spend:team_member:") :] + if ":" not in suffix: + return None + user_id, team_id = suffix.rsplit(":", 1) + return await TeamMembershipRepository(prisma_client).table.find_unique( + where={"user_id_team_id": {"user_id": user_id, "team_id": team_id}} + ) + if counter_key.startswith("spend:team:"): + return await TeamRepository(prisma_client).table.find_unique( + where={"team_id": counter_key[len("spend:team:") :]} + ) + if counter_key.startswith("spend:user:"): + return await UserRepository(prisma_client).table.find_unique( + where={"user_id": counter_key[len("spend:user:") :]} + ) + if counter_key.startswith("spend:org:"): + return await OrganizationRepository(prisma_client).table.find_unique( + where={"organization_id": counter_key[len("spend:org:") :]} + ) + if counter_key.startswith("spend:project:"): + return await ProjectRepository(prisma_client).table.find_unique( + where={"project_id": counter_key[len("spend:project:") :]} + ) + return None + @staticmethod async def end_user_from_db(prisma_client: Optional["PrismaClient"], counter_key: str) -> float | None: if prisma_client is None or not counter_key.startswith(END_USER_COUNTER_PREFIX): return None where: Final[LiteLLM_EndUserTableWhereUniqueInput] = {"user_id": counter_key[len(END_USER_COUNTER_PREFIX) :]} try: - row: Final = await EndUserRepository(prisma_client).table.find_unique(where=where) + row: Final = await bounded_db_lookup( + EndUserRepository(prisma_client).table.find_unique(where=where), name="end_user_spend" + ) except Exception: # noqa: BLE001 # a failed floor read falls back to the cached spend, like from_db verbose_proxy_logger.exception("SpendCounterReseed.end_user_from_db: failed for %s", counter_key) return None diff --git a/litellm/proxy/health_endpoints/_health_endpoints.py b/litellm/proxy/health_endpoints/_health_endpoints.py index 64fd59bbe44..f8801e65c82 100644 --- a/litellm/proxy/health_endpoints/_health_endpoints.py +++ b/litellm/proxy/health_endpoints/_health_endpoints.py @@ -16,7 +16,7 @@ from typing_extensions import ReadOnly import litellm from litellm._logging import verbose_logger, verbose_proxy_logger -from litellm.constants import HEALTH_CHECK_TIMEOUT_SECONDS +from litellm.constants import HEALTH_CHECK_TIMEOUT_SECONDS, PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS from litellm.integrations.SlackAlerting.ms_teams import ( MS_TEAMS_ALERT_HEADERS, build_ms_teams_payload, @@ -44,6 +44,7 @@ from litellm.proxy.auth.auth_utils import ( ) from litellm.proxy.auth.model_checks import get_key_models from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler from litellm.proxy.db.health_check_latest import ( LatestHealthCheckRow, @@ -1723,7 +1724,7 @@ async def _get_health_readiness_details( # check DB if prisma_client is not None: # if db passed in, check if it's connected - db_health_status: Final = await _db_health_readiness_check() + db_status: Final = _readiness_db_status(await _db_health_readiness_check()) # A configured DB that is not reachable means the worker cannot # serve requests that depend on persisted state (keys, budgets, # spend logs). Return 503 so orchestrators take this pod out of @@ -1733,13 +1734,13 @@ async def _get_health_readiness_details( # report the DB state through the body instead. if ( response is not None - and db_health_status["status"] != "connected" + and db_status != "connected" and not PrismaDBExceptionHandler.should_allow_request_on_db_unavailable() ): response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE return { "status": "healthy", - "db": db_health_status["status"], + "db": db_status, "cache": cache_type, "litellm_version": version, "success_callbacks": success_callback_names, @@ -1816,24 +1817,32 @@ def _authorize_drain_request(request: Request) -> None: ) +def _readiness_db_status(db_health_status: DBHealthCache) -> str: + """A pod whose pre-request lookups hit their deadline inside the stall window + reports "stalled" even though the ping succeeds: the ping is a fresh + connection, the stalled lookups are the ones requests actually wait on.""" + if db_health_status["status"] != "connected": + return db_health_status["status"] + if db_lookup_stall_tracker.stalled_within(PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS): + return "stalled" + return "connected" + + async def _resolve_public_readiness_db(response: Response) -> str: """ Return the db status string for the public probe and flip the response to - 503 when a configured DB is unreachable. Mirrors the legacy values: - "Not connected" (no DB configured), "connected", "disconnected". + 503 when a configured DB is unreachable or stalled. Mirrors the legacy values: + "Not connected" (no DB configured), "connected", "disconnected", plus "stalled". """ from litellm.proxy.proxy_server import prisma_client if prisma_client is None: return "Not connected" - db_health_status: Final = await _db_health_readiness_check() - if ( - db_health_status["status"] != "connected" - and not PrismaDBExceptionHandler.should_allow_request_on_db_unavailable() - ): + db_status: Final = _readiness_db_status(await _db_health_readiness_check()) + if db_status != "connected" and not PrismaDBExceptionHandler.should_allow_request_on_db_unavailable(): response.status_code = status.HTTP_503_SERVICE_UNAVAILABLE - return db_health_status["status"] + return db_status @router.get( diff --git a/litellm/proxy/hooks/proxy_track_cost_callback.py b/litellm/proxy/hooks/proxy_track_cost_callback.py index 5dae9e8bb10..4dfea5cd472 100644 --- a/litellm/proxy/hooks/proxy_track_cost_callback.py +++ b/litellm/proxy/hooks/proxy_track_cost_callback.py @@ -23,6 +23,7 @@ from litellm.proxy.auth.auth_checks import ( log_db_metrics, ) from litellm.proxy.auth.route_checks import RouteChecks +from litellm.proxy.db.db_lookup_gate import DBLookupDeadlineExceeded from litellm.proxy.db.db_spend_update_writer import ( DBSpendUpdateWriter, debitable_model_access_groups, @@ -186,8 +187,8 @@ class _ProxyDBLogger(CustomLogger): ) _metadata["error_information"] = _error_information - _metadata = await _ProxyDBLogger._enrich_failure_metadata_with_key_info( - metadata=_metadata, + _metadata = await _ProxyDBLogger._enrich_failure_metadata_unless_db_stalled( + metadata=_metadata, original_exception=original_exception ) existing_metadata: Final[dict] = request_data.get("metadata", None) or {} @@ -472,6 +473,12 @@ class _ProxyDBLogger(CustomLogger): spend_log_error("Error in tracking cost callback - %s", str(e), exc=e) + @staticmethod + async def _enrich_failure_metadata_unless_db_stalled(metadata: dict, original_exception: Exception) -> dict: + if isinstance(original_exception, DBLookupDeadlineExceeded): + return metadata + return await _ProxyDBLogger._enrich_failure_metadata_with_key_info(metadata=metadata) + @staticmethod async def _enrich_failure_metadata_with_key_info(metadata: dict, resolve_missing_key_identity: bool = True) -> dict: """ diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py index aa4bb2e49d3..30f5abdbb98 100644 --- a/tests/test_litellm/proxy/auth/test_auth_checks.py +++ b/tests/test_litellm/proxy/auth/test_auth_checks.py @@ -1,7 +1,7 @@ import asyncio import json import time -from collections.abc import Mapping +from collections.abc import Iterator, Mapping from types import SimpleNamespace from typing import TYPE_CHECKING, Final, Literal, Optional from unittest.mock import AsyncMock, MagicMock, patch @@ -646,6 +646,114 @@ async def test_fetch_key_object_from_db_bounds_in_flight_prisma_requests(): assert prisma.max_in_flight == PROXY_DB_LOOKUP_MAX_CONCURRENCY +@pytest.fixture +def _clear_db_lookup_stall() -> Iterator[None]: + from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker + + db_lookup_stall_tracker.clear() + yield + db_lookup_stall_tracker.clear() + + +class _StalledPrisma: + def __init__(self) -> None: + self.attempt_db_reconnect = AsyncMock(return_value=True) + self.db = MagicMock() + self.db.litellm_teamtable.find_unique = AsyncMock(side_effect=_stall_forever) + self.db.litellm_teamtable.update = AsyncMock(side_effect=_answer_slowly) + + async def get_data(self, token: str, table_name: str, parent_otel_span: None, proxy_logging_obj: None) -> None: + await _stall_forever() + + +async def _stall_forever(**kwargs: object) -> None: + await asyncio.Event().wait() + + +async def _answer_slowly(**kwargs: object) -> Mapping[str, object]: + await asyncio.sleep(0.15) + return {"team_id": "slow-write"} + + +@pytest.mark.asyncio +async def test_fetch_key_object_from_db_fails_a_stalled_burst_within_the_deadline_without_reconnecting( + _clear_db_lookup_stall, +): + """The incident: a stalled database parked every request in the pod with liveness + and readiness green until it OOMed. Every lookup in a burst larger than the gate, + the ones queued behind it included, must fail within one deadline, must not try to + reconnect (the transport is fine, the query is slow), and must leave every gate slot + free for the next burst.""" + from litellm.proxy.db.db_lookup_gate import DBLookupDeadlineExceeded + + prisma: Final = _StalledPrisma() + burst: Final = PROXY_DB_LOOKUP_MAX_CONCURRENCY * 3 + started: Final = time.monotonic() + + results: Final = await asyncio.gather( + *( + _fetch_key_object_from_db_with_reconnect( + hashed_token=f"hashed-token-{i}", + prisma_client=prisma, # pyright: ignore[reportArgumentType] # fake stands in for PrismaClient + parent_otel_span=None, + proxy_logging_obj=None, + deadline_seconds=0.2, + ) + for i in range(burst) + ), + return_exceptions=True, + ) + elapsed: Final = time.monotonic() - started + + assert len(results) == burst + assert all(isinstance(result, DBLookupDeadlineExceeded) for result in results) + assert all(PrismaDBExceptionHandler.is_database_service_unavailable_error(result) for result in results) + assert elapsed < 3 + prisma.attempt_db_reconnect.assert_not_awaited() + + recovered: Final = _InFlightCountingPrisma() + after: Final = await asyncio.wait_for( + asyncio.gather( + *( + _fetch_key_object_from_db_with_reconnect( + hashed_token=f"after-{i}", + prisma_client=recovered, # pyright: ignore[reportArgumentType] # fake stands in for PrismaClient + parent_otel_span=None, + proxy_logging_obj=None, + ) + for i in range(PROXY_DB_LOOKUP_MAX_CONCURRENCY) + ) + ), + timeout=5, + ) + assert {r.token for r in after if r is not None} == {f"after-{i}" for i in range(PROXY_DB_LOOKUP_MAX_CONCURRENCY)} + + +@pytest.mark.asyncio +async def test_team_lookup_fails_at_the_db_lookup_deadline_while_writes_stay_unbounded(_clear_db_lookup_stall): + """Team, user, budget, and membership reads share the key lookup's deadline through + the typed table wrappers; writes do not, since a slow write must land rather than + fail the request that already passed auth.""" + from litellm.proxy.auth.auth_checks import _team_table + from litellm.proxy.db.db_lookup_gate import DBLookupDeadlineExceeded + from litellm.repositories.table_repositories import TeamRepository + + prisma: Final = _StalledPrisma() + with patch( # test-quality-ok: lowers the module-level lookup deadline so the stalled-read test finishes fast + "litellm.proxy.db.db_lookup_gate.PROXY_DB_LOOKUP_DEADLINE_SECONDS", 0.05 + ): + started: Final = time.monotonic() + with pytest.raises(DBLookupDeadlineExceeded, match=r"team lookup did not answer within 0\.05s"): + await _get_team_db_check(team_id="stalled-team", prisma_client=prisma) # pyright: ignore[reportArgumentType] # fake stands in for PrismaClient + assert time.monotonic() - started < 2 + + written: Final = await _team_table(TeamRepository(prisma)).update( + where={"team_id": "slow-write"}, data={"spend": 1.0} + ) + + assert written == {"team_id": "slow-write"} + + def _fake_redis_cache(): fake_redis = MagicMock() fake_redis.async_get_cache = AsyncMock(return_value=None) @@ -6184,7 +6292,9 @@ async def test_get_org_object_for_request_serves_last_known_org_through_db_outag proxy_logging_obj=None, ) - with patch("litellm.proxy.proxy_server.general_settings", {}): # test-quality-ok: the outage fallback reads this module global; no dependency injection seam exists + with patch( + "litellm.proxy.proxy_server.general_settings", {} + ): # test-quality-ok: the outage fallback reads this module global; no dependency injection seam exists warm = await _lookup() assert warm is not None and warm.organization_alias == "platform-org" await user_api_key_cache.async_delete_cache("org_id:org-1:with_budget") @@ -8933,20 +9043,34 @@ async def test_access_group_model_fallback_uses_the_injected_database(channel: s reader: Final = AsyncMock(return_value=group) client: Final = MagicMock(db=MagicMock(litellm_accessgrouptable=MagicMock(find_unique=reader))) with ( - patch("litellm.proxy.proxy_server.prisma_client", None), # test-quality-ok: [TQ008] prove reads stay on the injected connection - patch("litellm.proxy.proxy_server.user_api_key_cache", UserApiKeyCache()), # test-quality-ok: [TQ008] isolate the process cache + patch( + "litellm.proxy.proxy_server.prisma_client", None + ), # test-quality-ok: [TQ008] prove reads stay on the injected connection + patch( + "litellm.proxy.proxy_server.user_api_key_cache", UserApiKeyCache() + ), # test-quality-ok: [TQ008] isolate the process cache ): if channel == "team": - assert await can_team_access_model( - model="allowed", team_object=LiteLLM_TeamTable(team_id="team-a", models=["other"], access_group_ids=["group-a"]), - llm_router=None, prisma_client=client, - ) is True + assert ( + await can_team_access_model( + model="allowed", + team_object=LiteLLM_TeamTable(team_id="team-a", models=["other"], access_group_ids=["group-a"]), + llm_router=None, + prisma_client=client, + ) + is True + ) else: - assert await can_key_call_model( - model="allowed", llm_model_list=None, - valid_token=UserAPIKeyAuth(models=["other"], access_group_ids=["group-a"]), - llm_router=None, prisma_client=client, - ) is True + assert ( + await can_key_call_model( + model="allowed", + llm_model_list=None, + valid_token=UserAPIKeyAuth(models=["other"], access_group_ids=["group-a"]), + llm_router=None, + prisma_client=client, + ) + is True + ) reader.assert_awaited_once_with(where={"access_group_id": "group-a"}) @@ -8967,6 +9091,7 @@ def test_jwt_team_role_reaches_the_gateway_token_endpoint_by_default(): litellm_proxy_roles=LiteLLM_JWTAuth(team_allowed_routes=[]), ) + def test_route_skips_budget_checks_marks_only_spend_free_routes() -> None: assert route_skips_budget_checks(route="/v1/models") is True assert route_skips_budget_checks(route="/spend/logs") is True @@ -9083,7 +9208,9 @@ async def test_team_member_budget_check_temp_budget_increase_extends_cap(): return fallback_spend with ( - patch("litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend), # test-quality-ok: [TQ008] no seam on the cross-pod spend counter + patch( + "litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend + ), # test-quality-ok: [TQ008] no seam on the cross-pod spend counter patch( # test-quality-ok: [TQ008] isolates the check from the DB fetch "litellm.proxy.auth.auth_checks.get_team_membership", new_callable=AsyncMock, @@ -9111,7 +9238,9 @@ async def test_team_member_budget_check_temp_budget_increase_extends_cap(): ), ) with ( - patch("litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend), # test-quality-ok: [TQ008] no seam on the cross-pod spend counter + patch( + "litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend + ), # test-quality-ok: [TQ008] no seam on the cross-pod spend counter patch( # test-quality-ok: [TQ008] isolates the check from the DB fetch "litellm.proxy.auth.auth_checks.get_team_membership", new_callable=AsyncMock, @@ -9173,7 +9302,9 @@ async def test_team_member_budget_check_adds_temp_increase_to_live_team_default( return fallback_spend with ( - patch("litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend), # test-quality-ok: [TQ008] no seam on the cross-pod spend counter + patch( + "litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend + ), # test-quality-ok: [TQ008] no seam on the cross-pod spend counter patch( # test-quality-ok: [TQ008] isolates the check from the DB fetch "litellm.proxy.auth.auth_checks.get_team_membership", new_callable=AsyncMock, diff --git a/tests/test_litellm/proxy/auth/test_user_api_key_auth.py b/tests/test_litellm/proxy/auth/test_user_api_key_auth.py index f03abe8f124..de669449f85 100644 --- a/tests/test_litellm/proxy/auth/test_user_api_key_auth.py +++ b/tests/test_litellm/proxy/auth/test_user_api_key_auth.py @@ -4,6 +4,7 @@ import logging import os import subprocess import sys +import time from collections.abc import Mapping from contextlib import contextmanager from datetime import datetime, timedelta, timezone @@ -186,11 +187,7 @@ async def test_disable_budget_reservation_does_not_log_per_request(caplog): general_settings={"disable_budget_reservation": True}, ) - records = [ - record - for record in caplog.records - if "disable_budget_reservation is enabled" in record.message - ] + records = [record for record in caplog.records if "disable_budget_reservation is enabled" in record.message] assert records == [] assert user_api_key_auth_obj.budget_reservation is None @@ -234,9 +231,7 @@ async def test_budget_reservation_runs_when_not_disabled(): ({}, False), ], ) -async def test_fail_closed_budget_enforcement_reaches_reservation( - general_settings, expected_flag -): +async def test_fail_closed_budget_enforcement_reaches_reservation(general_settings, expected_flag): """#33923: the strict flag must be threaded into reserve_budget_for_request so a failed reservation write can reject instead of failing open.""" user_api_key_auth_obj = UserAPIKeyAuth(token="test_token") @@ -259,10 +254,7 @@ async def test_fail_closed_budget_enforcement_reaches_reservation( general_settings=general_settings, ) - assert ( - mock_reserve.await_args.kwargs["fail_closed_budget_enforcement"] - is expected_flag - ) + assert mock_reserve.await_args.kwargs["fail_closed_budget_enforcement"] is expected_flag @pytest.mark.asyncio @@ -274,9 +266,7 @@ async def test_fail_closed_budget_enforcement_reaches_reservation( ({}, False), ], ) -async def test_apply_user_budget_to_team_keys_reaches_reservation( - general_settings, expected_flag -): +async def test_apply_user_budget_to_team_keys_reaches_reservation(general_settings, expected_flag): """The opt-in lives in general_settings but is consumed inside _get_budget_counters, so it has to be threaded through reserve_budget_for_request or the reservation path keeps exempting team keys while the read path enforces.""" @@ -300,9 +290,7 @@ async def test_apply_user_budget_to_team_keys_reaches_reservation( general_settings=general_settings, ) - assert ( - mock_reserve.await_args.kwargs["apply_user_budget_to_team_keys"] is expected_flag - ) + assert mock_reserve.await_args.kwargs["apply_user_budget_to_team_keys"] is expected_flag @pytest.mark.asyncio @@ -402,9 +390,7 @@ async def test_custom_auth_honors_key_level_model_access_restriction_allowed_wit "litellm.proxy.auth.user_api_key_auth.can_key_call_model", new_callable=AsyncMock, ) as mock_can_key, - patch( - "litellm.proxy.auth.user_api_key_auth.common_checks", new_callable=AsyncMock - ), + patch("litellm.proxy.auth.user_api_key_auth.common_checks", new_callable=AsyncMock), patch( "litellm.proxy.proxy_server.general_settings", {"custom_auth_run_common_checks": True}, @@ -435,9 +421,7 @@ async def test_custom_auth_enforces_key_model_access_from_file_route_header_with "litellm.proxy.auth.user_api_key_auth.can_key_call_model", new_callable=AsyncMock, ) as mock_can_key, - patch( - "litellm.proxy.auth.user_api_key_auth.common_checks", new_callable=AsyncMock - ), + patch("litellm.proxy.auth.user_api_key_auth.common_checks", new_callable=AsyncMock), patch( "litellm.proxy.proxy_server.general_settings", {"custom_auth_run_common_checks": True}, @@ -468,9 +452,7 @@ async def test_custom_auth_honors_key_level_model_access_restriction_denied_with "litellm.proxy.auth.user_api_key_auth.can_key_call_model", new_callable=AsyncMock, ) as mock_can_key, - patch( - "litellm.proxy.auth.user_api_key_auth.common_checks", new_callable=AsyncMock - ), + patch("litellm.proxy.auth.user_api_key_auth.common_checks", new_callable=AsyncMock), patch( "litellm.proxy.proxy_server.general_settings", {"custom_auth_run_common_checks": True}, @@ -506,9 +488,7 @@ def _proxy_server_attrs_for_custom_auth(*, user_custom_auth): mock_proxy_logging_obj = MagicMock() mock_proxy_logging_obj.internal_usage_cache = MagicMock() mock_proxy_logging_obj.internal_usage_cache.dual_cache = AsyncMock() - mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = ( - AsyncMock() - ) + mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock() mock_proxy_logging_obj.post_call_failure_hook = AsyncMock(return_value=None) return { @@ -770,9 +750,7 @@ async def test_enterprise_custom_auth_runs_post_custom_auth_checks_when_opt_in() litellm.enable_post_custom_auth_checks = original_flag -def _assert_get_api_key_with_custom_litellm_key_header( - custom_litellm_key_header, api_key, passed_in_key -): +def _assert_get_api_key_with_custom_litellm_key_header(custom_litellm_key_header, api_key, passed_in_key): assert get_api_key( custom_litellm_key_header=custom_litellm_key_header, api_key=None, @@ -829,9 +807,7 @@ def _assert_get_api_key_with_custom_litellm_key_header( ("App:LiteLLM", None, False, False), ], ) -def test_routing_selector_matches_claim_parametrized( - selector_value, claim_value, expected, split_space_delimited -): +def test_routing_selector_matches_claim_parametrized(selector_value, claim_value, expected, split_space_delimited): assert ( _routing_selector_matches_claim( selector_value=selector_value, @@ -925,10 +901,7 @@ def test_routing_selector_matches_claim_parametrized( ], ) def test_matches_routing_override_parametrized(override, token_claims, expected): - assert ( - _matches_routing_override(token_claims=token_claims, override=override) - is expected - ) + assert _matches_routing_override(token_claims=token_claims, override=override) is expected def test_get_api_key_with_custom_litellm_key_header_bearer_prefix(): @@ -1007,12 +980,9 @@ def test_team_metadata_with_tags_flows_through_jwt_auth(): ) # Verify team_metadata is set - assert ( - user_api_key_auth.team_metadata is not None - ), "team_metadata should be populated" + assert user_api_key_auth.team_metadata is not None, "team_metadata should be populated" assert user_api_key_auth.team_metadata == team_object.metadata, ( - f"team_metadata not correctly mapped. " - f"Expected: {team_object.metadata}, Got: {user_api_key_auth.team_metadata}" + f"team_metadata not correctly mapped. Expected: {team_object.metadata}, Got: {user_api_key_auth.team_metadata}" ) # Specifically verify tags are present @@ -1051,9 +1021,7 @@ def test_route_checks_is_llm_api_route(): ] for route in openai_routes: - assert RouteChecks.is_llm_api_route( - route=route - ), f"Route {route} should be identified as LLM API route" + assert RouteChecks.is_llm_api_route(route=route), f"Route {route} should be identified as LLM API route" # Test Anthropic routes anthropic_routes = [ @@ -1062,9 +1030,7 @@ def test_route_checks_is_llm_api_route(): ] for route in anthropic_routes: - assert RouteChecks.is_llm_api_route( - route=route - ), f"Route {route} should be identified as LLM API route" + assert RouteChecks.is_llm_api_route(route=route), f"Route {route} should be identified as LLM API route" # Test passthrough routes (this is the key improvement over the old route checking) passthrough_routes = [ @@ -1084,9 +1050,7 @@ def test_route_checks_is_llm_api_route(): ] for route in passthrough_routes: - assert RouteChecks.is_llm_api_route( - route=route - ), f"Route {route} should be identified as LLM API route" + assert RouteChecks.is_llm_api_route(route=route), f"Route {route} should be identified as LLM API route" # Test MCP routes mcp_routes = [ @@ -1096,9 +1060,7 @@ def test_route_checks_is_llm_api_route(): ] for route in mcp_routes: - assert RouteChecks.is_llm_api_route( - route=route - ), f"Route {route} should be identified as LLM API route" + assert RouteChecks.is_llm_api_route(route=route), f"Route {route} should be identified as LLM API route" # Test LiteLLM native RAG routes rag_routes = [ @@ -1108,9 +1070,7 @@ def test_route_checks_is_llm_api_route(): "/v1/rag/query", ] for route in rag_routes: - assert RouteChecks.is_llm_api_route( - route=route - ), f"Route {route} should be identified as LLM API route" + assert RouteChecks.is_llm_api_route(route=route), f"Route {route} should be identified as LLM API route" # Test routes with placeholders placeholder_routes = [ @@ -1125,9 +1085,7 @@ def test_route_checks_is_llm_api_route(): ] for route in placeholder_routes: - assert RouteChecks.is_llm_api_route( - route=route - ), f"Route {route} should be identified as LLM API route" + assert RouteChecks.is_llm_api_route(route=route), f"Route {route} should be identified as LLM API route" # Test Azure OpenAI routes azure_routes = [ @@ -1138,9 +1096,7 @@ def test_route_checks_is_llm_api_route(): ] for route in azure_routes: - assert RouteChecks.is_llm_api_route( - route=route - ), f"Route {route} should be identified as LLM API route" + assert RouteChecks.is_llm_api_route(route=route), f"Route {route} should be identified as LLM API route" # Test non-LLM routes (should return False) non_llm_routes = [ @@ -1159,9 +1115,7 @@ def test_route_checks_is_llm_api_route(): ] for route in non_llm_routes: - assert not RouteChecks.is_llm_api_route( - route=route - ), f"Route {route} should NOT be identified as LLM API route" + assert not RouteChecks.is_llm_api_route(route=route), f"Route {route} should NOT be identified as LLM API route" # Test invalid inputs invalid_inputs = [ @@ -1173,9 +1127,9 @@ def test_route_checks_is_llm_api_route(): ] for invalid_input in invalid_inputs: - assert not RouteChecks.is_llm_api_route( - route=invalid_input - ), f"Invalid input {invalid_input} should return False" + assert not RouteChecks.is_llm_api_route(route=invalid_input), ( + f"Invalid input {invalid_input} should return False" + ) @pytest.mark.asyncio @@ -1222,9 +1176,7 @@ async def test_proxy_admin_expired_key_from_cache(): mock_proxy_logging_obj = MagicMock() mock_proxy_logging_obj.internal_usage_cache = MagicMock() mock_proxy_logging_obj.internal_usage_cache.dual_cache = AsyncMock() - mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = ( - AsyncMock() - ) + mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock() # Mock post_call_failure_hook as async function returning None (no transformation) mock_proxy_logging_obj.post_call_failure_hook = AsyncMock(return_value=None) @@ -1261,9 +1213,7 @@ async def test_proxy_admin_expired_key_from_cache(): "jwt_handler": None, "litellm_proxy_admin_name": "admin", } - _original_values = { - attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set - } + _original_values = {attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set} try: for attr, val in _attrs_to_set.items(): setattr(_proxy_server_mod, attr, val) @@ -1287,36 +1237,30 @@ async def test_proxy_admin_expired_key_from_cache(): ) # Verify that ProxyException was raised with expired_key type - assert hasattr( - exc_info.value, "type" - ), "Exception should have 'type' attribute" - assert ( - exc_info.value.type == ProxyErrorTypes.expired_key - ), f"Expected expired_key error type, got {exc_info.value.type}" + assert hasattr(exc_info.value, "type"), "Exception should have 'type' attribute" + assert exc_info.value.type == ProxyErrorTypes.expired_key, ( + f"Expected expired_key error type, got {exc_info.value.type}" + ) assert int(exc_info.value.code) == status.HTTP_401_UNAUTHORIZED - assert "Expired Key" in str( - exc_info.value.message - ), f"Exception message should mention 'Expired Key', got: {exc_info.value.message}" + assert "Expired Key" in str(exc_info.value.message), ( + f"Exception message should mention 'Expired Key', got: {exc_info.value.message}" + ) # Verify that the param field does NOT leak the full API key (Issue #18731) # The param should be abbreviated like "sk-...XXXX" not the full plaintext key - assert ( - exc_info.value.param is not None - ), "Exception should have 'param' attribute" + assert exc_info.value.param is not None, "Exception should have 'param' attribute" assert exc_info.value.param != api_key, ( f"SECURITY: Full API key should NOT be in param field! " f"Got: {exc_info.value.param}, Expected abbreviated format like 'sk-...XXXX'" ) - assert exc_info.value.param.startswith( - "sk-..." - ), f"Param should be abbreviated to 'sk-...XXXX' format. Got: {exc_info.value.param}" + assert exc_info.value.param.startswith("sk-..."), ( + f"Param should be abbreviated to 'sk-...XXXX' format. Got: {exc_info.value.param}" + ) # Verify that cache deletion was called mock_delete_cache.assert_called_once() call_args = mock_delete_cache.call_args - assert ( - call_args[1]["hashed_token"] == hashed_key - ), "Cache deletion should be called with the hashed key" + assert call_args[1]["hashed_token"] == hashed_key, "Cache deletion should be called with the hashed key" finally: # Restore all module-level attributes so subsequent tests are not affected for attr, val in _original_values.items(): @@ -1354,9 +1298,7 @@ async def test_scim_deactivated_user_key_is_rejected(): mock_proxy_logging_obj = MagicMock() mock_proxy_logging_obj.internal_usage_cache = MagicMock() mock_proxy_logging_obj.internal_usage_cache.dual_cache = AsyncMock() - mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = ( - AsyncMock() - ) + mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock() mock_proxy_logging_obj.post_call_failure_hook = AsyncMock(return_value=None) mock_prisma_client = MagicMock() @@ -1377,9 +1319,7 @@ async def test_scim_deactivated_user_key_is_rejected(): "jwt_handler": None, "litellm_proxy_admin_name": "admin", } - _original_values = { - attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set - } + _original_values = {attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set} try: for attr, val in _attrs_to_set.items(): setattr(_proxy_server_mod, attr, val) @@ -1446,9 +1386,7 @@ async def test_cached_proxy_admin_key_sets_via_virtual_key_marker(): mock_proxy_logging_obj = MagicMock() mock_proxy_logging_obj.internal_usage_cache = MagicMock() mock_proxy_logging_obj.internal_usage_cache.dual_cache = AsyncMock() - mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = ( - AsyncMock() - ) + mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock() mock_proxy_logging_obj.post_call_failure_hook = AsyncMock(return_value=None) import litellm.proxy.proxy_server as _proxy_server_mod @@ -1467,9 +1405,7 @@ async def test_cached_proxy_admin_key_sets_via_virtual_key_marker(): "jwt_handler": None, "litellm_proxy_admin_name": "admin", } - _original_values = { - attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set - } + _original_values = {attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set} try: for attr, val in _attrs_to_set.items(): setattr(_proxy_server_mod, attr, val) @@ -1521,9 +1457,7 @@ async def test_master_key_auth_sets_via_virtual_key_marker(): mock_proxy_logging_obj = MagicMock() mock_proxy_logging_obj.internal_usage_cache = MagicMock() mock_proxy_logging_obj.internal_usage_cache.dual_cache = AsyncMock() - mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = ( - AsyncMock() - ) + mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock() mock_proxy_logging_obj.post_call_failure_hook = AsyncMock(return_value=None) import litellm.proxy.proxy_server as _proxy_server_mod @@ -1542,9 +1476,7 @@ async def test_master_key_auth_sets_via_virtual_key_marker(): "jwt_handler": None, "litellm_proxy_admin_name": "admin", } - _original_values = { - attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set - } + _original_values = {attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set} try: for attr, val in _attrs_to_set.items(): setattr(_proxy_server_mod, attr, val) @@ -1597,9 +1529,7 @@ async def test_db_virtual_key_auth_sets_via_virtual_key_marker(): mock_proxy_logging_obj = MagicMock() mock_proxy_logging_obj.internal_usage_cache = MagicMock() mock_proxy_logging_obj.internal_usage_cache.dual_cache = AsyncMock() - mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = ( - AsyncMock() - ) + mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock() mock_proxy_logging_obj.post_call_failure_hook = AsyncMock(return_value=None) mock_prisma_client = MagicMock() @@ -1620,9 +1550,7 @@ async def test_db_virtual_key_auth_sets_via_virtual_key_marker(): "jwt_handler": None, "litellm_proxy_admin_name": "admin", } - _original_values = { - attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set - } + _original_values = {attr: getattr(_proxy_server_mod, attr, None) for attr in _attrs_to_set} try: for attr, val in _attrs_to_set.items(): setattr(_proxy_server_mod, attr, val) @@ -2153,7 +2081,10 @@ async def test_auto_register_first_request_propagates_user_email(active: bool) - patch("litellm.proxy.proxy_server.master_key", "sk-master"), patch("litellm.proxy.proxy_server.prisma_client", prisma_client), patch("litellm.proxy.proxy_server.user_api_key_cache", user_api_key_cache), - patch("litellm.proxy.proxy_server.proxy_logging_obj", MagicMock(post_call_failure_hook=AsyncMock(return_value=None))), + patch( + "litellm.proxy.proxy_server.proxy_logging_obj", + MagicMock(post_call_failure_hook=AsyncMock(return_value=None)), + ), patch("litellm.proxy.proxy_server.jwt_handler", jwt_handler), patch( "litellm.proxy.auth.user_api_key_auth._resolve_jwt_to_virtual_key", @@ -2218,7 +2149,9 @@ async def test_auto_register_stamps_new_key_with_jwt_agent_id(): plaintext = "sk-auto-registered-agent" token_hash = hash_token(plaintext) persisted_principal = IdentityStore._principal_from_key( - UserAPIKeyAuth(token=token_hash, user_id="validated-user", team_id="validated-team", agent_id="canonical-agent-id"), + UserAPIKeyAuth( + token=token_hash, user_id="validated-user", team_id="validated-team", agent_id="canonical-agent-id" + ), auth_method=AuthMethod.API_KEY, credential_ref=CredentialRef(token_id=token_hash), ) @@ -2433,10 +2366,7 @@ class TestJWTOAuth2Coexistence: def test_is_jwt_detects_jwt_tokens(self): """JWT tokens have 3 dot-separated parts.""" assert JWTHandler.is_jwt("header.payload.signature") is True - assert ( - JWTHandler.is_jwt("eyJhbGciOiJSUzI1NiJ9.eyJzdWIiOiJ1c2VyMSJ9.sig123") - is True - ) + assert JWTHandler.is_jwt("eyJhbGciOiJSUzI1NiJ9.eyJzdWIiOiJ1c2VyMSJ9.sig123") is True def test_is_jwt_rejects_opaque_tokens(self): """Opaque OAuth2 tokens do not have 3 dot-separated parts.""" @@ -2545,10 +2475,7 @@ class TestJWTOAuth2Coexistence: assert exc_info.value.type == ProxyErrorTypes.auth_error assert exc_info.value.code == "403" - assert ( - "Oauth2 token validation is only available for premium users" - in exc_info.value.message - ) + assert "Oauth2 token validation is only available for premium users" in exc_info.value.message mock_oauth2.assert_not_called() @pytest.mark.asyncio @@ -2740,9 +2667,7 @@ class TestJWTOAuth2Coexistence: assert mock_auto_register.call_args.kwargs["team_id"] == "validated-team" assert mock_auto_register.call_args.kwargs["user_id"] == "validated-user" assert mock_auto_register.call_args.kwargs["org_id"] == "validated-org" - assert ( - mock_auto_register.call_args.kwargs["end_user_id"] == "validated-end-user" - ) + assert mock_auto_register.call_args.kwargs["end_user_id"] == "validated-end-user" assert result.org_id == "validated-org" assert result.user_email == "validated@example.com" @@ -2820,10 +2745,7 @@ class TestJWTOAuth2Coexistence: assert result.user_id == "mapped-user" assert result.user_email == "mapped@example.com" - assert ( - mock_get_user_object.call_args_list[0].kwargs["user_email"] - == "mapped@example.com" - ) + assert mock_get_user_object.call_args_list[0].kwargs["user_email"] == "mapped@example.com" @pytest.mark.asyncio async def test_mapped_virtual_key_does_not_backfill_mismatched_owner(self): @@ -2899,8 +2821,7 @@ class TestJWTOAuth2Coexistence: assert result.user_id == "other-owner" assert result.user_email is None assert all( - call.kwargs.get("user_email") != "principal@example.com" - for call in mock_get_user_object.call_args_list + call.kwargs.get("user_email") != "principal@example.com" for call in mock_get_user_object.call_args_list ) @pytest.mark.asyncio @@ -3705,9 +3626,7 @@ async def test_user_api_key_auth_builder_no_blocking_calls(): mock_proxy_logging_obj = MagicMock() mock_proxy_logging_obj.internal_usage_cache = MagicMock() mock_proxy_logging_obj.internal_usage_cache.dual_cache = AsyncMock() - mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = ( - AsyncMock() - ) + mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock() mock_proxy_logging_obj.post_call_failure_hook = AsyncMock(return_value=None) import litellm.proxy.proxy_server as _proxy_server_mod @@ -3839,9 +3758,7 @@ async def test_team_metadata_refreshed_from_team_object_during_auth(): mock_proxy_logging_obj = MagicMock() mock_proxy_logging_obj.internal_usage_cache = MagicMock() mock_proxy_logging_obj.internal_usage_cache.dual_cache = AsyncMock() - mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = ( - AsyncMock() - ) + mock_proxy_logging_obj.internal_usage_cache.dual_cache.async_delete_cache = AsyncMock() mock_proxy_logging_obj.post_call_failure_hook = AsyncMock(return_value=None) import litellm.proxy.proxy_server as _proxy_server_mod @@ -3891,9 +3808,9 @@ async def test_team_metadata_refreshed_from_team_object_during_auth(): request_data={}, ) - assert result.team_metadata == { - "guardrails": ["test-guardrail-333"] - }, f"team_metadata was not updated from fresh team object. Got: {result.team_metadata}" + assert result.team_metadata == {"guardrails": ["test-guardrail-333"]}, ( + f"team_metadata was not updated from fresh team object. Got: {result.team_metadata}" + ) finally: for k, v in _originals.items(): @@ -4218,9 +4135,7 @@ async def test_auth_flow_fallback_team_object_permission_none_when_unreadable(): # --------------------------------------------------------------------------- -def _proxy_attrs_for_centralized_checks( - user_custom_auth=None, flag=False, master_key="sk-test-master" -): +def _proxy_attrs_for_centralized_checks(user_custom_auth=None, flag=False, master_key="sk-test-master"): """Build the minimal proxy_server module attributes that _run_centralized_common_checks reads. @@ -4430,9 +4345,7 @@ async def _run_centralized_checks_with_key_end_user_budget( request = Request(scope={"type": "http"}) request._url = URL(url="/chat/completions") attrs = { - **_proxy_attrs_for_centralized_checks( - user_custom_auth=AsyncMock() if custom_auth else None, flag=custom_auth - ), + **_proxy_attrs_for_centralized_checks(user_custom_auth=AsyncMock() if custom_auth else None, flag=custom_auth), "prisma_client": prisma_client, "user_api_key_cache": user_api_key_cache if user_api_key_cache is not None else DualCache(), "proxy_logging_obj": proxy_logging_obj, @@ -4623,7 +4536,9 @@ async def test_centralized_common_checks_enforces_team_model_max_budget_from_the for k, v in attrs.items(): setattr(_proxy_server_mod, k, v) with ( - patch("litellm.proxy.auth.user_api_key_auth.common_checks", new_callable=AsyncMock), # test-quality-ok: stubs the sibling check so only the team model-budget gate is under test + patch( + "litellm.proxy.auth.user_api_key_auth.common_checks", new_callable=AsyncMock + ), # test-quality-ok: stubs the sibling check so only the team model-budget gate is under test patch( # test-quality-ok: stubs the budget reservation so only the team model-budget gate is under test "litellm.proxy.auth.user_api_key_auth._reserve_budget_after_common_checks", new_callable=AsyncMock, @@ -4656,9 +4571,7 @@ async def test_centralized_common_checks_skipped_for_custom_auth_without_flag(): request = Request(scope={"type": "http"}) request._url = URL(url="/chat/completions") - attrs = _proxy_attrs_for_centralized_checks( - user_custom_auth=AsyncMock(), flag=False - ) + attrs = _proxy_attrs_for_centralized_checks(user_custom_auth=AsyncMock(), flag=False) originals = {a: getattr(_proxy_server_mod, a, None) for a in attrs} try: for k, v in attrs.items(): @@ -5063,9 +4976,7 @@ async def test_centralized_common_checks_reserves_request_end_user_budget(): "applied_adjustment": 0.0, } ] - assert counter_cache.in_memory_cache.get_cache( - key="spend:end_user:alice" - ) == pytest.approx(0.6) + assert counter_cache.in_memory_cache.get_cache(key="spend:end_user:alice") == pytest.approx(0.6) @pytest.mark.asyncio @@ -5080,9 +4991,7 @@ async def test_centralized_common_checks_short_circuits_when_master_key_unset(): from litellm.proxy._types import LitellmUserRoles - token = UserAPIKeyAuth( - api_key="sk-test", user_id="u", user_role=LitellmUserRoles.INTERNAL_USER - ) + token = UserAPIKeyAuth(api_key="sk-test", user_id="u", user_role=LitellmUserRoles.INTERNAL_USER) request = Request(scope={"type": "http"}) request._url = URL(url="/get/config/callbacks") @@ -5883,9 +5792,7 @@ async def test_centralized_common_checks_user_http_exception_isolates_to_user_on request._url = URL(url="/chat/completions") request._body = json.dumps({"user": "alice", "model": "gpt-4o"}).encode() - fetched_team = LiteLLM_TeamTableCachedObj( - team_id="t1", max_budget=20.0, models=["gpt-4o"] - ) + fetched_team = LiteLLM_TeamTableCachedObj(team_id="t1", max_budget=20.0, models=["gpt-4o"]) fetched_end_user = LiteLLM_EndUserTable(user_id="alice", blocked=False, spend=1.0) fetched_project = LiteLLM_ProjectTableCachedObj( project_id="proj-1", @@ -6014,10 +5921,46 @@ async def test_centralized_common_checks_backfills_org_id_from_team(key_org_id, ("org-pinned", None, None, "preset", None, "success", False, False, "org-pinned", "preset", (None, None, None)), ("org-view", None, None, None, 3, "success", False, False, "org-view", None, (None, None, 3)), ("org-missing", None, None, None, None, "missing", False, False, "org-missing", None, (None, None, None)), - ("org-db-failure-allowed", None, None, None, None, "db_failure", True, False, "org-db-failure-allowed", None, (None, None, None)), - ("org-db-failure-denied", None, None, None, None, "db_failure", False, True, "org-db-failure-denied", None, (None, None, None)), + ( + "org-db-failure-allowed", + None, + None, + None, + None, + "db_failure", + True, + False, + "org-db-failure-allowed", + None, + (None, None, None), + ), + ( + "org-db-failure-denied", + None, + None, + None, + None, + "db_failure", + False, + True, + "org-db-failure-denied", + None, + (None, None, None), + ), ("org-bad-row", None, None, None, None, "bad_row", False, False, "org-bad-row", None, (None, None, None)), - ("org-nobudget", None, None, None, None, "no_budget", False, False, "org-nobudget", "acme-org", (None, None, None)), + ( + "org-nobudget", + None, + None, + None, + None, + "no_budget", + False, + False, + "org-nobudget", + "acme-org", + (None, None, None), + ), ], ) async def test_centralized_common_checks_inherits_org_identity( @@ -6326,9 +6269,7 @@ async def test_user_api_key_auth_sets_end_user_id_when_builder_skips_it(): } ) request._url = URL(url="/chat/completions") - request._body = json.dumps( - {"model": "gpt-4o", "user": "alice@example.com"} - ).encode() + request._body = json.dumps({"model": "gpt-4o", "user": "alice@example.com"}).encode() attrs = _proxy_attrs_for_centralized_checks(user_custom_auth=None) originals = {a: getattr(_proxy_server_mod, a, None) for a in attrs} @@ -6372,9 +6313,7 @@ async def test_user_api_key_auth_does_not_overwrite_end_user_id_set_by_builder() import litellm.proxy.proxy_server as _proxy_server_mod - builder_token = UserAPIKeyAuth( - api_key="sk-test", user_id="u1", end_user_id="builder-resolved-id" - ) + builder_token = UserAPIKeyAuth(api_key="sk-test", user_id="u1", end_user_id="builder-resolved-id") request = Request( scope={ @@ -6384,9 +6323,7 @@ async def test_user_api_key_auth_does_not_overwrite_end_user_id_set_by_builder() } ) request._url = URL(url="/chat/completions") - request._body = json.dumps( - {"model": "gpt-4o", "user": "different-id-from-body"} - ).encode() + request._body = json.dumps({"model": "gpt-4o", "user": "different-id-from-body"}).encode() attrs = _proxy_attrs_for_centralized_checks(user_custom_auth=None) originals = {a: getattr(_proxy_server_mod, a, None) for a in attrs} @@ -6717,6 +6654,83 @@ async def _run_builder_with_key_lookup(get_key_object_mock): setattr(_proxy_server_mod, k, v) +class _StalledKeyLookupPrisma: + """A database whose connection answers the readiness ping but whose key lookups + never return, which is what the incident's locked table looked like.""" + + def __init__(self) -> None: + self.health_check = AsyncMock(return_value=True) + self.attempt_db_reconnect = AsyncMock(return_value=True) + self.db = MagicMock() + + async def get_data(self, token: str, table_name: str, parent_otel_span: None, proxy_logging_obj: None) -> None: + await asyncio.Event().wait() + + +@pytest.mark.asyncio +async def test_burst_against_a_stalled_db_fails_fast_with_503_and_turns_readiness_red(): + """The incident, end to end: N requests into a proxy whose database stalls used to + park in the pod with readiness green until it OOMed. Now every one of them fails + within the lookup deadline as a 503, and the next readiness probe takes the pod out + of rotation.""" + import httpx + from fastapi import Depends, FastAPI + + import litellm.proxy.health_endpoints._health_endpoints as health_endpoints + import litellm.proxy.proxy_server as _proxy_server_mod + from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker + + app = FastAPI() + + @app.post("/chat/completions", dependencies=[Depends(user_api_key_auth)]) + async def chat_completions() -> Mapping[str, bool]: + return {"served": True} + + app.include_router(health_endpoints.router) + app.add_exception_handler(ProxyException, _proxy_server_mod.openai_exception_handler) + + attrs = {**_proxy_attrs_for_db_lookup(), "prisma_client": _StalledKeyLookupPrisma()} + originals = {a: getattr(_proxy_server_mod, a, None) for a in attrs} + health_endpoints.db_health_cache = {"status": "unknown", "last_updated": datetime.now() - timedelta(seconds=60)} + db_lookup_stall_tracker.clear() + burst = 60 + try: + for k, v in attrs.items(): + setattr(_proxy_server_mod, k, v) + with ( + patch( # test-quality-ok: lowers the module-level lookup deadline so the stalled burst finishes fast + "litellm.proxy.db.db_lookup_gate.PROXY_DB_LOOKUP_DEADLINE_SECONDS", 0.2 + ), + patch("litellm.proxy.auth.auth_exception_handler.seed_request_identity"), + ): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://t") as client: + started = time.monotonic() + responses = await asyncio.gather( + *( + client.post( + "/chat/completions", + json={"model": "gpt-5.5", "messages": [{"role": "user", "content": "hi"}]}, + headers={"Authorization": f"Bearer sk-stalled-{i}"}, + ) + for i in range(burst) + ) + ) + elapsed = time.monotonic() - started + readiness = await client.get("/health/readiness") + finally: + for k, v in originals.items(): + setattr(_proxy_server_mod, k, v) + db_lookup_stall_tracker.clear() + + assert len(responses) == burst + assert {r.status_code for r in responses} == {status.HTTP_503_SERVICE_UNAVAILABLE} + assert {r.json()["error"]["type"] for r in responses} == {ProxyErrorTypes.no_db_connection.value} + assert all("temporarily unreachable" in r.json()["error"]["message"] for r in responses) + assert elapsed < 5 + assert readiness.status_code == status.HTTP_503_SERVICE_UNAVAILABLE + assert readiness.json()["db"] == "stalled" + + @pytest.mark.asyncio async def test_builder_returns_503_when_db_lookup_raises_infra_error(): """End-to-end: a DB infrastructure failure during the key lookup must @@ -6784,9 +6798,7 @@ def _mint_cli_session_token(monkeypatch, *, user_id="cli-admin"): models=["gpt-3.5-turbo"], max_budget=100.0, ) - return ExperimentalUIJWTToken.get_cli_jwt_auth_token( - user_info, team_id="cli-team", team_alias="cli-team-alias" - ) + return ExperimentalUIJWTToken.get_cli_jwt_auth_token(user_info, team_id="cli-team", team_alias="cli-team-alias") @pytest.mark.asyncio @@ -6836,7 +6848,7 @@ async def test_random_non_sk_token_is_rejected(monkeypatch): patch("litellm.proxy.proxy_server.master_key", "sk-master"), patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), ): - with pytest.raises(Exception, match='LiteLLM Virtual Key expected\\.') as exc_info: + with pytest.raises(Exception, match="LiteLLM Virtual Key expected\\.") as exc_info: await user_api_key_auth( request=mock_request, api_key="Bearer not-a-real-token", @@ -6915,9 +6927,7 @@ async def test_non_admin_cli_session_token_reaches_production_auth_path(monkeypa user_role=LitellmUserRoles.INTERNAL_USER.value, models=[], ) - cli_token = ExperimentalUIJWTToken.get_cli_jwt_auth_token( - user_info, team_id="team-abc", team_alias="my-team" - ) + cli_token = ExperimentalUIJWTToken.get_cli_jwt_auth_token(user_info, team_id="team-abc", team_alias="my-team") import litellm.proxy.proxy_server as _proxy_server_mod from fastapi import Request @@ -7224,7 +7234,7 @@ async def test_real_jwt_still_requires_license_when_jwt_auth_enabled(monkeypatch patch("litellm.proxy.proxy_server.master_key", "sk-master"), patch("litellm.proxy.proxy_server.prisma_client", None), ): - with pytest.raises(Exception, match='JWT Auth is an enterprise only feature\\. You must be a') as exc_info: + with pytest.raises(Exception, match="JWT Auth is an enterprise only feature\\. You must be a") as exc_info: await user_api_key_auth( request=mock_request, api_key=f"Bearer {jwt_token}", @@ -7263,13 +7273,9 @@ async def test_auth_does_not_rewrite_cached_key_object_back_into_cache(): metadata={"model_rpm_limit": {"gpt-5.4-mini": 3}}, last_refreshed_at=1000.0, ) - await key_cache.async_set_cache( - key=hashed_key, value=stale_token, model_type=UserAPIKeyAuth - ) + await key_cache.async_set_cache(key=hashed_key, value=stale_token, model_type=UserAPIKeyAuth) - fetch_from_db = AsyncMock( - side_effect=AssertionError("cache-hit auth must not touch the DB") - ) + fetch_from_db = AsyncMock(side_effect=AssertionError("cache-hit auth must not touch the DB")) proxy_logging_obj = MagicMock() proxy_logging_obj.internal_usage_cache = MagicMock() @@ -7316,9 +7322,7 @@ async def test_auth_does_not_rewrite_cached_key_object_back_into_cache(): assert result.token == hashed_key fetch_from_db.assert_not_called() - cached_after = await key_cache.async_get_cache( - key=hashed_key, model_type=UserAPIKeyAuth - ) + cached_after = await key_cache.async_get_cache(key=hashed_key, model_type=UserAPIKeyAuth) assert cached_after is not None assert cached_after.last_refreshed_at == 1000.0 assert cached_after.metadata == {"model_rpm_limit": {"gpt-5.4-mini": 3}} @@ -7394,7 +7398,9 @@ class TestJWTAuthUserEmail: assert result.user_email == "resolved@example.com" @pytest.mark.asyncio - @pytest.mark.parametrize("route", ["/mcp-rest/tools/list", "/mcp-rest/tools/call", "/v1/chat/completions", "/user/info"]) + @pytest.mark.parametrize( + "route", ["/mcp-rest/tools/list", "/mcp-rest/tools/call", "/v1/chat/completions", "/user/info"] + ) @pytest.mark.parametrize("active", [False, True, None, "false", 0]) @pytest.mark.parametrize("is_admin", [False, True]) async def test_jwt_auth_rejects_deactivated_user( @@ -7469,9 +7475,7 @@ class TestCheckKeyModelBudgetWithFallback: @pytest.mark.asyncio async def test_within_budget_does_not_reroute(self): - valid_token = UserAPIKeyAuth( - token="test-key", budget_fallbacks={"gpt-4o": ["gpt-4o-mini"]} - ) + valid_token = UserAPIKeyAuth(token="test-key", budget_fallbacks={"gpt-4o": ["gpt-4o-mini"]}) limiter = AsyncMock() limiter.is_key_within_model_budget.return_value = True request_data = {"model": "gpt-4o"} @@ -7496,9 +7500,7 @@ class TestCheckKeyModelBudgetWithFallback: budget_fallbacks={"gpt-4o": ["gpt-4o-mini", "claude-haiku"]}, ) limiter = AsyncMock() - limiter.is_key_within_model_budget.side_effect = litellm.BudgetExceededError( - current_cost=10, max_budget=5 - ) + limiter.is_key_within_model_budget.side_effect = litellm.BudgetExceededError(current_cost=10, max_budget=5) limiter.get_fallback_model_within_budget.return_value = "gpt-4o-mini" request_data = {"model": "gpt-4o"} request = self._make_request() @@ -7512,9 +7514,7 @@ class TestCheckKeyModelBudgetWithFallback: ) assert request_data["model"] == "gpt-4o-mini" - limiter.get_fallback_model_within_budget.assert_awaited_once_with( - user_api_key_dict=valid_token, model="gpt-4o" - ) + limiter.get_fallback_model_within_budget.assert_awaited_once_with(user_api_key_dict=valid_token, model="gpt-4o") # the rerouted model must be visible to a later, separate # `_read_request_body` call on the same `request` (route handlers # re-parse the body from this cache instead of reusing the dict). @@ -7523,9 +7523,7 @@ class TestCheckKeyModelBudgetWithFallback: @pytest.mark.asyncio async def test_raises_when_every_fallback_also_exceeded(self): - valid_token = UserAPIKeyAuth( - token="test-key", budget_fallbacks={"gpt-4o": ["gpt-4o-mini"]} - ) + valid_token = UserAPIKeyAuth(token="test-key", budget_fallbacks={"gpt-4o": ["gpt-4o-mini"]}) limiter = AsyncMock() original_error = litellm.BudgetExceededError(current_cost=10, max_budget=5) limiter.is_key_within_model_budget.side_effect = original_error @@ -7595,9 +7593,7 @@ class TestCheckKeyModelBudgetWithFallback: budget_fallbacks={"gpt-4o": ["gpt-4o-mini"]}, ) limiter = AsyncMock() - limiter.is_key_within_model_budget.side_effect = litellm.BudgetExceededError( - current_cost=10, max_budget=5 - ) + limiter.is_key_within_model_budget.side_effect = litellm.BudgetExceededError(current_cost=10, max_budget=5) limiter.get_fallback_model_within_budget.return_value = "gpt-4o-mini" request_data = {"model": "gpt-4o"} request = self._make_request() @@ -7665,9 +7661,7 @@ class TestCheckKeyModelBudgetWithFallback: budget_fallbacks={"gpt-4o": ["gpt-4o-mini"]}, ) limiter = AsyncMock() - limiter.is_key_within_model_budget.side_effect = litellm.BudgetExceededError( - current_cost=10, max_budget=5 - ) + limiter.is_key_within_model_budget.side_effect = litellm.BudgetExceededError(current_cost=10, max_budget=5) limiter.get_fallback_model_within_budget.return_value = "gpt-4o-mini" request_data = {"model": "gpt-4o"} request = self._make_request() @@ -7747,9 +7741,7 @@ async def test_global_proxy_spend_reads_resettable_proxy_budget_row(): ) assert result == 42.5 - prisma_client.db.litellm_usertable.find_unique.assert_awaited_once_with( - where={"user_id": "litellm-proxy-budget"} - ) + prisma_client.db.litellm_usertable.find_unique.assert_awaited_once_with(where={"user_id": "litellm-proxy-budget"}) @pytest.mark.asyncio @@ -8124,9 +8116,7 @@ async def test_jwt_shaped_key_error_names_enable_jwt_auth_when_disabled(): Prometheus invalid-key filter and the admin UI both substring-match it. Keys that are not JWT-shaped must not pick up the hint. """ - jwt_error = await _proxy_exception_for_key( - "eyJhbGciOiJSUzI1NiJ9.eyJzdWIiOiJzdmMtMSJ9.c2lnbmF0dXJl", {}, True - ) + jwt_error = await _proxy_exception_for_key("eyJhbGciOiJSUzI1NiJ9.eyJzdWIiOiJzdmMtMSJ9.c2lnbmF0dXJl", {}, True) assert jwt_error.code == "401" assert "enable_jwt_auth" in jwt_error.message @@ -8136,9 +8126,7 @@ async def test_jwt_shaped_key_error_names_enable_jwt_auth_when_disabled(): assert "is a JWT" not in jwt_error.message opaque_error = await _proxy_exception_for_key("not-a-jwt-at-all", {}, True) - two_segment_error = await _proxy_exception_for_key( - "eyJhbGciOiJSUzI1NiJ9.eyJzdWIiOiJzdmMtMSJ9", {}, True - ) + two_segment_error = await _proxy_exception_for_key("eyJhbGciOiJSUzI1NiJ9.eyJzdWIiOiJzdmMtMSJ9", {}, True) assert "enable_jwt_auth" not in opaque_error.message assert "enable_jwt_auth" not in two_segment_error.message @@ -8167,9 +8155,7 @@ class TestLitellmReceivedAtStamping: on OTEL being configured to see a true request-arrival timestamp.""" def test_stamped_even_when_otel_is_not_configured(self, monkeypatch): - monkeypatch.setattr( - "litellm.proxy.proxy_server.open_telemetry_logger", None - ) + monkeypatch.setattr("litellm.proxy.proxy_server.open_telemetry_logger", None) request = MagicMock() request.state = SimpleNamespace() @@ -8201,7 +8187,7 @@ class TestLitellmReceivedAtStamping: _RECORDING_DDTRACE = dedent( - ''' + """ import functools import inspect @@ -8250,11 +8236,11 @@ _RECORDING_DDTRACE = dedent( tracer = _Tracer() - ''' + """ ) _DDTRACE_AUTH_PROBE = dedent( - ''' + """ import asyncio import json @@ -8293,7 +8279,7 @@ _DDTRACE_AUTH_PROBE = dedent( asyncio.run(main()) - ''' + """ ) @@ -8440,25 +8426,43 @@ async def test_jwt_builder_returns_every_team_grant_the_key_path_gets(is_proxy_a @pytest.mark.asyncio -@pytest.mark.parametrize("route", ["/v1/messages", "/messages", "/v1/chat/completions", "/chat/completions", "/v1/responses", "/responses"]) +@pytest.mark.parametrize( + "route", ["/v1/messages", "/messages", "/v1/chat/completions", "/chat/completions", "/v1/responses", "/responses"] +) async def test_claude_view_normalizes_before_model_access(monkeypatch, route): from starlette.requests import Request from litellm.proxy.auth.user_api_key_auth import _enforce_key_and_fallback_model_access source = "foo[1m]" encoded = "claude-router-" + source.encode().hex() + "[1m]" - router = litellm.Router(model_list=[{"model_name": source, "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}}]) + router = litellm.Router( + model_list=[{"model_name": source, "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}}] + ) monkeypatch.setattr(litellm.proxy.proxy_server, "llm_router", router) data = {"model": encoded, "messages": [{"role": "user", "content": "hi"}]} request = Request({"type": "http", "method": "POST", "path": route, "headers": [], "query_string": b""}) token = UserAPIKeyAuth(models=[source]) - await _enforce_key_and_fallback_model_access(valid_token=token, request_data=data, route=route, request=request, llm_model_list=router.model_list, llm_router=router) + await _enforce_key_and_fallback_model_access( + valid_token=token, + request_data=data, + route=route, + request=request, + llm_model_list=router.model_list, + llm_router=router, + ) assert data["model"] == source assert (await request.json())["model"] == source assert json.loads(await request.body())["model"] == source assert request.scope["parsed_body"][1]["model"] == source with pytest.raises(ProxyException): - await _enforce_key_and_fallback_model_access(valid_token=UserAPIKeyAuth(models=["other"]), request_data=data, route=route, request=request, llm_model_list=router.model_list, llm_router=router) + await _enforce_key_and_fallback_model_access( + valid_token=UserAPIKeyAuth(models=["other"]), + request_data=data, + route=route, + request=request, + llm_model_list=router.model_list, + llm_router=router, + ) @pytest.mark.asyncio @@ -8470,10 +8474,18 @@ async def test_claude_view_never_reinterprets_explicit_names(monkeypatch, layer) encoded = "claude-router-666f6f" names = ("foo", "other", encoded) if layer == "literal" else ("foo", "other") alias = {encoded: "other"} - router = litellm.Router(model_list=[{"model_name": name, "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}} for name in names], model_group_alias=alias if layer == "router" else None) + router = litellm.Router( + model_list=[ + {"model_name": name, "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-fake"}} for name in names + ], + model_group_alias=alias if layer == "router" else None, + ) monkeypatch.setattr(litellm.proxy.proxy_server, "llm_router", router) monkeypatch.setattr(litellm, "model_alias_map", alias if layer == "global" else {}) - token = UserAPIKeyAuth(aliases=alias if layer == "key" else {}, router_settings={"model_group_alias": alias} if layer == "hierarchical" else None) + token = UserAPIKeyAuth( + aliases=alias if layer == "key" else {}, + router_settings={"model_group_alias": alias} if layer == "hierarchical" else None, + ) data = {"model": encoded} request = Request({"type": "http", "method": "POST", "path": "/v1/messages", "headers": [], "query_string": b""}) await _normalize_claude_model(data, token, request, "/v1/messages") diff --git a/tests/test_litellm/proxy/db/test_db_lookup_gate.py b/tests/test_litellm/proxy/db/test_db_lookup_gate.py new file mode 100644 index 00000000000..68903170840 --- /dev/null +++ b/tests/test_litellm/proxy/db/test_db_lookup_gate.py @@ -0,0 +1,111 @@ +import asyncio +import time +from typing import Final + +import pytest + +from litellm.proxy.db.db_lookup_gate import DBLookupDeadlineExceeded, DBLookupStallTracker, bounded_db_lookup + + +async def _never_answers() -> None: + await asyncio.Event().wait() + + +class _FakeClock: + def __init__(self) -> None: + self.now = 1000.0 + + def __call__(self) -> float: + return self.now + + +@pytest.mark.asyncio +async def test_bounded_db_lookup_fails_a_stalled_lookup_at_the_deadline_and_records_the_hit(): + tracker: Final = DBLookupStallTracker() + started: Final = time.monotonic() + + with pytest.raises(DBLookupDeadlineExceeded) as exc_info: + await bounded_db_lookup(_never_answers(), name="team", deadline_seconds=0.05, tracker=tracker) + + assert time.monotonic() - started < 2 + assert exc_info.value.lookup == "team" + assert exc_info.value.deadline_seconds == 0.05 + assert str(exc_info.value) == "team lookup did not answer within 0.05s" + assert isinstance(exc_info.value, asyncio.TimeoutError) + assert tracker.stalled_within(30) is True + + +@pytest.mark.asyncio +async def test_bounded_db_lookup_returns_a_prompt_answer_without_recording_a_stall(): + tracker: Final = DBLookupStallTracker() + + async def answers() -> str: + return "row" + + assert await bounded_db_lookup(answers(), name="key", deadline_seconds=0.05, tracker=tracker) == "row" + assert tracker.stalled_within(30) is False + + +@pytest.mark.asyncio +async def test_bounded_db_lookup_fails_a_whole_stalled_burst_within_one_deadline(): + tracker: Final = DBLookupStallTracker() + burst: Final = 200 + started: Final = time.monotonic() + + results: Final = await asyncio.gather( + *( + bounded_db_lookup(_never_answers(), name=f"key-{i}", deadline_seconds=0.1, tracker=tracker) + for i in range(burst) + ), + return_exceptions=True, + ) + + assert time.monotonic() - started < 2 + assert len(results) == burst + assert all(isinstance(result, DBLookupDeadlineExceeded) for result in results) + assert tracker.stalled_within(30) is True + + +@pytest.mark.asyncio +async def test_bounded_db_lookup_fails_at_the_deadline_even_when_the_lookup_absorbs_the_cancel(): + tracker: Final = DBLookupStallTracker() + absorbed: Final = asyncio.Event() + let_go: Final = asyncio.Event() + + async def absorbs_the_cancel() -> str: + try: + await asyncio.Event().wait() + except asyncio.CancelledError: + absorbed.set() + await let_go.wait() + return "late row" + + started: Final = time.monotonic() + with pytest.raises(DBLookupDeadlineExceeded): + await asyncio.wait_for( + bounded_db_lookup(absorbs_the_cancel(), name="key", deadline_seconds=0.05, tracker=tracker), + timeout=2, + ) + + assert time.monotonic() - started < 1 + assert tracker.stalled_within(30) is True + await asyncio.wait_for(absorbed.wait(), timeout=1) + let_go.set() + await asyncio.sleep(0) + + +def test_stall_tracker_reports_a_stall_only_inside_the_window(): + clock: Final = _FakeClock() + tracker: Final = DBLookupStallTracker(clock=clock) + + assert tracker.stalled_within(30) is False + tracker.record_hit() + assert tracker.stalled_within(30) is True + assert tracker.stalled_within(0) is False + clock.now += 29.9 + assert tracker.stalled_within(30) is True + clock.now += 0.2 + assert tracker.stalled_within(30) is False + tracker.record_hit() + tracker.clear() + assert tracker.stalled_within(30) is False diff --git a/tests/test_litellm/proxy/db/test_exception_handler.py b/tests/test_litellm/proxy/db/test_exception_handler.py index 613ca847115..09f4d294ad0 100644 --- a/tests/test_litellm/proxy/db/test_exception_handler.py +++ b/tests/test_litellm/proxy/db/test_exception_handler.py @@ -774,3 +774,18 @@ def test_connection_error_answers_when_prisma_is_mocked_after_import(): with patch.dict(sys.modules, {"prisma": MagicMock()}): assert PrismaDBExceptionHandler.is_database_connection_error(Exception("x")) is False assert PrismaDBExceptionHandler.is_database_connection_error(httpx.ConnectError("refused")) is True + + +def test_db_lookup_deadline_is_a_connection_and_unavailability_error_but_never_a_transport_error(): + """A lookup that hit its deadline fails the request as a 503 and counts as a + DB outage for ``allow_requests_on_db_unavailable``, but it must not be read + as a broken transport: that would send every parked request into + ``attempt_db_reconnect`` and turn a slow database into a reconnect storm.""" + from litellm.proxy.db.db_lookup_gate import DBLookupDeadlineExceeded + + deadline: Final = DBLookupDeadlineExceeded("key", 10.0) + + assert PrismaDBExceptionHandler.is_database_connection_error(deadline) is True + assert PrismaDBExceptionHandler.is_database_service_unavailable_error(deadline) is True + assert PrismaDBExceptionHandler.is_database_transport_error(deadline) is False + assert "temporarily unreachable" in PrismaDBExceptionHandler.database_unavailable_message(deadline) diff --git a/tests/test_litellm/proxy/db/test_spend_counter_reseed.py b/tests/test_litellm/proxy/db/test_spend_counter_reseed.py index ff0b67d426b..ab931277313 100644 --- a/tests/test_litellm/proxy/db/test_spend_counter_reseed.py +++ b/tests/test_litellm/proxy/db/test_spend_counter_reseed.py @@ -12,12 +12,14 @@ from collections.abc import Mapping from datetime import datetime, timedelta, timezone from types import SimpleNamespace from typing import Final +from unittest.mock import AsyncMock import pytest from litellm.caching.dual_cache import DualCache from litellm.caching.in_memory_cache import InMemoryCache from litellm.constants import PROXY_DB_LOOKUP_MAX_CONCURRENCY +from litellm.proxy.db.db_lookup_gate import LoopBoundSemaphore, db_lookup_stall_tracker from litellm.proxy.db.spend_counter_reseed import SpendCounterReseed WINDOW_START = datetime(2026, 8, 1, tzinfo=timezone.utc) @@ -445,6 +447,31 @@ async def test_from_db_returns_none_for_a_missing_project_row(): assert await SpendCounterReseed.from_db(prisma_client=prisma, counter_key="spend:project:proj-1") is None +@pytest.mark.asyncio +async def test_from_db_deadline_covers_the_wait_for_a_gate_slot(monkeypatch: pytest.MonkeyPatch) -> None: + """A saturated gate must fail the lookup at the deadline instead of parking + the request on a gate slot outside the bounded window.""" + gate: Final = LoopBoundSemaphore(1) + monkeypatch.setattr("litellm.proxy.db.spend_counter_reseed.db_lookup_gate", gate) + monkeypatch.setattr("litellm.proxy.db.db_lookup_gate.PROXY_DB_LOOKUP_DEADLINE_SECONDS", 0.05) + find_unique: Final = AsyncMock() + prisma: Final = SimpleNamespace( + db=SimpleNamespace(litellm_verificationtoken=SimpleNamespace(find_unique=find_unique)) + ) + db_lookup_stall_tracker.clear() + try: + async with gate.current(): + result: Final = await asyncio.wait_for( + SpendCounterReseed.from_db(prisma_client=prisma, counter_key="spend:key:abc"), + timeout=1.0, + ) + assert result is None + assert db_lookup_stall_tracker.stalled_within(60.0) + find_unique.assert_not_called() + finally: + db_lookup_stall_tracker.clear() + + @pytest.mark.asyncio async def test_from_db_still_never_reads_the_end_user_row(): """A cold end-user counter keeps seeding from the cached end-user object the auth diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index 761cd0685f2..ee4c468a460 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -2703,6 +2703,134 @@ async def test_health_readiness_details_returns_200_when_db_down_and_allow_reque assert result["db"] == "disconnected" +@pytest.fixture +def _clear_db_lookup_stall() -> Iterator[None]: + from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker + + db_lookup_stall_tracker.clear() + yield + db_lookup_stall_tracker.clear() + + +def _connected_prisma() -> MagicMock: + mock_prisma = MagicMock() + mock_prisma.health_check = AsyncMock(return_value=True) + return mock_prisma + + +def _forget_db_health_cache() -> None: + _health_endpoints_module.db_health_cache = { + "status": "unknown", + "last_updated": datetime.now() - timedelta(seconds=60), + } + + +@pytest.mark.asyncio +async def test_health_readiness_returns_503_stalled_after_a_db_lookup_deadline_hit(_clear_db_lookup_stall): + """The incident's readiness stayed green while every request sat parked on the + database: the probe's own ping is a fresh connection that answers fine. A lookup + that hit its deadline inside the stall window must take the pod out of rotation.""" + from fastapi import Response + + from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker + from litellm.proxy.health_endpoints._health_endpoints import health_readiness + + _forget_db_health_cache() + db_lookup_stall_tracker.record_hit() + + response = Response() + with patch( # test-quality-ok: the readiness path reads the proxy-global DB client; it has no injection seam + "litellm.proxy.proxy_server.prisma_client", _connected_prisma() + ): + result = await health_readiness(response=response) + + assert response.status_code == 503 + assert result == {"status": "healthy", "db": "stalled"} + + +@pytest.mark.asyncio +async def test_health_readiness_details_returns_503_stalled_after_a_db_lookup_deadline_hit(_clear_db_lookup_stall): + from fastapi import Response + + from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker + from litellm.proxy.health_endpoints._health_endpoints import _get_health_readiness_details + + _forget_db_health_cache() + db_lookup_stall_tracker.record_hit() + + response = Response() + with patch( # test-quality-ok: the readiness path reads the proxy-global DB client; it has no injection seam + "litellm.proxy.proxy_server.prisma_client", _connected_prisma() + ): + result = await _get_health_readiness_details(response=response) + + assert response.status_code == 503 + assert result["db"] == "stalled" + + +@pytest.mark.asyncio +async def test_health_readiness_stays_200_with_stalled_body_when_requests_are_allowed_on_db_unavailable( + _clear_db_lookup_stall, +): + """The fail-open deployment keeps serving through a stalled database, so the pod + must stay in rotation and report the stall through the body, exactly as it does + for a disconnected one.""" + from fastapi import Response + + from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker + from litellm.proxy.health_endpoints._health_endpoints import health_readiness + + _forget_db_health_cache() + db_lookup_stall_tracker.record_hit() + + response = Response() + with ( + patch( # test-quality-ok: the readiness path reads the proxy-global DB client; it has no injection seam + "litellm.proxy.proxy_server.prisma_client", _connected_prisma() + ), + patch.dict( # test-quality-ok: the fail-open flag lives in the proxy-global general_settings; no injection seam + "litellm.proxy.proxy_server.general_settings", + {"allow_requests_on_db_unavailable": True}, + ), + ): + result = await health_readiness(response=response) + + assert response.status_code == 200 + assert result == {"status": "healthy", "db": "stalled"} + + +@pytest.mark.asyncio +@pytest.mark.parametrize("hit_recorded", [False, True]) +async def test_health_readiness_reports_connected_without_a_stall_inside_the_window( + _clear_db_lookup_stall, hit_recorded: bool +): + """No deadline hit, or a window of 0 (the opt-out), keeps the ordinary connected + answer, so a healthy pod never leaves rotation over the stall check.""" + from fastapi import Response + + from litellm.proxy.db.db_lookup_gate import db_lookup_stall_tracker + from litellm.proxy.health_endpoints._health_endpoints import health_readiness + + _forget_db_health_cache() + if hit_recorded: + db_lookup_stall_tracker.record_hit() + + response = Response() + with ( + patch( # test-quality-ok: the readiness path reads the proxy-global DB client; it has no injection seam + "litellm.proxy.proxy_server.prisma_client", _connected_prisma() + ), + patch( # test-quality-ok: lowers the module-level stall window to its opt-out value for the recorded-hit case + "litellm.proxy.health_endpoints._health_endpoints.PROXY_DB_LOOKUP_STALL_WINDOW_SECONDS", + 0.0 if hit_recorded else 30.0, + ), + ): + result = await health_readiness(response=response) + + assert response.status_code == 200 + assert result == {"status": "healthy", "db": "connected"} + + @pytest.mark.asyncio async def test_db_health_readiness_check_bounds_hung_health_check(): """ diff --git a/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py b/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py index 0e9c336a9eb..b5e594db701 100644 --- a/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py +++ b/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py @@ -5,12 +5,14 @@ from datetime import datetime from typing import Final from unittest.mock import AsyncMock, MagicMock, patch +import httpx import pytest from litellm._logging import verbose_proxy_logger from litellm.litellm_core_utils.internal_call_metadata import MODEL_ACCESS_GROUP_METADATA_KEY from litellm.proxy._types import SpendLogsPayload, UserAPIKeyAuth from litellm.proxy.collector import SpendEventConsumer +from litellm.proxy.db.db_lookup_gate import DBLookupDeadlineExceeded from litellm.proxy.db.db_spend_update_writer import DBSpendUpdateWriter from litellm.proxy.db.spend_log_tool_index import response_tool_call_names from litellm.proxy.hooks.proxy_track_cost_callback import ( @@ -1679,6 +1681,92 @@ async def test_async_post_call_failure_hook_enriches_auth_error_metadata(): assert metadata["user_api_key_team_alias"] == "my-team-alias" +@pytest.mark.asyncio +async def test_async_post_call_failure_hook_skips_the_key_lookup_when_the_failure_is_a_db_stall(): + logger = _ProxyDBLogger() + user_api_key_dict = UserAPIKeyAuth(api_key="hashed_key") + request_data = { + "model": "gpt-5.6", + "messages": [{"role": "user", "content": "Hello"}], + "metadata": {}, + "litellm_params": {}, + } + + with ( + patch( + "litellm.proxy.db.db_spend_update_writer.DBSpendUpdateWriter.update_database", + new_callable=AsyncMock, + ) as mock_update_database, + patch( + "litellm.proxy.hooks.proxy_track_cost_callback.get_key_object", + new_callable=AsyncMock, + ) as mock_get_key_object, + patch( + "litellm.proxy.hooks.proxy_track_cost_callback.get_team_object", + new_callable=AsyncMock, + ) as mock_get_team_object, + ): + await logger.async_post_call_failure_hook( + request_data=request_data, + original_exception=DBLookupDeadlineExceeded("key", 10.0), + user_api_key_dict=user_api_key_dict, + ) + + mock_get_key_object.assert_not_called() + mock_get_team_object.assert_not_called() + mock_update_database.assert_called_once() + metadata = mock_update_database.call_args[1]["kwargs"]["litellm_params"]["metadata"] + assert metadata["status"] == "failure" + assert metadata["user_api_key"] == "hashed_key" + assert metadata["user_api_key_alias"] is None + + +@pytest.mark.asyncio +async def test_async_post_call_failure_hook_still_enriches_metadata_for_a_non_stall_failure(): + """Only a DBLookupDeadlineExceeded skips the key lookup; a transport error + from the provider call must still resolve the key's alias for the failure row.""" + logger = _ProxyDBLogger() + user_api_key_dict = UserAPIKeyAuth(api_key="hashed_key") + request_data = { + "model": "gpt-5.6", + "messages": [{"role": "user", "content": "Hello"}], + "metadata": {}, + "litellm_params": {}, + } + + mock_key_obj = MagicMock() + mock_key_obj.key_alias = "my-key-alias" + mock_key_obj.user_id = "my-user-id" + mock_key_obj.team_id = "my-team-id" + mock_key_obj.org_id = None + mock_key_obj.project_id = None + + with ( + patch( + "litellm.proxy.db.db_spend_update_writer.DBSpendUpdateWriter.update_database", + new_callable=AsyncMock, + ) as mock_update_database, + patch( + "litellm.proxy.hooks.proxy_track_cost_callback.get_key_object", + new_callable=AsyncMock, + return_value=mock_key_obj, + ) as mock_get_key_object, + patch( + "litellm.proxy.hooks.proxy_track_cost_callback.get_team_object", + new_callable=AsyncMock, + ), + ): + await logger.async_post_call_failure_hook( + request_data=request_data, + original_exception=httpx.ConnectError("boom"), + user_api_key_dict=user_api_key_dict, + ) + + mock_get_key_object.assert_called_once() + metadata = mock_update_database.call_args[1]["kwargs"]["litellm_params"]["metadata"] + assert metadata["user_api_key_alias"] == "my-key-alias" + + @pytest.mark.asyncio async def test_async_post_call_failure_hook_enriches_missing_team_alias(): """ @@ -2035,9 +2123,15 @@ async def test_track_cost_callback_keeps_guardrail_cost_on_cache_hit(): } with ( - patch("litellm.proxy.proxy_server.increment_spend_counters", new_callable=AsyncMock) as mock_increment, # test-quality-ok: the callback imports this from proxy_server inside its body, so there is no injection seam - patch("litellm.proxy.proxy_server.update_cache", new_callable=AsyncMock), # test-quality-ok: same function-body import, no injection seam - patch("litellm.proxy.proxy_server.proxy_logging_obj") as mock_proxy_logging, # test-quality-ok: same function-body import, no injection seam + patch( + "litellm.proxy.proxy_server.increment_spend_counters", new_callable=AsyncMock + ) as mock_increment, # test-quality-ok: the callback imports this from proxy_server inside its body, so there is no injection seam + patch( + "litellm.proxy.proxy_server.update_cache", new_callable=AsyncMock + ), # test-quality-ok: same function-body import, no injection seam + patch( + "litellm.proxy.proxy_server.proxy_logging_obj" + ) as mock_proxy_logging, # test-quality-ok: same function-body import, no injection seam ): mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock() mock_proxy_logging.slack_alerting_instance.customer_spend_alert = AsyncMock() From 58d7cafb97007bc4ed9be04ccff40d8ea0f15d26 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 08:24:27 -0700 Subject: [PATCH 66/96] fix(cost-map): sync openrouter deepseek-v4-pro prices (#42969) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 6 +++--- model_prices_and_context_window.json | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 69a64de6e04..93e3a033578 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41242,21 +41242,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.24462e-07, + "input_cost_per_token": 9.19242e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.848924e-06, + "output_cost_per_token": 1.838484e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.70385e-08, + "cache_read_input_token_cost": 7.66035e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 69a64de6e04..93e3a033578 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41242,21 +41242,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.24462e-07, + "input_cost_per_token": 9.19242e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.848924e-06, + "output_cost_per_token": 1.838484e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.70385e-08, + "cache_read_input_token_cost": 7.66035e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, From 553f0b6ee7d59e709b8adf9c28ac0a55a4f420ae Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 08:27:42 -0700 Subject: [PATCH 67/96] feat(cost-map): sync azure models, add MAI-Image-2.6, deepseek-v4.1-flash, muse-spark-1.3 (#42970) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 69 +++++++++++++++++++ model_prices_and_context_window.json | 69 +++++++++++++++++++ 2 files changed, 138 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 93e3a033578..02d7c3c66eb 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -63384,6 +63384,73 @@ "supports_tool_choice": true, "supports_vision": true }, + "azure_ai/deepseek-v4.1-flash": { + "cache_read_input_token_cost": 8e-09, + "input_cost_per_token": 3.75e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 1000000, + "max_output_tokens": 384000, + "max_tokens": 384000, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "deprecation_date": "2026-12-15", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" + }, + "azure_ai/muse-spark-1.3": { + "cache_read_input_token_cost": 1.5e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "azure_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 4.25e-06, + "source": "https://ai.developer.meta.com/docs/pricing-rate-limits", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "azure_ai/MAI-Image-2.6": { + "input_cost_per_image_token": 8e-06, + "input_cost_per_token": 5e-06, + "litellm_provider": "azure_ai", + "mode": "image_generation", + "output_cost_per_image_token": 3.8e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/images/generations", + "/v1/images/edits" + ] + }, + "azure_ai/MAI-Image-2.6-Flash": { + "input_cost_per_image_token": 2.5e-06, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "azure_ai", + "mode": "image_generation", + "output_cost_per_image_token": 1.9e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/images/generations", + "/v1/images/edits" + ] + }, "azure_ai/FW-DeepSeek-V4.1-Flash": { "cache_read_input_token_cost": 8e-09, "input_cost_per_token": 3.75e-07, @@ -63445,6 +63512,7 @@ "supports_tool_choice": true }, "azure_ai/FW-GPT-OSS-120B": { + "deprecation_date": "2027-07-01", "cache_read_input_token_cost": 8.2e-08, "input_cost_per_token": 1.65e-07, "litellm_provider": "azure_ai", @@ -63462,6 +63530,7 @@ "supports_tool_choice": true }, "azure_ai/Cohere-command-a-plus-05-2026": { + "deprecation_date": "2026-10-16", "input_cost_per_token": 8e-07, "litellm_provider": "azure_ai", "max_input_tokens": 128000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 93e3a033578..02d7c3c66eb 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -63384,6 +63384,73 @@ "supports_tool_choice": true, "supports_vision": true }, + "azure_ai/deepseek-v4.1-flash": { + "cache_read_input_token_cost": 8e-09, + "input_cost_per_token": 3.75e-07, + "litellm_provider": "azure_ai", + "max_input_tokens": 1000000, + "max_output_tokens": 384000, + "max_tokens": 384000, + "mode": "chat", + "output_cost_per_token": 1.5e-06, + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "deprecation_date": "2026-12-15", + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" + }, + "azure_ai/muse-spark-1.3": { + "cache_read_input_token_cost": 1.5e-07, + "input_cost_per_token": 1.25e-06, + "litellm_provider": "azure_ai", + "max_input_tokens": 1048576, + "max_output_tokens": 131072, + "max_tokens": 131072, + "mode": "chat", + "output_cost_per_token": 4.25e-06, + "source": "https://ai.developer.meta.com/docs/pricing-rate-limits", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_vision": true + }, + "azure_ai/MAI-Image-2.6": { + "input_cost_per_image_token": 8e-06, + "input_cost_per_token": 5e-06, + "litellm_provider": "azure_ai", + "mode": "image_generation", + "output_cost_per_image_token": 3.8e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/images/generations", + "/v1/images/edits" + ] + }, + "azure_ai/MAI-Image-2.6-Flash": { + "input_cost_per_image_token": 2.5e-06, + "input_cost_per_token": 1.75e-06, + "litellm_provider": "azure_ai", + "mode": "image_generation", + "output_cost_per_image_token": 1.9e-05, + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supported_endpoints": [ + "/v1/images/generations", + "/v1/images/edits" + ] + }, "azure_ai/FW-DeepSeek-V4.1-Flash": { "cache_read_input_token_cost": 8e-09, "input_cost_per_token": 3.75e-07, @@ -63445,6 +63512,7 @@ "supports_tool_choice": true }, "azure_ai/FW-GPT-OSS-120B": { + "deprecation_date": "2027-07-01", "cache_read_input_token_cost": 8.2e-08, "input_cost_per_token": 1.65e-07, "litellm_provider": "azure_ai", @@ -63462,6 +63530,7 @@ "supports_tool_choice": true }, "azure_ai/Cohere-command-a-plus-05-2026": { + "deprecation_date": "2026-10-16", "input_cost_per_token": 8e-07, "litellm_provider": "azure_ai", "max_input_tokens": 128000, From cfa2830bde4b73c1b5e35d84fae1712490e9dd18 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 08:30:48 -0700 Subject: [PATCH 68/96] fix(bedrock): extrapolate global cris pricing for gpt-5.4 and gpt-5.5 (#42971) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 24 +++++++++---------- model_prices_and_context_window.json | 24 +++++++++---------- 2 files changed, 24 insertions(+), 24 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 02d7c3c66eb..14252b727ac 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -57189,12 +57189,12 @@ ] }, "global.openai.gpt-5.4": { - "input_cost_per_token": 2.75e-06, - "input_cost_per_token_above_272k_tokens": 5.5e-06, - "cache_read_input_token_cost": 2.75e-07, - "cache_read_input_token_cost_above_272k_tokens": 5.5e-07, - "output_cost_per_token": 1.65e-05, - "output_cost_per_token_above_272k_tokens": 2.475e-05, + "input_cost_per_token": 2.5e-06, + "input_cost_per_token_above_272k_tokens": 5e-06, + "cache_read_input_token_cost": 2.5e-07, + "cache_read_input_token_cost_above_272k_tokens": 5e-07, + "output_cost_per_token": 1.5e-05, + "output_cost_per_token_above_272k_tokens": 2.25e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, @@ -57251,12 +57251,12 @@ ] }, "global.openai.gpt-5.5": { - "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_272k_tokens": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_272k_tokens": 1.1e-06, - "output_cost_per_token": 3.3e-05, - "output_cost_per_token_above_272k_tokens": 4.95e-05, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 02d7c3c66eb..14252b727ac 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -57189,12 +57189,12 @@ ] }, "global.openai.gpt-5.4": { - "input_cost_per_token": 2.75e-06, - "input_cost_per_token_above_272k_tokens": 5.5e-06, - "cache_read_input_token_cost": 2.75e-07, - "cache_read_input_token_cost_above_272k_tokens": 5.5e-07, - "output_cost_per_token": 1.65e-05, - "output_cost_per_token_above_272k_tokens": 2.475e-05, + "input_cost_per_token": 2.5e-06, + "input_cost_per_token_above_272k_tokens": 5e-06, + "cache_read_input_token_cost": 2.5e-07, + "cache_read_input_token_cost_above_272k_tokens": 5e-07, + "output_cost_per_token": 1.5e-05, + "output_cost_per_token_above_272k_tokens": 2.25e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, @@ -57251,12 +57251,12 @@ ] }, "global.openai.gpt-5.5": { - "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_272k_tokens": 1.1e-05, - "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_272k_tokens": 1.1e-06, - "output_cost_per_token": 3.3e-05, - "output_cost_per_token_above_272k_tokens": 4.95e-05, + "input_cost_per_token": 5e-06, + "input_cost_per_token_above_272k_tokens": 1e-05, + "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "output_cost_per_token": 3e-05, + "output_cost_per_token_above_272k_tokens": 4.5e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, From 991c339946ad943633ad7961fe678ae507bed258 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 08:53:31 -0700 Subject: [PATCH 69/96] fix(cost-map): sync openrouter deepseek v4 flash, v4 pro and v4.1 flash prices (#42974) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 22 +++++++++---------- model_prices_and_context_window.json | 22 +++++++++---------- 2 files changed, 22 insertions(+), 22 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 14252b727ac..be43affab4e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41242,34 +41242,34 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.19242e-07, + "input_cost_per_token": 9.15936e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.838484e-06, + "output_cost_per_token": 1.831872e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.66035e-08, + "cache_read_input_token_cost": 7.6328e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, "supports_web_search": false }, "openrouter/deepseek/deepseek-v4.1-flash": { - "input_cost_per_token": 1.4e-07, - "output_cost_per_token": 4.2e-07, - "cache_read_input_token_cost": 4.2e-09, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1.2e-06, + "cache_read_input_token_cost": 6e-09, "litellm_provider": "openrouter", "max_input_tokens": 1048576, - "max_output_tokens": 943718, - "max_tokens": 943718, + "max_output_tokens": 393216, + "max_tokens": 393216, "mode": "chat", "off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":3e-9}, "source": "https://openrouter.ai/api/v1/models", @@ -66491,9 +66491,9 @@ "supports_web_search": true }, "openrouter/deepseek/deepseek-v4-flash": { - "input_cost_per_token": 8.554e-08, - "output_cost_per_token": 1.7108e-07, - "cache_read_input_token_cost": 1.7108e-08, + "input_cost_per_token": 8.4e-08, + "output_cost_per_token": 1.68e-07, + "cache_read_input_token_cost": 1.68e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 14252b727ac..be43affab4e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41242,34 +41242,34 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.19242e-07, + "input_cost_per_token": 9.15936e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.838484e-06, + "output_cost_per_token": 1.831872e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.66035e-08, + "cache_read_input_token_cost": 7.6328e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, "supports_web_search": false }, "openrouter/deepseek/deepseek-v4.1-flash": { - "input_cost_per_token": 1.4e-07, - "output_cost_per_token": 4.2e-07, - "cache_read_input_token_cost": 4.2e-09, + "input_cost_per_token": 3e-07, + "output_cost_per_token": 1.2e-06, + "cache_read_input_token_cost": 6e-09, "litellm_provider": "openrouter", "max_input_tokens": 1048576, - "max_output_tokens": 943718, - "max_tokens": 943718, + "max_output_tokens": 393216, + "max_tokens": 393216, "mode": "chat", "off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":3e-9}, "source": "https://openrouter.ai/api/v1/models", @@ -66491,9 +66491,9 @@ "supports_web_search": true }, "openrouter/deepseek/deepseek-v4-flash": { - "input_cost_per_token": 8.554e-08, - "output_cost_per_token": 1.7108e-07, - "cache_read_input_token_cost": 1.7108e-08, + "input_cost_per_token": 8.4e-08, + "output_cost_per_token": 1.68e-07, + "cache_read_input_token_cost": 1.68e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, From f0e671f7549665b39a562149d5e654682b9e6a51 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:02:02 -0700 Subject: [PATCH 70/96] refactor(types): replace Any with proven types in 13 files (#42937) * refactor(types): replace Any with proven types in 13 files Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(types): drop unused executor import from utils type-checking block Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(realtime): keep reserved-key filtering on azure realtime health params Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(realtime): pin reserved-key filtering in azure realtime health auth params Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(realtime): exercise the real azure header builder in the reserved-key test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/integrations/opentelemetry.py | 10 +-- litellm/litellm_core_utils/litellm_logging.py | 12 ++- .../prompt_templates/factory.py | 4 +- litellm/llms/custom_httpx/llm_http_handler.py | 58 +++++++++--- litellm/main.py | 7 +- litellm/proxy/common_request_processing.py | 12 +-- .../key_management_endpoints.py | 2 +- litellm/proxy/proxy_server.py | 12 +-- litellm/proxy/utils.py | 2 +- litellm/realtime_api/main.py | 14 +-- litellm/responses/main.py | 2 +- litellm/router.py | 10 +-- litellm/utils.py | 89 +++++++++++++++---- .../test_health_check_helpers.py | 17 ++++ 14 files changed, 177 insertions(+), 74 deletions(-) diff --git a/litellm/integrations/opentelemetry.py b/litellm/integrations/opentelemetry.py index 749f0ce4fcb..c1531f4e4ae 100644 --- a/litellm/integrations/opentelemetry.py +++ b/litellm/integrations/opentelemetry.py @@ -62,10 +62,10 @@ if TYPE_CHECKING: from litellm.proxy.proxy_server import UserAPIKeyAuth as _UserAPIKeyAuth Span = _Span | Any - Tracer = _Tracer | Any - Context = _Context | Any - SpanExporter = _SpanExporter | Any - UserAPIKeyAuth = _UserAPIKeyAuth | Any + Tracer = _Tracer + Context = _Context + SpanExporter = _SpanExporter + UserAPIKeyAuth = _UserAPIKeyAuth ManagementEndpointLoggingPayload = _ManagementEndpointLoggingPayload | Any else: Span = Any @@ -2730,7 +2730,7 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger): self.handle_callback_failure(callback_name=self.callback_name or "opentelemetry") verbose_logger.exception("OpenTelemetry logging error in set_attributes %s", str(e)) - def _cast_as_primitive_value_type(self, value) -> str | bool | int | float: + def _cast_as_primitive_value_type(self, value: object) -> str | bool | int | float: """ Casts the value to a primitive OTEL type if it is not already a primitive type. diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 2cecec729c2..a6391a2ae27 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -1566,14 +1566,14 @@ class Logging(LiteLLMLoggingBaseClass): attr = "debug" if json_logs: - callattr = getattr(verbose_logger, attr) + callattr = verbose_logger.warning if attr == "warning" else verbose_logger.debug callattr( "RAW RESPONSE:\n{}\n\n".format( self.model_call_details.get("original_response", self.model_call_details) ), ) else: - callattr = getattr(verbose_logger, attr) + callattr = verbose_logger.warning if attr == "warning" else verbose_logger.debug callattr( "RAW RESPONSE:\n{}\n\n".format( self.model_call_details.get("original_response", self.model_call_details) @@ -5882,7 +5882,7 @@ class StandardLoggingPayloadSetup: base_model: str | None, custom_pricing: bool | None, custom_llm_provider: str | None, - init_response_obj: Any | BaseModel | dict, + init_response_obj: object, api_base: str | None = None, ) -> StandardLoggingModelInformation: model_cost_name: Final = _select_model_name_for_cost_calc( @@ -5915,9 +5915,7 @@ class StandardLoggingPayloadSetup: return model_cost_information @staticmethod - def get_final_response_obj( - response_obj: dict, init_response_obj: Any | BaseModel | dict, kwargs: dict - ) -> dict | str | list | None: + def get_final_response_obj(response_obj: dict, init_response_obj: object, kwargs: dict) -> dict | str | list | None: """ Get final response object after redacting the message input/output from logging """ @@ -6360,7 +6358,7 @@ def _get_status_fields( def _extract_response_obj_and_hidden_params( - init_response_obj: Any | BaseModel | dict, + init_response_obj: object, original_exception: Exception | None, ) -> tuple[dict, dict | None]: """Extract response_obj and hidden_params from init_response_obj.""" diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 8424187dcbc..6fc319c26ae 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -446,7 +446,7 @@ def _render_chat_template(env, chat_template: str, bos_token: str, eos_token: st async def _afetch_and_extract_template( - model: str, chat_template: Any | None, get_config_fn, get_template_fn + model: str, chat_template: str | None, get_config_fn, get_template_fn ) -> tuple[str, str, str]: """ Async version: Fetch template and tokens from HuggingFace. @@ -500,7 +500,7 @@ async def _afetch_and_extract_template( def _fetch_and_extract_template( - model: str, chat_template: Any | None, get_config_fn, get_template_fn + model: str, chat_template: str | None, get_config_fn, get_template_fn ) -> tuple[str, str, str]: """ Sync version: Fetch template and tokens from HuggingFace. diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 052978c2680..dba0dee38fc 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -12,6 +12,7 @@ from typing import ( Literal, NamedTuple, Optional, + Protocol, TypedDict, TypeVar, Union, @@ -24,6 +25,7 @@ import httpx from httpx import USE_CLIENT_DEFAULT from httpx._types import FileContent from openai.types.file_deleted import FileDeleted +from typing_extensions import ReadOnly import litellm import litellm.litellm_core_utils @@ -206,6 +208,7 @@ if TYPE_CHECKING: FakeAnthropicMessagesStreamIterator, ) from litellm.llms.base_llm.passthrough.transformation import BasePassthroughConfig + from litellm.proxy._types import UserAPIKeyAuth from litellm.types.llms.openai_evals import ( CancelEvalResponse, CancelRunResponse, @@ -221,6 +224,21 @@ if TYPE_CHECKING: else: LiteLLMLoggingObj = Any + +class _RealtimeClientWebSocket(Protocol): + async def send_text(self, data: str) -> None: ... + + async def close(self, code: int = ..., reason: str | None = ...) -> None: ... + + +class _ResponsesClientWebSocket(Protocol): + async def send_text(self, data: str) -> None: ... + + async def receive_text(self) -> str: ... + + async def close(self, code: int = ..., reason: str | None = ...) -> None: ... + + _ResponseT = TypeVar("_ResponseT") @@ -237,6 +255,17 @@ class _MediaUploadKwargs(TypedDict, total=False): timeout: float | httpx.Timeout +class _SignedBodyKwargs(TypedDict, total=False): + data: ReadOnly[bytes] + json: ReadOnly[dict[str, object]] + + +def _signed_body_kwargs(*, signed_body: bytes | None, data: dict[str, object]) -> _SignedBodyKwargs: + if signed_body is not None: + return {"data": signed_body} + return {"json": data} + + def _google_genai_streaming_hidden_params( *, api_base: str, @@ -318,7 +347,9 @@ def _mask_presigned_request_headers(transformed_request: bytes | str | dict) -> } -def _aws_signing_overrides(optional_params: Mapping[str, Any], litellm_params: Mapping[str, Any]) -> Mapping[str, Any]: +def _aws_signing_overrides( + optional_params: Mapping[str, object], litellm_params: Mapping[str, object] +) -> Mapping[str, object]: return MappingProxyType( { key: litellm_params[key] @@ -2739,7 +2770,7 @@ class BaseLLMHTTPHandler: stream=stream, fake_stream=fake_stream, ) - body_kwargs: Final[dict[str, Any]] = {"data": signed_body} if signed_body is not None else {"json": data} + body_kwargs: Final = _signed_body_kwargs(signed_body=signed_body, data=data) ## LOGGING logging_obj.pre_call( @@ -2926,7 +2957,7 @@ class BaseLLMHTTPHandler: stream=stream, fake_stream=fake_stream, ) - body_kwargs: Final[dict[str, Any]] = {"data": signed_body} if signed_body is not None else {"json": data} + body_kwargs: Final = _signed_body_kwargs(signed_body=signed_body, data=data) ## LOGGING logging_obj.pre_call( @@ -4540,7 +4571,7 @@ class BaseLLMHTTPHandler: api_key=litellm_params.api_key, model=model, ) - body_kwargs: Final[dict[str, Any]] = {"data": signed_body} if signed_body is not None else {"json": data} + body_kwargs: Final = _signed_body_kwargs(signed_body=signed_body, data=data) ## LOGGING logging_obj.pre_call( @@ -4634,7 +4665,7 @@ class BaseLLMHTTPHandler: api_key=litellm_params.api_key, model=model, ) - body_kwargs: Final[dict[str, Any]] = {"data": signed_body} if signed_body is not None else {"json": data} + body_kwargs: Final = _signed_body_kwargs(signed_body=signed_body, data=data) ## LOGGING logging_obj.pre_call( @@ -6186,6 +6217,7 @@ class BaseLLMHTTPHandler: "BasePassthroughConfig", "BaseContainerConfig", BaseEvalsAPIConfig, + BaseRealtimeHTTPConfig, ], ): received_status_code: Final = ( @@ -6300,7 +6332,7 @@ class BaseLLMHTTPHandler: async def async_realtime( self, model: str, - websocket: Any, + websocket: _RealtimeClientWebSocket, logging_obj: LiteLLMLoggingObj, provider_config: BaseRealtimeConfig, headers: dict, @@ -6308,7 +6340,7 @@ class BaseLLMHTTPHandler: api_key: str | None = None, client: Any | None = None, timeout: float | None = None, - user_api_key_dict: Any | None = None, + user_api_key_dict: object | None = None, litellm_metadata: dict[str, object] | None = None, query_params: RealtimeQueryParams | None = None, ): @@ -6483,7 +6515,7 @@ class BaseLLMHTTPHandler: request_data: dict[str, object], logging_obj: LiteLLMLoggingObj, timeout: float | httpx.Timeout, - provider_config: Any | None = None, + provider_config: BaseRealtimeHTTPConfig | None = None, model: str | None = None, extra_headers: dict[str, object] | None = None, client: HTTPHandler | AsyncHTTPHandler | None = None, @@ -6555,7 +6587,7 @@ class BaseLLMHTTPHandler: sdp_body: bytes, logging_obj: LiteLLMLoggingObj, timeout: float | httpx.Timeout, - provider_config: Any | None = None, + provider_config: BaseRealtimeHTTPConfig | None = None, model: str | None = None, session_config: dict[str, object] | None = None, extra_headers: dict[str, object] | None = None, @@ -6633,13 +6665,13 @@ class BaseLLMHTTPHandler: async def async_responses_websocket( self, model: str, - websocket: Any, + websocket: _ResponsesClientWebSocket, logging_obj: LiteLLMLoggingObj, responses_api_provider_config: BaseResponsesAPIConfig | None, api_base: str | None = None, api_key: str | None = None, timeout: float | None = None, - user_api_key_dict: Any | None = None, + user_api_key_dict: "UserAPIKeyAuth | None" = None, litellm_metadata: dict[str, object] | None = None, custom_llm_provider: str | None = None, first_message: str | None = None, @@ -7850,7 +7882,7 @@ class BaseLLMHTTPHandler: def video_create_character_handler( self, name: str, - video: Any, + video: FileTypes, video_provider_config: BaseVideoConfig, custom_llm_provider: str, litellm_params, @@ -7934,7 +7966,7 @@ class BaseLLMHTTPHandler: async def async_video_create_character_handler( self, name: str, - video: Any, + video: FileTypes, video_provider_config: BaseVideoConfig, custom_llm_provider: str, litellm_params, diff --git a/litellm/main.py b/litellm/main.py index ceac729d3f0..98bb5126a90 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -57,6 +57,7 @@ from litellm.utils import ( # Logging is imported lazily when needed to avoid loading litellm_logging at import time if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging + from litellm.router import Router from litellm.types.utils import TokenCountResponse from litellm.constants import ( @@ -351,7 +352,7 @@ class LiteLLM: class Chat: - def __init__(self, params, router_obj: Any | None): + def __init__(self, params, router_obj: "Router | None"): self.params = params if self.params.get("acompletion", False) is True: self.params.pop("acompletion") @@ -361,7 +362,7 @@ class Chat: class Completions: - def __init__(self, params, router_obj: Any | None): + def __init__(self, params, router_obj: "Router | None"): self.params = params self.router_obj = router_obj @@ -377,7 +378,7 @@ class Completions: class AsyncCompletions: - def __init__(self, params, router_obj: Any | None): + def __init__(self, params, router_obj: "Router | None"): self.params = params self.router_obj = router_obj diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 904070cfadd..7e1989b0aab 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -218,7 +218,7 @@ ProxyRouteType: TypeAlias = Literal[ from litellm.llms.anthropic.chat.transformation import AnthropicConfig # Type alias for streaming chunk serializer (chunk after hooks + cost injection -> wire format) -StreamChunkSerializer = Callable[[Any], str] +StreamChunkSerializer = Callable[[object], str] # Type alias for streaming error serializer (ProxyException -> wire format) StreamErrorSerializer = Callable[[ProxyException], str] @@ -459,7 +459,7 @@ async def _bill_partial_streamed_spend_on_disconnect(request_data: dict, respons return True -async def _cancel_pending_gather_tasks(tasks: list["asyncio.Task[Any]"]) -> None: +async def _cancel_pending_gather_tasks(tasks: Sequence["asyncio.Task[object]"]) -> None: pending_tasks: Final = [task for task in tasks if not task.done()] for task in pending_tasks: task.cancel() @@ -3145,7 +3145,7 @@ class ProxyBaseLLMRequestProcessing: logging_obj._on_detached_stream_failure = _on_detached_stream_failure - def _is_streaming_response(self, response: Any) -> bool: + def _is_streaming_response(self, response: object) -> bool: """ Check if the response object is actually a streaming response by inspecting its type. @@ -3259,7 +3259,7 @@ class ProxyBaseLLMRequestProcessing: async def _handle_non_streaming_allm_passthrough_route( self, - response: Any, + response: _UpstreamHttpResponse, proxy_logging_obj: "ProxyLogging", user_api_key_dict: "UserAPIKeyAuth", custom_headers: Mapping[str, str], @@ -3852,7 +3852,7 @@ class ProxyBaseLLMRequestProcessing: @staticmethod async def async_streaming_data_generator( - response: Any, + response: object, user_api_key_dict: UserAPIKeyAuth, request_data: dict, proxy_logging_obj: ProxyLogging, @@ -3993,7 +3993,7 @@ class ProxyBaseLLMRequestProcessing: @staticmethod def async_sse_data_generator( - response: Any, + response: object, user_api_key_dict: UserAPIKeyAuth, request_data: dict, proxy_logging_obj: ProxyLogging, diff --git a/litellm/proxy/management_endpoints/key_management_endpoints.py b/litellm/proxy/management_endpoints/key_management_endpoints.py index 306ea90d7f1..2f1dc270cf6 100644 --- a/litellm/proxy/management_endpoints/key_management_endpoints.py +++ b/litellm/proxy/management_endpoints/key_management_endpoints.py @@ -453,7 +453,7 @@ def _regenerate_request_as_update_request(key: str, data: RegenerateKeyRequest) ) if not changed_fields: return None - return UpdateKeyRequest(key=key, **changed_fields) + return UpdateKeyRequest.model_validate(MappingProxyType({"key": key, **changed_fields})) class _LegacyDumpable(Protocol): diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 9fb72d2d1b5..c0071aa7c81 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -8700,9 +8700,9 @@ class ProxyConfig: @staticmethod def _merge_config_and_db_search_tools( - config_search_tools: list[SearchToolTypedDict], - db_search_tools: list[dict[str, Any]], - ) -> list[dict[str, Any]]: + config_search_tools: Sequence[SearchToolTypedDict], + db_search_tools: Sequence[dict[str, object]], + ) -> list[dict[str, object]]: db_tool_names: Final = {tool.get("search_tool_name") for tool in db_search_tools} return [ *[ @@ -9273,10 +9273,10 @@ _EMPTY_HEADERS: Final[Mapping[str, str]] = MappingProxyType({}) async def _iter_with_keepalive( - aiter: AsyncIterator[Any], + aiter: AsyncIterator[object], resolve_keepalive_seconds: Callable[[object], float], keepalive_seconds: float, -) -> AsyncGenerator[Any, None]: +) -> AsyncGenerator[object, None]: """Wrap `aiter` with idle-gap heartbeats, re-resolving the interval after each real chunk via `resolve_keepalive_seconds`. A mid-stream router fallback can swap in a deployment with a different keepalive policy, including one that @@ -9287,7 +9287,7 @@ async def _iter_with_keepalive( actually produced it, in both directions. While the interval is <= 0, no task is created and no timeout is awaited: a chunk is forwarded the moment it arrives, at the same cost as a bare `async for`.""" - pending: asyncio.Task[Any] | None = None # rebind-ok: rebound each loop iteration + pending: asyncio.Task[object] | None = None # rebind-ok: rebound each loop iteration current_keepalive_seconds = keepalive_seconds # rebind-ok: re-resolved after each chunk try: while True: diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index fe161f5d50b..c617047fad9 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -3956,7 +3956,7 @@ class ProxyLogging: request_data: dict, # mutable-ok: same request-payload shape the hooks mutate pipelines: "tuple[tuple[str, GuardrailPipeline], ...]", translation: "tuple[str, BaseTranslation]", - ) -> "AsyncGenerator[Any, None]": + ) -> "AsyncGenerator[object, None]": """ Execute post_call policy pipelines against a streamed response. diff --git a/litellm/realtime_api/main.py b/litellm/realtime_api/main.py index acc42c44c04..28814741852 100644 --- a/litellm/realtime_api/main.py +++ b/litellm/realtime_api/main.py @@ -127,15 +127,15 @@ def _get_realtime_http_provider_config( @wrapper_client async def acreate_realtime_client_secret( model: str | None = None, - session: dict[str, Any] | None = None, - expires_after: dict[str, Any] | None = None, + session: Mapping[str, object] | None = None, + expires_after: Mapping[str, object] | None = None, timeout: float | None = None, **kwargs, ): req: Final = RealtimeClientSecretRequest( model=model, - session=RealtimeSessionConfig(**session) if session else None, - expires_after=RealtimeExpiresAfter(**expires_after) if expires_after else None, + session=RealtimeSessionConfig.model_validate(session) if session else None, + expires_after=RealtimeExpiresAfter.model_validate(expires_after) if expires_after else None, ) model_name = (req.session.model if req.session is not None else None) or req.model or "gpt-4o-realtime-preview" litellm_logging_obj: Final[LiteLLMLogging] = kwargs.get("litellm_logging_obj") @@ -614,12 +614,14 @@ def _azure_realtime_health_protocol( def _realtime_health_check_auth_headers( - custom_llm_provider: str, api_key: str | None, model_params: Mapping[str, Any] + custom_llm_provider: str, api_key: str | None, model_params: Mapping[str, object] ) -> Mapping[str, str]: if custom_llm_provider == "azure": return azure_realtime.get_auth_headers( api_key=api_key, - azure_ad_token=(None if api_key else get_azure_ad_token(GenericLiteLLMParams(**model_params))), + azure_ad_token=( + None if api_key else get_azure_ad_token(GenericLiteLLMParams.model_validate(dict(model_params))) + ), ) if api_key is None: return _EMPTY_AUTH_HEADERS diff --git a/litellm/responses/main.py b/litellm/responses/main.py index 884ea7e217d..93c72bc2d3b 100644 --- a/litellm/responses/main.py +++ b/litellm/responses/main.py @@ -549,7 +549,7 @@ def _will_bridge_to_chat_completions( @contextmanager def _prompt_management_sees_a_provisional_message_list( - kwargs: dict[str, Any], # mutable-ok: the signal is read and popped out of the caller's own kwargs + kwargs: dict[str, object], # mutable-ok: the signal is read and popped out of the caller's own kwargs bridged: bool, ) -> Generator[None, None]: """Tell the cache-control hook that this layer's messages are not the ones sent upstream. diff --git a/litellm/router.py b/litellm/router.py index 62042e1c969..d328fbbb12f 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -4231,7 +4231,7 @@ class Router: models: Final = [m.strip() for m in model.split(",")] async def _async_completion_no_exceptions( - model_name: str, messages: list[dict[str, str]], stream: bool, **kwargs: Any + model_name: str, messages: list[dict[str, str]], stream: bool, **kwargs: object ) -> ModelResponse | CustomStreamWrapper | Exception: """ Wrapper around self.acompletion that catches exceptions and returns them as a result @@ -6736,7 +6736,7 @@ class Router: # Handle asynchronous call types async def async_wrapper( custom_llm_provider: str | None = None, - client: Any | None = None, + client: AsyncOpenAI | None = None, **kwargs, ): if call_type == "assistants": @@ -8441,7 +8441,7 @@ class Router: return self._has_content_policy_fallback(model, kwargs) def _should_raise_anthropic_refusal_error( - self, model: str, original_generic_function: Callable, response: object, kwargs: Mapping[str, Any] + self, model: str, original_generic_function: Callable, response: object, kwargs: Mapping[str, object] ) -> bool: """ The /v1/messages twin of _should_raise_content_policy_error: an Anthropic safeguard @@ -10318,7 +10318,7 @@ class Router: ) @staticmethod - def _widest_configured_limit(model_infos: Sequence[Mapping[str, Any]], field: str) -> int | None: + def _widest_configured_limit(model_infos: Sequence[Mapping[str, object]], field: str) -> int | None: """The largest usable value of ``field`` across a group's configured model_info blocks.""" limits: Final = tuple( limit @@ -13409,7 +13409,7 @@ class Router: self, model: str, request_kwargs: dict, - messages: list[dict[str, Any]] | None, + messages: list[dict[str, object]] | None, ) -> RoutingContext: """ Build a RoutingContext for `model`, run it through `self.routing_plugins` diff --git a/litellm/utils.py b/litellm/utils.py index 64097021dff..81258b8ca77 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -369,7 +369,7 @@ if TYPE_CHECKING: ) from litellm.litellm_core_utils.rules import Rules from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper - from litellm.litellm_core_utils.thread_pool_executor import executor + from litellm.litellm_core_utils.thread_pool_executor import BoundedLoggingThreadPoolExecutor from litellm.llms.base_llm.anthropic_messages.transformation import ( BaseAnthropicMessagesConfig, ) @@ -683,12 +683,12 @@ def load_credentials_from_list(kwargs: dict): Updates kwargs with the credentials if credential_name in kwarg """ # Access CredentialAccessor via module to trigger lazy loading if needed - CredentialAccessor: Final = getattr(sys.modules[__name__], "CredentialAccessor") + credential_accessor: Final[type[CredentialAccessor]] = getattr(sys.modules[__name__], "CredentialAccessor") credential_name: Final = kwargs.get("litellm_credential_name") if not credential_name: return - credential: Final = CredentialAccessor.find_credential(credential_name) + credential: Final = credential_accessor.find_credential(credential_name) if credential is None: verbose_logger.warning( "litellm_credential_name=%s matched none of the %d loaded credentials; the request runs without it", @@ -894,6 +894,45 @@ class _NamedFile(Protocol): def name(self) -> object: ... +class _LoggingClassGetter(Protocol): + def __call__(self) -> type[LiteLLMLoggingObject]: ... + + +class _ResponseMetadataUpdater(Protocol): + def __call__( + self, + result: object, + logging_obj: LiteLLMLoggingObject, + model: str | None, + kwargs: dict[str, object], + start_time: datetime.datetime, + end_time: datetime.datetime, + include_overhead: bool = True, + ) -> None: ... + + +class _SupportedOpenAIParamsGetter(Protocol): + def __call__( + self, + model: str, + custom_llm_provider: str | None = None, + request_type: Literal["chat_completion", "embeddings", "transcription"] = "chat_completion", + base_model: str | None = None, + ) -> list[str] | None: ... + + +class _NestedPathChecker(Protocol): + def __call__(self, path: str) -> bool: ... + + +class _NestedValueDeleter(Protocol): + def __call__(self, data: dict[str, object], path: str) -> dict[str, object]: ... + + +class _BaseModelFromMetadataGetter(Protocol): + def __call__(self, metadata: Mapping[str, object] | None) -> str | None: ... + + def _ocr_document_summary(document: object) -> str: if not isinstance(document, Mapping): return "default-message-value" @@ -978,7 +1017,9 @@ def function_setup( len(litellm.input_callback) > 0 or len(litellm.success_callback) > 0 or len(litellm.failure_callback) > 0 ) and len(callback_list) == 0: callback_list = list(set(litellm.input_callback + litellm.success_callback + litellm.failure_callback)) - get_set_callbacks: Final = getattr(sys.modules[__name__], "get_set_callbacks") + get_set_callbacks: Final[Callable[[], Callable[..., None]]] = getattr( + sys.modules[__name__], "get_set_callbacks" + ) get_set_callbacks()(callback_list=callback_list, function_id=function_id) ## ASYNC CALLBACKS - safety net for callbacks added via direct append if len(litellm.input_callback) > 0: @@ -1223,7 +1264,9 @@ def function_setup( call_type=call_type, ): stream = True - get_litellm_logging_class: Final = getattr(sys.modules[__name__], "get_litellm_logging_class") + get_litellm_logging_class: Final[_LoggingClassGetter] = getattr( + sys.modules[__name__], "get_litellm_logging_class" + ) # Victim for object pool logging_obj = get_litellm_logging_class()( # rebind-ok: 2nd assignment to logging_obj (see initial None above) model=model, @@ -1761,7 +1804,9 @@ def client(original_function): return litellm.stream_chunk_builder(chunks, messages=kwargs.get("messages", None)) else: # RETURN RESULT - update_response_metadata = getattr(sys.modules[__name__], "update_response_metadata") + update_response_metadata: _ResponseMetadataUpdater = getattr( + sys.modules[__name__], "update_response_metadata" + ) update_response_metadata( result=result, logging_obj=logging_obj, @@ -1802,7 +1847,9 @@ def client(original_function): kwargs=kwargs, ) - _update_response_metadata: Final = getattr(sys.modules[__name__], "update_response_metadata") + _update_response_metadata: Final[_ResponseMetadataUpdater] = getattr( + sys.modules[__name__], "update_response_metadata" + ) _update_response_metadata( result=result, logging_obj=logging_obj, @@ -1817,7 +1864,7 @@ def client(original_function): # Copy the current context to propagate it to the background thread # This is essential for OpenTelemetry span context propagation ctx: Final = contextvars.copy_context() - executor: Final = getattr(sys.modules[__name__], "executor") + executor: Final[BoundedLoggingThreadPoolExecutor] = getattr(sys.modules[__name__], "executor") executor.submit( ctx.run, logging_obj.success_handler, @@ -1910,7 +1957,9 @@ def client(original_function): print_args_passed_to_litellm(original_function, args, kwargs) start_time: Final = datetime.datetime.now() result = None - _update_response_metadata: Final = getattr(sys.modules[__name__], "update_response_metadata") + _update_response_metadata: Final[_ResponseMetadataUpdater] = getattr( + sys.modules[__name__], "update_response_metadata" + ) logging_obj: LiteLLMLoggingObject | None = kwargs.get("litellm_logging_obj", None) LLMCachingHandler: Final = _get_cached_llm_caching_handler() _llm_caching_handler: Final[LLMCachingHandler] = LLMCachingHandler( @@ -3678,7 +3727,9 @@ def get_optional_params_embeddings( **kwargs, ): # Lazy load get_supported_openai_params - get_supported_openai_params: Final = getattr(sys.modules[__name__], "get_supported_openai_params") + get_supported_openai_params: Final[_SupportedOpenAIParamsGetter] = getattr( + sys.modules[__name__], "get_supported_openai_params" + ) # retrieve all parameters passed to the function passed_params: Final = locals() @@ -4469,7 +4520,9 @@ def get_optional_params( message=f"{custom_llm_provider} does not support parameters: {list(unsupported_params.keys())}, for model={model}. To drop these, set `litellm.drop_params=True` or for proxy:\n\n`litellm_settings:\n drop_params: true`\n. \n If you want to use these params dynamically send allowed_openai_params={list(unsupported_params.keys())} in your request.", ) - get_supported_openai_params: Final = getattr(sys.modules[__name__], "get_supported_openai_params") + get_supported_openai_params: Final[_SupportedOpenAIParamsGetter] = getattr( + sys.modules[__name__], "get_supported_openai_params" + ) supported_params = get_supported_openai_params( model=model, custom_llm_provider=custom_llm_provider, base_model=base_model ) @@ -4640,9 +4693,9 @@ def get_optional_params( drop_params=bool(drop_params), ) elif custom_llm_provider == "bedrock": - BedrockModelInfo: Final = getattr(sys.modules[__name__], "BedrockModelInfo") - bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model) - bedrock_base_model: Final = BedrockModelInfo.get_base_model(model) + bedrock_model_info: Final[type[BedrockModelInfo]] = getattr(sys.modules[__name__], "BedrockModelInfo") + bedrock_route: Final = bedrock_model_info.get_bedrock_route(model) + bedrock_base_model: Final = bedrock_model_info.get_base_model(model) if bedrock_route == "converse" or bedrock_route == "converse_like": optional_params = litellm.AmazonConverseConfig().map_openai_params( model=model, @@ -4680,7 +4733,7 @@ def get_optional_params( drop_params=bool(drop_params), ) if bedrock_route == "claude_platform": - optional_params = BedrockModelInfo.map_claude_platform_auth_params( + optional_params = bedrock_model_info.map_claude_platform_auth_params( passed_params=passed_params, optional_params=optional_params ) elif custom_llm_provider == "cloudflare": @@ -4954,8 +5007,8 @@ def get_optional_params( # Apply nested drops from additional_drop_params if additional_drop_params: - is_nested_path: Final = getattr(sys.modules[__name__], "is_nested_path") - delete_nested_value: Final = getattr(sys.modules[__name__], "delete_nested_value") + is_nested_path: Final[_NestedPathChecker] = getattr(sys.modules[__name__], "is_nested_path") + delete_nested_value: Final[_NestedValueDeleter] = getattr(sys.modules[__name__], "delete_nested_value") nested_paths: Final = [p for p in additional_drop_params if is_nested_path(p)] for path in nested_paths: optional_params = delete_nested_value(optional_params, path) @@ -7852,7 +7905,7 @@ def _get_base_model_from_metadata(model_call_details=None): return _base_model metadata: Final = litellm_params.get("metadata") or {} - _get_base_model_from_litellm_call_metadata: Callable[..., str | None] = getattr( + _get_base_model_from_litellm_call_metadata: _BaseModelFromMetadataGetter = getattr( sys.modules[__name__], "_get_base_model_from_litellm_call_metadata" ) base_model_from_metadata: Final = _get_base_model_from_litellm_call_metadata(metadata=metadata) diff --git a/tests/test_litellm/litellm_core_utils/test_health_check_helpers.py b/tests/test_litellm/litellm_core_utils/test_health_check_helpers.py index 89b377af3a0..12134ba988a 100644 --- a/tests/test_litellm/litellm_core_utils/test_health_check_helpers.py +++ b/tests/test_litellm/litellm_core_utils/test_health_check_helpers.py @@ -2,6 +2,7 @@ import struct import zlib +from types import MappingProxyType from unittest.mock import AsyncMock, MagicMock, patch import pytest @@ -482,3 +483,19 @@ async def test_ocr_health_check_sends_the_document_kind_the_provider_config_acce document = mock_aocr.call_args.kwargs["document"] assert document["type"] == expected_document_type assert document[expected_document_type].startswith(expected_uri_prefix) + + +def test_realtime_health_check_azure_ad_params_drop_reserved_keys(): + from litellm.realtime_api import main as realtime_main + + seen = [] + with patch.object(realtime_main, "get_azure_ad_token", lambda params: seen.append(params) or "ad-token"): + headers = realtime_main._realtime_health_check_auth_headers( + "azure", + None, + MappingProxyType({"api_base": "https://x.openai.azure.com", "self": 1, "params": 2, "__class__": 3}), + ) + + assert dict(headers) == {"Authorization": "Bearer ad-token"} + assert seen[0].api_base == "https://x.openai.azure.com" + assert seen[0].model_extra == {} From cc7aaae7482d9cffbae66c34b3462a1df297715d Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:27:11 -0700 Subject: [PATCH 71/96] fix(cost-map): sync openrouter deepseek-v4-pro prices (#42978) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 6 +++--- model_prices_and_context_window.json | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index be43affab4e..926672b19a2 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41242,21 +41242,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.15936e-07, + "input_cost_per_token": 9.1263e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.831872e-06, + "output_cost_per_token": 1.82526e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.6328e-08, + "cache_read_input_token_cost": 7.60525e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index be43affab4e..926672b19a2 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41242,21 +41242,21 @@ "supports_web_search": false }, "openrouter/deepseek/deepseek-v4-pro": { - "input_cost_per_token": 9.15936e-07, + "input_cost_per_token": 9.1263e-07, "input_cost_per_token_cache_hit": 4.4e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, "max_tokens": 384000, "mode": "chat", - "output_cost_per_token": 1.831872e-06, + "output_cost_per_token": 1.82526e-06, "source": "https://openrouter.ai/api/v1/models", "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, - "cache_read_input_token_cost": 7.6328e-08, + "cache_read_input_token_cost": 7.60525e-08, "supports_audio_input": false, "supports_pdf_input": false, "supports_vision": false, From bd55afc0f87cfe88f5afe8eaba8888e1bdf9ac56 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:30:37 -0700 Subject: [PATCH 72/96] fix(cost-map): add vertex ai priority audio input prices for gemini flash rows (#42980) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 4 ++++ model_prices_and_context_window.json | 4 ++++ 2 files changed, 8 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 926672b19a2..47cdfecfc0f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -26063,6 +26063,7 @@ "input_cost_per_token_batches": 1.5e-07, "input_cost_per_token_flex": 1.5e-07, "input_cost_per_token_priority": 5.4e-07, + "input_cost_per_audio_token_priority": 1.8e-06, "output_cost_per_token_batches": 1.25e-06, "output_cost_per_token_flex": 1.25e-06, "output_cost_per_token_priority": 4.5e-06, @@ -26400,6 +26401,7 @@ "input_cost_per_token_batches": 1.25e-07, "input_cost_per_token_flex": 1.25e-07, "input_cost_per_token_priority": 4.5e-07, + "input_cost_per_audio_token_priority": 9e-07, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 1048576, "max_output_tokens": 65536, @@ -26595,6 +26597,7 @@ "input_cost_per_token_batches": 5e-08, "input_cost_per_token_flex": 5e-08, "input_cost_per_token_priority": 1.8e-07, + "input_cost_per_audio_token_priority": 5.4e-07, "output_cost_per_token_batches": 2e-07, "output_cost_per_token_flex": 2e-07, "output_cost_per_token_priority": 7.2e-07, @@ -49499,6 +49502,7 @@ "input_cost_per_token_batches": 1.25e-07, "input_cost_per_token_flex": 1.25e-07, "input_cost_per_token_priority": 4.5e-07, + "input_cost_per_audio_token_priority": 9e-07, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 1048576, "max_output_tokens": 65536, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 926672b19a2..47cdfecfc0f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -26063,6 +26063,7 @@ "input_cost_per_token_batches": 1.5e-07, "input_cost_per_token_flex": 1.5e-07, "input_cost_per_token_priority": 5.4e-07, + "input_cost_per_audio_token_priority": 1.8e-06, "output_cost_per_token_batches": 1.25e-06, "output_cost_per_token_flex": 1.25e-06, "output_cost_per_token_priority": 4.5e-06, @@ -26400,6 +26401,7 @@ "input_cost_per_token_batches": 1.25e-07, "input_cost_per_token_flex": 1.25e-07, "input_cost_per_token_priority": 4.5e-07, + "input_cost_per_audio_token_priority": 9e-07, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 1048576, "max_output_tokens": 65536, @@ -26595,6 +26597,7 @@ "input_cost_per_token_batches": 5e-08, "input_cost_per_token_flex": 5e-08, "input_cost_per_token_priority": 1.8e-07, + "input_cost_per_audio_token_priority": 5.4e-07, "output_cost_per_token_batches": 2e-07, "output_cost_per_token_flex": 2e-07, "output_cost_per_token_priority": 7.2e-07, @@ -49499,6 +49502,7 @@ "input_cost_per_token_batches": 1.25e-07, "input_cost_per_token_flex": 1.25e-07, "input_cost_per_token_priority": 4.5e-07, + "input_cost_per_audio_token_priority": 9e-07, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 1048576, "max_output_tokens": 65536, From 571ada0b0f09c0356055626160f78c583a9b2998 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:36:25 -0700 Subject: [PATCH 73/96] feat(rust_bridge): mark native streams with the x-litellm-rust header (#42758) Non-streaming responses served by the Rust core already carry x-litellm-rust: true through _hidden_params.additional_headers, which the SDK exposes and the gateway renders as a response header. Native streams did not, because the lifecycle Stream and SyncStream objects had nowhere to hold hidden params and the marker writer skips objects without them. Give both stream classes the same _hidden_params bag every other litellm response has, so the existing marker attaches without wrapping the stream or changing its identity. Co-authored-by: Yujong Lee --- litellm/rust_bridge/lifecycle.py | 2 + .../test_litellm/rust_bridge/test_runtime.py | 56 +++++++++++++++++++ .../messages/test_callbacks.py | 3 + 3 files changed, 61 insertions(+) diff --git a/litellm/rust_bridge/lifecycle.py b/litellm/rust_bridge/lifecycle.py index b3b7a1888c3..2f243e8c212 100644 --- a/litellm/rust_bridge/lifecycle.py +++ b/litellm/rust_bridge/lifecycle.py @@ -81,6 +81,7 @@ class Stream(AsyncIterator[object]): def __init__(self, execution: Execution) -> None: self._execution: Final = execution self._done = False + self._hidden_params: dict[str, object] = {} # mutable-ok: header writers mutate _hidden_params in place def __aiter__(self) -> Stream: return self @@ -117,6 +118,7 @@ class SyncStream(Iterator[object]): def __init__(self, execution: Execution) -> None: self._execution: Final = execution self._done = False + self._hidden_params: dict[str, object] = {} # mutable-ok: header writers mutate _hidden_params in place def __iter__(self) -> SyncStream: return self diff --git a/tests/test_litellm/rust_bridge/test_runtime.py b/tests/test_litellm/rust_bridge/test_runtime.py index bff7ded3114..bbe25e0de13 100644 --- a/tests/test_litellm/rust_bridge/test_runtime.py +++ b/tests/test_litellm/rust_bridge/test_runtime.py @@ -12,6 +12,7 @@ from litellm.router_utils.add_retry_fallback_headers import get_hidden_params_di from litellm.rust_bridge import bindings, configuration, runtime from litellm.rust_bridge.catalog import Delivery, Route, RouteContext, RouteRule from litellm.rust_bridge.configuration import Rollout +from litellm.rust_bridge.lifecycle import Complete, Open, Stream, SyncStream, Yield class RustBridgeDeclined(Exception): @@ -264,6 +265,61 @@ async def test_native_response_marker_reaches_caller_with_existing_metadata(shap } +class ScriptedStreamExecution: + def __init__(self, chunks: tuple[bytes, ...]) -> None: + self._steps: Final = iter((*(Yield(chunk) for chunk in chunks), Complete(None))) + self.closed = False + + def start(self) -> Open: + return Open(None) + + def resume_value(self, value: object) -> Yield | Complete: + return next(self._steps) + + def resume_error(self, error: BaseException) -> Complete: + return Complete(None) + + def close(self) -> None: + self.closed = True + + +@pytest.mark.asyncio +@pytest.mark.parametrize("asynchronous", (False, True)) +async def test_native_stream_marker_reaches_caller_without_wrapping_or_consuming_the_stream( + asynchronous: bool, +) -> None: + chunks: Final = (b"event: message_start\n\n", b"event: message_stop\n\n") + execution: Final = ScriptedStreamExecution(chunks) + stream: Final[Stream | SyncStream] = Stream(execution) if asynchronous else SyncStream(execution) + bound: Final[bindings.NativeBinding[Callable[[], object]]] = bindings.NativeBinding( + "messages", validate=lambda _: None + ) + bound.override(lambda: stream) + + def python() -> object: + pytest.fail("native success must not fall back") + + async def anative(fn: Callable[[], object]) -> object: + return fn() + + async def apython() -> object: + return python() + + result: Final = ( + await runtime.arun(CONTEXT, binding=bound, native=anative, python=apython, rules=rules(Rollout.RUST_REQUIRED)) + if asynchronous + else runtime.run( + CONTEXT, binding=bound, native=lambda fn: fn(), python=python, rules=rules(Rollout.RUST_REQUIRED) + ) + ) + assert result is stream + assert get_hidden_params_dict(result) == {"additional_headers": {"x-litellm-rust": "true"}} + assert not execution.closed + delivered: Final = tuple([chunk async for chunk in result]) if isinstance(result, Stream) else tuple(result) + assert delivered == chunks + assert execution.closed + + def test_upstream_error_maps_to_api_error_without_fallback() -> None: calls: Final = recorder(RustUpstreamError(429, "rate limited")) diff --git a/tests/test_litellm_rust/messages/test_callbacks.py b/tests/test_litellm_rust/messages/test_callbacks.py index d5dc5437398..19043780eb6 100644 --- a/tests/test_litellm_rust/messages/test_callbacks.py +++ b/tests/test_litellm_rust/messages/test_callbacks.py @@ -5,6 +5,7 @@ import pytest import litellm from litellm.integrations.custom_logger import CustomLogger +from litellm.router_utils.add_retry_fallback_headers import get_hidden_params_dict from litellm.rust_bridge import catalog from litellm.rust_bridge.catalog import Route, RouteRule from litellm.rust_bridge.configuration import Rollout @@ -126,6 +127,7 @@ async def test_native_messages_stream_relays_provider_events_and_logs_success_on **arguments(messages_server, stream=True, callbacks=[recorder]) ) assert isinstance(stream, AsyncIterator) + assert get_hidden_params_dict(stream) == {"additional_headers": {"x-litellm-rust": "true"}} first: Final = await anext(stream) await drain_logging() assert "async_log_success_event" not in recorder.names @@ -169,6 +171,7 @@ def test_native_sync_messages_stream_relays_provider_events_and_logs_success_onc stream: Final = litellm.anthropic.messages.create(**arguments(messages_server, stream=True, callbacks=[recorder])) assert isinstance(stream, Iterator) + assert get_hidden_params_dict(stream) == {"additional_headers": {"x-litellm-rust": "true"}} assert b"".join(stream) == sse_payload() assert_served_natively(messages_server) From 7a9bfb4f1c7e251b25ee8920d586640a764c8a07 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:44:33 -0700 Subject: [PATCH 74/96] feat(azure): add gpt-audio and gpt-realtime alias rows from the Azure model list (#42981) * feat(azure): add gpt-audio and gpt-realtime alias rows from the Azure model list Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(azure): keep existing catalog formatting untouched Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 136 ++++++++++++++++++ model_prices_and_context_window.json | 136 ++++++++++++++++++ 2 files changed, 272 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 47cdfecfc0f..e20cd12c9be 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -5603,6 +5603,39 @@ "supports_tool_choice": true, "supports_vision": true }, + "azure/gpt-audio": { + "deprecation_date": "2027-03-02", + "input_cost_per_audio_token": 4e-05, + "input_cost_per_token": 2.5e-06, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_audio_token": 8e-05, + "output_cost_per_token": 1e-05, + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": false, + "supports_reasoning": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false, + "source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure" + }, "azure/gpt-audio-2025-08-28": { "deprecation_date": "2027-03-02", "input_cost_per_audio_token": 4e-05, @@ -5635,6 +5668,39 @@ "supports_tool_choice": true, "supports_vision": false }, + "azure/gpt-audio-1.5": { + "deprecation_date": "2027-08-24", + "input_cost_per_audio_token": 4e-05, + "input_cost_per_token": 2.5e-06, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_audio_token": 8e-05, + "output_cost_per_token": 1e-05, + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": false, + "supports_reasoning": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false, + "source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure" + }, "azure/gpt-audio-1.5-2026-02-23": { "deprecation_date": "2027-08-24", "input_cost_per_audio_token": 4e-05, @@ -5849,6 +5915,41 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "azure/gpt-realtime": { + "cache_creation_input_audio_token_cost": 4e-06, + "cache_read_input_audio_token_cost": 4e-07, + "cache_read_input_token_cost": 4e-07, + "deprecation_date": "2027-03-02", + "input_cost_per_audio_token": 3.2e-05, + "input_cost_per_image_token": 5e-06, + "input_cost_per_token": 4e-06, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "realtime", + "output_cost_per_audio_token": 6.4e-05, + "output_cost_per_token": 1.6e-05, + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure" + }, "azure/gpt-realtime-2025-08-28": { "cache_creation_input_audio_token_cost": 4e-06, "cache_read_input_audio_token_cost": 4e-07, @@ -5883,6 +5984,41 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "azure/gpt-realtime-1.5": { + "cache_creation_input_audio_token_cost": 4e-06, + "cache_read_input_audio_token_cost": 4e-07, + "cache_read_input_token_cost": 4e-07, + "deprecation_date": "2027-08-24", + "input_cost_per_audio_token": 3.2e-05, + "input_cost_per_image_token": 5e-06, + "input_cost_per_token": 4e-06, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "realtime", + "output_cost_per_audio_token": 6.4e-05, + "output_cost_per_token": 1.6e-05, + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure" + }, "azure/gpt-realtime-1.5-2026-02-23": { "cache_creation_input_audio_token_cost": 4e-06, "cache_read_input_audio_token_cost": 4e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 47cdfecfc0f..e20cd12c9be 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -5603,6 +5603,39 @@ "supports_tool_choice": true, "supports_vision": true }, + "azure/gpt-audio": { + "deprecation_date": "2027-03-02", + "input_cost_per_audio_token": 4e-05, + "input_cost_per_token": 2.5e-06, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_audio_token": 8e-05, + "output_cost_per_token": 1e-05, + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": false, + "supports_reasoning": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false, + "source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure" + }, "azure/gpt-audio-2025-08-28": { "deprecation_date": "2027-03-02", "input_cost_per_audio_token": 4e-05, @@ -5635,6 +5668,39 @@ "supports_tool_choice": true, "supports_vision": false }, + "azure/gpt-audio-1.5": { + "deprecation_date": "2027-08-24", + "input_cost_per_audio_token": 4e-05, + "input_cost_per_token": 2.5e-06, + "litellm_provider": "azure", + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "max_tokens": 16384, + "mode": "chat", + "output_cost_per_audio_token": 8e-05, + "output_cost_per_token": 1e-05, + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supported_modalities": [ + "text", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": false, + "supports_reasoning": false, + "supports_response_schema": false, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": false, + "source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure" + }, "azure/gpt-audio-1.5-2026-02-23": { "deprecation_date": "2027-08-24", "input_cost_per_audio_token": 4e-05, @@ -5849,6 +5915,41 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "azure/gpt-realtime": { + "cache_creation_input_audio_token_cost": 4e-06, + "cache_read_input_audio_token_cost": 4e-07, + "cache_read_input_token_cost": 4e-07, + "deprecation_date": "2027-03-02", + "input_cost_per_audio_token": 3.2e-05, + "input_cost_per_image_token": 5e-06, + "input_cost_per_token": 4e-06, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "realtime", + "output_cost_per_audio_token": 6.4e-05, + "output_cost_per_token": 1.6e-05, + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure" + }, "azure/gpt-realtime-2025-08-28": { "cache_creation_input_audio_token_cost": 4e-06, "cache_read_input_audio_token_cost": 4e-07, @@ -5883,6 +5984,41 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "azure/gpt-realtime-1.5": { + "cache_creation_input_audio_token_cost": 4e-06, + "cache_read_input_audio_token_cost": 4e-07, + "cache_read_input_token_cost": 4e-07, + "deprecation_date": "2027-08-24", + "input_cost_per_audio_token": 3.2e-05, + "input_cost_per_image_token": 5e-06, + "input_cost_per_token": 4e-06, + "litellm_provider": "azure", + "max_input_tokens": 32000, + "max_output_tokens": 4096, + "max_tokens": 4096, + "mode": "realtime", + "output_cost_per_audio_token": 6.4e-05, + "output_cost_per_token": 1.6e-05, + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "source": "https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure" + }, "azure/gpt-realtime-1.5-2026-02-23": { "cache_creation_input_audio_token_cost": 4e-06, "cache_read_input_audio_token_cost": 4e-07, From 1ceb8fb08e778506816766c001d6fc9121c3d46e Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:45:09 -0700 Subject: [PATCH 75/96] fix(cost-map): align openrouter deepseek-v4-pro cache hit price with cache read (#42985) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 2 +- model_prices_and_context_window.json | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e20cd12c9be..65f90014151 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41382,7 +41382,7 @@ }, "openrouter/deepseek/deepseek-v4-pro": { "input_cost_per_token": 9.1263e-07, - "input_cost_per_token_cache_hit": 4.4e-08, + "input_cost_per_token_cache_hit": 7.60525e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e20cd12c9be..65f90014151 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41382,7 +41382,7 @@ }, "openrouter/deepseek/deepseek-v4-pro": { "input_cost_per_token": 9.1263e-07, - "input_cost_per_token_cache_hit": 4.4e-08, + "input_cost_per_token_cache_hit": 7.60525e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, From de314f8271788593087c3f6065b7d82e677a7a31 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:57:05 -0700 Subject: [PATCH 76/96] fix(cost-map): take the later azure deprecation date for gpt-4.1-nano, gpt-4o-transcribe and gpt-realtime-2.1 (#42986) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../model_prices_and_context_window_backup.json | 14 +++++++------- model_prices_and_context_window.json | 14 +++++++------- 2 files changed, 14 insertions(+), 14 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 65f90014151..8fa9740ee84 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -5442,7 +5442,7 @@ "supports_web_search": false }, "azure/gpt-4.1-nano": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -5476,7 +5476,7 @@ "supports_vision": true }, "azure/gpt-4.1-nano-2025-04-14": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -6057,7 +6057,7 @@ "cache_creation_input_audio_token_cost": 4e-07, "cache_read_input_audio_token_cost": 4e-07, "cache_read_input_token_cost": 4e-07, - "deprecation_date": "2027-06-25", + "deprecation_date": "2027-07-31", "input_cost_per_audio_token": 3.2e-05, "input_cost_per_image_token": 5e-06, "input_cost_per_token": 4e-06, @@ -6269,7 +6269,7 @@ "supports_tool_choice": true }, "azure/gpt-4o-transcribe": { - "deprecation_date": "2026-10-15", + "deprecation_date": "2026-12-31", "input_cost_per_audio_token": 2.5e-06, "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", @@ -10796,7 +10796,7 @@ "supports_web_search": false }, "azure/us/gpt-4.1-nano-2025-04-14": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -69100,7 +69100,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-4.1-nano": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -69535,7 +69535,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/us/gpt-4.1-nano": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 65f90014151..8fa9740ee84 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -5442,7 +5442,7 @@ "supports_web_search": false }, "azure/gpt-4.1-nano": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -5476,7 +5476,7 @@ "supports_vision": true }, "azure/gpt-4.1-nano-2025-04-14": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.5e-08, "input_cost_per_token": 1e-07, "input_cost_per_token_batches": 5e-08, @@ -6057,7 +6057,7 @@ "cache_creation_input_audio_token_cost": 4e-07, "cache_read_input_audio_token_cost": 4e-07, "cache_read_input_token_cost": 4e-07, - "deprecation_date": "2027-06-25", + "deprecation_date": "2027-07-31", "input_cost_per_audio_token": 3.2e-05, "input_cost_per_image_token": 5e-06, "input_cost_per_token": 4e-06, @@ -6269,7 +6269,7 @@ "supports_tool_choice": true }, "azure/gpt-4o-transcribe": { - "deprecation_date": "2026-10-15", + "deprecation_date": "2026-12-31", "input_cost_per_audio_token": 2.5e-06, "input_cost_per_token": 2.5e-06, "litellm_provider": "azure", @@ -10796,7 +10796,7 @@ "supports_web_search": false }, "azure/us/gpt-4.1-nano-2025-04-14": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -69100,7 +69100,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/eu/gpt-4.1-nano": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, @@ -69535,7 +69535,7 @@ "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" }, "azure/us/gpt-4.1-nano": { - "deprecation_date": "2026-10-14", + "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 2.8e-08, "input_cost_per_token": 1.1e-07, "input_cost_per_token_batches": 5.5e-08, From 793627ce83e875209d5705ce544ba52d0371bd62 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 09:57:52 -0700 Subject: [PATCH 77/96] fix(cost-map): add supports_reasoning to azure/eu/gpt-6-astra (#42989) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 3 ++- model_prices_and_context_window.json | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 8fa9740ee84..8088fa28b52 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -69288,7 +69288,8 @@ "mode": "chat", "output_cost_per_token": 5.5e-05, "output_cost_per_token_above_272k_tokens": 8.25e-05, - "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_reasoning": true }, "azure/eu/gpt-6-luna": { "deprecation_date": "2028-03-11", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 8fa9740ee84..8088fa28b52 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -69288,7 +69288,8 @@ "mode": "chat", "output_cost_per_token": 5.5e-05, "output_cost_per_token_above_272k_tokens": 8.25e-05, - "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'" + "source": "https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'", + "supports_reasoning": true }, "azure/eu/gpt-6-luna": { "deprecation_date": "2028-03-11", From 57eb3ff6b66acb87f9c8776292bb10fcbafe1c24 Mon Sep 17 00:00:00 2001 From: joshua-berri Date: Thu, 24 Sep 2026 17:23:30 +0000 Subject: [PATCH 78/96] fix(mcp): reject origins outside the configured allowlist (#42649) Co-authored-by: Joshua Valluru <326636767+joshua-berri@users.noreply.github.com> --- .../proxy/_experimental/mcp_server/server.py | 9 +++ .../mcp_server/test_mcp_server.py | 58 +++++++++++++++++++ 2 files changed, 67 insertions(+) diff --git a/litellm/proxy/_experimental/mcp_server/server.py b/litellm/proxy/_experimental/mcp_server/server.py index fe508fc22c7..6261f36983d 100644 --- a/litellm/proxy/_experimental/mcp_server/server.py +++ b/litellm/proxy/_experimental/mcp_server/server.py @@ -111,6 +111,13 @@ _MCP_DESTINATIONS_SCOPE_KEY: Final = "litellm_otel_request_destinations" _MCP_PROTOCOL_VERSION_HEADER: Final = b"mcp-protocol-version" +def reject_disallowed_mcp_origin(request: StarletteRequest) -> None: + from litellm.proxy.proxy_server import origins # noqa: PLC0415 # proxy imports this module during startup + + if "*" not in origins and any(origin not in origins for origin in request.headers.getlist("origin")): + raise HTTPException(status_code=403, detail="Invalid Origin header") + + def unsupported_protocol_version(scope: Scope) -> str | None: """Return the unsupported ``MCP-Protocol-Version`` header value, if any. @@ -1931,6 +1938,7 @@ if MCP_AVAILABLE: async def handle_streamable_http_mcp(scope: Scope, receive: Receive, send: Send) -> None: """Handle MCP requests through StreamableHTTP.""" try: + reject_disallowed_mcp_origin(StarletteRequest(scope)) bad_version: Final = unsupported_protocol_version(scope) if bad_version is not None: supported: Final = ", ".join(sorted(HANDSHAKE_PROTOCOL_VERSIONS)) @@ -2275,6 +2283,7 @@ if MCP_AVAILABLE: async def handle_sse_mcp(scope: Scope, receive: Receive, send: Send) -> None: """Handle MCP requests through SSE.""" try: + reject_disallowed_mcp_origin(StarletteRequest(scope)) bad_version: Final = unsupported_protocol_version(scope) if bad_version is not None: supported: Final = ", ".join(sorted(HANDSHAKE_PROTOCOL_VERSIONS)) diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py index ea82c9b069d..6887adf8283 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_mcp_server.py @@ -10201,6 +10201,64 @@ async def test_active_request_ctx_var_feeds_get_current_session(_mcp_request_ctx assert _get_current_session() is None +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("method", "path", "session_headers"), + ( + ("POST", "/mcp", ()), + ("GET", "/mcp", (("mcp-session-id", "existing-session"),)), + ("DELETE", "/mcp", (("mcp-session-id", "existing-session"),)), + ("POST", "/server/mcp", ()), + ("GET", "/sse", ()), + ("POST", "/sse/messages", ()), + ), +) +@pytest.mark.parametrize( + ("allowed_origins", "origin_headers", "expected_status"), + ( + (("https://allowed.example",), (("origin", "https://evil.example"),), 403), + (("https://allowed.example",), (("origin", "https://allowed.example.evil.example"),), 403), + (("https://allowed.example",), (("origin", "null"),), 403), + (("https://allowed.example",), (("origin", ""),), 403), + ( + ("https://allowed.example",), + (("origin", "https://allowed.example"), ("origin", "https://evil.example")), + 403, + ), + (("https://allowed.example",), (("origin", "https://allowed.example"),), 401), + (("https://allowed.example",), (), 401), + (("*",), (("origin", "https://another.example"),), 401), + ), +) +async def test_mcp_origin_admission_precedes_authentication( + method: str, + path: str, + session_headers: tuple[tuple[str, str], ...], + allowed_origins: tuple[str, ...], + origin_headers: tuple[tuple[str, str], ...], + expected_status: int, +) -> None: + import httpx + + from litellm.proxy._experimental.mcp_server import server + + authenticate: Final = AsyncMock(side_effect=HTTPException(status_code=401, detail="authentication required")) + with ( + patch("litellm.proxy.proxy_server.origins", allowed_origins), + patch.object(server, "extract_mcp_auth_context", authenticate), + ): + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=server.app), base_url="http://gateway") as client: + response: Final = await client.request(method, path, headers=(*session_headers, *origin_headers)) + + assert response.status_code == expected_status + if expected_status == 403: + assert response.json() == {"detail": "Invalid Origin header"} + authenticate.assert_not_awaited() + else: + assert response.json() == {"detail": "authentication required"} + authenticate.assert_awaited_once() + + @pytest.mark.asyncio async def test_active_request_ctx_var_feeds_auth_resolution_recording(_mcp_request_ctx) -> None: from starlette.requests import Request From 0fa1fe2059513f0bb387959894057d4a12a56657 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 10:37:59 -0700 Subject: [PATCH 79/96] fix(cost-map): drop stale cache_hit field from openrouter deepseek-v4-pro (#42994) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 1 - model_prices_and_context_window.json | 1 - 2 files changed, 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 8088fa28b52..a338c389ea4 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -41382,7 +41382,6 @@ }, "openrouter/deepseek/deepseek-v4-pro": { "input_cost_per_token": 9.1263e-07, - "input_cost_per_token_cache_hit": 7.60525e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 8088fa28b52..a338c389ea4 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -41382,7 +41382,6 @@ }, "openrouter/deepseek/deepseek-v4-pro": { "input_cost_per_token": 9.1263e-07, - "input_cost_per_token_cache_hit": 7.60525e-08, "litellm_provider": "openrouter", "max_input_tokens": 1048576, "max_output_tokens": 384000, From d3462c65b585eb7e51594330a5785669f6b7db51 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 11:08:09 -0700 Subject: [PATCH 80/96] fix(cost-map): add vertex cache read, batch and above 200k prices for gemini image preview rows (#42995) * fix(cost-map): add vertex cache read prices for gemini image preview rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost-map): add batch cache read and above 200k tiers to vertex gemini image preview rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...model_prices_and_context_window_backup.json | 18 ++++++++++++++++++ model_prices_and_context_window.json | 18 ++++++++++++++++++ 2 files changed, 36 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a338c389ea4..2600da22a77 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -26312,7 +26312,11 @@ }, "gemini-3-pro-image-preview": { "input_cost_per_image": 0.0011, + "cache_read_input_token_cost": 2e-07, + "cache_read_input_token_cost_above_200k_tokens": 4e-07, + "cache_read_input_token_cost_batches": 1e-07, "input_cost_per_token": 2e-06, + "input_cost_per_token_above_200k_tokens": 4e-06, "input_cost_per_token_batches": 1e-06, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 65536, @@ -26322,6 +26326,7 @@ "output_cost_per_image": 0.134, "output_cost_per_image_token": 0.00012, "output_cost_per_token": 1.2e-05, + "output_cost_per_token_above_200k_tokens": 1.8e-05, "output_cost_per_token_batches": 6e-06, "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_endpoints": [ @@ -26398,7 +26403,10 @@ }, "gemini-3.1-flash-image-preview": { "input_cost_per_image": 0.00056, + "cache_read_input_token_cost": 5e-08, + "cache_read_input_token_cost_batches": 2.5e-08, "input_cost_per_token": 5e-07, + "input_cost_per_token_batches": 2.5e-07, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 65536, "max_output_tokens": 32768, @@ -26407,6 +26415,7 @@ "output_cost_per_image": 0.0672, "output_cost_per_image_token": 6e-05, "output_cost_per_token": 3e-06, + "output_cost_per_token_batches": 1.5e-06, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models", "supported_endpoints": [ "/v1/chat/completions", @@ -49484,7 +49493,11 @@ }, "vertex_ai/gemini-3-pro-image-preview": { "input_cost_per_image": 0.0011, + "cache_read_input_token_cost": 2e-07, + "cache_read_input_token_cost_above_200k_tokens": 4e-07, + "cache_read_input_token_cost_batches": 1e-07, "input_cost_per_token": 2e-06, + "input_cost_per_token_above_200k_tokens": 4e-06, "input_cost_per_token_batches": 1e-06, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 65536, @@ -49494,6 +49507,7 @@ "output_cost_per_image": 0.134, "output_cost_per_image_token": 0.00012, "output_cost_per_token": 1.2e-05, + "output_cost_per_token_above_200k_tokens": 1.8e-05, "output_cost_per_token_batches": 6e-06, "supports_reasoning": false, "source": "https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/3-pro-image" @@ -49522,7 +49536,10 @@ }, "vertex_ai/gemini-3.1-flash-image-preview": { "input_cost_per_image": 0.00056, + "cache_read_input_token_cost": 5e-08, + "cache_read_input_token_cost_batches": 2.5e-08, "input_cost_per_token": 5e-07, + "input_cost_per_token_batches": 2.5e-07, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 65536, "max_output_tokens": 32768, @@ -49531,6 +49548,7 @@ "output_cost_per_image": 0.0672, "output_cost_per_image_token": 6e-05, "output_cost_per_token": 3e-06, + "output_cost_per_token_batches": 1.5e-06, "supports_reasoning": false, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models" }, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a338c389ea4..2600da22a77 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -26312,7 +26312,11 @@ }, "gemini-3-pro-image-preview": { "input_cost_per_image": 0.0011, + "cache_read_input_token_cost": 2e-07, + "cache_read_input_token_cost_above_200k_tokens": 4e-07, + "cache_read_input_token_cost_batches": 1e-07, "input_cost_per_token": 2e-06, + "input_cost_per_token_above_200k_tokens": 4e-06, "input_cost_per_token_batches": 1e-06, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 65536, @@ -26322,6 +26326,7 @@ "output_cost_per_image": 0.134, "output_cost_per_image_token": 0.00012, "output_cost_per_token": 1.2e-05, + "output_cost_per_token_above_200k_tokens": 1.8e-05, "output_cost_per_token_batches": 6e-06, "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_endpoints": [ @@ -26398,7 +26403,10 @@ }, "gemini-3.1-flash-image-preview": { "input_cost_per_image": 0.00056, + "cache_read_input_token_cost": 5e-08, + "cache_read_input_token_cost_batches": 2.5e-08, "input_cost_per_token": 5e-07, + "input_cost_per_token_batches": 2.5e-07, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 65536, "max_output_tokens": 32768, @@ -26407,6 +26415,7 @@ "output_cost_per_image": 0.0672, "output_cost_per_image_token": 6e-05, "output_cost_per_token": 3e-06, + "output_cost_per_token_batches": 1.5e-06, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models", "supported_endpoints": [ "/v1/chat/completions", @@ -49484,7 +49493,11 @@ }, "vertex_ai/gemini-3-pro-image-preview": { "input_cost_per_image": 0.0011, + "cache_read_input_token_cost": 2e-07, + "cache_read_input_token_cost_above_200k_tokens": 4e-07, + "cache_read_input_token_cost_batches": 1e-07, "input_cost_per_token": 2e-06, + "input_cost_per_token_above_200k_tokens": 4e-06, "input_cost_per_token_batches": 1e-06, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 65536, @@ -49494,6 +49507,7 @@ "output_cost_per_image": 0.134, "output_cost_per_image_token": 0.00012, "output_cost_per_token": 1.2e-05, + "output_cost_per_token_above_200k_tokens": 1.8e-05, "output_cost_per_token_batches": 6e-06, "supports_reasoning": false, "source": "https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/3-pro-image" @@ -49522,7 +49536,10 @@ }, "vertex_ai/gemini-3.1-flash-image-preview": { "input_cost_per_image": 0.00056, + "cache_read_input_token_cost": 5e-08, + "cache_read_input_token_cost_batches": 2.5e-08, "input_cost_per_token": 5e-07, + "input_cost_per_token_batches": 2.5e-07, "litellm_provider": "vertex_ai-language-models", "max_input_tokens": 65536, "max_output_tokens": 32768, @@ -49531,6 +49548,7 @@ "output_cost_per_image": 0.0672, "output_cost_per_image_token": 6e-05, "output_cost_per_token": 3e-06, + "output_cost_per_token_batches": 1.5e-06, "supports_reasoning": false, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models" }, From 0ac435bd0195ec224c8399f0655ace3fd0d3c68d Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 24 Sep 2026 11:54:26 -0700 Subject: [PATCH 81/96] test(ollama): assert the images sent to Ollama instead of echoing them through response (#42905) test_ollama_image returned the request's images list as the mocked `response` field and read it back from message content. Since #42838 the completion transform validates `response` as a string, so the list is dropped and the test fails with "string index out of range". Ollama always sends a string there. The mock now returns a real string reply and the test asserts on the images the transform actually sent, which is what it was checking all along. --- tests/local_testing/test_completion.py | 17 ++++++++--------- 1 file changed, 8 insertions(+), 9 deletions(-) diff --git a/tests/local_testing/test_completion.py b/tests/local_testing/test_completion.py index 25c6c50251d..c6dd78c73b4 100644 --- a/tests/local_testing/test_completion.py +++ b/tests/local_testing/test_completion.py @@ -1361,16 +1361,14 @@ def test_ollama_image(): from PIL import Image + sent_images = [] + def mock_post(url, **kwargs): + sent_images.append(json.loads(kwargs["data"])["images"]) mock_response = MagicMock() mock_response.status_code = 200 mock_response.headers = {"Content-Type": "application/json"} - data_json = json.loads(kwargs["data"]) - mock_response.json.return_value = { - # return the image in the response so that it can be tested - # against the original - "response": data_json["images"] - } + mock_response.json.return_value = {"response": "a black pixel"} return mock_response def make_b64image(format): @@ -1399,9 +1397,10 @@ def test_ollama_image(): client = HTTPHandler() for test in tests: + sent_images.clear() try: with patch.object(client, "post", side_effect=mock_post): - response = completion( + completion( model="ollama/llava", messages=[ { @@ -1417,14 +1416,14 @@ def test_ollama_image(): ], client=client, ) + (image_data,) = sent_images[0] if not test[1]: # the conversion process may not always generate the same image, # so just check for a JPEG image when a conversion was done. - image_data = response["choices"][0]["message"]["content"][0] image = Image.open(io.BytesIO(base64.b64decode(image_data))) assert image.format == "JPEG" else: - assert response["choices"][0]["message"]["content"][0] == test[1] + assert image_data == test[1] except Exception as e: pytest.fail(f"Error occurred: {e}") From 12960f3edfb1b3e26cd50fe567f242d9df8c50e4 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 11:54:52 -0700 Subject: [PATCH 82/96] fix(ui): pass is_proxy_admin for proxy admins on the models page team drill-in (#43003) * fix(ui): pass is_proxy_admin for proxy admins on the models page team drill-in Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(ui): drop explanatory comments from the models page team drill-in tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(ui): exclude view-only admins from is_proxy_admin on the models page team drill-in Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(ui): add browser integration contract for the team guardrail kill switch on the models page Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../tests/integrationCritical/expected.json | 1 + .../teamGlobalGuardrailKillSwitch.spec.ts | 77 ++++++ .../page.integration.test.tsx | 243 ++++++++++++++++++ .../models-and-endpoints/page.test.tsx | 23 +- .../(dashboard)/models-and-endpoints/page.tsx | 2 +- 5 files changed, 340 insertions(+), 6 deletions(-) create mode 100644 tests/e2e/ui/tests/integrationCritical/teamGlobalGuardrailKillSwitch.spec.ts create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.integration.test.tsx diff --git a/tests/e2e/ui/tests/integrationCritical/expected.json b/tests/e2e/ui/tests/integrationCritical/expected.json index 0354c9f729c..189cdef9a93 100644 --- a/tests/e2e/ui/tests/integrationCritical/expected.json +++ b/tests/e2e/ui/tests/integrationCritical/expected.json @@ -1,5 +1,6 @@ [ "tests/e2e/ui/tests/integrationCritical/projectDetachment.spec.ts::project creation and explicit detachment preserve saved scope and restore serving", + "tests/e2e/ui/tests/integrationCritical/teamGlobalGuardrailKillSwitch.spec.ts::proxy admin can enable the global guardrail kill switch from the models page team drill-in", "tests/e2e/ui/tests/integrationCritical/mcpUserEnvVars.spec.ts::per-user MCP env var stays updatable and clearable from the card after it is set", "tests/e2e/ui/tests/integrationCritical/mcpUserEnvVars.spec.ts::cancelling the clear confirmation keeps the stored value and sends no delete", "tests/e2e/ui/tests/integrationCritical/mcpUserEnvVars.spec.ts::pressing Enter on Update opens the credentials modal instead of the server editor", diff --git a/tests/e2e/ui/tests/integrationCritical/teamGlobalGuardrailKillSwitch.spec.ts b/tests/e2e/ui/tests/integrationCritical/teamGlobalGuardrailKillSwitch.spec.ts new file mode 100644 index 00000000000..ac48dd1a770 --- /dev/null +++ b/tests/e2e/ui/tests/integrationCritical/teamGlobalGuardrailKillSwitch.spec.ts @@ -0,0 +1,77 @@ +import { + test, + expect, + APIRequestContext, + Page as PlaywrightPage, +} from "@playwright/test"; +import { randomUUID } from "node:crypto"; + +const master = process.env.LITELLM_MASTER_KEY ?? "sk-integration-master"; +const headers = { Authorization: `Bearer ${master}` }; + +async function createTeam(request: APIRequestContext): Promise { + const created = await request.post("/team/new", { + headers, + data: { + team_alias: `int_kill_switch_${randomUUID().replace(/-/g, "").slice(0, 12)}`, + }, + }); + expect(created.ok(), await created.text()).toBe(true); + return (await created.json()).team_id as string; +} + +async function loginAsAdmin(page: PlaywrightPage): Promise { + await page.goto("/ui/login"); + await page.getByPlaceholder("Enter your username").fill("admin"); + await page.getByPlaceholder("Enter your password").fill(master); + await page.getByRole("button", { name: "Login", exact: true }).click(); + await expect(page).toHaveURL( + (url) => url.pathname.startsWith("/ui") && !url.pathname.includes("login"), + ); +} + +test("proxy admin can enable the global guardrail kill switch from the models page team drill-in", async ({ + page, + request, +}) => { + const teamId = await createTeam(request); + try { + await loginAsAdmin(page); + await page.goto(`/ui/models-and-endpoints?team=${teamId}`); + await page.getByRole("tab", { name: "Settings" }).click(); + await page.getByRole("button", { name: /edit settings/i }).click(); + await expect(page.getByLabel(/Team Name/)).toBeVisible(); + const killSwitch = page.getByRole("switch", { + name: /disable all global guardrails/i, + }); + await expect(killSwitch).toBeVisible(); + await expect(killSwitch).not.toBeChecked(); + await killSwitch.click(); + await page.getByRole("button", { name: "Save Changes" }).click(); + await expect + .poll(async () => { + const response = await request.get(`/team/info?team_id=${teamId}`, { + headers, + }); + expect(response.ok(), await response.text()).toBe(true); + const json = await response.json(); + return json.team_info?.metadata?.disable_global_guardrails; + }) + .toBe(true); + await page.reload(); + await page.getByRole("tab", { name: "Settings" }).click(); + await page.getByRole("button", { name: /edit settings/i }).click(); + await expect(page.getByLabel(/Team Name/)).toBeVisible(); + await expect( + page.getByRole("switch", { name: /disable all global guardrails/i }), + ).toBeChecked(); + } finally { + const removed = await request.post("/team/delete", { + headers, + data: { team_ids: [teamId] }, + }); + expect(removed.ok() || removed.status() === 404, await removed.text()).toBe( + true, + ); + } +}); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.integration.test.tsx new file mode 100644 index 00000000000..865cc808627 --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.integration.test.tsx @@ -0,0 +1,243 @@ +/* @vitest-environment jsdom */ +import { screen } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import * as networking from "@/components/networking"; +import { renderWithProviders } from "../../../../tests/test-utils"; +import ModelsAndEndpointsPage from "./page"; + +vi.mock("./panels/AllModelsPanel", () => ({ default: () =>
})); +vi.mock("./panels/AddModelPanel", () => ({ default: () =>
})); +vi.mock("./panels/AutoRoutersTabPanel", () => ({ default: () =>
})); +vi.mock("./panels/LlmCredentialsPanel", () => ({ default: () =>
})); +vi.mock("./panels/PassThroughPanel", () => ({ default: () =>
})); +vi.mock("./panels/HealthStatusPanel", () => ({ default: () =>
})); +vi.mock("./panels/ModelRetrySettingsPanel", () => ({ default: () =>
})); +vi.mock("./panels/ModelGroupAliasPanel", () => ({ default: () =>
})); +vi.mock("./panels/PriceDataPanel", () => ({ default: () =>
})); +vi.mock("./panels/AccessGroupBudgetsPanel", () => ({ default: () =>
})); +vi.mock("@/components/molecules/cost_optimization_feedback_banner", () => ({ default: () => null })); +vi.mock("@/components/model_info_view", () => ({ + default: ({ modelId }: { modelId: string }) =>
model:{modelId}
, +})); +vi.mock("./useModelDashboardData", () => ({ + useModelDashboardData: () => ({ availableModelAccessGroups: [], allModelsOnProxy: [], availableModelGroups: [] }), +})); + +const authState = vi.hoisted(() => ({ userRole: "Admin", isViewOnly: false })); +vi.mock("@/app/(dashboard)/hooks/useAuthorized", () => ({ + default: () => ({ + token: "123", + accessToken: "123", + userId: "user-1", + userEmail: "admin@example.com", + userRole: authState.userRole, + premiumUser: true, + isViewOnly: authState.isViewOnly, + disabledPersonalKeyCreation: null, + showSSOBanner: false, + }), +})); + +vi.mock("next/navigation", () => ({ useRouter: () => ({ push: vi.fn() }) })); + +vi.mock("@/components/networking", () => ({ + serverRootPath: "", + teamInfoCall: vi.fn(), + teamMemberDeleteCall: vi.fn(), + teamMemberAddCall: vi.fn(), + teamMemberUpdateCall: vi.fn(), + teamUpdateCall: vi.fn(), + getGuardrailsList: vi.fn(), + getPoliciesList: vi.fn(), + getPolicyInfoWithGuardrails: vi.fn(), + fetchMCPAccessGroups: vi.fn(), + getTeamPermissionsCall: vi.fn(), + organizationInfoCall: vi.fn(), + getRouterSettingsCall: vi.fn().mockResolvedValue({ fields: [] }), + getPassThroughEndpointsCall: vi.fn().mockResolvedValue({ endpoints: [] }), + fetchMCPServers: vi.fn().mockResolvedValue([]), + fetchMCPToolsets: vi.fn().mockResolvedValue([]), + listMCPTools: vi.fn().mockResolvedValue({ tools: [] }), + vectorStoreListCall: vi.fn().mockResolvedValue({ data: [] }), + getAgentsList: vi.fn().mockResolvedValue({ agents: [] }), + getClaudeCodePluginsList: vi.fn().mockResolvedValue({ plugins: [], count: 0 }), +})); + +const can = vi.fn(); +vi.mock("@/app/(dashboard)/hooks/useCan", () => ({ + default: (...args: unknown[]) => can(...args), +})); + +vi.mock("@/components/utils/dataUtils", () => ({ + copyToClipboard: vi.fn().mockResolvedValue(true), + formatNumberWithCommas: vi.fn((value: number) => value.toLocaleString()), +})); + +vi.mock("@/app/(dashboard)/hooks/teams/useTeamMetadataSchema", () => ({ + useTeamMetadataSchema: vi.fn(() => ({ data: [], isLoading: false })), +})); + +vi.mock("@/app/(dashboard)/hooks/uiSettings/useUISettings", () => ({ + useUISettings: vi.fn(() => ({ data: { values: {} }, isLoading: false })), +})); + +vi.mock("@/app/(dashboard)/hooks/models/useModels", () => ({ + useAllProxyModels: vi.fn(() => ({ data: { data: [] }, isLoading: false })), +})); + +vi.mock("@/app/(dashboard)/hooks/teams/useTeams", () => ({ + useTeams: vi.fn(() => ({ data: [], isLoading: false })), + useTeam: vi.fn(() => ({ data: undefined, isLoading: false })), +})); + +vi.mock("@/app/(dashboard)/hooks/organizations/useOrganizations", () => ({ + organizationKeys: { all: ["organizations"] }, + useOrganization: vi.fn(() => ({ data: undefined, isLoading: false })), + useOrganizations: vi.fn().mockReturnValue({ data: [], isLoading: false }), +})); + +vi.mock("@/app/(dashboard)/hooks/users/useCurrentUser", () => ({ + useCurrentUser: vi.fn(() => ({ data: { models: [] }, isLoading: false })), +})); + +vi.mock("@/app/(dashboard)/hooks/mcpServers/useMCPServers", () => ({ + useMCPServers: vi.fn(() => ({ data: [], isLoading: false, isError: false })), +})); + +vi.mock("@/app/(dashboard)/hooks/mcpServers/useMCPToolsets", () => ({ + useMCPToolsets: vi.fn(() => ({ data: [], isLoading: false, isError: false })), +})); + +vi.mock("@/components/mcp_server_management/MCPServerSelector", () => ({ + default: () =>
mcp server selector
, +})); + +vi.mock("@/components/team/TeamMemberTab", () => ({ + default: vi.fn(() =>
member tab
), +})); + +vi.mock("@/components/common_components/user_search_modal", () => ({ + default: vi.fn(() => null), +})); + +vi.mock("@/components/team/EditMembership", () => ({ + default: vi.fn(() => null), +})); + +vi.mock("@/components/common_components/DeleteResourceModal", () => ({ + default: vi.fn(() => null), +})); + +vi.mock("@/components/team/member_permissions", () => ({ + default: vi.fn(() =>
Member Permissions
), +})); + +vi.mock("@/components/common_components/ModelAliasManager", () => ({ + default: vi.fn(() =>
alias manager
), +})); + +vi.mock("@/app/(dashboard)/hooks/accessGroups/useAccessGroups", () => ({ + useAccessGroups: vi.fn().mockReturnValue({ data: [], isLoading: false, isError: false }), +})); + +vi.mock("@/components/common_components/AccessGroupSelector", () => ({ + default: () =>
access group selector
, +})); + +vi.mock("@/app/(dashboard)/hooks/keys/useKeys", () => { + const keysResult = { + data: { keys: [], total_count: 0, current_page: 1, total_pages: 1 }, + isPending: false, + isFetching: false, + refetch: vi.fn(), + }; + return { useKeys: vi.fn(() => keysResult) }; +}); + +vi.mock("@/components/key_team_helpers/filter_helpers", () => ({ + fetchTeamFilterOptions: vi.fn().mockResolvedValue({ keyAliases: [], organizationIds: [], userIds: [] }), + fetchAllKeyAliases: vi.fn().mockResolvedValue([]), + fetchAllOrganizations: vi.fn().mockResolvedValue([]), +})); + +const createMockTeamData = (overrides = {}) => ({ + team_id: "team-a1b2", + team_info: { + team_alias: "Test Team", + team_id: "team-a1b2", + organization_id: null, + admins: ["admin@test.com"], + members: ["user1@test.com"], + members_with_roles: [ + { user_id: "user1@test.com", user_email: "user1@test.com", role: "member", spend: 0, budget_id: "budget1" }, + ], + metadata: { disable_global_guardrails: true }, + tpm_limit: null, + rpm_limit: null, + max_budget: null, + budget_duration: null, + models: [], + blocked: false, + spend: 0, + max_parallel_requests: null, + budget_reset_at: null, + model_id: null, + litellm_model_table: null, + created_at: "2024-01-01T00:00:00Z", + team_member_budget_table: null, + guardrails: [], + policies: [], + object_permission: null, + ...overrides, + }, + keys: [], + team_memberships: [], +}); + +describe("ModelsAndEndpointsPage ?team drill-in", () => { + beforeEach(() => { + authState.userRole = "Admin"; + authState.isViewOnly = false; + can.mockReturnValue(true); + vi.mocked(networking.getGuardrailsList).mockResolvedValue({ guardrails: [] }); + vi.mocked(networking.getPoliciesList).mockResolvedValue({ policies: [] }); + vi.mocked(networking.fetchMCPAccessGroups).mockResolvedValue([]); + vi.mocked(networking.getTeamPermissionsCall).mockResolvedValue({ + all_available_permissions: [], + team_member_permissions: [], + } as never); + vi.mocked(networking.teamInfoCall).mockResolvedValue(createMockTeamData() as never); + // eslint-disable-next-line @typescript-eslint/no-explicit-any -- jsdom has no ResizeObserver global to type against + (global as any).ResizeObserver = class { + observe() {} + unobserve() {} + disconnect() {} + }; + }); + + afterEach(() => { + vi.clearAllMocks(); + }); + + it("shows the Disable all global guardrails switch to a proxy admin session", async () => { + const user = userEvent.setup({ delay: null }); + renderWithProviders(, { searchParams: { team: "team-a1b2" } }); + + await user.click(await screen.findByRole("tab", { name: "Settings" })); + await user.click(await screen.findByRole("button", { name: /edit settings/i })); + await screen.findByLabelText(/Team Name/); + + expect(screen.getByRole("switch", { name: /Disable all global guardrails/i })).toBeChecked(); + }); + + it("keeps the switch hidden from an internal user session on the same team", async () => { + authState.userRole = "Internal User"; + renderWithProviders(, { searchParams: { team: "team-a1b2" } }); + + await screen.findByText("Test Team"); + + expect(screen.queryByRole("button", { name: /edit settings/i })).not.toBeInTheDocument(); + expect(screen.queryByText("Disable all global guardrails")).not.toBeInTheDocument(); + }); +}); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.test.tsx index 105f6ff3043..652a3e804db 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.test.tsx @@ -25,12 +25,16 @@ vi.mock("@/components/molecules/cost_optimization_feedback_banner", () => ({ def vi.mock("@/components/model_info_view", () => ({ default: ({ modelId }: { modelId: string }) =>
model:{modelId}
, })); +const teamInfoProps = vi.hoisted(() => vi.fn()); vi.mock("@/components/team/TeamInfo", () => ({ - default: ({ teamId, is_team_admin }: { teamId: string; is_team_admin: boolean }) => ( -
- team:{teamId} -
- ), + default: (props: { teamId: string; is_team_admin: boolean; is_proxy_admin: boolean }) => { + teamInfoProps(props); + return ( +
+ team:{props.teamId} +
+ ); + }, })); const mockUseAuthorized = vi.fn(); @@ -107,12 +111,21 @@ describe("ModelsAndEndpointsPage", () => { expect(screen.getByTestId("team-info")).toHaveAttribute("data-team-admin", "true"); }); + it("passes is_proxy_admin for an admin session on the ?team drill-in", () => { + detailState.teamId = "team-a1b2"; + renderPage(); + expect(teamInfoProps).toHaveBeenLastCalledWith( + expect.objectContaining({ is_proxy_admin: true, is_team_admin: true }), + ); + }); + it("opens the ?team drill-in without edit rights for a view-only admin", () => { mockUseAuthorized.mockReturnValue(VIEW_ONLY_ADMIN); detailState.teamId = "team-9"; renderPage(); expect(screen.getByTestId("team-info")).toHaveTextContent("team:team-9"); expect(screen.getByTestId("team-info")).toHaveAttribute("data-team-admin", "false"); + expect(teamInfoProps).toHaveBeenLastCalledWith(expect.objectContaining({ is_proxy_admin: false })); }); it("hides admin-only tabs for a non-admin user", () => { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.tsx b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.tsx index bbc803af700..d8952b88545 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/models-and-endpoints/page.tsx @@ -151,7 +151,7 @@ export default function ModelsAndEndpointsPage() { onClose={close} accessToken={accessToken} is_team_admin={userRole === "Admin" && !isViewOnly} - is_proxy_admin={userRole === "Proxy Admin"} + is_proxy_admin={userRole === "Admin" && !isViewOnly} userModels={allModelsOnProxy} editTeam={false} onUpdate={invalidateModels} From 259f954f5642cd1156ce6a5930ea44bd5761655d Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 24 Sep 2026 11:55:07 -0700 Subject: [PATCH 83/96] test(e2e/ui): give the logout specs their own admin session (#42930) #42463 made Logout revoke the dashboard session key on the server. Both logout specs ran on the shared ADMIN_STORAGE_PATH session that globalSetup mints once, so clicking Logout revoked the key every later admin spec reuses. The CircleCI run is serial, and from the auth/ folder on, every admin-session spec failed with "Invalid proxy server token passed" (80 failures, up from 7) while the internal-user, internal-viewer and team-admin specs kept passing. Each logout spec now starts from an empty storage state and logs in through the login page, so the session it revokes is its own. The login steps live in a shared logInThroughLoginPage helper next to the other onboarding helpers. --- tests/e2e/ui/helpers/userOnboarding.ts | 10 ++++++++++ tests/e2e/ui/tests/auth/logout.spec.ts | 8 ++++++-- tests/e2e/ui/tests/auth/proxyLogoutUrl.spec.ts | 12 ++++++++---- 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/tests/e2e/ui/helpers/userOnboarding.ts b/tests/e2e/ui/helpers/userOnboarding.ts index a1ea6e5e82b..14e2b0257b2 100644 --- a/tests/e2e/ui/helpers/userOnboarding.ts +++ b/tests/e2e/ui/helpers/userOnboarding.ts @@ -57,3 +57,13 @@ export async function expectUnrestrictedDashboard(page: Page): Promise { expect(info.ok(), `Read own user with dashboard session: HTTP ${info.status()}`).toBe(true); expect((await info.json()).user_id).toBe(session.user_id); } + +export async function logInThroughLoginPage(page: Page, email: string, password: string): Promise { + await page.goto(`${rootPath()}/ui/login`); + await page.getByPlaceholder("Enter your username").fill(email); + await page.getByPlaceholder("Enter your password").fill(password); + await page.getByRole("button", { name: "Login", exact: true }).click(); + await page.waitForURL((url) => url.pathname.startsWith(`${rootPath()}/ui`) && !url.pathname.includes("/login"), { + timeout: 30_000, + }); +} diff --git a/tests/e2e/ui/tests/auth/logout.spec.ts b/tests/e2e/ui/tests/auth/logout.spec.ts index 351ba91e8d7..92c31456353 100644 --- a/tests/e2e/ui/tests/auth/logout.spec.ts +++ b/tests/e2e/ui/tests/auth/logout.spec.ts @@ -1,10 +1,14 @@ import { test, expect } from "@playwright/test"; -import { ADMIN_STORAGE_PATH } from "../../constants"; +import { Role, users } from "../../fixtures/users"; +import { logInThroughLoginPage } from "../../helpers/userOnboarding"; test.describe("Logout", () => { - test.use({ storageState: ADMIN_STORAGE_PATH }); + test.use({ storageState: { cookies: [], origins: [] } }); test("Clicking Logout clears the session and forces re-login on a protected page", async ({ page }) => { + const admin = users[Role.ProxyAdmin]; + await logInThroughLoginPage(page, admin.email, admin.password); + await page.goto("/ui"); // Scope to the sidebar; the top-bar breadcrumb also shows "Virtual Keys". await expect(page.getByRole("complementary").getByText("Virtual Keys")).toBeVisible({ timeout: 10_000 }); diff --git a/tests/e2e/ui/tests/auth/proxyLogoutUrl.spec.ts b/tests/e2e/ui/tests/auth/proxyLogoutUrl.spec.ts index 7f6cc6f2f87..3a79ce82717 100644 --- a/tests/e2e/ui/tests/auth/proxyLogoutUrl.spec.ts +++ b/tests/e2e/ui/tests/auth/proxyLogoutUrl.spec.ts @@ -1,5 +1,6 @@ import { test, expect } from "@playwright/test"; -import { ADMIN_STORAGE_PATH } from "../../constants"; +import { Role, users } from "../../fixtures/users"; +import { logInThroughLoginPage } from "../../helpers/userOnboarding"; /** * Runs as part of the standard e2e suite: both `run_e2e.sh` and the CircleCI @@ -16,9 +17,12 @@ const LOGOUT_URL = process.env.PROXY_LOGOUT_URL ?? ""; test.skip(!LOGOUT_URL, "Requires PROXY_LOGOUT_URL env var"); test.describe("PROXY_LOGOUT_URL redirect", () => { - test.use({ storageState: ADMIN_STORAGE_PATH }); + test.use({ storageState: { cookies: [], origins: [] } }); test("Logout clears the session and redirects to PROXY_LOGOUT_URL", async ({ page }) => { + const admin = users[Role.ProxyAdmin]; + await logInThroughLoginPage(page, admin.email, admin.password); + const target = new URL(LOGOUT_URL); // Stub the external logout destination so the assertion doesn't depend on @@ -46,8 +50,8 @@ test.describe("PROXY_LOGOUT_URL redirect", () => { await expect(page.getByRole("complementary").getByText("Virtual Keys")).toBeVisible({ timeout: 15_000 }); await settingsLoaded; - // Pre-condition: we start authenticated. The admin storage state carries a - // `token` cookie, so a real logout has something to tear down. + // Pre-condition: we start authenticated. The fresh login set a `token` + // cookie, so a real logout has something to tear down. const tokensBefore = (await page.context().cookies()).filter((c) => c.name === "token"); expect(tokensBefore.length, "should start logged in with a token cookie").toBeGreaterThan(0); From 342f9c3bf551a9e8e73dac3ee022c3a245e07fd9 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 11:55:36 -0700 Subject: [PATCH 84/96] fix(azure): update gpt-audio-mini and gpt-5-chat deprecation dates from the retirement schedule (#43017) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 4 ++-- model_prices_and_context_window.json | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 2600da22a77..8f38866387a 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -5734,7 +5734,7 @@ "supports_vision": false }, "azure/gpt-audio-mini": { - "deprecation_date": "2027-04-06", + "deprecation_date": "2027-06-15", "input_cost_per_audio_token": 1e-05, "input_cost_per_token": 6e-07, "litellm_provider": "azure", @@ -6768,7 +6768,7 @@ }, "azure/gpt-5-chat": { "cache_read_input_token_cost": 1.25e-07, - "deprecation_date": "2026-05-13", + "deprecation_date": "2026-06-29", "input_cost_per_token": 1.25e-06, "litellm_provider": "azure", "max_input_tokens": 128000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 2600da22a77..8f38866387a 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -5734,7 +5734,7 @@ "supports_vision": false }, "azure/gpt-audio-mini": { - "deprecation_date": "2027-04-06", + "deprecation_date": "2027-06-15", "input_cost_per_audio_token": 1e-05, "input_cost_per_token": 6e-07, "litellm_provider": "azure", @@ -6768,7 +6768,7 @@ }, "azure/gpt-5-chat": { "cache_read_input_token_cost": 1.25e-07, - "deprecation_date": "2026-05-13", + "deprecation_date": "2026-06-29", "input_cost_per_token": 1.25e-06, "litellm_provider": "azure", "max_input_tokens": 128000, From f4f1a75a9cdea2db8e83287e6f449b89019d4fa0 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 24 Sep 2026 11:56:42 -0700 Subject: [PATCH 85/96] test(vertex_ai): run the files peak-memory guards without coverage tracing (#42914) The new CircleCI tests pipeline (#42773) runs tests/unit under pytest-cov on CPython 3.12.2, where coverage traces every line through sys.settrace. The two tracemalloc peak comparisons in test_vertex_ai_files_streaming.py drive 8000-row payloads through both pipelines and slow from ~10s to over 3 minutes under that tracer, so both hit the 90s pytest-timeout on every run. Mark them no_cover so pytest-cov pauses tracing for just these two. Their assertions are unchanged and every other test in the file still reports coverage. --- .../unit/llms/vertex_ai/files/test_vertex_ai_files_streaming.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/unit/llms/vertex_ai/files/test_vertex_ai_files_streaming.py b/tests/unit/llms/vertex_ai/files/test_vertex_ai_files_streaming.py index 29eaf38b427..6851591f8b5 100644 --- a/tests/unit/llms/vertex_ai/files/test_vertex_ai_files_streaming.py +++ b/tests/unit/llms/vertex_ai/files/test_vertex_ai_files_streaming.py @@ -269,6 +269,7 @@ class TestStreamingPeakMemory: measurement removes any garbage the previous run left behind. """ + @pytest.mark.no_cover def test_streaming_peak_well_below_list_pipeline(self): cfg = VertexAIFilesConfig() raw = _make_openai_jsonl_bytes(8000) @@ -345,6 +346,7 @@ class TestPathSourcedStreaming: first_labels = json.loads(lines[0])["request"]["labels"] assert _get_litellm_batch_custom_id_from_labels(first_labels) == "request-0" + @pytest.mark.no_cover def test_path_source_peak_stays_below_list_pipeline(self, tmp_path): cfg = VertexAIFilesConfig() path, raw = self._write_jsonl(tmp_path, 8000) From 2cbaa4c9b5a4f9a7b6dc62a91ceb7c4dabff9841 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 24 Sep 2026 11:56:56 -0700 Subject: [PATCH 86/96] test(mcp): stop a comprehension variable from shadowing the body() helper (#42906) test_malformed_bodies_missing_users_and_foreign_servers_are_rejected used `body` as a comprehension variable and then called the module-level `body()` helper a few lines later. CPython 3.12.2, which the CircleCI integration job runs, compiles that later call as a local read, so the test raised UnboundLocalError on every integration-mcp run since #42652. Newer 3.12 patch releases and 3.13 compile it as a global read, which is why it passes locally. Renaming the comprehension variable makes both reads unambiguous. --- tests/integration/mcp/test_mcp_user_env_vars.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/integration/mcp/test_mcp_user_env_vars.py b/tests/integration/mcp/test_mcp_user_env_vars.py index d9cecaccadb..0170eb77491 100644 --- a/tests/integration/mcp/test_mcp_user_env_vars.py +++ b/tests/integration/mcp/test_mcp_user_env_vars.py @@ -195,7 +195,7 @@ def test_malformed_bodies_missing_users_and_foreign_servers_are_rejected(gateway path: Final = f"/v1/mcp/server/{identity}/user-env-vars" payload: Final[dict[str, JsonValue]] = {"values": {TOKEN: "x"}} malformed: Final[tuple[dict[str, JsonValue], ...]] = ({"values": {TOKEN: 7}}, {"values": ["a"]}, {}) - assert [gateway.request("POST", path, body, key=key).status_code for body in malformed] == [422, 422, 422] + assert [gateway.request("POST", path, bad, key=key).status_code for bad in malformed] == [422, 422, 422] assert set_names(env_status(gateway, key, identity)) == {TOKEN: False} assert [gateway.client.request(method, path, json=payload).status_code for method in METHODS] == [401, 401, 401] no_user: Final = tuple(gateway.request(method, path, payload, key=userless) for method in METHODS) From 4c68a97c2585783443fb066fedcee2bd439102f7 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 24 Sep 2026 11:59:52 -0700 Subject: [PATCH 87/96] test(mcp): patch create_mcp_server_if_identifier_free in the store-model-in-db MCP tests (#42916) #42791 switched the MCP management endpoints from create_mcp_server to create_mcp_server_if_identifier_free, which keeps the same arguments and returns the created row on success. Two tests in tests/store_model_in_db_tests/test_mcp_servers.py still patched the old name, so mock.patch raised AttributeError before the tests ran and proxy_store_model_in_db_tests has been red on main since. --- tests/store_model_in_db_tests/test_mcp_servers.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/store_model_in_db_tests/test_mcp_servers.py b/tests/store_model_in_db_tests/test_mcp_servers.py index 735d5d71ad3..0e20880ede9 100644 --- a/tests/store_model_in_db_tests/test_mcp_servers.py +++ b/tests/store_model_in_db_tests/test_mcp_servers.py @@ -134,7 +134,7 @@ async def test_create_mcp_server_direct(): "litellm.proxy.management_endpoints.mcp_management_endpoints.get_prisma_client_or_throw" ) as mock_get_prisma, mock.patch( - "litellm.proxy.management_endpoints.mcp_management_endpoints.create_mcp_server", + "litellm.proxy.management_endpoints.mcp_management_endpoints.create_mcp_server_if_identifier_free", new_callable=mock.AsyncMock, ) as mock_create, mock.patch( @@ -345,7 +345,7 @@ async def test_create_mcp_server_invalid_alias(): "litellm.proxy.management_endpoints.mcp_management_endpoints.get_mcp_server" ) as mock_get_server, mock.patch( - "litellm.proxy.management_endpoints.mcp_management_endpoints.create_mcp_server" + "litellm.proxy.management_endpoints.mcp_management_endpoints.create_mcp_server_if_identifier_free" ) as mock_create, ): from litellm.proxy.management_endpoints.mcp_management_endpoints import ( From 759c216366dc98c7221e591ecf0665e22a058e1a Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 24 Sep 2026 12:00:10 -0700 Subject: [PATCH 88/96] test(guardrails): run the cache-hit redis outage test on the shared owned_redis helper (#42925) * test(guardrails): run the cache-hit redis outage test on the shared owned_redis helper test_redis_outage_keeps_serving_in_memory_hits (#42780) spawned `redis-server` straight from PATH. The CircleCI integration machine has no redis-server binary, so the test died with FileNotFoundError before reaching the proxy. Use integration._support.redis_process.owned_redis, which runs the local binary when there is one and otherwise the job's redis-cache container, and take the outage with its stop()/start() pair the way test_redis_recovery already does. The assertions are unchanged. * test(guardrails): describe the owned_redis stop as an outage, not a kill --- .../test_cache_hit_guardrail_metrics_chaos.py | 46 ++++++------------- 1 file changed, 13 insertions(+), 33 deletions(-) diff --git a/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py b/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py index faaa1fc7325..c77b8eebe33 100644 --- a/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py +++ b/tests/integration/observability/test_cache_hit_guardrail_metrics_chaos.py @@ -1,12 +1,9 @@ import json import signal -import socket -import subprocess import threading import uuid -from collections.abc import Callable, Generator +from collections.abc import Callable from concurrent.futures import ThreadPoolExecutor -from contextlib import contextmanager from pathlib import Path from typing import Final @@ -14,6 +11,7 @@ import httpx import psutil from integration._support.client import Gateway, eventually, object_value, string_value from integration._support.process import owned_proxy_process +from integration._support.redis_process import owned_redis from integration._support.wire import Reply, Request, wire_server from prometheus_client.parser import text_string_to_metric_families from test_cache_hit_guardrail_metrics import ( @@ -31,22 +29,6 @@ from test_cache_hit_guardrail_metrics import ( BURST: Final = 10 -@contextmanager -def _redis(port: int) -> Generator[subprocess.Popen[bytes], None, None]: - process: Final = subprocess.Popen(["redis-server", "--port", str(port), "--save", ""], stdout=subprocess.DEVNULL) - try: - yield process - finally: - process.kill() - process.wait(timeout=10) - - -def _free_port() -> int: - with socket.socket() as reserve: - reserve.bind(("127.0.0.1", 0)) - return reserve.getsockname()[1] - - def _deployment_id(candidate: Gateway, model_name: str) -> str: entries: Final = candidate.get("/model/info")["data"] assert isinstance(entries, list) @@ -224,18 +206,16 @@ def test_stalled_guardrail_sink_recovers_and_counts(gateway: Gateway, tmp_path: def test_redis_outage_keeps_serving_in_memory_hits(gateway: Gateway, tmp_path: Path) -> None: - """X2: the redis cache keeps an in-memory shadow, so a redis kill does not stop cache-hit rejects.""" + """X2: the redis cache keeps an in-memory shadow, so a redis outage does not stop cache-hit rejects.""" marker: Final = uuid.uuid4().hex - port: Final = _free_port() - with _redis(port) as redis_one: - with _rig(gateway, tmp_path, marker, env={"REDIS_HOST": "127.0.0.1", "REDIS_PORT": str(port)}) as rig: + with owned_redis(tmp_path) as cache: + with _rig(gateway, tmp_path, marker, env={"REDIS_HOST": cache.host, "REDIS_PORT": str(cache.port)}) as rig: bodies: Final = _burst_bodies(rig, marker, None)[:BURST] _warm(rig, bodies) reject: Final = rig.candidate.request("POST", *bodies[0]) assert reject.status_code == 400, reject.text warmed_hits: Final = rig.provider.received.qsize() - redis_one.kill() - redis_one.wait(timeout=10) + cache.stop() outcomes: Final = _fire(rig, bodies[1:]) assert all(status == 400 for status, _ in outcomes), outcomes assert rig.provider.received.qsize() == warmed_hits, ( @@ -243,13 +223,13 @@ def test_redis_outage_keeps_serving_in_memory_hits(gateway: Gateway, tmp_path: P warmed_hits, rig.provider.received.qsize(), ) - with _redis(port): - recovered: Final = rig.candidate.request( - "POST", - "/v1/chat/completions", - _chat_body(rig.model_name, "x2 rehit " + marker, rig.guardrail_name), - ) - assert recovered.status_code == 400, recovered.text + cache.start() + recovered: Final = rig.candidate.request( + "POST", + "/v1/chat/completions", + _chat_body(rig.model_name, "x2 rehit " + marker, rig.guardrail_name), + ) + assert recovered.status_code == 400, recovered.text _expect_exactly_once(rig, (rig.model_name,), (rig.deployment_id,), 1 + len(bodies)) From bdf854c3ea3ecbfd399c26a33710d2c9644eb616 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 19:08:32 +0000 Subject: [PATCH 89/96] feat(rust): shape Anthropic Messages requests natively (#42982) * test(rust): encode anthropic response serialization shape as rstest cases Co-Authored-By: Claude Opus 5.5 * feat(rust): shape Anthropic Messages requests natively The Rust Messages route only relayed the body. It now runs the request shaping the Python handler does for the direct Anthropic provider: history sanitizers (empty blocks, tool ids, replayed web search results, provider_specific_fields, encrypted reasoning, advisor blocks), reasoning_effort and adaptive/legacy thinking translation against the model's capability flags, the sampling and speed gates under drop_params, the metadata allowlist, additional_drop_params, reasoning auto summary, OAuth and ANTHROPIC_AUTH_TOKEN credentials, provider_specific_header merging and anthropic-beta injection. Capability flags and LiteLLM settings reach Rust through route_host.shaping(). A request the route rejects before the call now maps to BadRequestError instead of APIConnectionError Co-Authored-By: Claude Fable 5.1 * test(rust): port Anthropic Messages shaping tests and pin comment contracts as cases Every Python unit test that exercises the ported shaping for the direct Anthropic provider now has a named rstest counterpart, and every comment that stated a behavior contract is deleted in favor of a case that pins it. Measured with cargo-mutants over the touched files, all viable mutants are caught Porting the tests surfaced parity gaps, fixed here to match Python: every casing of a forwarded anthropic-beta header is merged, replayed web search results are rewritten from their own block (an empty result keeps its slot and a server_tool_use with a non-string query stays), an empty output_config.effort falls back to medium, speed and reasoning effort errors quote values the way Python does, additional_drop_params apply after metadata validation and the auto summary and never touch model or messages, and a non-string metadata.user_id is rejected before the call * fix(rust): resolve Messages credentials through the secret source and scope headers by resolved provider The native Messages route read ANTHROPIC_API_KEY, ANTHROPIC_AUTH_TOKEN and the base URL straight from the process environment, so a key or base held in a configured secret manager was never found. Each provider config now declares its secret names and the route resolves them through the same SecretSource the OCR route uses, with the Python bridge passing in litellm's configured manager provider_specific_header entries were scoped by the explicit custom_llm_provider only, falling back to anthropic, so an azure_ai/ model lost its azure_ai scoped headers. Scoping now happens in the route after the provider is resolved from the model, as Python's handler does The Azure config now adds the same anthropic-beta feature headers Python's Azure route adds, and the metadata allowlist, reasoning auto summary and history sanitizers move from the core route into the llms crate, mirroring their home in Python's messages handler * test(rust): escape the dot in the metadata.user_id match pattern * refactor(rust-bridge): project Messages capabilities without mutable dicts The capability flags and effort tiers were built as dict comprehensions, which the type-discipline gate counts as mutable construction, and the asdict call carried a mutable-ok suppression that suppressed nothing. The flags are now passed one by one and the effort tiers are a frozen dataclass, which asdict projects to the same map the native side reads --------- Co-authored-by: Yujong Lee Co-authored-by: Claude Opus 5.5 --- litellm-rust/Cargo.lock | 1 + .../core-utils/src/dot_notation_indexing.rs | 274 +++ .../src/get_provider_specific_headers.rs | 93 + litellm-rust/crates/core-utils/src/lib.rs | 2 + .../crates/core/src/messages/common_utils.rs | 2 +- .../crates/core/src/messages/error.rs | 28 + litellm-rust/crates/core/src/messages/mod.rs | 8 +- .../crates/core/src/messages/prepare.rs | 501 +++++- .../crates/core/src/messages/route.rs | 52 +- .../crates/core/src/messages/tests.rs | 149 +- .../crates/core/src/messages/types.rs | 104 +- .../crates/llms/src/anthropic/common_utils.rs | 1568 +++++++++++++++++ .../messages/handler.rs | 270 +++ .../messages/headers.rs | 643 +++++++ .../experimental_pass_through/messages/mod.rs | 3 + .../messages/thinking.rs | 1182 +++++++++++++ .../messages/transformation.rs | 808 ++++++++- litellm-rust/crates/llms/src/anthropic/mod.rs | 1 + .../anthropic/messages_transformation.rs | 145 +- .../anthropic_messages/transformation.rs | 232 ++- .../python-bridge/src/routes/messages/host.rs | 175 +- .../python-bridge/src/routes/messages/mod.rs | 3 +- litellm-rust/crates/types/Cargo.toml | 3 + .../anthropic_messages/anthropic_request.rs | 156 ++ .../anthropic_messages/anthropic_response.rs | 60 +- litellm-rust/crates/types/src/utils.rs | 15 + litellm/rust_bridge/messages/route_host.py | 110 +- tests/test_litellm/rust_bridge/AGENTS.md | 2 + .../rust_bridge/messages/test_route_host.py | 112 ++ .../rust_bridge/messages/test_secrets.py | 111 ++ .../messages/test_request_shaping.py | 237 +++ 31 files changed, 6891 insertions(+), 159 deletions(-) create mode 100644 litellm-rust/crates/core-utils/src/dot_notation_indexing.rs create mode 100644 litellm-rust/crates/core-utils/src/get_provider_specific_headers.rs create mode 100644 litellm-rust/crates/llms/src/anthropic/common_utils.rs create mode 100644 litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/handler.rs create mode 100644 litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/headers.rs create mode 100644 litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/thinking.rs create mode 100644 tests/test_litellm/rust_bridge/messages/test_route_host.py create mode 100644 tests/test_litellm/rust_bridge/messages/test_secrets.py create mode 100644 tests/test_litellm_rust/messages/test_request_shaping.py diff --git a/litellm-rust/Cargo.lock b/litellm-rust/Cargo.lock index ef032bfe55c..0a91f0759c2 100644 --- a/litellm-rust/Cargo.lock +++ b/litellm-rust/Cargo.lock @@ -3414,6 +3414,7 @@ dependencies = [ name = "litellm-types" version = "0.1.0" dependencies = [ + "rstest", "serde", "serde_json", ] diff --git a/litellm-rust/crates/core-utils/src/dot_notation_indexing.rs b/litellm-rust/crates/core-utils/src/dot_notation_indexing.rs new file mode 100644 index 00000000000..be81d900f57 --- /dev/null +++ b/litellm-rust/crates/core-utils/src/dot_notation_indexing.rs @@ -0,0 +1,274 @@ +use serde_json::Value; + +#[derive(Clone, Debug, PartialEq, Eq)] +enum Segment { + Field(String), + Every, + Index(usize), +} + +fn parse_segments(path: &str) -> Option> { + let mut segments = Vec::new(); + let mut rest = path; + while !rest.is_empty() { + if let Some(after_open) = rest.strip_prefix('[') { + let (inside, after) = after_open.split_once(']')?; + segments.push(match inside { + "*" => Segment::Every, + index => Segment::Index(index.trim().parse().ok()?), + }); + rest = after.strip_prefix('.').unwrap_or(after); + continue; + } + let end = rest.find(['.', '[']).unwrap_or(rest.len()); + let (field, after) = rest.split_at(end); + if !field.is_empty() { + segments.push(Segment::Field(field.to_string())); + } + rest = after.strip_prefix('.').unwrap_or(after); + } + Some(segments) +} + +fn without_path(value: Value, segments: &[Segment]) -> Value { + let Some((segment, tail)) = segments.split_first() else { + return value; + }; + match (segment, value) { + (Segment::Field(name), Value::Object(object)) => Value::Object( + object + .into_iter() + .filter_map(|(key, item)| { + if key != *name { + return Some((key, item)); + } + (!tail.is_empty()).then(|| (key, without_path(item, tail))) + }) + .collect(), + ), + (Segment::Every, Value::Array(items)) => Value::Array( + items + .into_iter() + .map(|item| without_path(item, tail)) + .collect(), + ), + (Segment::Index(index), Value::Array(items)) => Value::Array( + items + .into_iter() + .enumerate() + .map(|(position, item)| { + if position == *index { + without_path(item, tail) + } else { + item + } + }) + .collect(), + ), + (_, value) => value, + } +} + +pub fn delete_nested_value(value: Value, path: &str) -> Value { + match parse_segments(path) { + Some(segments) => without_path(value, &segments), + None => value, + } +} + +#[cfg(test)] +mod tests { + use rstest::{fixture, rstest}; + use serde_json::json; + + use super::*; + + #[fixture] + fn body() -> Value { + json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }) + } + + #[rstest] + #[case::top_level_field("top", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}} + }))] + #[case::whole_object("meta", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "top": 0.7 + }))] + #[case::nested_field("meta.inner.drop", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"keep": 2}}, + "top": 0.7 + }))] + #[case::trailing_dot("meta.inner.drop.", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"keep": 2}}, + "top": 0.7 + }))] + #[case::leading_and_doubled_dots(".meta..inner.drop", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"keep": 2}}, + "top": 0.7 + }))] + #[case::field_in_every_element("tools[*].examples", json!({ + "tools": [ + {"name": "t0", "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }))] + #[case::whole_array_field_in_every_element("tools[*].arr", json!({ + "tools": [ + {"name": "t0", "examples": ["a"]}, + {"name": "t1", "examples": ["b"]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }))] + #[case::field_in_indexed_element("tools[1].examples", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }))] + #[case::padded_index("tools[ 1 ].examples", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }))] + #[case::field_right_after_bracket("tools[0]examples", json!({ + "tools": [ + {"name": "t0", "arr": [{"f": 1, "k": 1}, {"f": 2, "k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }))] + #[case::index_then_wildcard("tools[0].arr[*].f", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"k": 1}, {"k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }))] + #[case::wildcard_then_index_only_where_it_exists("tools[*].arr[1].f", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"f": 1, "k": 1}, {"k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"f": 3, "k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }))] + #[case::nested_wildcards("tools[*].arr[*].f", json!({ + "tools": [ + {"name": "t0", "examples": ["a"], "arr": [{"k": 1}, {"k": 2}]}, + {"name": "t1", "examples": ["b"], "arr": [{"k": 3}]} + ], + "meta": {"user": "u", "inner": {"drop": 1, "keep": 2}}, + "top": 0.7 + }))] + fn deletes_the_addressed_field(body: Value, #[case] path: &str, #[case] expected: Value) { + assert_eq!(delete_nested_value(body, path), expected); + } + + #[rstest] + #[case::empty_path("")] + #[case::missing_field("missing")] + #[case::missing_parent("missing.field")] + #[case::field_through_a_scalar("top.value")] + #[case::field_on_an_array("tools.name")] + #[case::index_on_an_object("meta[0].user")] + #[case::wildcard_on_an_object("meta[*].user")] + #[case::wildcard_over_scalars("tools[*].examples[*].name")] + #[case::index_out_of_range("tools[5].name")] + #[case::every_element_itself("tools[*]")] + #[case::indexed_element_itself("tools[0]")] + #[case::nested_element_itself("tools[*].arr[0]")] + #[case::negative_index("tools[-1].name")] + #[case::non_numeric_index("tools[x].name")] + #[case::empty_index("tools[].name")] + #[case::unclosed_bracket("top[0")] + fn leaves_the_value_untouched(body: Value, #[case] path: &str) { + assert_eq!(delete_nested_value(body.clone(), path), body); + } + + #[rstest] + #[case::wildcards_indices_and_nesting( + json!({"tools": [ + {"name": "t0", "configs": [{"id": "c0", "remove_me": 1, "keep": 1}, {"id": "c1", "remove_me": 2, "keep": 2}], "metadata": {"drop_this": 1, "preserve": 1}}, + {"name": "t1", "configs": [{"id": "c0", "remove_me": 3, "keep": 3}, {"id": "c1", "remove_me": 4, "keep": 4}], "metadata": {"drop_this": 2, "preserve": 2}}, + {"name": "t2", "configs": [{"id": "c0", "remove_me": 5, "keep": 5}], "metadata": {"drop_this": 3, "preserve": 3}} + ]}), + &["tools[*].configs[1].remove_me", "tools[1].metadata.drop_this", "tools[*].configs[*].id"], + json!({"tools": [ + {"name": "t0", "configs": [{"remove_me": 1, "keep": 1}, {"keep": 2}], "metadata": {"drop_this": 1, "preserve": 1}}, + {"name": "t1", "configs": [{"remove_me": 3, "keep": 3}, {"keep": 4}], "metadata": {"preserve": 2}}, + {"name": "t2", "configs": [{"remove_me": 5, "keep": 5}], "metadata": {"drop_this": 3, "preserve": 3}} + ]}), + )] + #[case::simple_and_wildcard_nesting( + json!({ + "tools": [{"name": "t1", "simple_nested": {"remove": 1, "keep": 2}, "complex": [{"nested": {"remove": 3, "keep": 4}}]}], + "top_level_remove": "should_go", + "top_level_keep": "should_stay" + }), + &["tools[*].simple_nested.remove", "tools[*].complex[*].nested.remove"], + json!({ + "tools": [{"name": "t1", "simple_nested": {"keep": 2}, "complex": [{"nested": {"keep": 4}}]}], + "top_level_remove": "should_go", + "top_level_keep": "should_stay" + }), + )] + #[case::triple_nested_wildcards( + json!({"tools": [{"name": "t1", "arr1": [ + {"arr2": [{"field": 1, "keep": 1}, {"field": 2, "keep": 2}]}, + {"arr2": [{"field": 3, "keep": 3}]} + ]}]}), + &["tools[*].arr1[*].arr2[*].field"], + json!({"tools": [{"name": "t1", "arr1": [ + {"arr2": [{"keep": 1}, {"keep": 2}]}, + {"arr2": [{"keep": 3}]} + ]}]}), + )] + fn applies_paths_in_sequence( + #[case] value: Value, + #[case] paths: &[&str], + #[case] expected: Value, + ) { + let deleted = paths + .iter() + .fold(value, |value, path| delete_nested_value(value, path)); + assert_eq!(deleted, expected); + } +} diff --git a/litellm-rust/crates/core-utils/src/get_provider_specific_headers.rs b/litellm-rust/crates/core-utils/src/get_provider_specific_headers.rs new file mode 100644 index 00000000000..bfcd448e2d8 --- /dev/null +++ b/litellm-rust/crates/core-utils/src/get_provider_specific_headers.rs @@ -0,0 +1,93 @@ +use litellm_types::utils::{ProviderSpecificHeader, ProviderSpecificHeaders}; +use serde_json::{Map, Value}; + +pub fn get_provider_specific_headers( + provider_specific_header: Option<&ProviderSpecificHeaders>, + custom_llm_provider: &str, +) -> Map { + let entries: &[ProviderSpecificHeader] = match provider_specific_header { + None => &[], + Some(ProviderSpecificHeaders::One(entry)) => std::slice::from_ref(entry), + Some(ProviderSpecificHeaders::Many(entries)) => entries, + }; + entries + .iter() + .filter(|entry| { + entry + .custom_llm_provider + .split(',') + .any(|scoped| scoped.trim() == custom_llm_provider) + }) + .flat_map(|entry| entry.extra_headers.clone()) + .collect() +} + +#[cfg(test)] +mod tests { + use rstest::rstest; + use serde_json::json; + + use super::*; + + #[rstest] + #[case::single_entry_for_the_provider( + json!({"custom_llm_provider": "anthropic", "extra_headers": {"Authorization": "Bearer t", "Custom-Header": "v"}}), + json!({"Authorization": "Bearer t", "Custom-Header": "v"}), + )] + #[case::single_entry_for_another_provider( + json!({"custom_llm_provider": "openai", "extra_headers": {"Authorization": "Bearer t"}}), + json!({}), + )] + #[case::provider_in_a_comma_separated_scope( + json!({"custom_llm_provider": "bedrock,anthropic,vertex_ai", "extra_headers": {"anthropic-beta": "context-1m-2025-08-07"}}), + json!({"anthropic-beta": "context-1m-2025-08-07"}), + )] + #[case::provider_missing_from_a_comma_separated_scope( + json!({"custom_llm_provider": "bedrock,vertex_ai", "extra_headers": {"anthropic-beta": "test"}}), + json!({}), + )] + #[case::scope_with_spaces( + json!({"custom_llm_provider": "bedrock, anthropic , vertex_ai", "extra_headers": {"anthropic-beta": "test"}}), + json!({"anthropic-beta": "test"}), + )] + #[case::scope_names_must_match_exactly( + json!({"custom_llm_provider": "anthropic_text", "extra_headers": {"anthropic-beta": "test"}}), + json!({}), + )] + #[case::entries_scope_independently( + json!([ + {"custom_llm_provider": "anthropic,bedrock,vertex_ai", "extra_headers": {"anthropic-beta": "context-1m-2025-08-07"}}, + {"custom_llm_provider": "bedrock", "extra_headers": {"x-bedrock-only": "no"}}, + {"custom_llm_provider": "anthropic", "extra_headers": {"authorization": "Bearer sk-ant-oat01-fake-token"}} + ]), + json!({"anthropic-beta": "context-1m-2025-08-07", "authorization": "Bearer sk-ant-oat01-fake-token"}), + )] + #[case::later_entries_win( + json!([ + {"custom_llm_provider": "anthropic", "extra_headers": {"x-scoped": "first"}}, + {"custom_llm_provider": "anthropic", "extra_headers": {"x-scoped": "second"}} + ]), + json!({"x-scoped": "second"}), + )] + #[case::empty_list(json!([]), json!({}))] + #[case::entry_without_scope(json!({"extra_headers": {"x-scoped": "yes"}}), json!({}))] + #[case::entry_without_headers(json!({"custom_llm_provider": "anthropic"}), json!({}))] + fn provider_specific_headers_match_the_scoped_provider( + #[case] configured: Value, + #[case] expected: Value, + ) { + let configured: ProviderSpecificHeaders = serde_json::from_value(configured).unwrap(); + assert_eq!( + Value::Object(get_provider_specific_headers( + Some(&configured), + "anthropic" + )), + expected + ); + } + + #[test] + fn no_configured_headers_match_nothing() { + assert_eq!(get_provider_specific_headers(None, "anthropic"), Map::new()); + } +} diff --git a/litellm-rust/crates/core-utils/src/lib.rs b/litellm-rust/crates/core-utils/src/lib.rs index ceb0e9eb3f2..a937f55654e 100644 --- a/litellm-rust/crates/core-utils/src/lib.rs +++ b/litellm-rust/crates/core-utils/src/lib.rs @@ -1,7 +1,9 @@ pub mod call_arguments; pub mod core_helpers; +pub mod dot_notation_indexing; pub mod exception_mapping_utils; pub mod get_llm_provider_logic; +pub mod get_provider_specific_headers; pub mod params; pub mod prompt_templates; pub mod secret_redaction; diff --git a/litellm-rust/crates/core/src/messages/common_utils.rs b/litellm-rust/crates/core/src/messages/common_utils.rs index dcefa3ebffc..015e026f6da 100644 --- a/litellm-rust/crates/core/src/messages/common_utils.rs +++ b/litellm-rust/crates/core/src/messages/common_utils.rs @@ -1,5 +1,5 @@ use litellm_http::request::string_headers as shared_string_headers; -pub(super) use litellm_http::request::{has_bearer_auth, has_header, truncate_error_body}; +pub(super) use litellm_http::request::truncate_error_body; use litellm_llms::{ anthropic::experimental_pass_through::messages::transformation::ANTHROPIC_MESSAGES_CONFIG, azure_ai::anthropic::messages_transformation::AZURE_ANTHROPIC_MESSAGES_CONFIG, diff --git a/litellm-rust/crates/core/src/messages/error.rs b/litellm-rust/crates/core/src/messages/error.rs index 51fb764032c..2a9723beb38 100644 --- a/litellm-rust/crates/core/src/messages/error.rs +++ b/litellm-rust/crates/core/src/messages/error.rs @@ -1,3 +1,5 @@ +use std::sync::Arc; + use litellm_llms::base_llm::chat::transformation::Error as LlmError; #[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] @@ -18,8 +20,34 @@ pub enum Error { Transport(#[from] litellm_http::transport::Error), #[error(transparent)] Headers(#[from] litellm_http::request::HeaderError), + #[error(transparent)] + Secret(#[from] SecretError), } +#[derive(Clone, Debug, thiserror::Error)] +#[error(transparent)] +pub struct SecretError(Arc); + +impl SecretError { + pub fn source_error(&self) -> &litellm_secrets::Error { + &self.0 + } +} + +impl From for Error { + fn from(error: litellm_secrets::Error) -> Self { + Self::Secret(SecretError(Arc::new(error))) + } +} + +impl PartialEq for SecretError { + fn eq(&self, other: &Self) -> bool { + Arc::ptr_eq(&self.0, &other.0) + } +} + +impl Eq for SecretError {} + impl From for Error { fn from(error: LlmError) -> Self { match error { diff --git a/litellm-rust/crates/core/src/messages/mod.rs b/litellm-rust/crates/core/src/messages/mod.rs index 289f79109dd..8795d4f8507 100644 --- a/litellm-rust/crates/core/src/messages/mod.rs +++ b/litellm-rust/crates/core/src/messages/mod.rs @@ -12,6 +12,9 @@ mod common_utils; mod handler; mod prepare; pub mod route; +use std::sync::Arc; + +use litellm_secrets::source::EnvironmentSecrets; use litellm_types::llms::anthropic_messages::anthropic_response::AnthropicMessagesResponse; use route::{LocalMessagesHost, MessagesCall, MessagesOutput, messages_machine}; use serde_json::Value; @@ -31,9 +34,12 @@ pub async fn messages(request: MessagesRequest<'_>) -> Result Ok(*message), MessagesOutput::Streamed => Err(Error::Unsupported( "streamed responses need a streaming host", diff --git a/litellm-rust/crates/core/src/messages/prepare.rs b/litellm-rust/crates/core/src/messages/prepare.rs index 850f9108869..dc4b3562e3f 100644 --- a/litellm-rust/crates/core/src/messages/prepare.rs +++ b/litellm-rust/crates/core/src/messages/prepare.rs @@ -1,51 +1,102 @@ -use litellm_core_utils::get_llm_provider_logic::{CustomLlmProvider, get_custom_llm_provider}; -use litellm_llms::base_llm::anthropic_messages::transformation::{ - BaseAnthropicMessagesConfig, MessagesAuthStrategy, +use litellm_core_utils::{ + dot_notation_indexing::delete_nested_value, + get_llm_provider_logic::{CustomLlmProvider, get_custom_llm_provider}, + get_provider_specific_headers::get_provider_specific_headers, + settings::Lookup, +}; +use litellm_llms::{ + anthropic::experimental_pass_through::messages::handler::shape_anthropic_messages_request, + base_llm::anthropic_messages::transformation::{ + BaseAnthropicMessagesConfig, MessagesTransformContext, + }, }; use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest; use serde_json::{Map, Value}; use super::{ Error, - common_utils::{has_bearer_auth, has_header, messages_provider_config, string_headers}, + common_utils::{messages_provider_config, string_headers}, }; use crate::messages::types::{MessagesRequest, ProviderMessagesRequest}; -pub(super) fn prepare_provider_request( - request: MessagesRequest<'_>, -) -> Result { - let provider_info = get_custom_llm_provider(request.model, request.custom_llm_provider) +pub(super) struct ResolvedProvider<'a> { + pub(super) model: &'a str, + pub(super) provider: &'a str, + pub(super) config: &'static dyn BaseAnthropicMessagesConfig, +} + +pub(super) fn resolve_provider<'a>( + model: &'a str, + custom_llm_provider: Option<&'a str>, +) -> Result, Error> { + let CustomLlmProvider { + model, + custom_llm_provider: provider, + } = get_custom_llm_provider(model, custom_llm_provider) .or_else(|| { - request - .custom_llm_provider - .map(|provider| CustomLlmProvider { - model: request.model, - custom_llm_provider: provider, - }) + custom_llm_provider.map(|provider| CustomLlmProvider { + model, + custom_llm_provider: provider, + }) }) .ok_or_else(|| { Error::InvalidProvider( "unable to resolve custom_llm_provider for messages request".to_string(), ) })?; - let model = provider_info.model.to_string(); - let provider = provider_info.custom_llm_provider; - let config = messages_provider_config(provider) .ok_or_else(|| Error::InvalidProvider(provider.to_string()))?; - let env_lookup = |key: &str| std::env::var(key).ok(); + Ok(ResolvedProvider { + model, + provider, + config, + }) +} - let headers = - validate_environment(config, request.extra_headers, request.api_key, &env_lookup)?; +pub(super) fn prepare_provider_request( + request: MessagesRequest<'_>, + resolved: ResolvedProvider<'_>, + secrets: &dyn Lookup, +) -> Result { + let ResolvedProvider { + model, + provider, + config, + } = resolved; + let model = model.to_string(); + let env_lookup = |key: &str| secrets.get(key); let typed_request: AnthropicMessagesRequest = - serde_json::from_value(request.body).map_err(|err| { - Error::InvalidRequest(format!("invalid Anthropic messages request: {err}")) - })?; - let transformed = config.transform_anthropic_messages_request(AnthropicMessagesRequest { - model: model.clone(), - ..typed_request - })?; + serde_json::from_value(request.body).map_err(invalid_request)?; + let sanitized = shape_anthropic_messages_request( + AnthropicMessagesRequest { + model: model.clone(), + ..typed_request + }, + request.shaping.reasoning_auto_summary, + )?; + let trimmed = + without_additional_drop_params(sanitized, &request.shaping.additional_drop_params)?; + let transformed = config.transform_anthropic_messages_request( + trimmed, + &MessagesTransformContext::new(request.shaping.capabilities, request.shaping.drop_params), + )?; + + let scoped = get_provider_specific_headers(request.provider_specific_header.as_ref(), provider); + let forwarded = string_headers(Some( + request + .extra_headers + .into_iter() + .flatten() + .chain(scoped) + .collect(), + ))?; + let authenticated = config.authenticate(forwarded, request.api_key, &env_lookup)?; + let headers = config.request_headers( + with_default_headers(authenticated, config.default_headers()), + &transformed, + ); + let body = serde_json::to_value(transformed).map_err(|err| { Error::InvalidRequest(format!( "failed to serialize Anthropic messages request: {err}" @@ -65,33 +116,371 @@ pub(super) fn prepare_provider_request( }) } -fn validate_environment( - config: &dyn BaseAnthropicMessagesConfig, - extra_headers: Option>, - api_key: Option<&str>, - env_lookup: &dyn Fn(&str) -> Option, -) -> Result, Error> { - let mut headers = string_headers(extra_headers)?; - - let auth_strategy = config.auth_strategy(); - let already_authorized = has_header(&headers, auth_strategy.header_name()) - || (config.accepts_bearer_auth() && has_bearer_auth(&headers)); - if !already_authorized { - let api_key = config.resolve_api_key(api_key, env_lookup)?; - let auth_header = match auth_strategy { - MessagesAuthStrategy::Bearer => { - ("authorization".to_string(), format!("Bearer {api_key}")) - } - MessagesAuthStrategy::Header(name) => (name.to_string(), api_key), - }; - headers.push(auth_header); - } - - for (name, value) in config.default_headers() { - if !has_header(&headers, name) { - headers.push((name.to_string(), value.to_string())); - } - } - - Ok(headers) +fn invalid_request(err: serde_json::Error) -> Error { + Error::InvalidRequest(format!("invalid Anthropic messages request: {err}")) +} + +fn without_additional_drop_params( + request: AnthropicMessagesRequest, + paths: &[String], +) -> Result { + if paths.is_empty() { + return Ok(request); + } + let Value::Object(fields) = serde_json::to_value(request).map_err(invalid_request)? else { + return Err(Error::InvalidRequest( + "Anthropic messages request did not serialize to an object".to_string(), + )); + }; + let (required, optional): (Map, Map) = fields + .into_iter() + .partition(|(key, _)| matches!(key.as_str(), "model" | "messages")); + let trimmed = paths.iter().fold(Value::Object(optional), |body, path| { + delete_nested_value(body, path) + }); + let merged: Map = required + .into_iter() + .chain(trimmed.as_object().cloned().unwrap_or_default()) + .collect(); + serde_json::from_value(Value::Object(merged)).map_err(invalid_request) +} + +fn with_default_headers( + headers: Vec<(String, String)>, + defaults: &[(&str, &str)], +) -> Vec<(String, String)> { + let missing: Vec<(String, String)> = defaults + .iter() + .filter(|(name, _)| { + !headers + .iter() + .any(|(header, _)| header.eq_ignore_ascii_case(name)) + }) + .map(|(name, value)| ((*name).to_string(), (*value).to_string())) + .collect(); + headers.into_iter().chain(missing).collect() +} + +#[cfg(test)] +mod tests { + use litellm_types::utils::ProviderSpecificHeaders; + use rstest::{fixture, rstest}; + use serde_json::json; + + use super::*; + use crate::messages::types::MessagesShaping; + + #[fixture] + fn shaping() -> MessagesShaping { + MessagesShaping::default() + } + + fn prepare(request: MessagesRequest<'_>) -> Result { + prepare_with_secrets(request, &|_: &str| None) + } + + fn prepare_with_secrets( + request: MessagesRequest<'_>, + secrets: &dyn Lookup, + ) -> Result { + let resolved = resolve_provider(request.model, request.custom_llm_provider)?; + prepare_provider_request(request, resolved, secrets) + } + + #[rstest] + #[case::api_key( + &[("ANTHROPIC_API_KEY", "sk-secret")], + &[("x-api-key", "sk-secret")], + "https://api.anthropic.com/v1/messages" + )] + #[case::auth_token( + &[("ANTHROPIC_AUTH_TOKEN", "token")], + &[("authorization", "Bearer token")], + "https://api.anthropic.com/v1/messages" + )] + #[case::api_base( + &[("ANTHROPIC_API_KEY", "sk-secret"), ("ANTHROPIC_API_BASE", "https://gateway.test")], + &[("x-api-key", "sk-secret")], + "https://gateway.test/v1/messages" + )] + #[case::sdk_base_url( + &[("ANTHROPIC_API_KEY", "sk-secret"), ("ANTHROPIC_BASE_URL", "https://sdk.test")], + &[("x-api-key", "sk-secret")], + "https://sdk.test/v1/messages" + )] + fn credentials_and_base_come_from_the_resolved_secrets( + shaping: MessagesShaping, + #[case] secrets: &[(&str, &str)], + #[case] expected_auth: &[(&str, &str)], + #[case] expected_url: &str, + ) { + let lookup = |name: &str| { + secrets + .iter() + .find(|(key, _)| *key == name) + .map(|(_, value)| value.to_string()) + }; + let prepared = prepare_with_secrets( + MessagesRequest { + model: "claude-test", + body: json!({"model": "claude-test", "messages": [{"role": "user", "content": "hi"}], "max_tokens": 16}), + api_key: None, + api_base: None, + custom_llm_provider: Some("anthropic"), + extra_headers: None, + provider_specific_header: None, + timeout: None, + shaping, + }, + &lookup, + ) + .unwrap(); + let auth: Vec<(&str, &str)> = prepared + .upstream_headers + .iter() + .filter(|(name, _)| matches!(name.as_str(), "x-api-key" | "authorization")) + .map(|(name, value)| (name.as_str(), value.as_str())) + .collect(); + assert_eq!( + (auth.as_slice(), prepared.url.as_str()), + (expected_auth, expected_url) + ); + } + + fn prepared_body(body: Value, shaping: MessagesShaping) -> Result { + prepare(MessagesRequest { + model: "anthropic/claude-test", + body, + api_key: Some("sk-test"), + api_base: Some("https://anthropic.test"), + custom_llm_provider: Some("anthropic"), + extra_headers: None, + provider_specific_header: None, + timeout: None, + shaping, + }) + .map(|prepared| prepared.body) + } + + #[rstest] + #[case::nothing_forwarded( + &[], + &[("x-version", "1"), ("content-type", "application/json")], + &[("x-version", "1"), ("content-type", "application/json")], + )] + #[case::forwarded_header_wins_in_any_case( + &[("X-Version", "custom"), ("x-api-key", "k")], + &[("x-version", "1"), ("content-type", "application/json")], + &[("X-Version", "custom"), ("x-api-key", "k"), ("content-type", "application/json")], + )] + #[case::no_defaults(&[("x-api-key", "k")], &[], &[("x-api-key", "k")])] + fn default_headers_fill_only_missing_names( + #[case] forwarded: &[(&str, &str)], + #[case] defaults: &[(&str, &str)], + #[case] expected: &[(&str, &str)], + ) { + let owned = |headers: &[(&str, &str)]| -> Vec<(String, String)> { + headers + .iter() + .map(|(name, value)| ((*name).to_string(), (*value).to_string())) + .collect() + }; + assert_eq!( + with_default_headers(owned(forwarded), defaults), + owned(expected) + ); + } + + #[rstest] + #[case::top_level_and_nested_paths( + json!({ + "max_tokens": 1024, + "thinking": {"type": "enabled", "budget_tokens": 2048}, + "context_management": {"edits": [{"type": "clear_thinking_20251015"}]}, + "metadata": {"user_id": "u1"}, + "tools": [{"name": "lookup", "input_schema": {"type": "object"}, "input_examples": [{"q": "x"}]}] + }), + &["thinking", "context_management", "tools[*].input_examples"], + json!({ + "max_tokens": 1024, + "metadata": {"user_id": "u1"}, + "tools": [{"name": "lookup", "input_schema": {"type": "object"}}] + }), + )] + #[case::no_paths( + json!({"max_tokens": 16, "safeguards": [{"type": "dangerous_tool_use"}]}), + &[], + json!({"max_tokens": 16, "safeguards": [{"type": "dangerous_tool_use"}]}), + )] + #[case::model_and_messages_are_never_dropped( + json!({"max_tokens": 16}), + &["model", "messages", "messages[0].content"], + json!({"max_tokens": 16}), + )] + fn prepared_body_drops_configured_paths( + shaping: MessagesShaping, + #[case] fields: Value, + #[case] additional_drop_params: &[&str], + #[case] expected_fields: Value, + ) { + let with_messages = |fields: Value| -> Value { + let Value::Object(fields) = fields else { + unreachable!() + }; + Value::Object( + [ + ("model".to_string(), json!("claude-test")), + ( + "messages".to_string(), + json!([{"role": "user", "content": "hi"}]), + ), + ] + .into_iter() + .chain(fields) + .collect(), + ) + }; + let shaping = MessagesShaping { + additional_drop_params: additional_drop_params + .iter() + .map(ToString::to_string) + .collect(), + ..shaping + }; + assert_eq!( + prepared_body(with_messages(fields), shaping), + Ok(with_messages(expected_fields)) + ); + } + + #[rstest] + #[case::model_prefix_picks_the_provider( + "azure_ai/claude-test", + None, + &[("x-priority", "extra"), ("x-scoped", "azure_ai")] + )] + #[case::explicit_provider( + "claude-test", + Some("anthropic"), + &[("x-priority", "scoped"), ("x-scoped", "anthropic")] + )] + #[case::provider_prefix_on_an_anthropic_model( + "anthropic/claude-test", + None, + &[("x-priority", "scoped"), ("x-scoped", "anthropic")] + )] + fn provider_specific_headers_follow_the_resolved_provider( + shaping: MessagesShaping, + #[case] model: &str, + #[case] custom_llm_provider: Option<&str>, + #[case] expected: &[(&str, &str)], + ) { + let configured: ProviderSpecificHeaders = serde_json::from_value(json!([ + {"custom_llm_provider": "azure_ai", "extra_headers": {"x-scoped": "azure_ai"}}, + {"custom_llm_provider": "anthropic", "extra_headers": {"x-scoped": "anthropic", "x-priority": "scoped"}} + ])) + .unwrap(); + let prepared = prepare(MessagesRequest { + model, + body: json!({"model": model, "messages": [{"role": "user", "content": "hi"}], "max_tokens": 16}), + api_key: Some("sk-test"), + api_base: Some("https://resource.services.ai.azure.com"), + custom_llm_provider, + extra_headers: Some(serde_json::from_value(json!({"x-priority": "extra"})).unwrap()), + provider_specific_header: Some(configured), + timeout: None, + shaping, + }) + .unwrap(); + let caller_headers: Vec<(&str, &str)> = prepared + .upstream_headers + .iter() + .filter(|(name, _)| matches!(name.as_str(), "x-priority" | "x-scoped")) + .map(|(name, value)| (name.as_str(), value.as_str())) + .collect(); + assert_eq!(caller_headers, expected); + } + + #[rstest] + fn prepared_body_carries_the_provider_stripped_model(shaping: MessagesShaping) { + assert_eq!( + prepared_body( + json!({ + "model": "anthropic/claude-test", + "messages": [{"role": "user", "content": "hi"}], + "max_tokens": 16 + }), + shaping, + ), + Ok(json!({ + "model": "claude-test", + "messages": [{"role": "user", "content": "hi"}], + "max_tokens": 16 + })) + ); + } + + #[rstest] + fn dropped_thinking_display_is_not_restored_by_auto_summary(shaping: MessagesShaping) { + let shaping = MessagesShaping { + reasoning_auto_summary: true, + additional_drop_params: vec!["thinking.display".to_string()], + ..shaping + }; + assert_eq!( + prepared_body( + json!({ + "model": "claude-test", + "messages": [{"role": "user", "content": "hi"}], + "max_tokens": 4096, + "thinking": {"type": "enabled", "budget_tokens": 2048} + }), + shaping, + ), + Ok(json!({ + "model": "claude-test", + "messages": [{"role": "user", "content": "hi"}], + "max_tokens": 4096, + "thinking": {"type": "enabled", "budget_tokens": 2048} + })) + ); + } + + #[rstest] + fn dropping_an_invalid_metadata_user_id_does_not_skip_its_validation(shaping: MessagesShaping) { + let shaping = MessagesShaping { + additional_drop_params: vec!["metadata.user_id".to_string()], + ..shaping + }; + assert!(matches!( + prepared_body( + json!({ + "model": "claude-test", + "messages": [{"role": "user", "content": "hi"}], + "max_tokens": 16, + "metadata": {"user_id": 123} + }), + shaping, + ), + Err(Error::InvalidRequest(_)) + )); + } + + #[rstest] + fn prepared_body_rejects_invalid_metadata_before_the_call(shaping: MessagesShaping) { + assert_eq!( + prepared_body( + json!({ + "model": "claude-test", + "messages": [{"role": "user", "content": "hi"}], + "max_tokens": 16, + "metadata": {"user_id": 123} + }), + shaping, + ), + Err(Error::InvalidRequest( + "metadata.user_id must be a string, got 123".to_string() + )) + ); + } } diff --git a/litellm-rust/crates/core/src/messages/route.rs b/litellm-rust/crates/core/src/messages/route.rs index 838b56fcb4b..8cd3eaf3aa3 100644 --- a/litellm-rust/crates/core/src/messages/route.rs +++ b/litellm-rust/crates/core/src/messages/route.rs @@ -1,4 +1,7 @@ -use std::{sync::Mutex, time::Duration}; +use std::{ + sync::{Arc, Mutex}, + time::Duration, +}; use bytes::Bytes; use litellm_auth::SecretValue; @@ -9,15 +12,19 @@ use litellm_host::{ machine::{HostChannel, MachineFault, RouteMachine}, route::Route, }; -use litellm_types::llms::anthropic_messages::anthropic_response::AnthropicMessagesResponse; +use litellm_secrets::source::SecretSource; +use litellm_types::{ + llms::anthropic_messages::anthropic_response::AnthropicMessagesResponse, + utils::ProviderSpecificHeaders, +}; use serde_json::{Map, Value}; use super::{ Error, common_utils::messages_provider_config, handler::{decode_response, network, provider_error, send}, - prepare::prepare_provider_request, - types::MessagesRequest, + prepare::{prepare_provider_request, resolve_provider}, + types::{MessagesRequest, MessagesShaping}, }; use crate::constants::ANTHROPIC_MESSAGES_PROVIDER; @@ -38,7 +45,9 @@ pub struct MessagesCall { pub api_base: Option, pub custom_llm_provider: Option, pub extra_headers: Option>, + pub provider_specific_header: Option, pub timeout: Option, + pub shaping: MessagesShaping, } impl MessagesCall { @@ -120,22 +129,33 @@ impl Host for LocalMessagesHost { } } -pub fn messages_machine() -> MessagesMachine { - RouteMachine::new(|host| Box::pin(execute(host))) +pub fn messages_machine(secrets: Arc) -> MessagesMachine { + RouteMachine::new(move |host| Box::pin(execute(host, secrets.clone()))) } -async fn execute(host: MessagesHost) -> Result { +async fn execute( + host: MessagesHost, + secrets: Arc, +) -> Result { let MessagesOpResult::Request(call) = host.route(MessagesOp::ProjectRequest).await?; let stream = call.streams(); - let request = prepare_provider_request(MessagesRequest { - model: &call.model, - body: Value::Object(call.body.clone()), - api_key: call.api_key.as_deref(), - api_base: call.api_base.as_deref(), - custom_llm_provider: call.custom_llm_provider.as_deref(), - extra_headers: call.extra_headers.clone(), - timeout: call.timeout, - })?; + let resolved = resolve_provider(&call.model, call.custom_llm_provider.as_deref())?; + let secrets = secrets.resolve(resolved.config.secret_names()).await?; + let request = prepare_provider_request( + MessagesRequest { + model: &call.model, + body: Value::Object(call.body.clone()), + api_key: call.api_key.as_deref(), + api_base: call.api_base.as_deref(), + custom_llm_provider: call.custom_llm_provider.as_deref(), + extra_headers: call.extra_headers.clone(), + provider_specific_header: call.provider_specific_header.clone(), + timeout: call.timeout, + shaping: call.shaping.clone(), + }, + resolved, + secrets.as_ref(), + )?; if stream && request.provider != ANTHROPIC_MESSAGES_PROVIDER { return Err(Error::Unsupported("streaming messages for this provider")); } diff --git a/litellm-rust/crates/core/src/messages/tests.rs b/litellm-rust/crates/core/src/messages/tests.rs index 057b42a316c..ce48752864a 100644 --- a/litellm-rust/crates/core/src/messages/tests.rs +++ b/litellm-rust/crates/core/src/messages/tests.rs @@ -1,5 +1,8 @@ -use std::time::Duration; +use std::{sync::Arc, time::Duration}; +use futures_util::future::BoxFuture; +use litellm_http::request::{has_bearer_auth, has_header}; +use litellm_secrets::{SecretValue, source::SecretSource}; use serde_json::{Map, Value, json}; use tokio::{ io::{AsyncReadExt, AsyncWriteExt}, @@ -8,12 +11,132 @@ use tokio::{ use super::{ Error, - common_utils::{ - has_bearer_auth, has_header, messages_provider_config, string_headers, truncate_error_body, - }, + common_utils::{messages_provider_config, string_headers, truncate_error_body}, messages, + route::{LocalMessagesHost, MessagesCall, MessagesOutput, messages_machine}, }; -use crate::messages::types::MessagesRequest; +use crate::messages::types::{MessagesRequest, MessagesShaping}; + +struct RecordingSecrets { + values: Vec<(&'static str, String)>, + fails: bool, + requested: std::sync::Mutex>, +} + +impl RecordingSecrets { + fn new(values: Vec<(&'static str, String)>, fails: bool) -> Self { + Self { + values, + fails, + requested: std::sync::Mutex::new(Vec::new()), + } + } +} + +impl SecretSource for RecordingSecrets { + fn get_secret_str<'a>( + &'a self, + name: &'a str, + ) -> BoxFuture<'a, Result, litellm_secrets::Error>> { + Box::pin(async move { + self.requested.lock().unwrap().push(name.to_string()); + if self.fails { + return Err(litellm_secrets::Error::ManagedSecretMissing); + } + Ok(self + .values + .iter() + .find(|(key, _)| *key == name) + .map(|(_, value)| SecretValue::new(value.clone()))) + }) + } +} + +fn secrets_call() -> MessagesCall { + let Value::Object(body) = json!({ + "model": "claude-sonnet-4-5", + "max_tokens": 16, + "messages": [{"role": "user", "content": "hi"}] + }) else { + unreachable!("literal object") + }; + MessagesCall { + model: "claude-sonnet-4-5".into(), + body, + api_key: None, + api_base: None, + custom_llm_provider: Some("anthropic".into()), + extra_headers: None, + provider_specific_header: None, + timeout: Some(Duration::from_secs(5)), + shaping: MessagesShaping::default(), + } +} + +#[tokio::test] +async fn route_reads_the_provider_credential_and_base_from_the_secret_source() { + let listener = TcpListener::bind("127.0.0.1:0").await.expect("binds"); + let addr = listener.local_addr().expect("addr"); + let server = tokio::spawn(async move { + let (mut socket, _) = listener.accept().await.expect("accepts request"); + let request = read_http_request(&mut socket).await; + let response_body = r#"{"id":"msg_1","type":"message","role":"assistant","content":[],"model":"claude-sonnet-4-5","stop_reason":"end_turn","usage":{"input_tokens":1,"output_tokens":1}}"#; + socket + .write_all(write_response(response_body).as_bytes()) + .await + .expect("writes response"); + request + }); + let secrets = Arc::new(RecordingSecrets::new( + vec![ + ("ANTHROPIC_API_KEY", "sk-from-manager".to_string()), + ("ANTHROPIC_BASE_URL", format!("http://{addr}")), + ], + false, + )); + + let output = litellm_host::run::run( + messages_machine(secrets.clone()), + &LocalMessagesHost::new(secrets_call()), + ) + .await + .expect("messages request succeeds"); + + assert!(matches!(output, MessagesOutput::Message(_))); + let request = server.await.expect("server task completes"); + assert!( + request + .to_ascii_lowercase() + .contains("x-api-key: sk-from-manager"), + "{request}" + ); + let requested = secrets.requested.lock().unwrap().clone(); + assert_eq!( + requested, + messages_provider_config("anthropic") + .unwrap() + .secret_names() + .iter() + .map(ToString::to_string) + .collect::>() + ); +} + +#[tokio::test] +async fn route_surfaces_a_secret_manager_failure_before_the_call() { + let Err(error) = litellm_host::run::run( + messages_machine(Arc::new(RecordingSecrets::new(Vec::new(), true))), + &LocalMessagesHost::new(secrets_call()), + ) + .await + else { + panic!("a secret manager failure fails the call"); + }; + assert!( + matches!(&error, Error::Secret(source) if matches!(source.source_error(), litellm_secrets::Error::ManagedSecretMissing)), + "{error:?}" + ); +} async fn read_http_request(socket: &mut TcpStream) -> String { let mut request = Vec::new(); @@ -159,7 +282,9 @@ async fn messages_round_trip_builds_azure_request_and_passes_response_through() api_base: Some(&format!("http://{addr}")), custom_llm_provider: Some("azure_ai"), extra_headers: None, + provider_specific_header: None, timeout: Some(Duration::from_secs(5)), + shaping: MessagesShaping::default(), }) .await .expect("messages request succeeds"); @@ -215,7 +340,9 @@ async fn messages_round_trip_builds_native_anthropic_request() { api_base: Some(&format!("http://{addr}")), custom_llm_provider: Some("anthropic"), extra_headers: None, + provider_specific_header: None, timeout: Some(Duration::from_secs(5)), + shaping: MessagesShaping::default(), }) .await .expect("messages request succeeds"); @@ -268,7 +395,9 @@ async fn messages_does_not_duplicate_auth_when_x_api_key_supplied() { api_base: Some(&format!("http://{addr}")), custom_llm_provider: Some("azure_ai"), extra_headers: Some(headers), + provider_specific_header: None, timeout: Some(Duration::from_secs(5)), + shaping: MessagesShaping::default(), }) .await .expect("messages request succeeds"); @@ -322,7 +451,9 @@ async fn messages_forwards_entra_id_bearer_without_requiring_api_key() { api_base: Some(&format!("http://{addr}")), custom_llm_provider: Some("azure_ai"), extra_headers: Some(headers), + provider_specific_header: None, timeout: Some(Duration::from_secs(5)), + shaping: MessagesShaping::default(), }) .await .expect("entra id request succeeds without api key"); @@ -346,7 +477,9 @@ async fn messages_requires_auth_when_no_key_and_no_header() { api_base: Some("http://127.0.0.1:1"), custom_llm_provider: Some("azure_ai"), extra_headers: None, + provider_specific_header: None, timeout: Some(Duration::from_millis(50)), + shaping: MessagesShaping::default(), }) .await .expect_err("missing auth errors"); @@ -384,7 +517,9 @@ async fn messages_ignores_malformed_authorization_and_uses_api_key() { api_base: Some(&format!("http://{addr}")), custom_llm_provider: Some("azure_ai"), extra_headers: Some(headers), + provider_specific_header: None, timeout: Some(Duration::from_secs(5)), + shaping: MessagesShaping::default(), }) .await .expect("falls back to api key"); @@ -425,7 +560,9 @@ async fn messages_maps_provider_error_status_to_http_error() { api_base: Some(&format!("http://{addr}")), custom_llm_provider: Some("azure_ai"), extra_headers: None, + provider_specific_header: None, timeout: Some(Duration::from_secs(5)), + shaping: MessagesShaping::default(), }) .await .expect_err("provider error propagates"); @@ -445,7 +582,9 @@ async fn messages_rejects_unsupported_provider() { api_base: Some("http://127.0.0.1:1"), custom_llm_provider: Some("openai"), extra_headers: None, + provider_specific_header: None, timeout: Some(Duration::from_millis(50)), + shaping: MessagesShaping::default(), }) .await .expect_err("unsupported provider errors"); diff --git a/litellm-rust/crates/core/src/messages/types.rs b/litellm-rust/crates/core/src/messages/types.rs index a73ceffad7a..4a5dd2926e0 100644 --- a/litellm-rust/crates/core/src/messages/types.rs +++ b/litellm-rust/crates/core/src/messages/types.rs @@ -1,8 +1,25 @@ use std::time::Duration; -use litellm_llms::base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig; +use litellm_llms::{ + anthropic::common_utils::AnthropicModelCapabilities, + base_llm::anthropic_messages::transformation::BaseAnthropicMessagesConfig, +}; +use litellm_types::utils::ProviderSpecificHeaders; +use serde::{Deserialize, Serialize}; use serde_json::{Map, Value}; +#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)] +pub struct MessagesShaping { + #[serde(default)] + pub capabilities: AnthropicModelCapabilities, + #[serde(default)] + pub drop_params: bool, + #[serde(default)] + pub reasoning_auto_summary: bool, + #[serde(default)] + pub additional_drop_params: Vec, +} + pub struct MessagesRequest<'a> { pub model: &'a str, pub body: Value, @@ -10,7 +27,9 @@ pub struct MessagesRequest<'a> { pub api_base: Option<&'a str>, pub custom_llm_provider: Option<&'a str>, pub extra_headers: Option>, + pub provider_specific_header: Option, pub timeout: Option, + pub shaping: MessagesShaping, } pub struct ProviderMessagesRequest { @@ -22,3 +41,86 @@ pub struct ProviderMessagesRequest { pub upstream_headers: Vec<(String, String)>, pub timeout: Option, } + +#[cfg(test)] +mod tests { + use litellm_llms::anthropic::common_utils::SupportedEffortTiers; + use rstest::rstest; + use serde_json::json; + + use super::*; + + #[rstest] + #[case::nothing_projected(json!({}), MessagesShaping::default())] + #[case::only_drop_params( + json!({"drop_params": true}), + MessagesShaping { drop_params: true, ..MessagesShaping::default() }, + )] + #[case::only_reasoning_auto_summary( + json!({"reasoning_auto_summary": true}), + MessagesShaping { reasoning_auto_summary: true, ..MessagesShaping::default() }, + )] + #[case::only_additional_drop_params( + json!({"additional_drop_params": ["tools[*].input_examples"]}), + MessagesShaping { + additional_drop_params: vec!["tools[*].input_examples".to_string()], + ..MessagesShaping::default() + }, + )] + #[case::partial_capabilities( + json!({"capabilities": {"supports_reasoning": true}}), + MessagesShaping { + capabilities: AnthropicModelCapabilities { + supports_reasoning: true, + ..AnthropicModelCapabilities::default() + }, + ..MessagesShaping::default() + }, + )] + #[case::everything_the_python_host_projects( + json!({ + "capabilities": { + "supports_reasoning": true, + "supports_adaptive_thinking": true, + "thinking_always_on": false, + "supports_legacy_thinking": false, + "supports_output_config": true, + "supports_sampling_params": false, + "supports_speed": true, + "effort_tiers": {"minimal": false, "low": true, "medium": true, "high": true, "xhigh": true, "max": false} + }, + "drop_params": true, + "reasoning_auto_summary": true, + "additional_drop_params": ["metadata.user_id", "thinking"] + }), + MessagesShaping { + capabilities: AnthropicModelCapabilities { + supports_reasoning: true, + supports_adaptive_thinking: true, + thinking_always_on: false, + supports_legacy_thinking: false, + supports_output_config: true, + supports_sampling_params: false, + supports_speed: true, + effort_tiers: SupportedEffortTiers { + minimal: false, + low: true, + medium: true, + high: true, + xhigh: true, + max: false, + }, + }, + drop_params: true, + reasoning_auto_summary: true, + additional_drop_params: vec!["metadata.user_id".to_string(), "thinking".to_string()], + }, + )] + fn shaping_deserializes_with_defaults_for_absent_fields( + #[case] projected: Value, + #[case] expected: MessagesShaping, + ) { + let shaping: MessagesShaping = serde_json::from_value(projected).unwrap(); + assert_eq!(shaping, expected); + } +} diff --git a/litellm-rust/crates/llms/src/anthropic/common_utils.rs b/litellm-rust/crates/llms/src/anthropic/common_utils.rs new file mode 100644 index 00000000000..a2234e0df03 --- /dev/null +++ b/litellm-rust/crates/llms/src/anthropic/common_utils.rs @@ -0,0 +1,1568 @@ +use litellm_types::llms::anthropic_messages::anthropic_request::{ + AnthropicMessage, ContentBlock, MessageContent, +}; +use serde::{Deserialize, Serialize}; +use serde_json::Value; + +use crate::anthropic::ANTHROPIC_OAUTH_TOKEN_PREFIX; + +pub const ANTHROPIC_OAUTH_BETA_HEADER: &str = "oauth-2025-04-20"; +pub const ANTHROPIC_ADVISOR_TOOL_TYPE: &str = "advisor_20260301"; +pub const ANTHROPIC_TOOL_SEARCH_TOOL_TYPES: [&str; 2] = [ + "tool_search_tool_regex_20251119", + "tool_search_tool_bm25_20251119", +]; +pub const ENCRYPTED_REASONING_SIGNATURE_PREFIX: &str = "litellm_encrypted_reasoning:"; +const THOUGHT_SIGNATURE_SEPARATOR: &str = "__thought__"; + +pub mod beta { + pub const CONTEXT_MANAGEMENT_2025_06_27: &str = "context-management-2025-06-27"; + pub const COMPACT_2026_01_12: &str = "compact-2026-01-12"; + pub const COMPACT_2026_09_04: &str = "compact-2026-09-04"; + pub const STRUCTURED_OUTPUT: &str = "structured-outputs-2025-11-13"; + pub const ADVANCED_TOOL_USE_2025_11_20: &str = "advanced-tool-use-2025-11-20"; + pub const FAST_MODE_2026_02_01: &str = "fast-mode-2026-02-01"; + pub const ADVISOR_TOOL_2026_03_01: &str = "advisor-tool-2026-03-01"; + pub const PER_TURN_CONTROL_2026_07_01: &str = "per-turn-control-2026-07-01"; +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum EffortLevel { + Low, + Medium, + High, + Xhigh, + Max, +} + +impl EffortLevel { + pub fn as_str(self) -> &'static str { + match self { + Self::Low => "low", + Self::Medium => "medium", + Self::High => "high", + Self::Xhigh => "xhigh", + Self::Max => "max", + } + } + + pub fn parse(value: &str) -> Option { + match value { + "low" => Some(Self::Low), + "medium" => Some(Self::Medium), + "high" => Some(Self::High), + "xhigh" => Some(Self::Xhigh), + "max" => Some(Self::Max), + _ => None, + } + } +} + +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct SupportedEffortTiers { + #[serde(default)] + pub minimal: bool, + #[serde(default)] + pub low: bool, + #[serde(default)] + pub medium: bool, + #[serde(default)] + pub high: bool, + #[serde(default)] + pub xhigh: bool, + #[serde(default)] + pub max: bool, +} + +impl SupportedEffortTiers { + pub fn any(self) -> bool { + self.minimal || self.low || self.medium || self.high || self.xhigh || self.max + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct AnthropicModelCapabilities { + #[serde(default)] + pub supports_reasoning: bool, + #[serde(default)] + pub supports_adaptive_thinking: bool, + #[serde(default)] + pub thinking_always_on: bool, + #[serde(default)] + pub supports_legacy_thinking: bool, + #[serde(default)] + pub supports_output_config: bool, + #[serde(default = "default_true")] + pub supports_sampling_params: bool, + #[serde(default)] + pub supports_speed: bool, + #[serde(default)] + pub effort_tiers: SupportedEffortTiers, +} + +fn default_true() -> bool { + true +} + +impl Default for AnthropicModelCapabilities { + fn default() -> Self { + Self { + supports_reasoning: false, + supports_adaptive_thinking: false, + thinking_always_on: false, + supports_legacy_thinking: false, + supports_output_config: false, + supports_sampling_params: true, + supports_speed: false, + effort_tiers: SupportedEffortTiers::default(), + } + } +} + +impl AnthropicModelCapabilities { + pub fn supports_effort_tier(&self, level: EffortLevel) -> bool { + match level { + EffortLevel::Low => self.effort_tiers.low, + EffortLevel::Medium => self.effort_tiers.medium, + EffortLevel::High => self.effort_tiers.high, + EffortLevel::Xhigh => self.effort_tiers.xhigh, + EffortLevel::Max => self.effort_tiers.max, + } + } + + pub fn supports_effort_param(&self) -> bool { + self.supports_output_config || self.effort_tiers.any() + } + + pub fn effort_level_rejection(&self, effort: &str, model: &str) -> Option { + match effort { + "max" if !(self.supports_adaptive_thinking || self.effort_tiers.max) => Some(format!( + "effort='max' is not supported by this model. Got model: {model}" + )), + "xhigh" if !self.effort_tiers.xhigh => Some(format!( + "effort='xhigh' is not supported by this model. Got model: {model}" + )), + _ => None, + } + } +} + +pub fn is_anthropic_oauth_key(value: &str) -> bool { + value + .strip_prefix("Bearer ") + .unwrap_or(value) + .starts_with(ANTHROPIC_OAUTH_TOKEN_PREFIX) +} + +pub fn split_beta_values(header: Option<&str>) -> impl Iterator + '_ { + header + .into_iter() + .flat_map(|value| value.split(',')) + .map(str::trim) + .filter(|piece| !piece.is_empty()) + .map(str::to_string) +} + +pub fn join_beta_values(values: impl IntoIterator) -> String { + let mut values: Vec = values.into_iter().collect(); + values.sort(); + values.dedup(); + values.join(",") +} + +pub fn is_tool_search_used(tools: Option<&[Value]>) -> bool { + tools.into_iter().flatten().any(|tool| { + tool.get("type") + .and_then(Value::as_str) + .is_some_and(|tool_type| ANTHROPIC_TOOL_SEARCH_TOOL_TYPES.contains(&tool_type)) + }) +} + +pub fn has_advisor_tool(tools: Option<&[Value]>) -> bool { + tools + .into_iter() + .flatten() + .any(|tool| tool.get("type").and_then(Value::as_str) == Some(ANTHROPIC_ADVISOR_TOOL_TYPE)) +} + +pub fn requires_native_compaction_beta( + compaction: Option<&Value>, + messages: &[AnthropicMessage], +) -> bool { + compaction.is_some() + || messages + .iter() + .flat_map(AnthropicMessage::blocks) + .any(|block| { + block.is_type("compaction") + && block.signature.as_deref().is_some_and(|s| !s.is_empty()) + }) +} + +fn is_blank(text: Option<&str>) -> bool { + text.is_none_or(|text| text.trim().is_empty()) +} + +fn is_empty_text_block(block: &ContentBlock) -> bool { + block.is_type("text") && is_blank(block.text.as_deref()) +} + +pub fn is_empty_thinking_block(block: &ContentBlock) -> bool { + block.is_type("thinking") && is_blank(block.thinking.as_deref()) +} + +fn retain_blocks( + messages: Vec, + keep: impl Fn(&ContentBlock) -> bool, +) -> Vec { + messages + .into_iter() + .filter_map(|message| match message.content { + MessageContent::Text(_) => Some(message), + MessageContent::Blocks(ref blocks) => { + let kept: Vec = + blocks.iter().filter(|block| keep(block)).cloned().collect(); + if kept.len() == blocks.len() { + return Some(message); + } + (!kept.is_empty()).then(|| message.with_blocks(kept)) + } + }) + .collect() +} + +pub fn strip_empty_content_blocks(messages: Vec) -> Vec { + retain_blocks(messages, |block| { + !is_empty_text_block(block) && !is_empty_thinking_block(block) + }) +} + +pub fn normalize_anthropic_tool_use_id(raw_id: &str) -> String { + let base = raw_id + .split_once(THOUGHT_SIGNATURE_SEPARATOR) + .map_or(raw_id, |(base, _)| base); + let sanitized: String = base + .chars() + .map(|character| { + if character.is_ascii_alphanumeric() || matches!(character, '_' | '-') { + character + } else { + '_' + } + }) + .collect(); + if sanitized.is_empty() { + "tool_use_id".to_string() + } else { + sanitized + } +} + +fn normalized_if_changed(raw_id: Option<&str>) -> Option { + let raw_id = raw_id?; + let normalized = normalize_anthropic_tool_use_id(raw_id); + (normalized != raw_id).then_some(normalized) +} + +fn sanitize_tool_use_id_block(block: ContentBlock) -> ContentBlock { + match block.block_type.as_deref() { + Some("tool_use" | "server_tool_use") => match normalized_if_changed(block.id.as_deref()) { + Some(id) => ContentBlock { + id: Some(id), + ..block + }, + None => block, + }, + Some("tool_result") => match normalized_if_changed(block.tool_use_id.as_deref()) { + Some(tool_use_id) => ContentBlock { + tool_use_id: Some(tool_use_id), + ..block + }, + None => block, + }, + _ => block, + } +} + +fn map_blocks( + messages: Vec, + rewrite: impl Fn(Vec) -> Vec, +) -> Vec { + messages + .into_iter() + .map(|message| match message.content { + MessageContent::Blocks(blocks) => AnthropicMessage { + content: MessageContent::Blocks(rewrite(blocks)), + ..message + }, + MessageContent::Text(_) => message, + }) + .collect() +} + +pub fn sanitize_tool_use_ids(messages: Vec) -> Vec { + map_blocks(messages, |blocks| { + blocks.into_iter().map(sanitize_tool_use_id_block).collect() + }) +} + +pub fn strip_provider_specific_fields(messages: Vec) -> Vec { + map_blocks(messages, |blocks| { + blocks + .into_iter() + .map(|block| ContentBlock { + provider_specific_fields: None, + ..block + }) + .collect() + }) +} + +pub fn is_encrypted_reasoning_block(block: &ContentBlock) -> bool { + let field = match block.block_type.as_deref() { + Some("thinking") => block.signature.as_deref(), + Some("redacted_thinking") => block.data.as_deref(), + _ => None, + }; + field.is_some_and(|value| value.starts_with(ENCRYPTED_REASONING_SIGNATURE_PREFIX)) +} + +pub fn strip_encrypted_reasoning_blocks(messages: Vec) -> Vec { + retain_blocks(messages, |block| !is_encrypted_reasoning_block(block)) +} + +fn is_advisor_use(block: &ContentBlock) -> bool { + block.is_type("server_tool_use") + && block.name.as_deref() == Some("advisor") + && block.id.as_deref().is_some_and(|id| !id.is_empty()) +} + +pub fn strip_advisor_blocks(messages: Vec) -> Vec { + messages + .into_iter() + .map(|message| { + if message.role != "assistant" { + return message; + } + let MessageContent::Blocks(blocks) = &message.content else { + return message; + }; + let advisor_ids: Vec<&str> = blocks + .iter() + .filter(|block| is_advisor_use(block)) + .filter_map(|block| block.id.as_deref()) + .collect(); + if advisor_ids.is_empty() { + return message; + } + let kept: Vec = blocks + .iter() + .filter(|block| { + let is_result = block.is_type("advisor_tool_result") + && block + .tool_use_id + .as_deref() + .is_some_and(|id| advisor_ids.contains(&id)); + !is_advisor_use(block) && !is_result + }) + .cloned() + .collect(); + message.with_blocks(kept) + }) + .collect() +} + +#[derive(Deserialize)] +struct ReplayedWebSearchResult { + #[serde(default)] + url: String, + #[serde(default)] + title: String, + #[serde(default)] + snippet: String, + #[serde(default)] + encrypted_content: String, +} + +#[derive(Deserialize)] +#[serde(tag = "type")] +enum ReplayedWebSearchContent { + #[serde(rename = "web_search_tool_result_error")] + Error { + #[serde(default)] + error_code: String, + }, +} + +enum WebSearchResults { + Results(Vec), + Error(String), +} + +fn flattenable_web_search_results(block: &ContentBlock) -> Option<(&str, WebSearchResults)> { + if !block.is_type("web_search_tool_result") { + return None; + } + let tool_use_id = block.tool_use_id.as_deref()?; + let results = match block.content.as_ref()? { + Value::Array(items) => { + let results = items + .iter() + .map(|item| { + (item.get("type").and_then(Value::as_str) == Some("web_search_result")) + .then(|| { + serde_json::from_value::(item.clone()).ok() + }) + .flatten() + }) + .collect::>>()?; + if results + .iter() + .any(|result| !result.encrypted_content.is_empty()) + { + return None; + } + WebSearchResults::Results(results) + } + error @ Value::Object(_) => match serde_json::from_value(error.clone()).ok()? { + ReplayedWebSearchContent::Error { error_code } => WebSearchResults::Error(error_code), + }, + _ => return None, + }; + Some((tool_use_id, results)) +} + +fn render_web_search_results(query: &str, results: &WebSearchResults) -> String { + let header = if query.is_empty() { + "Web search results:".to_string() + } else { + format!("Web search results for '{query}':") + }; + match results { + WebSearchResults::Error(code) => { + let code = if code.is_empty() { "unavailable" } else { code }; + format!("{header}\n\nSearch failed: {code}") + } + WebSearchResults::Results(results) if results.is_empty() => { + format!("{header}\n\nNo results were returned.") + } + WebSearchResults::Results(results) => { + let body = results + .iter() + .map(|result| { + [ + (!result.title.is_empty()).then(|| format!("Title: {}", result.title)), + (!result.url.is_empty()).then(|| format!("URL: {}", result.url)), + (!result.snippet.is_empty()) + .then(|| format!("Snippet: {}", result.snippet)), + ] + .into_iter() + .flatten() + .collect::>() + .join("\n") + }) + .collect::>() + .join("\n\n"); + if body.is_empty() { + header + } else { + format!("{header}\n\n{body}") + } + } + } +} + +fn server_tool_use_query(block: &ContentBlock) -> Option<(&str, &str)> { + if !block.is_type("server_tool_use") { + return None; + } + let id = block.id.as_deref()?; + let query = match block.input.as_ref() { + None => "", + Some(Value::Object(input)) => match input.get("query") { + None => "", + Some(query) => query.as_str()?, + }, + Some(_) => return None, + }; + Some((id, query)) +} + +fn flatten_web_search_results_in_blocks(blocks: Vec) -> Vec { + let flattenable_ids: Vec<&str> = blocks + .iter() + .filter_map(flattenable_web_search_results) + .map(|(tool_use_id, _)| tool_use_id) + .collect(); + if flattenable_ids.is_empty() { + return blocks; + } + let queries: Vec<(&str, &str)> = blocks.iter().filter_map(server_tool_use_query).collect(); + blocks + .iter() + .filter_map(|block| { + if let Some((tool_use_id, results)) = flattenable_web_search_results(block) { + let query = queries + .iter() + .rfind(|(id, _)| *id == tool_use_id) + .map_or("", |(_, query)| query); + return Some(ContentBlock::text(render_web_search_results( + query, &results, + ))); + } + if let Some((id, _)) = server_tool_use_query(block) + && flattenable_ids.contains(&id) + { + return None; + } + Some(block.clone()) + }) + .collect() +} + +pub fn flatten_unencrypted_web_search_results( + messages: Vec, +) -> Vec { + map_blocks(messages, flatten_web_search_results_in_blocks) +} + +#[cfg(test)] +mod tests { + use rstest::{fixture, rstest}; + use serde_json::json; + + use super::*; + + const ALL_LEVELS: [EffortLevel; 5] = [ + EffortLevel::Low, + EffortLevel::Medium, + EffortLevel::High, + EffortLevel::Xhigh, + EffortLevel::Max, + ]; + + fn apply( + sanitizer: fn(Vec) -> Vec, + messages: Value, + ) -> Value { + let parsed: Vec = serde_json::from_value(messages).unwrap(); + serde_json::to_value(sanitizer(parsed)).unwrap() + } + + fn block(value: Value) -> ContentBlock { + serde_json::from_value(value).unwrap() + } + + fn history(messages: Value) -> Vec { + serde_json::from_value(messages).unwrap() + } + + fn tools(value: Option) -> Option> { + value.map(|tools| tools.as_array().unwrap().clone()) + } + + fn tagged(encrypted: &str) -> String { + format!("{ENCRYPTED_REASONING_SIGNATURE_PREFIX}{encrypted}") + } + + fn tiers( + minimal: bool, + low: bool, + medium: bool, + high: bool, + xhigh: bool, + max: bool, + ) -> SupportedEffortTiers { + SupportedEffortTiers { + minimal, + low, + medium, + high, + xhigh, + max, + } + } + + fn replayed_search_turn(results: Value) -> Value { + json!([ + {"role": "user", "content": "when was Rome founded?"}, + {"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {"query": "when"}}, + {"type": "web_search_tool_result", "tool_use_id": "srvtoolu_1", "content": results}, + {"type": "text", "text": "753 BC."} + ]} + ]) + } + + #[fixture] + fn unmapped() -> AnthropicModelCapabilities { + AnthropicModelCapabilities::default() + } + + #[rstest] + #[case::empty_text(json!({"type": "thinking", "thinking": ""}), true)] + #[case::whitespace_only(json!({"type": "thinking", "thinking": " \n\t "}), true)] + #[case::null_text(json!({"type": "thinking", "thinking": null}), true)] + #[case::missing_text(json!({"type": "thinking"}), true)] + #[case::empty_text_despite_signature(json!({"type": "thinking", "thinking": "", "signature": "sig_abc"}), true)] + #[case::real_thinking(json!({"type": "thinking", "thinking": "plan", "signature": "sig"}), false)] + #[case::padded_real_thinking(json!({"type": "thinking", "thinking": " plan "}), false)] + #[case::redacted_thinking_is_a_different_type(json!({"type": "redacted_thinking", "data": "opaque"}), false)] + #[case::empty_text_block(json!({"type": "text", "text": ""}), false)] + #[case::untyped_block(json!({"thinking": ""}), false)] + fn empty_thinking_block_detection(#[case] input: Value, #[case] expected: bool) { + assert_eq!(is_empty_thinking_block(&block(input)), expected); + } + + #[rstest] + #[case::empty_text_beside_tool_use( + json!([{"role": "assistant", "content": [ + {"type": "text", "text": ""}, + {"type": "tool_use", "id": "x", "name": "Bash", "input": {}} + ]}]), + json!([{"role": "assistant", "content": [{"type": "tool_use", "id": "x", "name": "Bash", "input": {}}]}]) + )] + #[case::whitespace_text_beside_tool_use( + json!([{"role": "assistant", "content": [ + {"type": "text", "text": " \n "}, + {"type": "tool_use", "id": "x", "name": "Bash", "input": {}} + ]}]), + json!([{"role": "assistant", "content": [{"type": "tool_use", "id": "x", "name": "Bash", "input": {}}]}]) + )] + #[case::null_text( + json!([{"role": "user", "content": [ + {"type": "text", "text": null}, + {"type": "tool_result", "tool_use_id": "x", "content": "y"} + ]}]), + json!([{"role": "user", "content": [{"type": "tool_result", "tool_use_id": "x", "content": "y"}]}]) + )] + #[case::missing_text( + json!([{"role": "user", "content": [ + {"type": "text"}, + {"type": "tool_result", "tool_use_id": "x", "content": "y"} + ]}]), + json!([{"role": "user", "content": [{"type": "tool_result", "tool_use_id": "x", "content": "y"}]}]) + )] + #[case::empty_signed_thinking_beside_tool_use( + json!([ + {"role": "user", "content": "weather?"}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "", "signature": "sig_abc"}, + {"type": "tool_use", "id": "toolu_01A", "name": "get_weather", "input": {"city": "Paris"}} + ]} + ]), + json!([ + {"role": "user", "content": "weather?"}, + {"role": "assistant", "content": [ + {"type": "tool_use", "id": "toolu_01A", "name": "get_weather", "input": {"city": "Paris"}} + ]} + ]) + )] + #[case::whitespace_thinking_beside_real_and_redacted_thinking( + json!([{"role": "assistant", "content": [ + {"type": "thinking", "thinking": " \n "}, + {"type": "thinking", "thinking": "real plan", "signature": "sig"}, + {"type": "redacted_thinking", "data": "opaque"} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "thinking", "thinking": "real plan", "signature": "sig"}, + {"type": "redacted_thinking", "data": "opaque"} + ]}]) + )] + #[case::blank_text_beside_real_thinking( + json!([{"role": "assistant", "content": [ + {"type": "thinking", "thinking": "plan", "signature": "sig"}, + {"type": "text", "text": ""} + ]}]), + json!([{"role": "assistant", "content": [{"type": "thinking", "thinking": "plan", "signature": "sig"}]}]) + )] + #[case::message_left_without_blocks_is_dropped( + json!([ + {"role": "user", "content": "hello"}, + {"role": "assistant", "content": [{"type": "text", "text": ""}]}, + {"role": "assistant", "content": [{"type": "thinking", "thinking": ""}]} + ]), + json!([{"role": "user", "content": "hello"}]) + )] + fn strip_empty_content_blocks_rewrites(#[case] input: Value, #[case] expected: Value) { + assert_eq!(apply(strip_empty_content_blocks, input), expected); + } + + #[rstest] + #[case::non_empty_text(json!([{"role": "assistant", "content": [{"type": "text", "text": "hi"}]}]))] + #[case::padded_text(json!([{"role": "assistant", "content": [{"type": "text", "text": " hi "}]}]))] + #[case::empty_string_content(json!([{"role": "user", "content": ""}]))] + #[case::textless_non_text_block(json!([{"role": "user", "content": [ + {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": "AA=="}} + ]}]))] + #[case::encrypted_reasoning_left_for_the_responses_bridge(json!([{"role": "assistant", "content": [ + {"type": "thinking", "thinking": "plan", "signature": tagged("gAAAA_1")}, + {"type": "redacted_thinking", "data": tagged("gAAAA_2")}, + {"type": "text", "text": "The answer."} + ]}]))] + fn strip_empty_content_blocks_leaves_untouched(#[case] input: Value) { + assert_eq!(apply(strip_empty_content_blocks, input.clone()), input); + } + + #[rstest] + #[case::replayed_provider_id("functions.Bash:0", "functions_Bash_0")] + #[case::thought_signature_suffix("call_abc123__thought__CiIBDDnWx+/a==", "call_abc123")] + #[case::splits_at_first_thought_separator("call_1__thought__a__thought__b", "call_1")] + #[case::valid_id("toolu_01-A_b", "toolu_01-A_b")] + #[case::non_ascii_letter("café", "caf_")] + #[case::only_invalid_characters("::", "__")] + #[case::empty("", "tool_use_id")] + #[case::thought_signature_only("__thought__CiIB", "tool_use_id")] + fn normalize_anthropic_tool_use_id_cases(#[case] raw: &str, #[case] expected: &str) { + assert_eq!(normalize_anthropic_tool_use_id(raw), expected); + } + + #[rstest] + #[case::tool_use_and_its_result( + json!([ + {"role": "assistant", "content": [{"type": "tool_use", "id": "functions.Bash:0", "name": "Bash", "input": {}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions.Bash:0", "content": "ok"}]} + ]), + json!([ + {"role": "assistant", "content": [{"type": "tool_use", "id": "functions_Bash_0", "name": "Bash", "input": {}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions_Bash_0", "content": "ok"}]} + ]) + )] + #[case::server_tool_use( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "srv.1", "name": "web_search", "input": {}} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "srv_1", "name": "web_search", "input": {}} + ]}]) + )] + #[case::tool_use_rewrites_only_its_id( + json!([{"role": "assistant", "content": [ + {"type": "tool_use", "id": "a.b", "tool_use_id": "c.d", "name": "Bash", "input": {}} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "tool_use", "id": "a_b", "tool_use_id": "c.d", "name": "Bash", "input": {}} + ]}]) + )] + #[case::tool_result_rewrites_only_its_tool_use_id( + json!([{"role": "user", "content": [ + {"type": "tool_result", "id": "a.b", "tool_use_id": "c.d", "content": "ok"} + ]}]), + json!([{"role": "user", "content": [ + {"type": "tool_result", "id": "a.b", "tool_use_id": "c_d", "content": "ok"} + ]}]) + )] + fn sanitize_tool_use_ids_rewrites(#[case] input: Value, #[case] expected: Value) { + assert_eq!(apply(sanitize_tool_use_ids, input), expected); + } + + #[rstest] + #[case::valid_ids(json!([ + {"role": "assistant", "content": [{"type": "tool_use", "id": "toolu_01", "name": "Bash", "input": {}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "toolu_01", "content": "ok"}]} + ]))] + #[case::id_mentioned_in_text(json!([{"role": "user", "content": [{"type": "text", "text": "id: functions.Bash:0"}]}]))] + #[case::tool_use_without_id(json!([{"role": "assistant", "content": [{"type": "tool_use", "name": "Bash", "input": {}}]}]))] + #[case::tool_result_without_tool_use_id(json!([{"role": "user", "content": [{"type": "tool_result", "content": "ok"}]}]))] + #[case::string_content(json!([{"role": "user", "content": "functions.Bash:0"}]))] + fn sanitize_tool_use_ids_leaves_untouched(#[case] input: Value) { + assert_eq!(apply(sanitize_tool_use_ids, input.clone()), input); + } + + #[rstest] + #[case::thinking_block( + json!([{"role": "assistant", "content": [ + {"type": "thinking", "thinking": "hm", "signature": "s", "provider_specific_fields": {"a": 1}} + ]}]), + json!([{"role": "assistant", "content": [{"type": "thinking", "thinking": "hm", "signature": "s"}]}]) + )] + #[case::every_block_of_every_message( + json!([ + {"role": "assistant", "content": [ + {"type": "text", "text": "a", "provider_specific_fields": {"x": 1}}, + {"type": "tool_use", "id": "t1", "name": "f", "input": {}, "provider_specific_fields": {"y": 2}} + ]}, + {"role": "user", "content": [ + {"type": "tool_result", "tool_use_id": "t1", "content": "ok", "provider_specific_fields": {}} + ]} + ]), + json!([ + {"role": "assistant", "content": [ + {"type": "text", "text": "a"}, + {"type": "tool_use", "id": "t1", "name": "f", "input": {}} + ]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "t1", "content": "ok"}]} + ]) + )] + fn strip_provider_specific_fields_rewrites(#[case] input: Value, #[case] expected: Value) { + assert_eq!(apply(strip_provider_specific_fields, input), expected); + } + + #[rstest] + #[case::string_content(json!([{"role": "user", "content": "provider_specific_fields"}]))] + #[case::blocks_without_the_field(json!([{"role": "assistant", "content": [{"type": "text", "text": "a"}]}]))] + fn strip_provider_specific_fields_leaves_untouched(#[case] input: Value) { + assert_eq!(apply(strip_provider_specific_fields, input.clone()), input); + } + + #[rstest] + #[case::tagged_thinking_signature(json!({"type": "thinking", "thinking": "x", "signature": tagged("g")}), true)] + #[case::tagged_redacted_data(json!({"type": "redacted_thinking", "data": tagged("g")}), true)] + #[case::bare_tag_signature(json!({"type": "thinking", "thinking": "x", "signature": tagged("")}), true)] + #[case::bare_tag_data(json!({"type": "redacted_thinking", "data": tagged("")}), true)] + #[case::anthropic_signature(json!({"type": "thinking", "thinking": "x", "signature": "ErcBCkgIValid"}), false)] + #[case::anthropic_data(json!({"type": "redacted_thinking", "data": "EmwKAhgBEgy"}), false)] + #[case::unsigned_thinking(json!({"type": "thinking", "thinking": "x"}), false)] + #[case::tag_in_text_block(json!({"type": "text", "text": tagged("g")}), false)] + #[case::tag_in_thinking_data(json!({"type": "thinking", "thinking": "x", "data": tagged("g")}), false)] + #[case::tag_in_redacted_signature( + json!({"type": "redacted_thinking", "data": "EmwKAhgBEgy", "signature": tagged("g")}), + false + )] + #[case::tag_not_at_start(json!({"type": "thinking", "thinking": "x", "signature": format!("x{}", tagged("g"))}), false)] + fn encrypted_reasoning_block_detection(#[case] input: Value, #[case] expected: bool) { + assert_eq!(is_encrypted_reasoning_block(&block(input)), expected); + } + + #[rstest] + #[case::only_the_bridge_tagged_blocks( + json!([ + {"role": "user", "content": "Solve it."}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "plan", "signature": tagged("gAAAA_1")}, + {"type": "redacted_thinking", "data": tagged("gAAAA_2")} + ]}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "plan", "signature": tagged("gAAAA_3")}, + {"type": "thinking", "thinking": "native", "signature": "EqQBCkYIAxgCIkA_anthropic_signed"}, + {"type": "redacted_thinking", "data": "EmwKAhgBEgy_anthropic_minted"}, + {"type": "text", "text": "The answer."} + ]} + ]), + json!([ + {"role": "user", "content": "Solve it."}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "native", "signature": "EqQBCkYIAxgCIkA_anthropic_signed"}, + {"type": "redacted_thinking", "data": "EmwKAhgBEgy_anthropic_minted"}, + {"type": "text", "text": "The answer."} + ]} + ]) + )] + #[case::bridge_turn_keeps_its_text( + json!([ + {"role": "user", "content": "Solve it."}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "plan", "signature": tagged("gAAAA_1")}, + {"type": "redacted_thinking", "data": tagged("gAAAA_2")}, + {"type": "text", "text": "The answer."} + ]}, + {"role": "user", "content": "And the next one?"} + ]), + json!([ + {"role": "user", "content": "Solve it."}, + {"role": "assistant", "content": [{"type": "text", "text": "The answer."}]}, + {"role": "user", "content": "And the next one?"} + ]) + )] + fn strip_encrypted_reasoning_blocks_rewrites(#[case] input: Value, #[case] expected: Value) { + assert_eq!(apply(strip_encrypted_reasoning_blocks, input), expected); + } + + #[rstest] + #[case::anthropic_signed_blocks(json!([ + {"role": "user", "content": "Solve it."}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "plan", "signature": "EqQBCkYIAxgCIkA_anthropic_signed"}, + {"type": "redacted_thinking", "data": "EmwKAhgBEgy_anthropic_minted"}, + {"type": "text", "text": "The answer."} + ]} + ]))] + #[case::string_content(json!([{"role": "user", "content": tagged("g")}]))] + fn strip_encrypted_reasoning_blocks_leaves_untouched(#[case] input: Value) { + assert_eq!( + apply(strip_encrypted_reasoning_blocks, input.clone()), + input + ); + } + + #[rstest] + #[case::advisor_exchange_between_texts( + json!([ + {"role": "user", "content": "Build a worker pool."}, + {"role": "assistant", "content": [ + {"type": "text", "text": "Let me consult the advisor."}, + {"type": "server_tool_use", "id": "srvtoolu_abc123", "name": "advisor", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "srvtoolu_abc123", + "content": {"type": "advisor_result", "text": "Use channels."}}, + {"type": "text", "text": "Here is the implementation."} + ]} + ]), + json!([ + {"role": "user", "content": "Build a worker pool."}, + {"role": "assistant", "content": [ + {"type": "text", "text": "Let me consult the advisor."}, + {"type": "text", "text": "Here is the implementation."} + ]} + ]) + )] + #[case::only_results_of_this_turns_advisor_calls( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "adv_1", "name": "advisor", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "adv_1", "content": "advice"}, + {"type": "advisor_tool_result", "tool_use_id": "other", "content": "kept"}, + {"type": "tool_result", "tool_use_id": "adv_1", "content": "kept"}, + {"type": "text", "text": "answer"} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "advisor_tool_result", "tool_use_id": "other", "content": "kept"}, + {"type": "tool_result", "tool_use_id": "adv_1", "content": "kept"}, + {"type": "text", "text": "answer"} + ]}]) + )] + #[case::advisor_call_without_result( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "adv_1", "name": "advisor", "input": {}}, + {"type": "text", "text": "answer"} + ]}]), + json!([{"role": "assistant", "content": [{"type": "text", "text": "answer"}]}]) + )] + #[case::advisor_only_turn_keeps_an_empty_block_list( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "adv_1", "name": "advisor", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "adv_1", "content": "advice"} + ]}]), + json!([{"role": "assistant", "content": []}]) + )] + fn strip_advisor_blocks_rewrites(#[case] input: Value, #[case] expected: Value) { + assert_eq!(apply(strip_advisor_blocks, input), expected); + } + + #[rstest] + #[case::no_advisor_blocks(json!([ + {"role": "user", "content": "Hello"}, + {"role": "assistant", "content": [ + {"type": "text", "text": "Hi there"}, + {"type": "tool_use", "id": "toolu_abc", "name": "get_weather", "input": {"location": "SF"}} + ]} + ]))] + #[case::user_turn(json!([{"role": "user", "content": [ + {"type": "server_tool_use", "id": "adv_2", "name": "advisor", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "adv_2", "content": "advice"} + ]}]))] + #[case::other_server_tool(json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "s1", "content": "advice"} + ]}]))] + #[case::client_tool_named_advisor(json!([{"role": "assistant", "content": [ + {"type": "tool_use", "id": "t1", "name": "advisor", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "t1", "content": "advice"} + ]}]))] + #[case::advisor_call_with_empty_id(json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "", "name": "advisor", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "", "content": "advice"} + ]}]))] + #[case::advisor_call_without_id(json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "name": "advisor", "input": {}} + ]}]))] + #[case::string_content(json!([{"role": "assistant", "content": "advisor"}]))] + fn strip_advisor_blocks_leaves_untouched(#[case] input: Value) { + assert_eq!(apply(strip_advisor_blocks, input.clone()), input); + } + + #[rstest] + #[case::results_keep_their_evidence( + json!([ + {"role": "user", "content": "latest version?"}, + {"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {"query": "latest version"}}, + {"type": "web_search_tool_result", "tool_use_id": "srvtoolu_1", "content": [ + {"type": "web_search_result", "url": "https://example.com/releases", "title": "Releases", + "page_age": null, "encrypted_content": "", "snippet": "Latest release v1.95.0"} + ]}, + {"type": "text", "text": "v1.95.0"} + ]} + ]), + json!([ + {"role": "user", "content": "latest version?"}, + {"role": "assistant", "content": [ + {"type": "text", "text": "Web search results for 'latest version':\n\nTitle: Releases\nURL: https://example.com/releases\nSnippet: Latest release v1.95.0"}, + {"type": "text", "text": "v1.95.0"} + ]} + ]) + )] + #[case::each_result_lists_only_its_present_fields( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "q"}}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [ + {"type": "web_search_result", "title": "A"}, + {"type": "web_search_result", "snippet": "b"}, + {"type": "web_search_result", "url": "https://c"} + ]} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results for 'q':\n\nTitle: A\n\nSnippet: b\n\nURL: https://c"} + ]}]) + )] + #[case::result_without_fields_renders_the_header_only( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "q"}}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [{"type": "web_search_result"}]} + ]}]), + json!([{"role": "assistant", "content": [{"type": "text", "text": "Web search results for 'q':"}]}]) + )] + #[case::resultless_search( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {"query": "who won"}}, + {"type": "web_search_tool_result", "tool_use_id": "srvtoolu_1", "content": []}, + {"type": "text", "text": "I could not find that."} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results for 'who won':\n\nNo results were returned."}, + {"type": "text", "text": "I could not find that."} + ]}]) + )] + #[case::failed_search( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {"query": "q"}}, + {"type": "web_search_tool_result", "tool_use_id": "srvtoolu_1", + "content": {"type": "web_search_tool_result_error", "error_code": "max_uses_exceeded"}} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results for 'q':\n\nSearch failed: max_uses_exceeded"} + ]}]) + )] + #[case::failed_search_without_error_code( + json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "e1", "content": {"type": "web_search_tool_result_error"}} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results:\n\nSearch failed: unavailable"} + ]}]) + )] + #[case::server_tool_use_without_query( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search"}, + {"type": "server_tool_use", "id": "s2", "name": "web_search", "input": {}}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": []}, + {"type": "web_search_tool_result", "tool_use_id": "s2", "content": []} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results:\n\nNo results were returned."}, + {"type": "text", "text": "Web search results:\n\nNo results were returned."} + ]}]) + )] + #[case::genuine_results_in_the_same_turn_stay( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "rust"}}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [ + {"type": "web_search_result", "url": "https://r", "title": "Rust", "snippet": "fast"} + ]}, + {"type": "server_tool_use", "id": "s2", "name": "web_search", "input": {"query": "real"}}, + {"type": "web_search_tool_result", "tool_use_id": "s2", "content": [ + {"type": "web_search_result", "url": "https://a", "title": "A", "snippet": "b", "encrypted_content": "enc"} + ]}, + {"type": "text", "text": "done"} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results for 'rust':\n\nTitle: Rust\nURL: https://r\nSnippet: fast"}, + {"type": "server_tool_use", "id": "s2", "name": "web_search", "input": {"query": "real"}}, + {"type": "web_search_tool_result", "tool_use_id": "s2", "content": [ + {"type": "web_search_result", "url": "https://a", "title": "A", "snippet": "b", "encrypted_content": "enc"} + ]}, + {"type": "text", "text": "done"} + ]}]) + )] + #[case::other_blocks_sharing_the_tool_use_id_stay( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "q"}}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": []}, + {"type": "tool_result", "tool_use_id": "s1", "content": "x"} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results for 'q':\n\nNo results were returned."}, + {"type": "tool_result", "tool_use_id": "s1", "content": "x"} + ]}]) + )] + #[case::query_lookup_stays_within_the_message( + json!([ + {"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "q"}} + ]}, + {"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": []} + ]} + ]), + json!([ + {"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "q"}} + ]}, + {"role": "assistant", "content": [ + {"type": "text", "text": "Web search results:\n\nNo results were returned."} + ]} + ]) + )] + #[case::result_without_any_field_keeps_its_slot( + json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [ + {"type": "web_search_result", "url": "https://a", "title": "A"}, + {"type": "web_search_result"}, + {"type": "web_search_result", "title": "B"} + ]} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results:\n\nTitle: A\nURL: https://a\n\n\n\nTitle: B"} + ]}]) + )] + #[case::non_string_query_keeps_its_server_tool_use( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": 123}}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": []} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": 123}}, + {"type": "text", "text": "Web search results:\n\nNo results were returned."} + ]}]) + )] + #[case::non_object_input_keeps_its_server_tool_use( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": "q"}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": []} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": "q"}, + {"type": "text", "text": "Web search results:\n\nNo results were returned."} + ]}]) + )] + #[case::repeated_tool_use_id_renders_each_block_from_its_own_results( + json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": []}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": {"type": "web_search_tool_result_error", "error_code": "max_uses"}} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results:\n\nNo results were returned."}, + {"type": "text", "text": "Web search results:\n\nSearch failed: max_uses"} + ]}]) + )] + #[case::encrypted_block_sharing_a_replayed_id_stays( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "q"}}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": []}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [ + {"type": "web_search_result", "url": "https://a", "encrypted_content": "enc"} + ]} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results for 'q':\n\nNo results were returned."}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [ + {"type": "web_search_result", "url": "https://a", "encrypted_content": "enc"} + ]} + ]}]) + )] + #[case::last_query_wins_for_a_repeated_server_tool_use_id( + json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "first"}}, + {"type": "server_tool_use", "id": "s1", "name": "web_search", "input": {"query": "second"}}, + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": []} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "text", "text": "Web search results for 'second':\n\nNo results were returned."} + ]}]) + )] + fn flatten_unencrypted_web_search_results_rewrites( + #[case] input: Value, + #[case] expected: Value, + ) { + assert_eq!( + apply(flatten_unencrypted_web_search_results, input), + expected + ); + } + + #[rstest] + #[case::anthropic_issued_results(json!([{"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {"query": "q"}}, + {"type": "web_search_tool_result", "tool_use_id": "srvtoolu_1", "content": [ + {"type": "web_search_result", "url": "https://example.com", "title": "Example", + "page_age": null, "encrypted_content": "EqgfCioIARgBIiQ4"} + ]} + ]}]))] + #[case::any_encrypted_result_marks_the_block_genuine(json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [ + {"type": "web_search_result", "url": "https://a", "encrypted_content": ""}, + {"type": "web_search_result", "url": "https://b", "encrypted_content": "enc"} + ]} + ]}]))] + #[case::result_without_tool_use_id(json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "content": []} + ]}]))] + #[case::foreign_item_in_results(json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [ + {"type": "web_search_result", "url": "https://a"}, + {"type": "text", "text": "x"} + ]} + ]}]))] + #[case::result_with_null_url(json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": [ + {"type": "web_search_result", "url": null, "title": "A"} + ]} + ]}]))] + #[case::string_result_content(json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": "oops"} + ]}]))] + #[case::object_content_that_is_not_an_error(json!([{"role": "assistant", "content": [ + {"type": "web_search_tool_result", "tool_use_id": "s1", "content": {"type": "web_search_result", "url": "https://a"}} + ]}]))] + #[case::string_content(json!([{"role": "assistant", "content": "web_search_tool_result"}]))] + fn flatten_unencrypted_web_search_results_leaves_untouched(#[case] input: Value) { + assert_eq!( + apply(flatten_unencrypted_web_search_results, input.clone()), + input + ); + } + + #[rstest] + #[case::with_results(json!([{"type": "web_search_result", "url": "u", "title": "Rome", "snippet": "s", "page_age": null}]))] + #[case::without_results(json!([]))] + fn flatten_unencrypted_web_search_results_is_idempotent(#[case] results: Value) { + let input = replayed_search_turn(results); + let once = apply(flatten_unencrypted_web_search_results, input.clone()); + let twice = apply(flatten_unencrypted_web_search_results, once.clone()); + assert_ne!(once, input); + assert_eq!(twice, once); + } + + #[rstest] + #[case::no_existing_header(None, "b", "b")] + #[case::empty_existing_header(Some(""), "b", "b")] + #[case::whitespace_existing_header(Some(" "), "b", "b")] + #[case::sorted_after_merge(Some("c,a"), "b", "a,b,c")] + #[case::already_present(Some("a,b"), "a", "a,b")] + #[case::trimmed_and_deduplicated(Some("b, a ,b"), "c", "a,b,c")] + #[case::blank_pieces_skipped(Some("a,,b"), "c", "a,b,c")] + fn beta_values_merge_sorted_and_deduplicated( + #[case] existing: Option<&str>, + #[case] new_beta: &str, + #[case] expected: &str, + ) { + assert_eq!( + join_beta_values(split_beta_values(existing).chain([new_beta.to_string()])), + expected + ); + } + + #[rstest] + #[case::raw_token("sk-ant-oat01-abc123", true)] + #[case::bearer_token("Bearer sk-ant-oat02-xyz789", true)] + #[case::bare_prefix(ANTHROPIC_OAUTH_TOKEN_PREFIX, true)] + #[case::api_key("sk-ant-api01-abc123", false)] + #[case::bearer_api_key("Bearer sk-ant-api01-abc123", false)] + #[case::empty("", false)] + #[case::uppercase_prefix("sk-ant-OAT01-abc123", false)] + #[case::shouting_prefix("SK-ANT-OAT01-abc123", false)] + #[case::lowercase_bearer("bearer sk-ant-oat01-abc123", false)] + #[case::bearer_stripped_once("Bearer Bearer sk-ant-oat01-abc123", false)] + #[case::prefix_not_at_start(" sk-ant-oat01-abc123", false)] + fn anthropic_oauth_key_detection(#[case] value: &str, #[case] expected: bool) { + assert_eq!(is_anthropic_oauth_key(value), expected); + } + + #[rstest] + #[case::regex_tool(Some(json!([{"type": ANTHROPIC_TOOL_SEARCH_TOOL_TYPES[0], "name": "tool_search_tool_regex"}])), true)] + #[case::bm25_tool(Some(json!([{"type": ANTHROPIC_TOOL_SEARCH_TOOL_TYPES[1], "name": "tool_search_tool_bm25"}])), true)] + #[case::after_other_tools( + Some(json!([{"name": "get_weather", "input_schema": {}}, {"type": ANTHROPIC_TOOL_SEARCH_TOOL_TYPES[1]}])), + true + )] + #[case::function_tool(Some(json!([{"type": "function", "function": {"name": "get_weather"}}])), false)] + #[case::name_without_type(Some(json!([{"name": ANTHROPIC_TOOL_SEARCH_TOOL_TYPES[0]}])), false)] + #[case::empty_tools(Some(json!([])), false)] + #[case::no_tools(None, false)] + fn tool_search_detection(#[case] input: Option, #[case] expected: bool) { + assert_eq!(is_tool_search_used(tools(input).as_deref()), expected); + } + + #[rstest] + #[case::advisor_tool(Some(json!([{"type": ANTHROPIC_ADVISOR_TOOL_TYPE, "name": "advisor"}])), true)] + #[case::after_other_tools(Some(json!([{"name": "f", "input_schema": {}}, {"type": ANTHROPIC_ADVISOR_TOOL_TYPE}])), true)] + #[case::tool_named_advisor(Some(json!([{"name": "advisor", "input_schema": {}}])), false)] + #[case::other_server_tool(Some(json!([{"type": "web_search_20250305", "name": "web_search"}])), false)] + #[case::empty_tools(Some(json!([])), false)] + #[case::no_tools(None, false)] + fn advisor_tool_detection(#[case] input: Option, #[case] expected: bool) { + assert_eq!(has_advisor_tool(tools(input).as_deref()), expected); + } + + #[rstest] + #[case::param_without_history(Some(json!({})), json!([]), true)] + #[case::param_with_unsigned_history( + Some(json!({"trigger": 1})), + json!([{"role": "assistant", "content": [{"type": "compaction", "content": "c"}]}]), + true + )] + #[case::signed_block(None, json!([{"role": "assistant", "content": [{"type": "compaction", "content": "c", "signature": "s"}]}]), true)] + #[case::signed_block_later_in_history( + None, + json!([ + {"role": "user", "content": "hi"}, + {"role": "assistant", "content": [{"type": "text", "text": "a"}, {"type": "compaction", "content": "c", "signature": "s"}]} + ]), + true + )] + #[case::unsigned_block(None, json!([{"role": "assistant", "content": [{"type": "compaction", "content": "c"}]}]), false)] + #[case::empty_signature(None, json!([{"role": "assistant", "content": [{"type": "compaction", "content": "c", "signature": ""}]}]), false)] + #[case::signed_non_compaction_block( + None, + json!([{"role": "assistant", "content": [{"type": "thinking", "thinking": "t", "signature": "s"}]}]), + false + )] + #[case::string_content(None, json!([{"role": "user", "content": "compaction"}]), false)] + #[case::neither(None, json!([]), false)] + fn native_compaction_beta_requirement( + #[case] compaction: Option, + #[case] messages: Value, + #[case] expected: bool, + ) { + assert_eq!( + requires_native_compaction_beta(compaction.as_ref(), &history(messages)), + expected + ); + } + + #[rstest] + #[case::low(EffortLevel::Low, "low")] + #[case::medium(EffortLevel::Medium, "medium")] + #[case::high(EffortLevel::High, "high")] + #[case::xhigh(EffortLevel::Xhigh, "xhigh")] + #[case::max(EffortLevel::Max, "max")] + fn effort_level_names_agree_across_str_parse_and_serde( + #[case] level: EffortLevel, + #[case] name: &str, + ) { + assert_eq!(level.as_str(), name); + assert_eq!(EffortLevel::parse(name), Some(level)); + assert_eq!(serde_json::to_value(level).unwrap(), json!(name)); + assert_eq!( + serde_json::from_value::(json!(name)).unwrap(), + level + ); + } + + #[rstest] + #[case::unknown("ultra")] + #[case::minimal_is_not_an_output_config_level("minimal")] + #[case::uppercase("HIGH")] + #[case::empty("")] + fn effort_level_parse_rejects(#[case] value: &str) { + assert_eq!(EffortLevel::parse(value), None); + } + + #[rstest] + #[case::minimal_only(tiers(true, false, false, false, false, false), [false, false, false, false, false])] + #[case::low_only(tiers(false, true, false, false, false, false), [true, false, false, false, false])] + #[case::medium_only(tiers(false, false, true, false, false, false), [false, true, false, false, false])] + #[case::high_only(tiers(false, false, false, true, false, false), [false, false, true, false, false])] + #[case::xhigh_only(tiers(false, false, false, false, true, false), [false, false, false, true, false])] + #[case::max_only(tiers(false, false, false, false, false, true), [false, false, false, false, true])] + fn supports_effort_tier_reads_the_matching_flag( + #[case] effort_tiers: SupportedEffortTiers, + #[case] expected: [bool; 5], + unmapped: AnthropicModelCapabilities, + ) { + let capabilities = AnthropicModelCapabilities { + effort_tiers, + ..unmapped + }; + assert_eq!( + ALL_LEVELS.map(|level| capabilities.supports_effort_tier(level)), + expected + ); + } + + #[rstest] + #[case::unmapped(false, false, false, SupportedEffortTiers::default(), false)] + #[case::reasoning_and_adaptive_thinking_alone( + true, + true, + false, + SupportedEffortTiers::default(), + false + )] + #[case::output_config_without_tiers(false, false, true, SupportedEffortTiers::default(), true)] + #[case::minimal_tier( + false, + false, + false, + tiers(true, false, false, false, false, false), + true + )] + #[case::low_tier( + false, + false, + false, + tiers(false, true, false, false, false, false), + true + )] + #[case::medium_tier( + false, + false, + false, + tiers(false, false, true, false, false, false), + true + )] + #[case::high_tier( + false, + false, + false, + tiers(false, false, false, true, false, false), + true + )] + #[case::xhigh_tier( + false, + false, + false, + tiers(false, false, false, false, true, false), + true + )] + #[case::max_tier( + false, + false, + false, + tiers(false, false, false, false, false, true), + true + )] + fn supports_effort_param_cases( + #[case] supports_reasoning: bool, + #[case] supports_adaptive_thinking: bool, + #[case] supports_output_config: bool, + #[case] effort_tiers: SupportedEffortTiers, + #[case] expected: bool, + unmapped: AnthropicModelCapabilities, + ) { + let capabilities = AnthropicModelCapabilities { + supports_reasoning, + supports_adaptive_thinking, + supports_output_config, + effort_tiers, + ..unmapped + }; + assert_eq!(capabilities.supports_effort_param(), expected); + } + + #[rstest] + #[case::max_on_adaptive_thinking_model(true, SupportedEffortTiers::default(), "max", None)] + #[case::max_on_max_tier_model( + false, + tiers(false, false, false, false, false, true), + "max", + None + )] + #[case::max_on_output_config_only_model( + false, + SupportedEffortTiers::default(), + "max", + Some("effort='max' is not supported by this model. Got model: claude-test") + )] + #[case::max_on_xhigh_tier_model( + false, + tiers(false, false, false, false, true, false), + "max", + Some("effort='max' is not supported by this model. Got model: claude-test") + )] + #[case::xhigh_on_xhigh_tier_model( + false, + tiers(false, false, false, false, true, false), + "xhigh", + None + )] + #[case::xhigh_on_adaptive_thinking_model( + true, + SupportedEffortTiers::default(), + "xhigh", + Some("effort='xhigh' is not supported by this model. Got model: claude-test") + )] + #[case::xhigh_on_max_tier_model( + false, + tiers(false, false, false, false, false, true), + "xhigh", + Some("effort='xhigh' is not supported by this model. Got model: claude-test") + )] + #[case::high_on_unmapped_model(false, SupportedEffortTiers::default(), "high", None)] + #[case::low_on_unmapped_model(false, SupportedEffortTiers::default(), "low", None)] + #[case::unknown_level_is_left_to_other_validation( + false, + SupportedEffortTiers::default(), + "ultra", + None + )] + fn effort_level_rejection_cases( + #[case] supports_adaptive_thinking: bool, + #[case] effort_tiers: SupportedEffortTiers, + #[case] effort: &str, + #[case] expected: Option<&str>, + unmapped: AnthropicModelCapabilities, + ) { + let capabilities = AnthropicModelCapabilities { + supports_output_config: true, + supports_adaptive_thinking, + effort_tiers, + ..unmapped + }; + assert_eq!( + capabilities + .effort_level_rejection(effort, "claude-test") + .as_deref(), + expected + ); + } + + #[rstest] + fn unmapped_model_has_no_reasoning_features_but_accepts_sampling_params( + unmapped: AnthropicModelCapabilities, + ) { + assert_eq!( + unmapped, + AnthropicModelCapabilities { + supports_reasoning: false, + supports_adaptive_thinking: false, + thinking_always_on: false, + supports_legacy_thinking: false, + supports_output_config: false, + supports_sampling_params: true, + supports_speed: false, + effort_tiers: tiers(false, false, false, false, false, false), + } + ); + assert_eq!( + serde_json::from_value::(json!({})).unwrap(), + unmapped + ); + } + + #[rstest] + #[case::sampling_params_removed( + json!({"supports_sampling_params": false}), + AnthropicModelCapabilities { supports_sampling_params: false, ..AnthropicModelCapabilities::default() } + )] + #[case::fast_mode( + json!({"supports_speed": true}), + AnthropicModelCapabilities { supports_speed: true, ..AnthropicModelCapabilities::default() } + )] + #[case::partial_effort_tiers( + json!({"supports_reasoning": true, "effort_tiers": {"xhigh": true}}), + AnthropicModelCapabilities { + supports_reasoning: true, + effort_tiers: tiers(false, false, false, false, true, false), + ..AnthropicModelCapabilities::default() + } + )] + fn capabilities_fill_missing_flags_with_unmapped_defaults( + #[case] input: Value, + #[case] expected: AnthropicModelCapabilities, + ) { + assert_eq!( + serde_json::from_value::(input).unwrap(), + expected + ); + } +} diff --git a/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/handler.rs b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/handler.rs new file mode 100644 index 00000000000..0e2ab97956a --- /dev/null +++ b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/handler.rs @@ -0,0 +1,270 @@ +use litellm_types::llms::anthropic_messages::anthropic_request::{ + AnthropicMessage, AnthropicMessagesRequest, +}; +use serde_json::{Value, json}; + +use crate::{ + anthropic::common_utils::{ + flatten_unencrypted_web_search_results, sanitize_tool_use_ids, strip_empty_content_blocks, + strip_provider_specific_fields, + }, + base_llm::chat::transformation::Error, +}; + +pub fn shape_anthropic_messages_request( + request: AnthropicMessagesRequest, + reasoning_auto_summary: bool, +) -> Result { + Ok(AnthropicMessagesRequest { + messages: sanitize_anthropic_messages(request.messages), + metadata: request + .metadata + .as_ref() + .map(validate_anthropic_api_metadata) + .transpose()?, + thinking: with_reasoning_auto_summary(request.thinking, reasoning_auto_summary), + ..request + }) +} + +fn sanitize_anthropic_messages(messages: Vec) -> Vec { + strip_provider_specific_fields(flatten_unencrypted_web_search_results( + sanitize_tool_use_ids(strip_empty_content_blocks(messages)), + )) +} + +fn validate_anthropic_api_metadata(metadata: &Value) -> Result { + let Value::Object(fields) = metadata else { + return Err(Error::InvalidRequest(format!( + "metadata must be an object, got {metadata}" + ))); + }; + match fields.get("user_id") { + None | Some(Value::Null) => Ok(json!({})), + Some(Value::String(user_id)) => Ok(json!({"user_id": user_id})), + Some(other) => Err(Error::InvalidRequest(format!( + "metadata.user_id must be a string, got {other}" + ))), + } +} + +fn with_reasoning_auto_summary(thinking: Option, enabled: bool) -> Option { + let Some(Value::Object(thinking)) = thinking else { + return thinking; + }; + if !enabled || thinking.get("type").and_then(Value::as_str) == Some("disabled") { + return Some(Value::Object(thinking)); + } + Some(Value::Object( + thinking + .into_iter() + .filter(|(key, _)| key != "display") + .chain([("display".to_string(), json!("summarized"))]) + .collect(), + )) +} + +#[cfg(test)] +mod tests { + use rstest::rstest; + + use super::*; + + fn messages(value: Value) -> Vec { + serde_json::from_value(value).unwrap() + } + + fn request(body: Value) -> AnthropicMessagesRequest { + serde_json::from_value(body).unwrap() + } + + #[rstest] + #[case::empty_text_next_to_a_tool_use( + json!([{"role": "assistant", "content": [ + {"type": "text", "text": " "}, + {"type": "tool_use", "id": "t", "name": "B", "input": {}} + ]}]), + json!([{"role": "assistant", "content": [ + {"type": "tool_use", "id": "t", "name": "B", "input": {}} + ]}]), + )] + #[case::cross_provider_tool_ids( + json!([ + {"role": "assistant", "content": [{"type": "tool_use", "id": "functions.Bash:0", "name": "Bash", "input": {}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions.Bash:0", "content": "ok"}]} + ]), + json!([ + {"role": "assistant", "content": [{"type": "tool_use", "id": "functions_Bash_0", "name": "Bash", "input": {}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions_Bash_0", "content": "ok"}]} + ]), + )] + #[case::replayed_unencrypted_web_search_results( + json!([ + {"role": "user", "content": "latest litellm version?"}, + {"role": "assistant", "content": [ + {"type": "server_tool_use", "id": "srvtoolu_1", "name": "web_search", "input": {"query": "latest litellm version"}}, + {"type": "web_search_tool_result", "tool_use_id": "srvtoolu_1", "content": [{ + "type": "web_search_result", + "url": "https://github.com/BerriAI/litellm/releases", + "title": "Releases", + "page_age": null, + "encrypted_content": "", + "snippet": "Latest release v1.95.0" + }]} + ]}, + {"role": "user", "content": "which version?"} + ]), + json!([ + {"role": "user", "content": "latest litellm version?"}, + {"role": "assistant", "content": [{ + "type": "text", + "text": "Web search results for 'latest litellm version':\n\nTitle: Releases\nURL: https://github.com/BerriAI/litellm/releases\nSnippet: Latest release v1.95.0" + }]}, + {"role": "user", "content": "which version?"} + ]), + )] + #[case::replayed_provider_specific_fields( + json!([ + {"role": "assistant", "content": [{ + "type": "tool_use", "id": "toolu_01", "name": "get_weather", "input": {"city": "Paris"}, + "provider_specific_fields": {"signature": "sig_abc"} + }]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "toolu_01", "content": "Sunny"}]} + ]), + json!([ + {"role": "assistant", "content": [{"type": "tool_use", "id": "toolu_01", "name": "get_weather", "input": {"city": "Paris"}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "toolu_01", "content": "Sunny"}]} + ]), + )] + #[case::ids_are_normalized_before_web_search_results_flatten( + json!([ + {"role": "user", "content": "run it"}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "", "signature": "sig"}, + {"type": "text", "text": ""}, + {"type": "tool_use", "id": "functions.Bash:0", "name": "Bash", "input": {}, "provider_specific_fields": {"x": 1}}, + {"type": "server_tool_use", "id": "srv.1", "name": "web_search", "input": {"query": "q"}, "provider_specific_fields": {"x": 2}}, + {"type": "web_search_tool_result", "tool_use_id": "srv.1", "provider_specific_fields": {"x": 3}, "content": [ + {"type": "web_search_result", "url": "u", "title": "", "encrypted_content": "", "provider_specific_fields": {"x": 4}} + ]} + ]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions.Bash:0", "content": "ok"}]}, + {"role": "assistant", "content": [{"type": "text", "text": " "}]} + ]), + json!([ + {"role": "user", "content": "run it"}, + {"role": "assistant", "content": [ + {"type": "tool_use", "id": "functions_Bash_0", "name": "Bash", "input": {}}, + {"type": "server_tool_use", "id": "srv_1", "name": "web_search", "input": {"query": "q"}}, + {"type": "text", "text": "Web search results:\n\nURL: u"} + ]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions_Bash_0", "content": "ok"}]} + ]), + )] + fn sanitize_anthropic_messages_cleans_replayed_history( + #[case] history: Value, + #[case] expected: Value, + ) { + assert_eq!( + serde_json::to_value(sanitize_anthropic_messages(messages(history))).unwrap(), + expected + ); + } + + #[rstest] + #[case::keeps_only_user_id(json!({"user_id": "u-1", "trace_id": "internal"}), Ok(json!({"user_id": "u-1"})))] + #[case::null_user_id(json!({"user_id": null, "trace_id": "internal"}), Ok(json!({})))] + #[case::no_user_id(json!({"trace_id": "internal"}), Ok(json!({})))] + #[case::empty(json!({}), Ok(json!({})))] + #[case::numeric_user_id( + json!({"user_id": 123}), + Err(Error::InvalidRequest("metadata.user_id must be a string, got 123".to_string())), + )] + #[case::boolean_user_id( + json!({"user_id": true}), + Err(Error::InvalidRequest("metadata.user_id must be a string, got true".to_string())), + )] + #[case::not_an_object( + json!(["u-1"]), + Err(Error::InvalidRequest(r#"metadata must be an object, got ["u-1"]"#.to_string())), + )] + fn validate_anthropic_api_metadata_passes_only_a_string_user_id( + #[case] metadata: Value, + #[case] expected: Result, + ) { + assert_eq!(validate_anthropic_api_metadata(&metadata), expected); + } + + #[rstest] + #[case::adaptive( + Some(json!({"type": "adaptive", "budget_tokens": 5000})), + true, + Some(json!({"type": "adaptive", "budget_tokens": 5000, "display": "summarized"})), + )] + #[case::enabled( + Some(json!({"type": "enabled", "budget_tokens": 10000})), + true, + Some(json!({"type": "enabled", "budget_tokens": 10000, "display": "summarized"})), + )] + #[case::no_type(Some(json!({})), true, Some(json!({"display": "summarized"})))] + #[case::display_omitted_is_overridden( + Some(json!({"type": "enabled", "budget_tokens": 10000, "display": "omitted"})), + true, + Some(json!({"type": "enabled", "budget_tokens": 10000, "display": "summarized"})), + )] + #[case::display_summarized_is_kept( + Some(json!({"type": "enabled", "display": "summarized"})), + true, + Some(json!({"type": "enabled", "display": "summarized"})), + )] + #[case::disabled_thinking(Some(json!({"type": "disabled"})), true, Some(json!({"type": "disabled"})))] + #[case::flag_off( + Some(json!({"type": "enabled", "budget_tokens": 10000})), + false, + Some(json!({"type": "enabled", "budget_tokens": 10000})), + )] + #[case::flag_off_keeps_callers_display( + Some(json!({"type": "enabled", "display": "omitted"})), + false, + Some(json!({"type": "enabled", "display": "omitted"})), + )] + #[case::no_thinking(None, true, None)] + #[case::non_object_thinking(Some(json!("enabled")), true, Some(json!("enabled")))] + fn reasoning_auto_summary_marks_active_thinking_as_summarized( + #[case] thinking: Option, + #[case] enabled: bool, + #[case] expected: Option, + ) { + assert_eq!(with_reasoning_auto_summary(thinking, enabled), expected); + } + + #[test] + fn shaping_cleans_messages_metadata_and_thinking() { + let sanitized = shape_anthropic_messages_request( + request(json!({ + "model": "m", + "messages": [{"role": "assistant", "content": [ + {"type": "text", "text": ""}, + {"type": "tool_use", "id": "functions.Bash:0", "name": "Bash", "input": {}} + ]}], + "metadata": {"user_id": "u", "trace_id": "t"}, + "thinking": {"type": "enabled", "budget_tokens": 1024}, + "safeguards": [{"type": "dangerous_tool_use"}] + })), + true, + ) + .unwrap(); + assert_eq!( + serde_json::to_value(sanitized).unwrap(), + json!({ + "model": "m", + "messages": [{"role": "assistant", "content": [ + {"type": "tool_use", "id": "functions_Bash_0", "name": "Bash", "input": {}} + ]}], + "metadata": {"user_id": "u"}, + "thinking": {"type": "enabled", "budget_tokens": 1024, "display": "summarized"}, + "safeguards": [{"type": "dangerous_tool_use"}] + }) + ); + } +} diff --git a/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/headers.rs b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/headers.rs new file mode 100644 index 00000000000..8d48d7a0f5c --- /dev/null +++ b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/headers.rs @@ -0,0 +1,643 @@ +use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest; +use serde_json::Value; + +use crate::{ + anthropic::{ + ANTHROPIC_OAUTH_TOKEN_PREFIX, + common_utils::{ + ANTHROPIC_OAUTH_BETA_HEADER, beta, has_advisor_tool, is_anthropic_oauth_key, + is_tool_search_used, join_beta_values, requires_native_compaction_beta, + split_beta_values, + }, + }, + base_llm::anthropic_messages::transformation::Headers, +}; + +const ANTHROPIC_API_KEY_ENV: &str = "ANTHROPIC_API_KEY"; +const ANTHROPIC_AUTH_TOKEN_ENV: &str = "ANTHROPIC_AUTH_TOKEN"; +const BETA_HEADER: &str = "anthropic-beta"; +const AUTHORIZATION: &str = "authorization"; +const API_KEY_HEADER: &str = "x-api-key"; +const DIRECT_BROWSER_ACCESS_HEADER: &str = "anthropic-dangerous-direct-browser-access"; + +fn header_value<'a>(headers: &'a [(String, String)], name: &str) -> Option<&'a str> { + headers + .iter() + .find(|(header, _)| header.eq_ignore_ascii_case(name)) + .map(|(_, value)| value.as_str()) +} + +fn without(headers: Headers, names: &[&str]) -> Headers { + headers + .into_iter() + .filter(|(header, _)| !names.iter().any(|name| header.eq_ignore_ascii_case(name))) + .collect() +} + +fn existing_betas(headers: &[(String, String)]) -> impl Iterator + '_ { + headers + .iter() + .filter(|(header, _)| header.eq_ignore_ascii_case(BETA_HEADER)) + .flat_map(|(_, value)| split_beta_values(Some(value))) +} + +fn with_oauth_bearer(headers: Headers, bearer: String) -> Headers { + let beta = + join_beta_values(existing_betas(&headers).chain([ANTHROPIC_OAUTH_BETA_HEADER.to_string()])); + without(headers, &[API_KEY_HEADER, AUTHORIZATION, BETA_HEADER]) + .into_iter() + .chain([ + (AUTHORIZATION.to_string(), bearer), + (BETA_HEADER.to_string(), beta), + (DIRECT_BROWSER_ACCESS_HEADER.to_string(), "true".to_string()), + ]) + .collect() +} + +fn non_empty(value: Option<&str>) -> Option<&str> { + value.map(str::trim).filter(|value| !value.is_empty()) +} + +pub fn authenticate( + headers: Headers, + api_key: Option<&str>, + env_lookup: &dyn Fn(&str) -> Option, +) -> Result { + if let Some(forwarded) = header_value(&headers, AUTHORIZATION) + && forwarded + .strip_prefix("Bearer ") + .is_some_and(|token| token.starts_with(ANTHROPIC_OAUTH_TOKEN_PREFIX)) + { + let bearer = forwarded.to_string(); + return Ok(with_oauth_bearer(headers, bearer)); + } + if let Some(key) = api_key.filter(|key| key.starts_with(ANTHROPIC_OAUTH_TOKEN_PREFIX)) { + return Ok(with_oauth_bearer(headers, format!("Bearer {key}"))); + } + if header_value(&headers, API_KEY_HEADER).is_some() + || header_value(&headers, AUTHORIZATION).is_some() + { + return Ok(headers); + } + let resolved_key = non_empty(api_key) + .map(str::to_string) + .or_else(|| env_lookup(ANTHROPIC_API_KEY_ENV).filter(|value| !value.trim().is_empty())); + let auth = match resolved_key { + Some(key) if is_anthropic_oauth_key(&key) => { + (AUTHORIZATION.to_string(), format!("Bearer {key}")) + } + Some(key) => (API_KEY_HEADER.to_string(), key), + None => match env_lookup(ANTHROPIC_AUTH_TOKEN_ENV).filter(|value| !value.trim().is_empty()) + { + Some(token) => (AUTHORIZATION.to_string(), format!("Bearer {token}")), + None => { + return Err(litellm_auth::Error::MissingApiKey { + provider: "Anthropic", + environment_variable: ANTHROPIC_API_KEY_ENV, + }); + } + }, + }; + Ok(headers.into_iter().chain([auth]).collect()) +} + +fn context_management_betas( + context_management: Option<&Value>, +) -> impl Iterator { + let edits = context_management + .and_then(|value| value.get("edits")) + .and_then(Value::as_array) + .map(Vec::as_slice) + .unwrap_or(&[]); + let (compact, other) = edits.iter().fold((false, false), |(compact, other), edit| { + match edit.get("type").and_then(Value::as_str) { + Some("compact_20260112") => (true, other), + _ => (compact, true), + } + }); + compact + .then_some(beta::COMPACT_2026_01_12) + .into_iter() + .chain(other.then_some(beta::CONTEXT_MANAGEMENT_2025_06_27)) +} + +fn uses_structured_output(request: &AnthropicMessagesRequest) -> bool { + request.output_format.is_some() + || request + .output_config + .as_ref() + .and_then(|config| config.get("format")) + .is_some_and(|format| !format.is_null()) +} + +fn messages_carry_output_config(request: &AnthropicMessagesRequest) -> bool { + request + .messages + .iter() + .any(|message| message.extra.contains_key("output_config")) +} + +pub fn feature_betas(request: &AnthropicMessagesRequest) -> Vec<&'static str> { + let tools = request.tools.as_deref(); + [ + requires_native_compaction_beta(request.compaction.as_ref(), &request.messages) + .then_some(beta::COMPACT_2026_09_04), + uses_structured_output(request).then_some(beta::STRUCTURED_OUTPUT), + (request.speed.as_deref() == Some("fast")).then_some(beta::FAST_MODE_2026_02_01), + messages_carry_output_config(request).then_some(beta::PER_TURN_CONTROL_2026_07_01), + has_advisor_tool(tools).then_some(beta::ADVISOR_TOOL_2026_03_01), + is_tool_search_used(tools).then_some(beta::ADVANCED_TOOL_USE_2025_11_20), + ] + .into_iter() + .flatten() + .chain(context_management_betas( + request.context_management.as_ref(), + )) + .collect() +} + +pub fn with_feature_betas(headers: Headers, request: &AnthropicMessagesRequest) -> Headers { + let existing = existing_betas(&headers).collect::>(); + let features = feature_betas(request); + if existing.is_empty() && features.is_empty() { + return headers; + } + let merged = join_beta_values( + existing + .into_iter() + .chain(features.into_iter().map(str::to_string)), + ); + without(headers, &[BETA_HEADER]) + .into_iter() + .chain([(BETA_HEADER.to_string(), merged)]) + .collect() +} + +#[cfg(test)] +mod tests { + use rstest::{fixture, rstest}; + use serde_json::json; + + use super::*; + + const OAUTH_TOKEN: &str = "sk-ant-oat01-token"; + const OAUTH_BEARER: &str = "Bearer sk-ant-oat01-token"; + const REGULAR_KEY: &str = "sk-ant-api03-regular"; + const BROWSER_ACCESS: (&str, &str) = ("anthropic-dangerous-direct-browser-access", "true"); + + type Env = &'static [(&'static str, &'static str)]; + + fn request(fields: Value) -> AnthropicMessagesRequest { + let mut body = + json!({"model": "claude", "messages": [{"role": "user", "content": "Hello"}]}); + body.as_object_mut() + .unwrap() + .extend(fields.as_object().unwrap().clone()); + serde_json::from_value(body).unwrap() + } + + fn headers(pairs: &[(&str, &str)]) -> Headers { + pairs + .iter() + .map(|(name, value)| (name.to_string(), value.to_string())) + .collect() + } + + fn betas(values: &[&str]) -> String { + values.join(",") + } + + #[fixture] + fn no_env() -> Env { + &[] + } + + #[fixture] + fn full_env() -> Env { + &[ + ("ANTHROPIC_API_KEY", "sk-env"), + ("ANTHROPIC_AUTH_TOKEN", "env-token"), + ] + } + + fn authenticate_with( + forwarded: &[(&str, &str)], + api_key: Option<&str>, + env: Env, + ) -> Result { + let lookup = |name: &str| { + env.iter() + .find(|(key, _)| *key == name) + .map(|(_, value)| value.to_string()) + }; + authenticate(headers(forwarded), api_key, &lookup) + } + + #[rstest] + #[case::forwarded_bearer_drops_forwarded_and_deployment_keys( + &[("X-Api-Key", REGULAR_KEY), ("Authorization", OAUTH_BEARER)], + Some(REGULAR_KEY), + OAUTH_BEARER, + &[], + )] + #[case::forwarded_bearer_in_uppercase_authorization_header( + &[("AUTHORIZATION", OAUTH_BEARER)], + None, + OAUTH_BEARER, + &[], + )] + #[case::forwarded_bearer_keeps_unrelated_headers_in_place( + &[("anthropic-version", "2023-06-01"), ("authorization", OAUTH_BEARER)], + None, + OAUTH_BEARER, + &[("anthropic-version", "2023-06-01")], + )] + #[case::forwarded_bearer_wins_over_an_oauth_api_key( + &[("authorization", OAUTH_BEARER)], + Some("sk-ant-oat01-deployment"), + OAUTH_BEARER, + &[], + )] + #[case::api_key_authenticates_as_a_bearer(&[], Some(OAUTH_TOKEN), OAUTH_BEARER, &[])] + #[case::api_key_removes_a_forwarded_x_api_key( + &[("x-api-key", OAUTH_TOKEN)], + Some(OAUTH_TOKEN), + OAUTH_BEARER, + &[], + )] + #[case::api_key_replaces_a_forwarded_non_oauth_bearer( + &[("Authorization", "Bearer some-proxy-token")], + Some(OAUTH_TOKEN), + OAUTH_BEARER, + &[], + )] + fn oauth_token_is_the_whole_credential( + #[case] forwarded: &[(&str, &str)], + #[case] api_key: Option<&str>, + #[case] expected_bearer: &str, + #[case] kept: &[(&str, &str)], + full_env: Env, + ) { + let expected = kept + .iter() + .copied() + .chain([ + ("authorization", expected_bearer), + ("anthropic-beta", ANTHROPIC_OAUTH_BETA_HEADER), + BROWSER_ACCESS, + ]) + .collect::>(); + assert_eq!( + authenticate_with(forwarded, api_key, full_env).unwrap(), + headers(&expected) + ); + } + + #[rstest] + #[case::forwarded_bearer_merges_a_differently_cased_beta_header( + &[("Anthropic-Beta", "web-search-2025-03-05"), ("authorization", OAUTH_BEARER)], + None, + )] + #[case::forwarded_bearer_dedupes_an_existing_oauth_beta( + &[("anthropic-beta", "web-search-2025-03-05, oauth-2025-04-20"), ("authorization", OAUTH_BEARER)], + None, + )] + #[case::api_key_merges_the_existing_beta_header( + &[("anthropic-beta", " web-search-2025-03-05 ,")], + Some(OAUTH_TOKEN), + )] + #[case::forwarded_bearer_unions_every_beta_header_casing( + &[("anthropic-beta", "oauth-2025-04-20"), ("ANTHROPIC-BETA", "web-search-2025-03-05"), ("authorization", OAUTH_BEARER)], + None, + )] + fn oauth_beta_merges_into_existing_betas( + #[case] forwarded: &[(&str, &str)], + #[case] api_key: Option<&str>, + no_env: Env, + ) { + assert_eq!( + authenticate_with(forwarded, api_key, no_env).unwrap(), + headers(&[ + ("authorization", OAUTH_BEARER), + ( + "anthropic-beta", + &betas(&[ANTHROPIC_OAUTH_BETA_HEADER, "web-search-2025-03-05"]) + ), + BROWSER_ACCESS, + ]) + ); + } + + #[rstest] + #[case::x_api_key_over_the_deployment_key(&[("x-api-key", "caller-key")], Some("sk-other"))] + #[case::uppercase_x_api_key(&[("X-API-KEY", "caller-key")], None)] + #[case::non_oauth_bearer(&[("Authorization", "Bearer some-proxy-token")], None)] + #[case::non_oauth_bearer_over_a_regular_api_key( + &[("authorization", "Bearer sk-ant-api03-forwarded")], + Some(REGULAR_KEY), + )] + #[case::oauth_token_without_the_bearer_scheme(&[("authorization", OAUTH_TOKEN)], None)] + #[case::oauth_token_behind_a_lowercase_bearer_scheme( + &[("authorization", "bearer sk-ant-oat01-token")], + None, + )] + fn forwarded_auth_header_is_kept_untouched( + #[case] forwarded: &[(&str, &str)], + #[case] api_key: Option<&str>, + full_env: Env, + ) { + assert_eq!( + authenticate_with(forwarded, api_key, full_env).unwrap(), + headers(forwarded) + ); + } + + #[rstest] + #[case::api_key_param(Some("sk-param"), &[], ("x-api-key", "sk-param"))] + #[case::api_key_param_over_env_key_and_auth_token( + Some("sk-param"), + &[("ANTHROPIC_API_KEY", "sk-env"), ("ANTHROPIC_AUTH_TOKEN", "env-token")], + ("x-api-key", "sk-param"), + )] + #[case::env_key_without_a_param(None, &[("ANTHROPIC_API_KEY", "sk-env")], ("x-api-key", "sk-env"))] + #[case::env_key_when_the_param_is_empty(Some(""), &[("ANTHROPIC_API_KEY", "sk-env")], ("x-api-key", "sk-env"))] + #[case::env_key_when_the_param_is_whitespace( + Some(" "), + &[("ANTHROPIC_API_KEY", "sk-env")], + ("x-api-key", "sk-env"), + )] + #[case::env_key_over_auth_token( + None, + &[("ANTHROPIC_API_KEY", "sk-env"), ("ANTHROPIC_AUTH_TOKEN", "env-token")], + ("x-api-key", "sk-env"), + )] + #[case::auth_token_as_a_bearer( + None, + &[("ANTHROPIC_AUTH_TOKEN", "env-token")], + ("authorization", "Bearer env-token"), + )] + #[case::auth_token_when_the_env_key_is_whitespace( + None, + &[("ANTHROPIC_API_KEY", " \t"), ("ANTHROPIC_AUTH_TOKEN", "env-token")], + ("authorization", "Bearer env-token"), + )] + #[case::oauth_env_key_as_a_plain_bearer( + None, + &[("ANTHROPIC_API_KEY", "sk-ant-oat01-env")], + ("authorization", "Bearer sk-ant-oat01-env"), + )] + fn credential_is_resolved_after_the_existing_headers( + #[case] api_key: Option<&str>, + #[case] env: Env, + #[case] expected: (&str, &str), + ) { + let forwarded = [("anthropic-beta", "web-search-2025-03-05")]; + assert_eq!( + authenticate_with(&forwarded, api_key, env).unwrap(), + headers(&[forwarded[0], expected]) + ); + } + + #[rstest] + #[case::no_credentials(&[], None, &[])] + #[case::empty_api_key(&[], Some(""), &[])] + #[case::whitespace_only_env_values( + &[], + None, + &[("ANTHROPIC_API_KEY", " "), ("ANTHROPIC_AUTH_TOKEN", " \t")], + )] + #[case::unrelated_forwarded_headers(&[("anthropic-beta", "web-search-2025-03-05")], None, &[])] + fn missing_credentials_are_an_auth_error( + #[case] forwarded: &[(&str, &str)], + #[case] api_key: Option<&str>, + #[case] env: Env, + ) { + assert!(matches!( + authenticate_with(forwarded, api_key, env), + Err(litellm_auth::Error::MissingApiKey { + provider: "Anthropic", + environment_variable: "ANTHROPIC_API_KEY", + }) + )); + } + + #[rstest] + #[case::no_features(json!({}), &[])] + #[case::output_format(json!({"output_format": {"type": "json_schema"}}), &[beta::STRUCTURED_OUTPUT])] + #[case::null_output_format(json!({"output_format": null}), &[])] + #[case::output_config_format( + json!({"output_config": {"format": {"type": "json_schema"}, "effort": "xhigh"}}), + &[beta::STRUCTURED_OUTPUT] + )] + #[case::null_output_config_format(json!({"output_config": {"format": null}}), &[])] + #[case::top_level_output_config_without_format(json!({"output_config": {"effort": "high"}}), &[])] + #[case::fast_speed(json!({"speed": "fast"}), &[beta::FAST_MODE_2026_02_01])] + #[case::standard_speed(json!({"speed": "standard"}), &[])] + #[case::compaction_param(json!({"compaction": {"enabled": true}}), &[beta::COMPACT_2026_09_04])] + #[case::empty_compaction_param(json!({"compaction": {}}), &[beta::COMPACT_2026_09_04])] + #[case::signed_compaction_block_in_history( + json!({"messages": [ + {"role": "assistant", "content": [{"type": "compaction", "content": "summary", "signature": "sig"}]}, + {"role": "user", "content": "Continue"}, + ]}), + &[beta::COMPACT_2026_09_04] + )] + #[case::unsigned_compaction_block_in_history( + json!({"messages": [ + {"role": "assistant", "content": [{"type": "compaction", "content": "summary", "signature": ""}]}, + {"role": "user", "content": "Continue"}, + ]}), + &[] + )] + #[case::advisor_tool( + json!({"tools": [{"type": "advisor_20260301", "name": "advisor", "model": "claude-opus-4-6"}]}), + &[beta::ADVISOR_TOOL_2026_03_01] + )] + #[case::no_tools(json!({"tools": []}), &[])] + #[case::regex_tool_search( + json!({"tools": [{"type": "tool_search_tool_regex_20251119"}]}), + &[beta::ADVANCED_TOOL_USE_2025_11_20] + )] + #[case::bm25_tool_search( + json!({"tools": [{"type": "tool_search_tool_bm25_20251119"}]}), + &[beta::ADVANCED_TOOL_USE_2025_11_20] + )] + #[case::unrelated_server_tool(json!({"tools": [{"type": "web_search_20250305", "name": "web_search"}]}), &[])] + #[case::only_compact_edits( + json!({"context_management": {"edits": [{"type": "compact_20260112"}]}}), + &[beta::COMPACT_2026_01_12] + )] + #[case::only_other_edits( + json!({"context_management": {"edits": [{"type": "clear_tool_uses_20250919", "keep": {"type": "tool_uses", "value": 3}}]}}), + &[beta::CONTEXT_MANAGEMENT_2025_06_27] + )] + #[case::compact_and_other_edits( + json!({"context_management": {"edits": [{"type": "compact_20260112"}, {"type": "clear_tool_uses_20250919"}]}}), + &[beta::COMPACT_2026_01_12, beta::CONTEXT_MANAGEMENT_2025_06_27] + )] + #[case::edit_without_a_type(json!({"context_management": {"edits": [{}]}}), &[beta::CONTEXT_MANAGEMENT_2025_06_27])] + #[case::empty_edits(json!({"context_management": {"edits": []}}), &[])] + #[case::context_management_without_edits(json!({"context_management": {}}), &[])] + #[case::per_message_output_config( + json!({"messages": [{"role": "user", "content": "hi", "output_config": {"effort": "low"}}]}), + &[beta::PER_TURN_CONTROL_2026_07_01] + )] + #[case::per_message_null_output_config( + json!({"messages": [{"role": "user", "content": "hi", "output_config": null}]}), + &[beta::PER_TURN_CONTROL_2026_07_01] + )] + fn feature_betas_follow_the_request(#[case] fields: Value, #[case] expected: &[&str]) { + assert_eq!(feature_betas(&request(fields)), expected); + } + + #[rstest] + #[case::no_betas(&[("x-api-key", "k"), ("anthropic-version", "2023-06-01")], json!({}))] + #[case::blank_beta_header(&[("Anthropic-Beta", " , "), ("x-api-key", "k")], json!({}))] + fn headers_without_any_beta_value_are_untouched( + #[case] input: &[(&str, &str)], + #[case] fields: Value, + ) { + assert_eq!( + with_feature_betas(headers(input), &request(fields)), + headers(input) + ); + } + + #[rstest] + #[case::feature_beta_is_appended( + &[("x-api-key", "k")], + json!({"speed": "fast"}), + &[("x-api-key", "k"), ("anthropic-beta", beta::FAST_MODE_2026_02_01)], + )] + #[case::existing_betas_are_normalized_without_features( + &[("Anthropic-Beta", "web-search-2025-03-05, interleaved-thinking-2025-05-14 ,web-search-2025-03-05"), ("x-api-key", "k")], + json!({}), + &[("x-api-key", "k"), ("anthropic-beta", "interleaved-thinking-2025-05-14,web-search-2025-03-05")], + )] + #[case::existing_advisor_beta_is_kept_without_an_advisor_tool( + &[("anthropic-beta", beta::ADVISOR_TOOL_2026_03_01)], + json!({"tools": []}), + &[("anthropic-beta", beta::ADVISOR_TOOL_2026_03_01)], + )] + #[case::feature_already_sent_is_not_duplicated( + &[("anthropic-beta", beta::FAST_MODE_2026_02_01)], + json!({"speed": "fast"}), + &[("anthropic-beta", beta::FAST_MODE_2026_02_01)], + )] + fn feature_betas_merge_into_the_headers( + #[case] input: &[(&str, &str)], + #[case] fields: Value, + #[case] expected: &[(&str, &str)], + ) { + assert_eq!( + with_feature_betas(headers(input), &request(fields)), + headers(expected) + ); + } + + #[test] + fn differently_cased_beta_header_is_replaced_by_one_sorted_header() { + let merged = with_feature_betas( + headers(&[("Anthropic-Beta", "interleaved-thinking-2025-05-14")]), + &request( + json!({"messages": [{"role": "system", "content": "env", "output_config": {"effort": "low"}}]}), + ), + ); + assert_eq!( + merged, + headers(&[( + "anthropic-beta", + &betas(&[ + "interleaved-thinking-2025-05-14", + beta::PER_TURN_CONTROL_2026_07_01 + ]) + )]) + ); + } + + #[test] + fn every_beta_header_casing_is_unioned_into_one_header() { + let merged = with_feature_betas( + headers(&[ + ("anthropic-beta", "interleaved-thinking-2025-05-14"), + ("Anthropic-Beta", "web-search-2025-03-05"), + ]), + &request(json!({"speed": "fast"})), + ); + assert_eq!( + merged, + headers(&[( + "anthropic-beta", + &betas(&[ + beta::FAST_MODE_2026_02_01, + "interleaved-thinking-2025-05-14", + "web-search-2025-03-05" + ]) + )]) + ); + } + + #[test] + fn unknown_client_betas_survive_alongside_the_added_one() { + let client_betas = [ + "claude-code-20250219", + "interleaved-thinking-2025-05-14", + beta::CONTEXT_MANAGEMENT_2025_06_27, + beta::PER_TURN_CONTROL_2026_07_01, + "effort-2025-11-24", + ]; + let merged = with_feature_betas( + headers(&[("anthropic-beta", &betas(&client_betas))]), + &request( + json!({"messages": [{"role": "user", "content": "hi", "output_config": {"effort": "low"}}]}), + ), + ); + assert_eq!( + merged, + headers(&[( + "anthropic-beta", + &betas(&[ + "claude-code-20250219", + beta::CONTEXT_MANAGEMENT_2025_06_27, + "effort-2025-11-24", + "interleaved-thinking-2025-05-14", + beta::PER_TURN_CONTROL_2026_07_01, + ]) + )]) + ); + } + + #[test] + fn every_feature_merges_with_the_oauth_beta_sorted_and_last() { + let oauth_headers = authenticate_with(&[], Some(OAUTH_TOKEN), &[]).unwrap(); + let all_features = request(json!({ + "compaction": {"enabled": true}, + "output_format": {"type": "json_schema"}, + "speed": "fast", + "tools": [{"type": "advisor_20260301"}, {"type": "tool_search_tool_bm25_20251119"}], + "context_management": {"edits": [{"type": "compact_20260112"}, {"type": "clear_thinking_20251015"}]}, + "messages": [{"role": "user", "content": "hi", "output_config": {"effort": "low"}}], + })); + assert_eq!( + with_feature_betas(oauth_headers, &all_features), + headers(&[ + ("authorization", OAUTH_BEARER), + BROWSER_ACCESS, + ( + "anthropic-beta", + &betas(&[ + beta::ADVANCED_TOOL_USE_2025_11_20, + beta::ADVISOR_TOOL_2026_03_01, + beta::COMPACT_2026_01_12, + beta::COMPACT_2026_09_04, + beta::CONTEXT_MANAGEMENT_2025_06_27, + beta::FAST_MODE_2026_02_01, + ANTHROPIC_OAUTH_BETA_HEADER, + beta::PER_TURN_CONTROL_2026_07_01, + beta::STRUCTURED_OUTPUT, + ]) + ), + ]) + ); + } +} diff --git a/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/mod.rs b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/mod.rs index 481d98c4e9d..5adf5fda16f 100644 --- a/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/mod.rs +++ b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/mod.rs @@ -1,2 +1,5 @@ +pub mod handler; +pub mod headers; pub mod streaming_iterator; +pub mod thinking; pub mod transformation; diff --git a/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/thinking.rs b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/thinking.rs new file mode 100644 index 00000000000..ffa4c8ffeb8 --- /dev/null +++ b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/thinking.rs @@ -0,0 +1,1182 @@ +use litellm_core_utils::settings::Lookup; +use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest; +use serde_json::{Map, Value, json}; + +use crate::{ + anthropic::common_utils::AnthropicModelCapabilities, base_llm::chat::transformation::Error, +}; + +pub const ANTHROPIC_MIN_THINKING_BUDGET_TOKENS: u64 = 1024; + +const EFFORT_NAMES: &str = "'minimal', 'low', 'medium', 'high', 'xhigh', 'max', 'none'"; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ThinkingBudgets { + pub minimal: u64, + pub low: u64, + pub medium: u64, + pub high: u64, + pub xhigh: u64, + pub max: u64, +} + +impl Default for ThinkingBudgets { + fn default() -> Self { + Self { + minimal: 128, + low: 1024, + medium: 2048, + high: 4096, + xhigh: 8192, + max: 16384, + } + } +} + +impl ThinkingBudgets { + pub fn from_lookup(env: &impl Lookup) -> Self { + let defaults = Self::default(); + let tier = |name: &str, default: u64| { + env.parsed::(&format!("DEFAULT_REASONING_EFFORT_{name}_THINKING_BUDGET")) + .unwrap_or(default) + }; + Self { + minimal: tier("MINIMAL", defaults.minimal), + low: tier("LOW", defaults.low), + medium: tier("MEDIUM", defaults.medium), + high: tier("HIGH", defaults.high), + xhigh: tier("XHIGH", defaults.xhigh), + max: tier("MAX", defaults.max), + } + } + + fn for_effort(&self, reasoning_effort: &str) -> Option { + match reasoning_effort { + "low" => Some(self.low), + "medium" => Some(self.medium), + "high" => Some(self.high), + "xhigh" => Some(self.xhigh), + "max" => Some(self.max), + "minimal" => Some(self.minimal.max(ANTHROPIC_MIN_THINKING_BUDGET_TOKENS)), + _ => None, + } + } + + fn effort_for_budget( + &self, + budget_tokens: u64, + capabilities: &AnthropicModelCapabilities, + ) -> &'static str { + if budget_tokens >= self.xhigh && capabilities.effort_tiers.xhigh { + return "xhigh"; + } + if budget_tokens >= self.high { + return "high"; + } + if budget_tokens >= self.medium { + return "medium"; + } + "low" + } +} + +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct ThinkingContext { + pub capabilities: AnthropicModelCapabilities, + pub budgets: ThinkingBudgets, +} + +fn bad_request(message: String) -> Error { + Error::InvalidRequest(message) +} + +fn thinking_type(thinking: Option<&Value>) -> Option<&str> { + thinking?.get("type")?.as_str() +} + +fn output_config_effort(output_config: Option<&Value>) -> Option<&str> { + output_config?.get("effort")?.as_str() +} + +fn enabled_thinking(budget_tokens: u64) -> Value { + json!({"type": "enabled", "budget_tokens": budget_tokens}) +} + +fn map_reasoning_effort( + reasoning_effort: &str, + context: &ThinkingContext, +) -> Result, Error> { + if reasoning_effort == "none" { + return Ok(None); + } + if context.capabilities.supports_adaptive_thinking { + return Ok(Some(json!({"type": "adaptive", "display": "summarized"}))); + } + context + .budgets + .for_effort(reasoning_effort) + .map(|budget| Some(enabled_thinking(budget))) + .ok_or_else(|| { + bad_request(format!( + "Unmapped reasoning effort: '{reasoning_effort}'. Must be one of: {EFFORT_NAMES}." + )) + }) +} + +fn cap_thinking_budget_to_max_tokens(thinking: Value, max_tokens: Option) -> Option { + let (Some(max_tokens), Some(budget)) = ( + max_tokens, + thinking.get("budget_tokens").and_then(Value::as_u64), + ) else { + return Some(thinking); + }; + if max_tokens <= ANTHROPIC_MIN_THINKING_BUDGET_TOKENS { + return None; + } + if budget < max_tokens { + return Some(thinking); + } + Some(enabled_thinking(max_tokens - 1)) +} + +fn reasoning_effort_to_output_config_effort(reasoning_effort: &str) -> Option<&'static str> { + match reasoning_effort { + "low" | "minimal" => Some("low"), + "medium" => Some("medium"), + "high" => Some("high"), + "xhigh" => Some("xhigh"), + "max" => Some("max"), + _ => None, + } +} + +fn with_default_effort(output_config: Option, effort: &str) -> Value { + let mut config = match output_config { + Some(Value::Object(config)) => config, + _ => Map::new(), + }; + if !config.contains_key("effort") { + config.insert("effort".to_string(), Value::String(effort.to_string())); + } + Value::Object(config) +} + +fn translate_reasoning_effort( + request: AnthropicMessagesRequest, + context: &ThinkingContext, +) -> Result { + let Some(reasoning_effort) = request.reasoning_effort.clone() else { + return Ok(request); + }; + let request = AnthropicMessagesRequest { + reasoning_effort: None, + ..request + }; + let Some(mapped) = map_reasoning_effort(&reasoning_effort, context)? else { + return Ok(AnthropicMessagesRequest { + thinking: None, + output_config: None, + ..request + }); + }; + let Some(fitted) = cap_thinking_budget_to_max_tokens(mapped, request.max_tokens) else { + return Ok(request); + }; + let thinking = Some(request.thinking.clone().unwrap_or(fitted)); + if !context.capabilities.supports_adaptive_thinking { + return Ok(AnthropicMessagesRequest { + thinking, + ..request + }); + } + let effort = reasoning_effort_to_output_config_effort(&reasoning_effort).ok_or_else(|| { + bad_request(format!( + "Invalid reasoning_effort: '{reasoning_effort}'. Must be one of: {EFFORT_NAMES}" + )) + })?; + if let Some(rejection) = context + .capabilities + .effort_level_rejection(effort, &request.model) + { + return Err(bad_request(rejection)); + } + Ok(AnthropicMessagesRequest { + thinking, + output_config: Some(with_default_effort(request.output_config.clone(), effort)), + ..request + }) +} + +fn drop_disabled_thinking( + request: AnthropicMessagesRequest, + context: &ThinkingContext, +) -> AnthropicMessagesRequest { + if !context.capabilities.thinking_always_on + || thinking_type(request.thinking.as_ref()) != Some("disabled") + { + return request; + } + AnthropicMessagesRequest { + thinking: None, + ..request + } +} + +fn translate_legacy_thinking_for_adaptive_model( + request: AnthropicMessagesRequest, + context: &ThinkingContext, +) -> AnthropicMessagesRequest { + let capabilities = &context.capabilities; + if !capabilities.supports_adaptive_thinking + || capabilities.supports_legacy_thinking + || thinking_type(request.thinking.as_ref()) != Some("enabled") + { + return request; + } + let budget = request + .thinking + .as_ref() + .and_then(|thinking| thinking.get("budget_tokens")) + .and_then(Value::as_u64) + .unwrap_or(0); + let effort = context.budgets.effort_for_budget(budget, capabilities); + AnthropicMessagesRequest { + thinking: Some(json!({"type": "adaptive"})), + output_config: Some(with_default_effort(request.output_config.clone(), effort)), + ..request + } +} + +fn output_config_without_effort(output_config: Option) -> Option { + let Some(Value::Object(config)) = output_config else { + return output_config; + }; + if !config.contains_key("effort") { + return Some(Value::Object(config)); + } + let residual: Map = config + .into_iter() + .filter(|(key, _)| key != "effort") + .collect(); + (!residual.is_empty()).then_some(Value::Object(residual)) +} + +fn translate_adaptive_effort_for_non_adaptive_model( + request: AnthropicMessagesRequest, + context: &ThinkingContext, +) -> Result { + let capabilities = &context.capabilities; + if capabilities.supports_adaptive_thinking { + return Ok(request); + } + let effort = output_config_effort(request.output_config.as_ref()).map(str::to_string); + let adaptive_thinking = thinking_type(request.thinking.as_ref()) == Some("adaptive"); + if effort.is_none() && !adaptive_thinking { + return Ok(request); + } + let level_supported = effort.as_deref().is_none_or(|effort| { + capabilities + .effort_level_rejection(effort, &request.model) + .is_none() + }); + if capabilities.supports_effort_param() && (!adaptive_thinking || level_supported) { + return Ok(AnthropicMessagesRequest { + thinking: if adaptive_thinking { + None + } else { + request.thinking.clone() + }, + ..request + }); + } + let legacy = if capabilities.supports_reasoning { + map_reasoning_effort( + effort + .as_deref() + .filter(|effort| !effort.is_empty()) + .unwrap_or("medium"), + context, + )? + } else { + None + }; + let capped = + legacy.and_then(|thinking| cap_thinking_budget_to_max_tokens(thinking, request.max_tokens)); + Ok(AnthropicMessagesRequest { + thinking: capped, + output_config: output_config_without_effort(request.output_config.clone()), + ..request + }) +} + +fn drop_incompatible_temperature_for_thinking( + request: AnthropicMessagesRequest, + context: &ThinkingContext, +) -> AnthropicMessagesRequest { + if context.capabilities.supports_adaptive_thinking { + return request; + } + let pinned = request + .temperature + .is_some_and(|temperature| temperature != 1.0); + let thinking_enabled = thinking_type(request.thinking.as_ref()) == Some("enabled"); + let effort_enabled = output_config_effort(request.output_config.as_ref()).is_some(); + if !pinned || !(thinking_enabled || effort_enabled) { + return request; + } + AnthropicMessagesRequest { + temperature: None, + ..request + } +} + +pub fn translate_thinking( + request: AnthropicMessagesRequest, + context: &ThinkingContext, +) -> Result { + let request = translate_reasoning_effort(request, context)?; + let request = drop_disabled_thinking(request, context); + let request = translate_legacy_thinking_for_adaptive_model(request, context); + let request = translate_adaptive_effort_for_non_adaptive_model(request, context)?; + Ok(drop_incompatible_temperature_for_thinking(request, context)) +} + +#[cfg(test)] +mod tests { + use rstest::{fixture, rstest}; + + use super::*; + use crate::anthropic::common_utils::SupportedEffortTiers; + + const EFFORT_CHOICES: &str = "'minimal', 'low', 'medium', 'high', 'xhigh', 'max', 'none'"; + + fn request(fields: Value) -> AnthropicMessagesRequest { + let mut body = + json!({"model": "claude", "messages": [{"role": "user", "content": "Hello"}]}); + body.as_object_mut() + .unwrap() + .extend(fields.as_object().unwrap().clone()); + serde_json::from_value(body).unwrap() + } + + fn context(capabilities: AnthropicModelCapabilities) -> ThinkingContext { + ThinkingContext { + capabilities, + budgets: ThinkingBudgets::default(), + } + } + + fn translate( + capabilities: AnthropicModelCapabilities, + fields: Value, + ) -> Result { + translate_thinking(request(fields), &context(capabilities)) + } + + fn overridden_budgets(overrides: &[(&str, &str)]) -> ThinkingBudgets { + let env = |name: &str| { + overrides + .iter() + .find(|(tier, _)| { + name == format!("DEFAULT_REASONING_EFFORT_{tier}_THINKING_BUDGET") + }) + .map(|(_, value)| value.to_string()) + }; + ThinkingBudgets::from_lookup(&env) + } + + fn claude_code_payload(effort: &str, max_tokens: u64) -> Value { + json!({"max_tokens": max_tokens, "thinking": {"type": "adaptive"}, "output_config": {"effort": effort}}) + } + + fn with_temperature(fields: Value, temperature: f64) -> Value { + let mut fields = fields; + fields + .as_object_mut() + .unwrap() + .insert("temperature".to_string(), json!(temperature)); + fields + } + + #[fixture] + fn haiku_3_5() -> AnthropicModelCapabilities { + AnthropicModelCapabilities::default() + } + + #[fixture] + fn haiku_4_5() -> AnthropicModelCapabilities { + AnthropicModelCapabilities { + supports_reasoning: true, + ..Default::default() + } + } + + #[fixture] + fn opus_4_5() -> AnthropicModelCapabilities { + AnthropicModelCapabilities { + supports_reasoning: true, + supports_output_config: true, + ..Default::default() + } + } + + #[fixture] + fn sonnet_4_6() -> AnthropicModelCapabilities { + AnthropicModelCapabilities { + supports_reasoning: true, + supports_adaptive_thinking: true, + supports_legacy_thinking: true, + supports_output_config: true, + effort_tiers: SupportedEffortTiers { + max: true, + ..Default::default() + }, + ..Default::default() + } + } + + #[fixture] + fn opus_4_7() -> AnthropicModelCapabilities { + AnthropicModelCapabilities { + supports_reasoning: true, + supports_adaptive_thinking: true, + supports_output_config: true, + effort_tiers: SupportedEffortTiers { + xhigh: true, + max: true, + ..Default::default() + }, + ..Default::default() + } + } + + #[fixture] + fn fable_5_1() -> AnthropicModelCapabilities { + AnthropicModelCapabilities { + thinking_always_on: true, + ..opus_4_7() + } + } + + #[fixture] + fn newfamily_6() -> AnthropicModelCapabilities { + AnthropicModelCapabilities { + supports_reasoning: true, + supports_adaptive_thinking: true, + ..Default::default() + } + } + + #[rstest] + #[case::minimal_maps_to_low(opus_4_7(), "minimal", "low")] + #[case::low(opus_4_7(), "low", "low")] + #[case::medium(opus_4_7(), "medium", "medium")] + #[case::high(opus_4_7(), "high", "high")] + #[case::xhigh_with_xhigh_tier(opus_4_7(), "xhigh", "xhigh")] + #[case::max(opus_4_7(), "max", "max")] + #[case::minimal_maps_to_low_on_4_6(sonnet_4_6(), "minimal", "low")] + #[case::low_on_4_6(sonnet_4_6(), "low", "low")] + #[case::max_without_max_tier_is_allowed_on_adaptive_models(newfamily_6(), "max", "max")] + fn reasoning_effort_on_adaptive_model_becomes_summarized_adaptive_thinking_and_effort( + #[case] capabilities: AnthropicModelCapabilities, + #[case] reasoning_effort: &str, + #[case] expected_effort: &str, + ) { + assert_eq!( + translate( + capabilities, + json!({"max_tokens": 1024, "reasoning_effort": reasoning_effort}) + ), + Ok(request(json!({ + "max_tokens": 1024, + "thinking": {"type": "adaptive", "display": "summarized"}, + "output_config": {"effort": expected_effort} + }))) + ); + } + + #[rstest] + #[case::adaptive_shape_is_not_dropped_for_small_max_tokens( + opus_4_7(), + json!({"max_tokens": 64, "reasoning_effort": "high"}), + json!({"max_tokens": 64, "thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "high"}}) + )] + #[case::caller_output_config_effort_wins( + opus_4_7(), + json!({"max_tokens": 1024, "reasoning_effort": "low", "output_config": {"effort": "max"}}), + json!({"max_tokens": 1024, "thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "max"}}) + )] + #[case::effort_merges_into_caller_output_config( + opus_4_7(), + json!({"max_tokens": 1024, "reasoning_effort": "high", "output_config": {"format": {"type": "json_schema"}}}), + json!({ + "max_tokens": 1024, + "thinking": {"type": "adaptive", "display": "summarized"}, + "output_config": {"format": {"type": "json_schema"}, "effort": "high"} + }) + )] + #[case::non_object_output_config_is_replaced( + opus_4_7(), + json!({"max_tokens": 1024, "reasoning_effort": "high", "output_config": "bogus"}), + json!({"max_tokens": 1024, "thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "high"}}) + )] + #[case::caller_thinking_and_output_config_win( + sonnet_4_6(), + json!({ + "max_tokens": 16000, + "reasoning_effort": "low", + "thinking": {"type": "enabled", "budget_tokens": 8000}, + "output_config": {"effort": "high"} + }), + json!({ + "max_tokens": 16000, + "thinking": {"type": "enabled", "budget_tokens": 8000}, + "output_config": {"effort": "high"} + }) + )] + #[case::caller_legacy_thinking_is_then_translated_while_reasoning_effort_level_stays( + opus_4_7(), + json!({"max_tokens": 16000, "reasoning_effort": "low", "thinking": {"type": "enabled", "budget_tokens": 8000}}), + json!({"max_tokens": 16000, "thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}}) + )] + #[case::caller_disabled_thinking_is_kept_then_omitted_on_always_on_model( + fable_5_1(), + json!({"max_tokens": 1024, "reasoning_effort": "high", "thinking": {"type": "disabled"}}), + json!({"max_tokens": 1024, "output_config": {"effort": "high"}}) + )] + #[case::non_adaptive_model_gets_no_output_config( + opus_4_5(), + json!({"max_tokens": 8192, "reasoning_effort": "high"}), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}}) + )] + #[case::caller_thinking_wins_on_non_adaptive_model( + opus_4_5(), + json!({"max_tokens": 16000, "reasoning_effort": "low", "thinking": {"type": "enabled", "budget_tokens": 8000}}), + json!({"max_tokens": 16000, "thinking": {"type": "enabled", "budget_tokens": 8000}}) + )] + #[case::caller_thinking_survives_when_mapped_budget_cannot_fit( + opus_4_5(), + json!({"max_tokens": 1024, "reasoning_effort": "low", "thinking": {"type": "enabled", "budget_tokens": 8000}}), + json!({"max_tokens": 1024, "thinking": {"type": "enabled", "budget_tokens": 8000}}) + )] + #[case::missing_max_tokens_leaves_budget_uncapped( + haiku_4_5(), + json!({"reasoning_effort": "high"}), + json!({"thinking": {"type": "enabled", "budget_tokens": 4096}}) + )] + #[case::budget_below_max_tokens_is_kept( + haiku_4_5(), + json!({"max_tokens": 4097, "reasoning_effort": "high"}), + json!({"max_tokens": 4097, "thinking": {"type": "enabled", "budget_tokens": 4096}}) + )] + #[case::budget_equal_to_max_tokens_is_capped( + haiku_4_5(), + json!({"max_tokens": 4096, "reasoning_effort": "high"}), + json!({"max_tokens": 4096, "thinking": {"type": "enabled", "budget_tokens": 4095}}) + )] + #[case::budget_above_max_tokens_is_capped( + haiku_4_5(), + json!({"max_tokens": 4000, "reasoning_effort": "xhigh"}), + json!({"max_tokens": 4000, "thinking": {"type": "enabled", "budget_tokens": 3999}}) + )] + #[case::max_tokens_just_above_min_budget_caps_to_min_budget( + haiku_4_5(), + json!({"max_tokens": 1025, "reasoning_effort": "xhigh"}), + json!({"max_tokens": 1025, "thinking": {"type": "enabled", "budget_tokens": 1024}}) + )] + #[case::max_tokens_at_min_budget_drops_thinking( + haiku_4_5(), + json!({"max_tokens": 1024, "reasoning_effort": "xhigh"}), + json!({"max_tokens": 1024}) + )] + #[case::pinned_temperature_is_dropped_after_thinking_is_synthesized( + haiku_4_5(), + json!({"max_tokens": 8192, "reasoning_effort": "low", "temperature": 0}), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 1024}}) + )] + fn reasoning_effort_is_translated( + #[case] capabilities: AnthropicModelCapabilities, + #[case] input: Value, + #[case] expected: Value, + ) { + assert_eq!(translate(capabilities, input), Ok(request(expected))); + } + + #[rstest] + #[case::minimal_floors_at_min_budget("minimal", 1024)] + #[case::low("low", 1024)] + #[case::medium("medium", 2048)] + #[case::high("high", 4096)] + #[case::xhigh("xhigh", 8192)] + #[case::max("max", 16384)] + fn reasoning_effort_on_non_adaptive_model_uses_the_tier_budget( + haiku_4_5: AnthropicModelCapabilities, + #[case] reasoning_effort: &str, + #[case] expected_budget: u64, + ) { + assert_eq!( + translate( + haiku_4_5, + json!({"max_tokens": 32000, "reasoning_effort": reasoning_effort}) + ), + Ok(request(json!({ + "max_tokens": 32000, + "thinking": {"type": "enabled", "budget_tokens": expected_budget} + }))) + ); + } + + #[rstest] + #[case::adaptive_model(opus_4_7())] + #[case::effort_capable_model(opus_4_5())] + #[case::budget_model(haiku_4_5())] + fn reasoning_effort_none_clears_thinking_and_output_config( + #[case] capabilities: AnthropicModelCapabilities, + ) { + assert_eq!( + translate( + capabilities, + json!({ + "max_tokens": 1024, + "reasoning_effort": "none", + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "high"} + }) + ), + Ok(request(json!({"max_tokens": 1024}))) + ); + } + + #[rstest] + #[case::bogus_on_budget_model( + opus_4_5(), + json!({"max_tokens": 1024, "reasoning_effort": "bogus"}), + format!("Unmapped reasoning effort: 'bogus'. Must be one of: {EFFORT_CHOICES}.") + )] + #[case::disabled_on_budget_model( + haiku_4_5(), + json!({"max_tokens": 1024, "reasoning_effort": "disabled"}), + format!("Unmapped reasoning effort: 'disabled'. Must be one of: {EFFORT_CHOICES}.") + )] + #[case::empty_on_budget_model( + haiku_4_5(), + json!({"max_tokens": 1024, "reasoning_effort": ""}), + format!("Unmapped reasoning effort: ''. Must be one of: {EFFORT_CHOICES}.") + )] + #[case::invalid_on_adaptive_model( + opus_4_7(), + json!({"max_tokens": 1024, "reasoning_effort": "invalid"}), + format!("Invalid reasoning_effort: 'invalid'. Must be one of: {EFFORT_CHOICES}") + )] + #[case::disabled_on_adaptive_model( + opus_4_7(), + json!({"max_tokens": 1024, "reasoning_effort": "disabled"}), + format!("Invalid reasoning_effort: 'disabled'. Must be one of: {EFFORT_CHOICES}") + )] + #[case::empty_on_adaptive_model( + opus_4_7(), + json!({"max_tokens": 1024, "reasoning_effort": ""}), + format!("Invalid reasoning_effort: ''. Must be one of: {EFFORT_CHOICES}") + )] + #[case::xhigh_without_xhigh_tier_on_4_6( + sonnet_4_6(), + json!({"max_tokens": 1024, "reasoning_effort": "xhigh"}), + "effort='xhigh' is not supported by this model. Got model: claude".to_string() + )] + #[case::xhigh_without_xhigh_tier_on_unmapped_adaptive_model( + newfamily_6(), + json!({"max_tokens": 1024, "reasoning_effort": "xhigh"}), + "effort='xhigh' is not supported by this model. Got model: claude".to_string() + )] + #[case::unrecognized_adaptive_effort_on_budget_model( + haiku_4_5(), + claude_code_payload("turbo", 8192), + format!("Unmapped reasoning effort: 'turbo'. Must be one of: {EFFORT_CHOICES}.") + )] + fn unsupported_effort_is_a_request_error( + #[case] capabilities: AnthropicModelCapabilities, + #[case] input: Value, + #[case] expected_message: String, + ) { + assert_eq!( + translate(capabilities, input), + Err(Error::InvalidRequest(expected_message)) + ); + } + + #[rstest] + #[case::omitted_on_always_on_model(fable_5_1(), json!({"type": "disabled"}), None)] + #[case::kept_on_adaptive_model(opus_4_7(), json!({"type": "disabled"}), Some(json!({"type": "disabled"})))] + #[case::kept_on_budget_model(haiku_4_5(), json!({"type": "disabled"}), Some(json!({"type": "disabled"})))] + #[case::adaptive_kept_on_always_on_model( + fable_5_1(), + json!({"type": "adaptive"}), + Some(json!({"type": "adaptive"})) + )] + fn disabled_thinking_is_omitted_only_for_always_on_models( + #[case] capabilities: AnthropicModelCapabilities, + #[case] thinking: Value, + #[case] expected_thinking: Option, + ) { + let expected = match expected_thinking { + Some(thinking) => json!({"max_tokens": 64, "thinking": thinking}), + None => json!({"max_tokens": 64}), + }; + assert_eq!( + translate( + capabilities, + json!({"max_tokens": 64, "thinking": thinking}) + ), + Ok(request(expected)) + ); + } + + #[rstest] + #[case::far_above_xhigh_budget(opus_4_7(), json!(16384), "xhigh")] + #[case::at_xhigh_budget(opus_4_7(), json!(8192), "xhigh")] + #[case::below_xhigh_budget(opus_4_7(), json!(8191), "high")] + #[case::xhigh_budget_without_xhigh_tier(newfamily_6(), json!(8192), "high")] + #[case::large_budget_without_xhigh_tier(newfamily_6(), json!(31999), "high")] + #[case::at_high_budget(opus_4_7(), json!(4096), "high")] + #[case::below_high_budget(opus_4_7(), json!(4095), "medium")] + #[case::at_medium_budget(opus_4_7(), json!(2048), "medium")] + #[case::below_medium_budget(opus_4_7(), json!(2047), "low")] + #[case::tiny_budget(opus_4_7(), json!(1), "low")] + #[case::missing_budget(opus_4_7(), Value::Null, "low")] + #[case::always_on_model(fable_5_1(), json!(24000), "xhigh")] + fn legacy_thinking_is_bucketed_into_adaptive_effort_on_adaptive_only_models( + #[case] capabilities: AnthropicModelCapabilities, + #[case] budget_tokens: Value, + #[case] expected_effort: &str, + ) { + let thinking = match budget_tokens { + Value::Null => json!({"type": "enabled"}), + budget_tokens => json!({"type": "enabled", "budget_tokens": budget_tokens}), + }; + assert_eq!( + translate( + capabilities, + json!({"max_tokens": 1024, "thinking": thinking}) + ), + Ok(request(json!({ + "max_tokens": 1024, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": expected_effort} + }))) + ); + } + + #[rstest] + #[case::verbatim_on_model_accepting_legacy_thinking( + sonnet_4_6(), + json!({"max_tokens": 1024, "thinking": {"type": "enabled", "budget_tokens": 31999}}), + json!({"max_tokens": 1024, "thinking": {"type": "enabled", "budget_tokens": 31999}}) + )] + #[case::verbatim_with_explicit_output_config_on_model_accepting_legacy_thinking( + sonnet_4_6(), + json!({"max_tokens": 1024, "thinking": {"type": "enabled", "budget_tokens": 31999}, "output_config": {"effort": "low"}}), + json!({"max_tokens": 1024, "thinking": {"type": "enabled", "budget_tokens": 31999}, "output_config": {"effort": "low"}}) + )] + #[case::verbatim_on_non_adaptive_model( + opus_4_5(), + json!({"max_tokens": 1024, "thinking": {"type": "enabled", "budget_tokens": 31999}}), + json!({"max_tokens": 1024, "thinking": {"type": "enabled", "budget_tokens": 31999}}) + )] + #[case::caller_output_config_effort_wins( + opus_4_7(), + json!({ + "max_tokens": 32000, + "thinking": {"type": "enabled", "budget_tokens": 31999}, + "output_config": {"effort": "low", "format": {"type": "json_schema"}} + }), + json!({ + "max_tokens": 32000, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "low", "format": {"type": "json_schema"}} + }) + )] + #[case::effort_merges_into_caller_output_config( + opus_4_7(), + json!({ + "max_tokens": 32000, + "thinking": {"type": "enabled", "budget_tokens": 4096}, + "output_config": {"format": {"type": "json_schema"}} + }), + json!({ + "max_tokens": 32000, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "high", "format": {"type": "json_schema"}} + }) + )] + #[case::adaptive_thinking_is_left_alone( + opus_4_7(), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive", "display": "summarized"}}), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive", "display": "summarized"}}) + )] + fn legacy_thinking_on_adaptive_capable_models( + #[case] capabilities: AnthropicModelCapabilities, + #[case] input: Value, + #[case] expected: Value, + ) { + assert_eq!(translate(capabilities, input), Ok(request(expected))); + } + + #[rstest] + #[case::bare_adaptive_becomes_medium_budget_on_budget_model( + haiku_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}}), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}}) + )] + #[case::medium_effort_becomes_medium_budget_on_budget_model( + haiku_4_5(), + claude_code_payload("medium", 8192), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}}) + )] + #[case::empty_effort_becomes_medium_budget_on_budget_model( + haiku_4_5(), + claude_code_payload("", 8192), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}}) + )] + #[case::high_effort_becomes_high_budget_on_budget_model( + haiku_4_5(), + claude_code_payload("high", 8192), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}}) + )] + #[case::effort_only_becomes_budget_on_budget_model( + haiku_4_5(), + json!({"max_tokens": 8192, "output_config": {"effort": "high"}}), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}}) + )] + #[case::effort_replaces_caller_legacy_budget_on_budget_model( + haiku_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 3000}, "output_config": {"effort": "high"}}), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}}) + )] + #[case::residual_output_config_survives_effort_translation( + haiku_4_5(), + json!({ + "max_tokens": 8192, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "medium", "format": {"type": "json_schema"}} + }), + json!({ + "max_tokens": 8192, + "thinking": {"type": "enabled", "budget_tokens": 2048}, + "output_config": {"format": {"type": "json_schema"}} + }) + )] + #[case::effortless_output_config_is_kept( + haiku_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}, "output_config": {"format": {"type": "json_schema"}}}), + json!({ + "max_tokens": 8192, + "thinking": {"type": "enabled", "budget_tokens": 2048}, + "output_config": {"format": {"type": "json_schema"}} + }) + )] + #[case::empty_output_config_is_kept( + haiku_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}, "output_config": {}}), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}, "output_config": {}}) + )] + #[case::missing_max_tokens_leaves_budget_uncapped( + haiku_4_5(), + json!({"thinking": {"type": "adaptive"}}), + json!({"thinking": {"type": "enabled", "budget_tokens": 2048}}) + )] + #[case::budget_is_capped_below_max_tokens( + haiku_4_5(), + claude_code_payload("high", 3000), + json!({"max_tokens": 3000, "thinking": {"type": "enabled", "budget_tokens": 2999}}) + )] + #[case::max_tokens_just_above_min_budget_caps_to_min_budget( + haiku_4_5(), + claude_code_payload("medium", 1025), + json!({"max_tokens": 1025, "thinking": {"type": "enabled", "budget_tokens": 1024}}) + )] + #[case::max_tokens_at_min_budget_drops_thinking_and_effort( + haiku_4_5(), + claude_code_payload("medium", 1024), + json!({"max_tokens": 1024}) + )] + #[case::max_tokens_below_min_budget_drops_thinking_and_effort( + haiku_4_5(), + claude_code_payload("medium", 512), + json!({"max_tokens": 512}) + )] + #[case::bare_adaptive_is_dropped_on_non_reasoning_model( + haiku_3_5(), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}}), + json!({"max_tokens": 8192}) + )] + #[case::adaptive_and_effort_are_dropped_on_non_reasoning_model( + haiku_3_5(), + claude_code_payload("medium", 8192), + json!({"max_tokens": 8192}) + )] + #[case::effort_only_is_dropped_on_non_reasoning_model( + haiku_3_5(), + json!({"max_tokens": 8192, "output_config": {"effort": "high", "format": {"type": "json_schema"}}}), + json!({"max_tokens": 8192, "output_config": {"format": {"type": "json_schema"}}}) + )] + #[case::supported_effort_is_kept_and_adaptive_thinking_dropped_on_effort_model( + opus_4_5(), + claude_code_payload("medium", 8192), + json!({"max_tokens": 8192, "output_config": {"effort": "medium"}}) + )] + #[case::bare_adaptive_is_dropped_on_effort_model( + opus_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}}), + json!({"max_tokens": 8192}) + )] + #[case::effort_only_is_left_alone_on_effort_model( + opus_4_5(), + json!({"max_tokens": 8192, "output_config": {"effort": "high"}}), + json!({"max_tokens": 8192, "output_config": {"effort": "high"}}) + )] + #[case::unsupported_effort_only_is_left_for_provider_normalization( + opus_4_5(), + json!({"max_tokens": 4096, "output_config": {"effort": "xhigh"}}), + json!({"max_tokens": 4096, "output_config": {"effort": "xhigh"}}) + )] + #[case::legacy_thinking_is_kept_beside_native_effort_on_effort_model( + opus_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}, "output_config": {"effort": "high"}}), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}, "output_config": {"effort": "high"}}) + )] + #[case::unsupported_xhigh_with_adaptive_thinking_falls_back_to_budget( + opus_4_5(), + claude_code_payload("xhigh", 64000), + json!({"max_tokens": 64000, "thinking": {"type": "enabled", "budget_tokens": 8192}}) + )] + #[case::unsupported_max_with_adaptive_thinking_falls_back_to_budget( + opus_4_5(), + claude_code_payload("max", 64000), + json!({"max_tokens": 64000, "thinking": {"type": "enabled", "budget_tokens": 16384}}) + )] + #[case::bare_adaptive_is_native_on_4_6( + sonnet_4_6(), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}}), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}}) + )] + #[case::adaptive_payload_is_native_on_4_6( + sonnet_4_6(), + claude_code_payload("high", 8192), + claude_code_payload("high", 8192) + )] + #[case::request_without_adaptive_interface_is_left_alone( + haiku_4_5(), + json!({"max_tokens": 1024}), + json!({"max_tokens": 1024}) + )] + fn adaptive_interface_is_reshaped_for_non_adaptive_models( + #[case] capabilities: AnthropicModelCapabilities, + #[case] input: Value, + #[case] expected: Value, + ) { + assert_eq!(translate(capabilities, input), Ok(request(expected))); + } + + #[rstest] + #[case::adaptive_downgraded_to_enabled_thinking( + haiku_4_5(), + claude_code_payload("medium", 8192), + 0.0, + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}}) + )] + #[case::bare_adaptive_downgraded_to_enabled_thinking( + haiku_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "adaptive"}}), + 0.0, + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}}) + )] + #[case::reasoning_effort_synthesized_enabled_thinking( + haiku_4_5(), + json!({"max_tokens": 8192, "reasoning_effort": "high"}), + 0.2, + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}}) + )] + #[case::above_one_with_enabled_thinking( + haiku_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}}), + 1.5, + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}}) + )] + #[case::native_effort_kept_on_effort_model( + opus_4_5(), + claude_code_payload("medium", 8192), + 0.0, + json!({"max_tokens": 8192, "output_config": {"effort": "medium"}}) + )] + #[case::effort_only_on_effort_model( + opus_4_5(), + json!({"max_tokens": 8192, "output_config": {"effort": "high"}}), + 0.0, + json!({"max_tokens": 8192, "output_config": {"effort": "high"}}) + )] + fn pinned_temperature_is_dropped_when_thinking_or_effort_survives_on_non_adaptive_model( + #[case] capabilities: AnthropicModelCapabilities, + #[case] input: Value, + #[case] temperature: f64, + #[case] expected: Value, + ) { + assert_eq!( + translate(capabilities, with_temperature(input, temperature)), + Ok(request(expected)) + ); + } + + #[rstest] + #[case::temperature_one_with_enabled_thinking( + haiku_4_5(), + claude_code_payload("medium", 8192), + 1.0, + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 2048}}) + )] + #[case::thinking_dropped_for_small_max_tokens( + haiku_4_5(), + claude_code_payload("medium", 512), + 0.0, + json!({"max_tokens": 512}) + )] + #[case::thinking_dropped_on_non_reasoning_model( + haiku_3_5(), + claude_code_payload("medium", 8192), + 0.0, + json!({"max_tokens": 8192}) + )] + #[case::disabled_thinking( + haiku_4_5(), + json!({"max_tokens": 8192, "thinking": {"type": "disabled"}}), + 0.0, + json!({"max_tokens": 8192, "thinking": {"type": "disabled"}}) + )] + #[case::no_thinking(haiku_4_5(), json!({"max_tokens": 8192}), 0.0, json!({"max_tokens": 8192}))] + #[case::output_config_without_effort( + haiku_4_5(), + json!({"max_tokens": 8192, "output_config": {"format": {"type": "json_schema"}}}), + 0.0, + json!({"max_tokens": 8192, "output_config": {"format": {"type": "json_schema"}}}) + )] + #[case::adaptive_model( + opus_4_7(), + claude_code_payload("medium", 8192), + 0.0, + claude_code_payload("medium", 8192) + )] + #[case::legacy_thinking_on_adaptive_model( + sonnet_4_6(), + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}}), + 0.0, + json!({"max_tokens": 8192, "thinking": {"type": "enabled", "budget_tokens": 4096}}) + )] + fn temperature_is_kept( + #[case] capabilities: AnthropicModelCapabilities, + #[case] input: Value, + #[case] temperature: f64, + #[case] expected: Value, + ) { + assert_eq!( + translate(capabilities, with_temperature(input, temperature)), + Ok(request(with_temperature(expected, temperature))) + ); + } + + #[rstest] + #[case::minimal("MINIMAL", ThinkingBudgets { minimal: 5000, ..ThinkingBudgets::default() })] + #[case::low("LOW", ThinkingBudgets { low: 5000, ..ThinkingBudgets::default() })] + #[case::medium("MEDIUM", ThinkingBudgets { medium: 5000, ..ThinkingBudgets::default() })] + #[case::high("HIGH", ThinkingBudgets { high: 5000, ..ThinkingBudgets::default() })] + #[case::xhigh("XHIGH", ThinkingBudgets { xhigh: 5000, ..ThinkingBudgets::default() })] + #[case::max("MAX", ThinkingBudgets { max: 5000, ..ThinkingBudgets::default() })] + fn each_tier_budget_reads_only_its_own_environment_override( + #[case] tier: &str, + #[case] expected: ThinkingBudgets, + ) { + assert_eq!(overridden_budgets(&[(tier, "5000")]), expected); + } + + #[rstest] + #[case::whitespace_is_trimmed(" 6000 ", 6000)] + #[case::unparseable_value_keeps_default("lots", 4096)] + fn environment_override_parsing(#[case] raw: &str, #[case] expected_high: u64) { + assert_eq!( + overridden_budgets(&[("HIGH", raw)]), + ThinkingBudgets { + high: expected_high, + ..ThinkingBudgets::default() + } + ); + } + + #[rstest] + #[case::reasoning_effort_uses_overridden_budget( + &[("HIGH", "6000")], + haiku_4_5(), + json!({"max_tokens": 32000, "reasoning_effort": "high"}), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 6000}}) + )] + #[case::minimal_override_below_min_budget_is_floored( + &[("MINIMAL", "512")], + haiku_4_5(), + json!({"max_tokens": 32000, "reasoning_effort": "minimal"}), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 1024}}) + )] + #[case::minimal_override_above_min_budget_is_used( + &[("MINIMAL", "2000")], + haiku_4_5(), + json!({"max_tokens": 32000, "reasoning_effort": "minimal"}), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 2000}}) + )] + #[case::adaptive_fallback_uses_overridden_medium_budget( + &[("MEDIUM", "3000")], + haiku_4_5(), + json!({"max_tokens": 32000, "thinking": {"type": "adaptive"}}), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 3000}}) + )] + #[case::legacy_bucket_below_overridden_high_budget( + &[("HIGH", "6000")], + opus_4_7(), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 5999}}), + json!({"max_tokens": 32000, "thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}}) + )] + #[case::legacy_bucket_at_overridden_high_budget( + &[("HIGH", "6000")], + opus_4_7(), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 6000}}), + json!({"max_tokens": 32000, "thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}}) + )] + #[case::legacy_bucket_below_overridden_xhigh_budget( + &[("XHIGH", "20000")], + opus_4_7(), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 19999}}), + json!({"max_tokens": 32000, "thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}}) + )] + #[case::legacy_bucket_at_overridden_medium_budget( + &[("MEDIUM", "3000")], + opus_4_7(), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 3000}}), + json!({"max_tokens": 32000, "thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}}) + )] + #[case::legacy_bucket_below_overridden_medium_budget( + &[("MEDIUM", "3000")], + opus_4_7(), + json!({"max_tokens": 32000, "thinking": {"type": "enabled", "budget_tokens": 2999}}), + json!({"max_tokens": 32000, "thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}}) + )] + fn translation_honors_budget_overrides( + #[case] overrides: &[(&str, &str)], + #[case] capabilities: AnthropicModelCapabilities, + #[case] input: Value, + #[case] expected: Value, + ) { + let context = ThinkingContext { + capabilities, + budgets: overridden_budgets(overrides), + }; + assert_eq!( + translate_thinking(request(input), &context), + Ok(request(expected)) + ); + } +} diff --git a/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/transformation.rs b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/transformation.rs index c791749ac6d..59280c04a70 100644 --- a/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/transformation.rs +++ b/litellm-rust/crates/llms/src/anthropic/experimental_pass_through/messages/transformation.rs @@ -1,9 +1,28 @@ -use crate::base_llm::{ - anthropic_messages::transformation::BaseAnthropicMessagesConfig, chat::transformation::Error, +use litellm_core_utils::settings::{Lookup, ProcessEnvironment}; +use litellm_types::llms::anthropic_messages::anthropic_request::AnthropicMessagesRequest; +use serde_json::{Map, Value, json}; + +use super::{ + headers::{authenticate, with_feature_betas}, + thinking::{ThinkingBudgets, ThinkingContext, translate_thinking}, +}; +use crate::{ + anthropic::common_utils::{ + AnthropicModelCapabilities, has_advisor_tool, strip_advisor_blocks, + strip_encrypted_reasoning_blocks, + }, + base_llm::{ + anthropic_messages::transformation::{ + BaseAnthropicMessagesConfig, Headers, MessagesTransformContext, + }, + chat::transformation::Error, + }, }; const ANTHROPIC_API_KEY_ENV: &str = "ANTHROPIC_API_KEY"; +const ANTHROPIC_AUTH_TOKEN_ENV: &str = "ANTHROPIC_AUTH_TOKEN"; const ANTHROPIC_API_BASE_ENV: &str = "ANTHROPIC_API_BASE"; +const ANTHROPIC_BASE_URL_ENV: &str = "ANTHROPIC_BASE_URL"; const DEFAULT_ANTHROPIC_API_BASE: &str = "https://api.anthropic.com"; const MESSAGES_PATH_SUFFIX: &str = "/v1/messages"; @@ -11,6 +30,26 @@ pub struct AnthropicMessagesConfig; pub const ANTHROPIC_MESSAGES_CONFIG: AnthropicMessagesConfig = AnthropicMessagesConfig; +impl MessagesTransformContext { + pub fn new(capabilities: AnthropicModelCapabilities, drop_params: bool) -> Self { + Self::with_lookup(capabilities, drop_params, &ProcessEnvironment) + } + + pub fn with_lookup( + capabilities: AnthropicModelCapabilities, + drop_params: bool, + env: &impl Lookup, + ) -> Self { + Self { + thinking: ThinkingContext { + capabilities, + budgets: ThinkingBudgets::from_lookup(env), + }, + drop_params, + } + } +} + impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig { fn get_complete_url( &self, @@ -21,6 +60,35 @@ impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig { Ok(complete_anthropic_url(api_base, env_lookup)) } + fn transform_anthropic_messages_request( + &self, + request: AnthropicMessagesRequest, + context: &MessagesTransformContext, + ) -> Result { + if request.max_tokens.is_none() { + return Err(Error::InvalidRequest( + "max_tokens is required for Anthropic /v1/messages API".to_string(), + )); + } + let request = drop_unsupported_params(request, context)?; + let request = translate_thinking(request, &context.thinking)?; + let context_management = request + .context_management + .as_ref() + .and_then(map_openai_context_management_to_anthropic) + .or_else(|| request.context_management.clone()); + let messages = if has_advisor_tool(request.tools.as_deref()) { + request.messages + } else { + strip_advisor_blocks(request.messages) + }; + Ok(AnthropicMessagesRequest { + messages: strip_encrypted_reasoning_blocks(messages), + context_management, + ..request + }) + } + fn resolve_api_key( &self, api_key: Option<&str>, @@ -28,6 +96,113 @@ impl BaseAnthropicMessagesConfig for AnthropicMessagesConfig { ) -> Result { resolve_anthropic_api_key(api_key, env_lookup).map_err(Error::from) } + + fn secret_names(&self) -> &'static [&'static str] { + &[ + ANTHROPIC_API_KEY_ENV, + ANTHROPIC_AUTH_TOKEN_ENV, + ANTHROPIC_API_BASE_ENV, + ANTHROPIC_BASE_URL_ENV, + ] + } + + fn authenticate( + &self, + headers: Headers, + api_key: Option<&str>, + env_lookup: &dyn Fn(&str) -> Option, + ) -> Result { + authenticate(headers, api_key, env_lookup).map_err(Error::from) + } + + fn request_headers(&self, headers: Headers, request: &AnthropicMessagesRequest) -> Headers { + with_feature_betas(headers, request) + } +} + +fn unsupported_param(model: &str, param: &str, value: &str, hint: &str) -> Error { + Error::InvalidRequest(format!( + "{model} does not support {param}={value}. {hint}To drop unsupported params, set `litellm.drop_params = True`." + )) +} + +fn drop_unsupported_params( + request: AnthropicMessagesRequest, + context: &MessagesTransformContext, +) -> Result { + let capabilities = &context.thinking.capabilities; + let model = request.model.clone(); + let reject = |param: &str, value: String, hint: &str| -> Result<(), Error> { + if context.drop_params { + return Ok(()); + } + Err(unsupported_param(&model, param, &value, hint)) + }; + let speed = match request.speed.as_deref() { + Some(speed) if !capabilities.supports_speed => { + reject("speed", format!("'{speed}'"), "")?; + None + } + _ => request.speed.clone(), + }; + if capabilities.supports_sampling_params { + return Ok(AnthropicMessagesRequest { speed, ..request }); + } + let temperature = match request.temperature { + Some(temperature) if temperature != 1.0 => { + reject( + "temperature", + json!(temperature).to_string(), + "Only temperature=1 is supported. ", + )?; + None + } + temperature => temperature, + }; + if let Some(top_p) = request.top_p { + reject("top_p", json!(top_p).to_string(), "")?; + } + if let Some(top_k) = request.top_k { + reject("top_k", json!(top_k).to_string(), "")?; + } + Ok(AnthropicMessagesRequest { + speed, + temperature, + top_p: None, + top_k: None, + ..request + }) +} + +pub fn map_openai_context_management_to_anthropic(context_management: &Value) -> Option { + match context_management { + Value::Object(edits) if edits.contains_key("edits") => Some(context_management.clone()), + Value::Array(entries) => { + let edits: Vec = entries + .iter() + .filter_map(Value::as_object) + .filter(|entry| entry.get("type").and_then(Value::as_str) == Some("compaction")) + .map(|entry| { + let trigger = entry.get("compact_threshold").and_then(Value::as_f64).map( + |threshold| json!({"type": "input_tokens", "value": threshold as i64}), + ); + let passthrough = entry + .iter() + .filter(|(key, _)| !matches!(key.as_str(), "type" | "compact_threshold")) + .map(|(key, value)| (key.clone(), value.clone())); + Value::Object( + [("type".to_string(), json!("compact_20260112"))] + .into_iter() + .chain(trigger.map(|trigger| ("trigger".to_string(), trigger))) + .chain(passthrough) + .collect::>(), + ) + }) + .collect(); + (!edits.is_empty()).then(|| json!({"edits": edits})) + } + _ => None, + } } pub fn non_empty(value: Option<&str>) -> Option<&str> { @@ -64,70 +239,619 @@ pub fn resolve_anthropic_api_base( api_base: Option<&str>, env_lookup: &dyn Fn(&str) -> Option, ) -> String { + let env = |name: &str| env_lookup(name).filter(|value| !value.trim().is_empty()); non_empty(api_base) .map(str::to_string) - .or_else(|| env_lookup(ANTHROPIC_API_BASE_ENV).filter(|value| !value.trim().is_empty())) + .or_else(|| env(ANTHROPIC_API_BASE_ENV)) + .or_else(|| env(ANTHROPIC_BASE_URL_ENV)) .unwrap_or_else(|| DEFAULT_ANTHROPIC_API_BASE.to_string()) } #[cfg(test)] mod tests { + use std::process::Command; + + use rstest::{fixture, rstest}; + use super::*; + use crate::anthropic::common_utils::{ENCRYPTED_REASONING_SIGNATURE_PREFIX, beta}; - #[test] - fn url_defaults_to_public_anthropic_endpoint() { + type Env = &'static [(&'static str, &'static str)]; + + const BOTH_BASE_ENVS: Env = &[ + (ANTHROPIC_API_BASE_ENV, "https://api-base.example.com"), + (ANTHROPIC_BASE_URL_ENV, "https://base-url.example.com"), + ]; + const API_KEY_ENV: Env = &[(ANTHROPIC_API_KEY_ENV, "sk-env")]; + const MISSING_API_KEY: &str = + "Missing Anthropic API Key - Set `api_key` or the ANTHROPIC_API_KEY environment variable"; + const LOW_BUDGET_ENV: &str = "DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET"; + const PROCESS_ENV_PROBE: &str = "LITELLM_MESSAGES_TRANSFORM_CONTEXT_PROBE"; + + fn merged(base: Value, fields: Value) -> Value { + Value::Object( + base.as_object() + .unwrap() + .clone() + .into_iter() + .chain(fields.as_object().unwrap().clone()) + .collect(), + ) + } + + fn body(fields: Value) -> Value { + merged( + json!({ + "model": "claude", + "max_tokens": 1024, + "messages": [{"role": "user", "content": "Hello"}] + }), + fields, + ) + } + + fn request(fields: Value) -> AnthropicMessagesRequest { + serde_json::from_value(body(fields)).unwrap() + } + + fn no_env(_: &str) -> Option { + None + } + + fn env(vars: Env) -> impl Fn(&str) -> Option { + move |name| { + vars.iter() + .find(|(key, _)| *key == name) + .map(|(_, value)| value.to_string()) + } + } + + fn headers(pairs: &[(&str, &str)]) -> Headers { + pairs + .iter() + .map(|(name, value)| (name.to_string(), value.to_string())) + .collect() + } + + fn transform( + fields: Value, + capabilities: AnthropicModelCapabilities, + drop_params: bool, + ) -> Result { + ANTHROPIC_MESSAGES_CONFIG + .transform_anthropic_messages_request( + request(fields), + &MessagesTransformContext::with_lookup(capabilities, drop_params, &no_env), + ) + .map(|transformed| serde_json::to_value(transformed).unwrap()) + } + + fn invalid(message: &str) -> Result { + Err(Error::InvalidRequest(message.to_string())) + } + + fn advisor_history() -> Value { + json!([ + {"role": "user", "content": "Build a worker pool."}, + {"role": "assistant", "content": [ + {"type": "text", "text": "Let me consult the advisor."}, + {"type": "server_tool_use", "id": "srvtoolu_abc123", "name": "advisor", "input": {}}, + {"type": "advisor_tool_result", "tool_use_id": "srvtoolu_abc123", "content": {"type": "advisor_result", "text": "Use channels."}}, + {"type": "text", "text": "Here is the implementation."} + ]} + ]) + } + + #[fixture] + fn unmapped() -> AnthropicModelCapabilities { + AnthropicModelCapabilities::default() + } + + #[fixture] + fn sampling_removed() -> AnthropicModelCapabilities { + AnthropicModelCapabilities { + supports_sampling_params: false, + ..Default::default() + } + } + + #[fixture] + fn fast_mode() -> AnthropicModelCapabilities { + AnthropicModelCapabilities { + supports_speed: true, + ..Default::default() + } + } + + #[rstest] + #[case::alone(json!({"max_tokens": null}))] + #[case::ahead_of_the_param_gate(json!({"max_tokens": null, "speed": "fast"}))] + fn missing_max_tokens_is_rejected(#[case] fields: Value, unmapped: AnthropicModelCapabilities) { assert_eq!( - complete_anthropic_url(None, &|_| None), - "https://api.anthropic.com/v1/messages" + transform(fields, unmapped, false), + invalid("max_tokens is required for Anthropic /v1/messages API") + ); + } + + #[rstest] + #[case::sampling_params_on_a_sampling_model( + unmapped(), + false, + json!({"temperature": 0.3, "top_p": 0.9, "top_k": 40}) + )] + #[case::sampling_params_on_a_sampling_model_under_drop_params( + unmapped(), + true, + json!({"temperature": 0.3, "top_p": 0.9, "top_k": 40}) + )] + #[case::unit_temperature_on_a_sampling_removed_model( + sampling_removed(), + false, + json!({"temperature": 1.0}) + )] + #[case::unit_temperature_on_a_sampling_removed_model_under_drop_params( + sampling_removed(), + true, + json!({"temperature": 1.0}) + )] + #[case::speed_on_a_fast_mode_model(fast_mode(), false, json!({"speed": "fast"}))] + #[case::speed_on_a_fast_mode_model_under_drop_params(fast_mode(), true, json!({"speed": "fast"}))] + #[case::native_context_management_edits(unmapped(), false, json!({"context_management": {"edits": [{ + "type": "clear_tool_uses_20250919", + "trigger": {"type": "input_tokens", "value": 30000}, + "keep": {"type": "tool_uses", "value": 3}, + "clear_at_least": {"type": "input_tokens", "value": 5000}, + "exclude_tools": ["web_search"], + "clear_tool_inputs": false + }]}}))] + #[case::first_party_billing_header_system_block(unmapped(), false, json!({"system": [ + {"type": "text", "text": "x-anthropic-billing-header: cc_version=1"}, + {"type": "text", "text": "real system prompt"} + ]}))] + #[case::anthropic_signed_reasoning_history(unmapped(), false, json!({"messages": [ + {"role": "user", "content": "Solve it."}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "plan", "signature": "EqQBCkYIAxgCIkA_anthropic_signed"}, + {"type": "redacted_thinking", "data": "EmwKAhgBEgy_anthropic_minted"}, + {"type": "text", "text": "The answer."} + ]} + ]}))] + #[case::advisor_history_alongside_the_advisor_tool(unmapped(), false, json!({ + "messages": advisor_history(), + "tools": [{"type": "advisor_20260301", "name": "advisor"}] + }))] + fn request_is_forwarded_unchanged( + #[case] capabilities: AnthropicModelCapabilities, + #[case] drop_params: bool, + #[case] fields: Value, + ) { + assert_eq!( + transform(fields.clone(), capabilities, drop_params), + Ok(body(fields)) + ); + } + + #[rstest] + #[case::temperature(sampling_removed(), json!({"temperature": 0.3}), json!({}))] + #[case::top_p(sampling_removed(), json!({"top_p": 0.9}), json!({}))] + #[case::top_k(sampling_removed(), json!({"top_k": 40}), json!({}))] + #[case::every_sampling_param_keeping_the_rest( + sampling_removed(), + json!({"temperature": 0.3, "top_p": 0.9, "top_k": 40, "stream": true}), + json!({"stream": true}) + )] + #[case::speed_on_a_sampling_model( + unmapped(), + json!({"speed": "fast", "temperature": 0.5}), + json!({"temperature": 0.5}) + )] + #[case::speed_on_a_sampling_removed_model( + sampling_removed(), + json!({"speed": "fast", "temperature": 1.0}), + json!({"temperature": 1.0}) + )] + fn removed_params_are_dropped_under_drop_params( + #[case] capabilities: AnthropicModelCapabilities, + #[case] fields: Value, + #[case] expected: Value, + ) { + assert_eq!(transform(fields, capabilities, true), Ok(body(expected))); + } + + #[rstest] + #[case::temperature( + sampling_removed(), + json!({"temperature": 0.3}), + "claude does not support temperature=0.3. Only temperature=1 is supported. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::temperature_just_below_one( + sampling_removed(), + json!({"temperature": 0.99}), + "claude does not support temperature=0.99. Only temperature=1 is supported. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::whole_number_temperature_keeps_its_decimal( + sampling_removed(), + json!({"temperature": 2.0}), + "claude does not support temperature=2.0. Only temperature=1 is supported. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::top_p( + sampling_removed(), + json!({"top_p": 0.9}), + "claude does not support top_p=0.9. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::top_k( + sampling_removed(), + json!({"top_k": 5}), + "claude does not support top_k=5. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::top_k_next_to_unit_temperature( + sampling_removed(), + json!({"temperature": 1.0, "top_k": 5}), + "claude does not support top_k=5. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::temperature_ahead_of_top_k( + sampling_removed(), + json!({"temperature": 0.5, "top_k": 5}), + "claude does not support temperature=0.5. Only temperature=1 is supported. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::top_p_ahead_of_top_k( + sampling_removed(), + json!({"top_p": 0.9, "top_k": 5}), + "claude does not support top_p=0.9. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::speed( + unmapped(), + json!({"speed": "fast"}), + "claude does not support speed='fast'. To drop unsupported params, set `litellm.drop_params = True`." + )] + #[case::speed_ahead_of_sampling_params( + sampling_removed(), + json!({"speed": "fast", "temperature": 0.5}), + "claude does not support speed='fast'. To drop unsupported params, set `litellm.drop_params = True`." + )] + fn removed_params_are_rejected_without_drop_params( + #[case] capabilities: AnthropicModelCapabilities, + #[case] fields: Value, + #[case] message: &str, + ) { + assert_eq!(transform(fields, capabilities, false), invalid(message)); + } + + #[rstest] + #[case::compaction_threshold( + json!([{"type": "compaction", "compact_threshold": 200000}]), + Some(json!({"edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 200000}}]})) + )] + #[case::other_keys_pass_through( + json!([{"type": "compaction", "compact_threshold": 150000, "instructions": "Focus on preserving code snippets"}]), + Some(json!({"edits": [{ + "type": "compact_20260112", + "trigger": {"type": "input_tokens", "value": 150000}, + "instructions": "Focus on preserving code snippets" + }]})) + )] + #[case::float_threshold_is_truncated( + json!([{"type": "compaction", "compact_threshold": 150000.9}]), + Some(json!({"edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}]})) + )] + #[case::compaction_without_threshold( + json!([{"type": "compaction"}]), + Some(json!({"edits": [{"type": "compact_20260112"}]})) + )] + #[case::non_numeric_threshold_is_dropped( + json!([{"type": "compaction", "compact_threshold": "150000"}]), + Some(json!({"edits": [{"type": "compact_20260112"}]})) + )] + #[case::non_object_entries_are_skipped( + json!([42, "compaction", null, [], {"type": "compaction", "compact_threshold": 1000}]), + Some(json!({"edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 1000}}]})) + )] + #[case::only_compaction_entries_are_mapped_in_order( + json!([ + {"type": "compaction", "compact_threshold": 1000}, + {"type": "other", "compact_threshold": 5}, + {"type": "compaction", "instructions": "second"} + ]), + Some(json!({"edits": [ + {"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 1000}}, + {"type": "compact_20260112", "instructions": "second"} + ]})) + )] + #[case::list_without_compaction(json!([{"type": "other"}]), None)] + #[case::empty_list(json!([]), None)] + #[case::anthropic_edits_pass_through( + json!({"edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}]}), + Some(json!({"edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 150000}}]})) + )] + #[case::object_without_edits(json!({"type": "compaction"}), None)] + #[case::scalar(json!("compaction"), None)] + fn openai_context_management_maps_to_anthropic_edits( + #[case] context_management: Value, + #[case] expected: Option, + ) { + assert_eq!( + map_openai_context_management_to_anthropic(&context_management), + expected + ); + } + + #[rstest] + #[case::openai_list_is_mapped( + json!([{"type": "compaction", "compact_threshold": 200000}]), + json!({"edits": [{"type": "compact_20260112", "trigger": {"type": "input_tokens", "value": 200000}}]}) + )] + #[case::unmappable_list_is_kept(json!([{"type": "other"}]), json!([{"type": "other"}]))] + #[case::unmappable_object_is_kept(json!({"type": "other"}), json!({"type": "other"}))] + fn context_management_reaches_the_wire( + #[case] context_management: Value, + #[case] expected: Value, + unmapped: AnthropicModelCapabilities, + ) { + assert_eq!( + transform( + json!({"context_management": context_management}), + unmapped, + false + ), + Ok(body(json!({"context_management": expected}))) + ); + } + + #[rstest] + #[case::without_tools(json!({}))] + #[case::with_only_other_tools(json!({"tools": [{"name": "get_weather", "input_schema": {"type": "object"}}]}))] + fn advisor_history_is_stripped_without_the_advisor_tool( + #[case] tools: Value, + unmapped: AnthropicModelCapabilities, + ) { + let stripped = json!([ + {"role": "user", "content": "Build a worker pool."}, + {"role": "assistant", "content": [ + {"type": "text", "text": "Let me consult the advisor."}, + {"type": "text", "text": "Here is the implementation."} + ]} + ]); + assert_eq!( + transform( + merged(tools.clone(), json!({"messages": advisor_history()})), + unmapped, + false + ), + Ok(body(merged(tools, json!({"messages": stripped})))) + ); + } + + #[rstest] + fn bridge_minted_reasoning_is_stripped_from_the_wire(unmapped: AnthropicModelCapabilities) { + let messages = json!([ + {"role": "user", "content": "Solve it."}, + {"role": "assistant", "content": [ + {"type": "thinking", "thinking": "plan", "signature": format!("{ENCRYPTED_REASONING_SIGNATURE_PREFIX}gAAAA_1")}, + {"type": "redacted_thinking", "data": format!("{ENCRYPTED_REASONING_SIGNATURE_PREFIX}gAAAA_2")}, + {"type": "text", "text": "The answer."} + ]}, + {"role": "user", "content": "And the next one?"} + ]); + assert_eq!( + transform(json!({"messages": messages}), unmapped, false), + Ok(body(json!({"messages": [ + {"role": "user", "content": "Solve it."}, + {"role": "assistant", "content": [{"type": "text", "text": "The answer."}]}, + {"role": "user", "content": "And the next one?"} + ]}))) ); } #[test] - fn url_appends_messages_suffix_to_custom_base() { + fn thinking_is_translated_with_the_context_budgets() { + let context = MessagesTransformContext::with_lookup( + AnthropicModelCapabilities { + supports_reasoning: true, + ..Default::default() + }, + false, + &env(&[(LOW_BUDGET_ENV, "2000")]), + ); + let transformed = ANTHROPIC_MESSAGES_CONFIG + .transform_anthropic_messages_request( + request(json!({"max_tokens": 4096, "reasoning_effort": "low"})), + &context, + ) + .map(|transformed| serde_json::to_value(transformed).unwrap()); assert_eq!( - complete_anthropic_url(Some("https://proxy.internal"), &|_| None), - "https://proxy.internal/v1/messages" + transformed, + Ok(body(json!({ + "max_tokens": 4096, + "thinking": {"type": "enabled", "budget_tokens": 2000} + }))) ); } #[test] - fn url_leaves_complete_messages_endpoint_untouched() { + fn new_reads_thinking_budgets_from_the_process_environment() { + if std::env::var_os(PROCESS_ENV_PROBE).is_some() { + assert_eq!( + MessagesTransformContext::new(sampling_removed(), true), + MessagesTransformContext { + thinking: ThinkingContext { + capabilities: sampling_removed(), + budgets: ThinkingBudgets { + low: 2000, + ..ThinkingBudgets::default() + }, + }, + drop_params: true, + } + ); + return; + } + let (_, test_path) = concat!( + module_path!(), + "::new_reads_thinking_budgets_from_the_process_environment" + ) + .split_once("::") + .unwrap(); + let other_tiers = ["MINIMAL", "MEDIUM", "HIGH", "XHIGH", "MAX"] + .map(|tier| format!("DEFAULT_REASONING_EFFORT_{tier}_THINKING_BUDGET")); + let output = other_tiers + .iter() + .fold( + Command::new(std::env::current_exe().unwrap()), + |mut command, name| { + command.env_remove(name); + command + }, + ) + .args([test_path, "--exact"]) + .env(PROCESS_ENV_PROBE, "1") + .env(LOW_BUDGET_ENV, "2000") + .output() + .unwrap(); + let stdout = String::from_utf8_lossy(&output.stdout); + assert!( + output.status.success() && stdout.contains("1 passed"), + "{stdout}{}", + String::from_utf8_lossy(&output.stderr) + ); + } + + #[rstest] + #[case::public_endpoint_by_default(None, &[], "https://api.anthropic.com")] + #[case::explicit_api_base_beats_env( + Some("https://explicit.example.com"), + BOTH_BASE_ENVS, + "https://explicit.example.com" + )] + #[case::explicit_api_base_is_trimmed( + Some(" https://explicit.example.com "), + &[], + "https://explicit.example.com" + )] + #[case::blank_api_base_falls_back_to_env( + Some(" "), + BOTH_BASE_ENVS, + "https://api-base.example.com" + )] + #[case::api_base_env_beats_base_url_env(None, BOTH_BASE_ENVS, "https://api-base.example.com")] + #[case::base_url_env_without_api_base_env( + None, + &[(ANTHROPIC_BASE_URL_ENV, "https://base-url.example.com")], + "https://base-url.example.com" + )] + #[case::blank_api_base_env_falls_back_to_base_url_env( + None, + &[(ANTHROPIC_API_BASE_ENV, " \t "), (ANTHROPIC_BASE_URL_ENV, "https://base-url.example.com")], + "https://base-url.example.com" + )] + #[case::blank_envs_fall_back_to_public_endpoint( + None, + &[(ANTHROPIC_API_BASE_ENV, ""), (ANTHROPIC_BASE_URL_ENV, " ")], + "https://api.anthropic.com" + )] + fn api_base_resolution( + #[case] api_base: Option<&str>, + #[case] vars: Env, + #[case] expected: &str, + ) { + assert_eq!(resolve_anthropic_api_base(api_base, &env(vars)), expected); + } + + #[rstest] + #[case::public_endpoint(None, &[], "https://api.anthropic.com/v1/messages")] + #[case::base_url_env( + None, + &[(ANTHROPIC_BASE_URL_ENV, "https://custom.example.com")], + "https://custom.example.com/v1/messages" + )] + #[case::custom_base(Some("https://proxy.internal"), &[], "https://proxy.internal/v1/messages")] + #[case::trailing_slash(Some("https://proxy.internal/"), &[], "https://proxy.internal/v1/messages")] + #[case::complete_endpoint( + Some("https://proxy.internal/v1/messages"), + &[], + "https://proxy.internal/v1/messages" + )] + #[case::complete_endpoint_with_trailing_slash( + Some("https://proxy.internal/v1/messages/"), + &[], + "https://proxy.internal/v1/messages" + )] + fn complete_url_ends_in_the_messages_path( + #[case] api_base: Option<&str>, + #[case] vars: Env, + #[case] expected: &str, + ) { assert_eq!( - complete_anthropic_url(Some("https://proxy.internal/v1/messages"), &|_| None), - "https://proxy.internal/v1/messages" + ANTHROPIC_MESSAGES_CONFIG.get_complete_url(api_base, "claude", &env(vars)), + Ok(expected.to_string()) + ); + } + + #[rstest] + #[case::param_beats_env(Some("sk-param"), API_KEY_ENV, Ok("sk-param"))] + #[case::param_is_trimmed(Some(" sk-param "), &[], Ok("sk-param"))] + #[case::blank_param_falls_back_to_env(Some(" "), API_KEY_ENV, Ok("sk-env"))] + #[case::env_without_param(None, API_KEY_ENV, Ok("sk-env"))] + #[case::blank_env_is_missing(None, &[(ANTHROPIC_API_KEY_ENV, " ")], Err(MISSING_API_KEY))] + #[case::nothing_is_missing(None, &[], Err(MISSING_API_KEY))] + fn api_key_resolution( + #[case] api_key: Option<&str>, + #[case] vars: Env, + #[case] expected: Result<&str, &str>, + ) { + assert_eq!( + resolve_anthropic_api_key(api_key, &env(vars)).map_err(|error| error.to_string()), + expected.map(str::to_string).map_err(str::to_string) ); } #[test] - fn url_falls_back_to_env_base() { - let with_env = |key: &str| { - (key == ANTHROPIC_API_BASE_ENV).then(|| "https://env.anthropic".to_string()) - }; + fn config_reports_a_missing_key_as_an_auth_error() { assert_eq!( - complete_anthropic_url(Some(" "), &with_env), - "https://env.anthropic/v1/messages" + ANTHROPIC_MESSAGES_CONFIG.resolve_api_key(None, &no_env), + Err(Error::Auth(litellm_auth::Error::MissingApiKey { + provider: "Anthropic", + environment_variable: ANTHROPIC_API_KEY_ENV, + })) ); } #[test] - fn api_key_prefers_param_then_env_then_errors() { + fn config_authenticates_with_the_anthropic_auth_token() { assert_eq!( - resolve_anthropic_api_key(Some("sk-param"), &|_| None).unwrap(), - "sk-param" + ANTHROPIC_MESSAGES_CONFIG.authenticate( + vec![], + None, + &env(&[("ANTHROPIC_AUTH_TOKEN", "auth-token")]) + ), + Ok(headers(&[("authorization", "Bearer auth-token")])) ); - let with_env = |key: &str| (key == ANTHROPIC_API_KEY_ENV).then(|| "sk-env".to_string()); + } + + #[test] + fn config_requests_the_betas_the_request_features_need() { assert_eq!( - resolve_anthropic_api_key(Some(" "), &with_env).unwrap(), - "sk-env" - ); - assert_eq!( - resolve_anthropic_api_key(None, &|_| None) - .expect_err("missing key") - .to_string(), - "Missing Anthropic API Key - Set `api_key` or the ANTHROPIC_API_KEY environment variable" + ANTHROPIC_MESSAGES_CONFIG.request_headers( + headers(&[("x-api-key", "sk")]), + &request(json!({"speed": "fast"})) + ), + headers(&[ + ("x-api-key", "sk"), + ("anthropic-beta", beta::FAST_MODE_2026_02_01) + ]) ); } + #[rstest] + #[case::absent(None, None)] + #[case::blank(Some(" \t "), None)] + #[case::padded(Some(" value "), Some("value"))] + fn non_empty_trims_and_drops_blank_values( + #[case] value: Option<&str>, + #[case] expected: Option<&str>, + ) { + assert_eq!(non_empty(value), expected); + } + #[test] fn auth_strategy_and_default_headers_match_anthropic() { assert_eq!( @@ -142,4 +866,26 @@ mod tests { ] ); } + + #[test] + fn secret_names_cover_every_credential_and_base_lookup() { + let requested = std::cell::RefCell::new(Vec::::new()); + let record = |name: &str| -> Option { + requested.borrow_mut().push(name.to_string()); + None + }; + let _ = ANTHROPIC_MESSAGES_CONFIG.authenticate(Vec::new(), None, &record); + let _ = ANTHROPIC_MESSAGES_CONFIG.get_complete_url(None, "claude", &record); + let requested = requested.into_inner(); + assert!(!requested.is_empty()); + let undeclared: Vec<&String> = requested + .iter() + .filter(|name| { + !ANTHROPIC_MESSAGES_CONFIG + .secret_names() + .contains(&name.as_str()) + }) + .collect(); + assert_eq!(undeclared, Vec::<&String>::new()); + } } diff --git a/litellm-rust/crates/llms/src/anthropic/mod.rs b/litellm-rust/crates/llms/src/anthropic/mod.rs index d181ceaca3c..755bc7d1907 100644 --- a/litellm-rust/crates/llms/src/anthropic/mod.rs +++ b/litellm-rust/crates/llms/src/anthropic/mod.rs @@ -1,5 +1,6 @@ pub mod batches; pub mod chat; +pub mod common_utils; pub mod count_tokens; pub mod experimental_pass_through; diff --git a/litellm-rust/crates/llms/src/azure_ai/anthropic/messages_transformation.rs b/litellm-rust/crates/llms/src/azure_ai/anthropic/messages_transformation.rs index 99f55f18afc..c409f7f687e 100644 --- a/litellm-rust/crates/llms/src/azure_ai/anthropic/messages_transformation.rs +++ b/litellm-rust/crates/llms/src/azure_ai/anthropic/messages_transformation.rs @@ -4,14 +4,15 @@ use litellm_types::llms::anthropic_messages::{ }, anthropic_response::AnthropicMessagesResponse, }; -use serde_json::{Map, Value}; use crate::{ anthropic::experimental_pass_through::messages::transformation::{ ANTHROPIC_MESSAGES_CONFIG, AnthropicMessagesConfig, non_empty, }, base_llm::{ - anthropic_messages::transformation::{BaseAnthropicMessagesConfig, MessagesAuthStrategy}, + anthropic_messages::transformation::{ + BaseAnthropicMessagesConfig, Headers, MessagesAuthStrategy, MessagesTransformContext, + }, chat::transformation::Error, }, }; @@ -21,7 +22,6 @@ const AZURE_API_BASE_ENV: &str = "AZURE_API_BASE"; const ANTHROPIC_PATH_SEGMENT: &str = "/anthropic"; const MESSAGES_PATH_SUFFIX: &str = "/v1/messages"; const SYSTEM_ROLE: &str = "system"; -const TEXT_BLOCK_TYPE: &str = "text"; pub struct AzureAnthropicMessagesConfig { anthropic: AnthropicMessagesConfig, @@ -45,6 +45,7 @@ impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig { fn transform_anthropic_messages_request( &self, request: AnthropicMessagesRequest, + context: &MessagesTransformContext, ) -> Result { let mut request = fold_system_role_messages(request); if let Some(system) = request.system.as_mut() { @@ -54,7 +55,8 @@ impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig { .messages .iter_mut() .for_each(strip_scope_from_message); - self.anthropic.transform_anthropic_messages_request(request) + self.anthropic + .transform_anthropic_messages_request(request, context) } fn transform_anthropic_messages_response( @@ -74,6 +76,10 @@ impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig { resolve_azure_api_key(api_key, env_lookup) } + fn secret_names(&self) -> &'static [&'static str] { + &[AZURE_API_KEY_ENV, AZURE_API_BASE_ENV] + } + fn auth_strategy(&self) -> MessagesAuthStrategy { self.anthropic.auth_strategy() } @@ -85,6 +91,10 @@ impl BaseAnthropicMessagesConfig for AzureAnthropicMessagesConfig { fn default_headers(&self) -> &'static [(&'static str, &'static str)] { self.anthropic.default_headers() } + + fn request_headers(&self, headers: Headers, request: &AnthropicMessagesRequest) -> Headers { + self.anthropic.request_headers(headers, request) + } } pub fn resolve_azure_api_key( @@ -143,17 +153,7 @@ fn strip_scope_from_message(message: &mut AnthropicMessage) { } fn text_content_block(text: String) -> ContentBlock { - let extra = Map::from_iter([ - ( - "type".to_string(), - Value::String(TEXT_BLOCK_TYPE.to_string()), - ), - ("text".to_string(), Value::String(text)), - ]); - ContentBlock { - cache_control: None, - extra, - } + ContentBlock::text(text) } fn content_into_blocks(content: MessageContent) -> Vec { @@ -202,6 +202,7 @@ mod tests { use serde_json::json; use super::*; + use crate::anthropic::common_utils::AnthropicModelCapabilities; fn request_from(value: serde_json::Value) -> AnthropicMessagesRequest { serde_json::from_value(value).expect("valid request") @@ -346,7 +347,7 @@ mod tests { let transformed = to_value( AZURE_ANTHROPIC_MESSAGES_CONFIG - .transform_anthropic_messages_request(request) + .transform_anthropic_messages_request(request, &MessagesTransformContext::default()) .expect("request transforms"), ); @@ -373,10 +374,13 @@ mod tests { "messages": [{"role": "user", "content": "hi"}] })); let once = AZURE_ANTHROPIC_MESSAGES_CONFIG - .transform_anthropic_messages_request(request) + .transform_anthropic_messages_request(request, &MessagesTransformContext::default()) .expect("request transforms"); let twice = AZURE_ANTHROPIC_MESSAGES_CONFIG - .transform_anthropic_messages_request(once.clone()) + .transform_anthropic_messages_request( + once.clone(), + &MessagesTransformContext::default(), + ) .expect("request transforms"); assert_eq!(once, twice); assert_eq!(to_value(once)["system"], json!("plain string system")); @@ -408,9 +412,21 @@ mod tests { "inference_geo": "us", "litellm_metadata": {"trace": "abc"} }); + let context = MessagesTransformContext::with_lookup( + AnthropicModelCapabilities { + supports_reasoning: true, + supports_adaptive_thinking: true, + supports_legacy_thinking: true, + supports_output_config: true, + supports_speed: true, + ..Default::default() + }, + false, + &|_: &str| None, + ); let transformed = to_value( AZURE_ANTHROPIC_MESSAGES_CONFIG - .transform_anthropic_messages_request(request_from(body.clone())) + .transform_anthropic_messages_request(request_from(body.clone()), &context) .expect("request transforms"), ); assert_eq!(transformed, body); @@ -430,7 +446,7 @@ mod tests { let transformed = to_value( AZURE_ANTHROPIC_MESSAGES_CONFIG - .transform_anthropic_messages_request(request) + .transform_anthropic_messages_request(request, &MessagesTransformContext::default()) .expect("request transforms"), ); @@ -460,7 +476,7 @@ mod tests { let transformed = to_value( AZURE_ANTHROPIC_MESSAGES_CONFIG - .transform_anthropic_messages_request(request) + .transform_anthropic_messages_request(request, &MessagesTransformContext::default()) .expect("request transforms"), ); @@ -485,9 +501,21 @@ mod tests { {"role": "assistant", "content": "hello"} ] }); + let context = MessagesTransformContext::with_lookup( + AnthropicModelCapabilities { + supports_reasoning: true, + supports_adaptive_thinking: true, + supports_legacy_thinking: true, + supports_output_config: true, + supports_speed: true, + ..Default::default() + }, + false, + &|_: &str| None, + ); let transformed = to_value( AZURE_ANTHROPIC_MESSAGES_CONFIG - .transform_anthropic_messages_request(request_from(body.clone())) + .transform_anthropic_messages_request(request_from(body.clone()), &context) .expect("request transforms"), ); assert_eq!(transformed, body); @@ -500,6 +528,57 @@ mod tests { assert!(err.is_data()); } + #[rstest::rstest] + #[case::compact_context_management_edit( + json!({"context_management": {"edits": [{"type": "compact_20260112"}]}}), + &[], + &[("x-api-key", "k"), ("anthropic-beta", "compact-2026-01-12")] + )] + #[case::forwarded_beta_merged_with_structured_output( + json!({"output_config": {"format": {"type": "json_schema"}}}), + &[("anthropic-beta", "web-search-2025-03-05")], + &[("x-api-key", "k"), ("anthropic-beta", "structured-outputs-2025-11-13,web-search-2025-03-05")] + )] + #[case::no_feature_needs_a_beta(json!({}), &[], &[("x-api-key", "k")])] + fn request_headers_carry_the_anthropic_feature_betas( + #[case] fields: serde_json::Value, + #[case] forwarded: &[(&str, &str)], + #[case] expected: &[(&str, &str)], + ) { + let pairs = |pairs: &[(&str, &str)]| -> Vec<(String, String)> { + pairs + .iter() + .map(|(name, value)| (name.to_string(), value.to_string())) + .collect() + }; + let serde_json::Value::Object(fields) = fields else { + panic!("case fields are an object") + }; + let request = request_from(serde_json::Value::Object( + [ + ("model".to_string(), json!("claude-sonnet")), + ("max_tokens".to_string(), json!(16)), + ( + "messages".to_string(), + json!([{"role": "user", "content": "hi"}]), + ), + ] + .into_iter() + .chain(fields) + .collect(), + )); + assert_eq!( + AZURE_ANTHROPIC_MESSAGES_CONFIG.request_headers( + pairs(&[("x-api-key", "k")]) + .into_iter() + .chain(pairs(forwarded)) + .collect(), + &request + ), + pairs(expected) + ); + } + #[test] fn transform_response_passes_through() { let response: AnthropicMessagesResponse = serde_json::from_value(json!({ @@ -521,4 +600,26 @@ mod tests { assert_eq!(value["stop_sequence"], json!(null)); assert_eq!(value["content"][0]["text"], json!("hello")); } + + #[test] + fn secret_names_cover_every_credential_and_base_lookup() { + let requested = std::cell::RefCell::new(Vec::::new()); + let record = |name: &str| -> Option { + requested.borrow_mut().push(name.to_string()); + None + }; + let _ = AZURE_ANTHROPIC_MESSAGES_CONFIG.authenticate(Vec::new(), None, &record); + let _ = AZURE_ANTHROPIC_MESSAGES_CONFIG.get_complete_url(None, "claude", &record); + let requested = requested.into_inner(); + assert!(!requested.is_empty()); + let undeclared: Vec<&String> = requested + .iter() + .filter(|name| { + !AZURE_ANTHROPIC_MESSAGES_CONFIG + .secret_names() + .contains(&name.as_str()) + }) + .collect(); + assert_eq!(undeclared, Vec::<&String>::new()); + } } diff --git a/litellm-rust/crates/llms/src/base_llm/anthropic_messages/transformation.rs b/litellm-rust/crates/llms/src/base_llm/anthropic_messages/transformation.rs index 5b4afb601d2..8db14687214 100644 --- a/litellm-rust/crates/llms/src/base_llm/anthropic_messages/transformation.rs +++ b/litellm-rust/crates/llms/src/base_llm/anthropic_messages/transformation.rs @@ -1,8 +1,14 @@ +use litellm_http::request::{has_bearer_auth, has_header}; use litellm_types::llms::anthropic_messages::{ anthropic_request::AnthropicMessagesRequest, anthropic_response::AnthropicMessagesResponse, }; -use crate::base_llm::chat::transformation::Error; +use crate::{ + anthropic::experimental_pass_through::messages::thinking::ThinkingContext, + base_llm::chat::transformation::Error, +}; + +pub type Headers = Vec<(String, String)>; #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum MessagesAuthStrategy { @@ -19,6 +25,12 @@ impl MessagesAuthStrategy { } } +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct MessagesTransformContext { + pub thinking: ThinkingContext, + pub drop_params: bool, +} + pub trait BaseAnthropicMessagesConfig: Sync { fn get_complete_url( &self, @@ -30,6 +42,7 @@ pub trait BaseAnthropicMessagesConfig: Sync { fn transform_anthropic_messages_request( &self, request: AnthropicMessagesRequest, + _context: &MessagesTransformContext, ) -> Result { Ok(request) } @@ -48,6 +61,8 @@ pub trait BaseAnthropicMessagesConfig: Sync { env_lookup: &dyn Fn(&str) -> Option, ) -> Result; + fn secret_names(&self) -> &'static [&'static str]; + fn auth_strategy(&self) -> MessagesAuthStrategy { MessagesAuthStrategy::Header("x-api-key") } @@ -56,10 +71,225 @@ pub trait BaseAnthropicMessagesConfig: Sync { false } + fn authenticate( + &self, + headers: Headers, + api_key: Option<&str>, + env_lookup: &dyn Fn(&str) -> Option, + ) -> Result { + let strategy = self.auth_strategy(); + if has_header(&headers, strategy.header_name()) + || (self.accepts_bearer_auth() && has_bearer_auth(&headers)) + { + return Ok(headers); + } + let api_key = self.resolve_api_key(api_key, env_lookup)?; + let auth_header = match strategy { + MessagesAuthStrategy::Bearer => { + ("authorization".to_string(), format!("Bearer {api_key}")) + } + MessagesAuthStrategy::Header(name) => (name.to_string(), api_key), + }; + Ok(headers.into_iter().chain([auth_header]).collect()) + } + fn default_headers(&self) -> &'static [(&'static str, &'static str)] { &[ ("anthropic-version", "2023-06-01"), ("content-type", "application/json"), ] } + + fn request_headers(&self, headers: Headers, _request: &AnthropicMessagesRequest) -> Headers { + headers + } +} + +#[cfg(test)] +mod tests { + use rstest::rstest; + + use super::*; + + const X_API_KEY: MessagesAuthStrategy = MessagesAuthStrategy::Header("x-api-key"); + + struct StubConfig { + strategy: MessagesAuthStrategy, + accepts_bearer: bool, + } + + impl BaseAnthropicMessagesConfig for StubConfig { + fn secret_names(&self) -> &'static [&'static str] { + &[] + } + + fn get_complete_url( + &self, + _api_base: Option<&str>, + _model: &str, + _env_lookup: &dyn Fn(&str) -> Option, + ) -> Result { + Ok(String::new()) + } + + fn resolve_api_key( + &self, + api_key: Option<&str>, + _env_lookup: &dyn Fn(&str) -> Option, + ) -> Result { + api_key + .map(str::to_string) + .ok_or(Error::MissingField("api_key")) + } + + fn auth_strategy(&self) -> MessagesAuthStrategy { + self.strategy + } + + fn accepts_bearer_auth(&self) -> bool { + self.accepts_bearer + } + } + + struct DefaultsConfig; + + impl BaseAnthropicMessagesConfig for DefaultsConfig { + fn secret_names(&self) -> &'static [&'static str] { + &[] + } + + fn get_complete_url( + &self, + _api_base: Option<&str>, + _model: &str, + _env_lookup: &dyn Fn(&str) -> Option, + ) -> Result { + Ok(String::new()) + } + + fn resolve_api_key( + &self, + api_key: Option<&str>, + _env_lookup: &dyn Fn(&str) -> Option, + ) -> Result { + api_key + .map(str::to_string) + .ok_or(Error::MissingField("api_key")) + } + } + + #[test] + fn default_config_adds_its_key_next_to_a_forwarded_bearer() { + assert_eq!( + DefaultsConfig.authenticate( + headers(&[("authorization", "Bearer forwarded")]), + Some("sk"), + &|_| None + ), + Ok(headers(&[ + ("authorization", "Bearer forwarded"), + ("x-api-key", "sk") + ])) + ); + } + + #[test] + fn default_request_headers_are_the_given_headers() { + let request: AnthropicMessagesRequest = serde_json::from_value(serde_json::json!({ + "model": "claude", + "max_tokens": 16, + "speed": "fast", + "messages": [{"role": "user", "content": "hi"}] + })) + .unwrap(); + assert_eq!( + DefaultsConfig.request_headers(headers(&[("x-api-key", "sk")]), &request), + headers(&[("x-api-key", "sk")]) + ); + } + + fn headers(pairs: &[(&str, &str)]) -> Headers { + pairs + .iter() + .map(|(name, value)| (name.to_string(), value.to_string())) + .collect() + } + + #[rstest] + #[case::own_header_is_kept( + X_API_KEY, + false, + headers(&[("x-api-key", "forwarded")]), + None, + Ok(headers(&[("x-api-key", "forwarded")])) + )] + #[case::own_header_in_any_casing_is_kept( + X_API_KEY, + false, + headers(&[("X-Api-Key", "forwarded")]), + None, + Ok(headers(&[("X-Api-Key", "forwarded")])) + )] + #[case::accepted_bearer_is_kept( + X_API_KEY, + true, + headers(&[("authorization", "Bearer forwarded")]), + None, + Ok(headers(&[("authorization", "Bearer forwarded")])) + )] + #[case::bearer_the_provider_does_not_accept_gets_the_key_too( + X_API_KEY, + false, + headers(&[("authorization", "Bearer forwarded")]), + Some("sk"), + Ok(headers(&[("authorization", "Bearer forwarded"), ("x-api-key", "sk")])) + )] + #[case::blank_bearer_gets_the_key( + X_API_KEY, + true, + headers(&[("authorization", "Bearer ")]), + Some("sk"), + Ok(headers(&[("authorization", "Bearer "), ("x-api-key", "sk")])) + )] + #[case::key_goes_in_the_provider_header( + X_API_KEY, + false, + headers(&[("content-type", "application/json")]), + Some("sk"), + Ok(headers(&[("content-type", "application/json"), ("x-api-key", "sk")])) + )] + #[case::key_goes_in_a_bearer( + MessagesAuthStrategy::Bearer, + false, + headers(&[]), + Some("sk"), + Ok(headers(&[("authorization", "Bearer sk")])) + )] + #[case::bearer_strategy_keeps_a_forwarded_authorization( + MessagesAuthStrategy::Bearer, + false, + headers(&[("authorization", "Bearer forwarded")]), + None, + Ok(headers(&[("authorization", "Bearer forwarded")])) + )] + #[case::missing_key_is_an_error( + X_API_KEY, + false, + headers(&[]), + None, + Err(Error::MissingField("api_key")) + )] + fn default_authenticate_applies_the_key_unless_a_credential_is_forwarded( + #[case] strategy: MessagesAuthStrategy, + #[case] accepts_bearer: bool, + #[case] forwarded: Headers, + #[case] api_key: Option<&str>, + #[case] expected: Result, + ) { + let config = StubConfig { + strategy, + accepts_bearer, + }; + assert_eq!(config.authenticate(forwarded, api_key, &|_| None), expected); + } } diff --git a/litellm-rust/crates/python-bridge/src/routes/messages/host.rs b/litellm-rust/crates/python-bridge/src/routes/messages/host.rs index 1a9b170f661..9d97094aeda 100644 --- a/litellm-rust/crates/python-bridge/src/routes/messages/host.rs +++ b/litellm-rust/crates/python-bridge/src/routes/messages/host.rs @@ -2,9 +2,11 @@ use bytes::Bytes; use litellm_core::messages::{ Error, route::{Messages, MessagesCall, MessagesOp, MessagesOpResult, MessagesOutput}, + types::MessagesShaping, }; use litellm_host_python::{InvokeError, RouteHost, from_py, lookup, to_py}; use litellm_http::transport::Error as TransportError; +use litellm_types::utils::ProviderSpecificHeaders; use pyo3::{ exceptions::{PyException, PyValueError}, gc::{PyTraverseError, PyVisit}, @@ -18,9 +20,10 @@ use crate::{ marshal::{optional_timeout, python_timeout_seconds}, }; -/// The Anthropic Messages body fields a caller may pass besides `model` and `messages`, -/// as `AnthropicMessagesRequestOptionalParams` declares them. -const BODY_FIELDS: [&str; 20] = [ +const ROUTE_HOST_MODULE: &str = "litellm.rust_bridge.messages.route_host"; +const REQUEST_ERROR_MARKER: &str = "messages_request_error"; + +const BODY_FIELDS: [&str; 22] = [ "max_tokens", "metadata", "stop_sequences", @@ -35,14 +38,46 @@ const BODY_FIELDS: [&str; 20] = [ "top_p", "mcp_servers", "context_management", + "compaction", "container", "output_format", "speed", "output_config", "cache_control", "reasoning_effort", + "safeguards", ]; +fn merge_headers( + forwarded: Option>, + extra_headers: Option>, +) -> Option> { + let merged: Map = forwarded + .into_iter() + .flatten() + .chain(extra_headers.into_iter().flatten()) + .collect(); + (!merged.is_empty()).then_some(merged) +} + +fn native_error(py: Python<'_>, error: Error) -> PyResult { + match error { + Error::Transport(TransportError::Http { status, body }) => { + let error = RustUpstreamError::new_err((status, body)); + error + .value(py) + .setattr("headers", Vec::<(String, String)>::new())?; + Ok(error) + } + Error::InvalidRequest(message) => { + let error = PyValueError::new_err(message); + error.value(py).setattr(REQUEST_ERROR_MARKER, true)?; + Ok(error) + } + other => Ok(messages_error_to_pyerr(other)), + } +} + /// The Python side of the Messages route: projects the prepared arguments and builds the /// public response, chunks and exceptions. pub(super) struct MessagesRouteHost { @@ -84,19 +119,65 @@ impl MessagesRouteHost { .map(|value| python_timeout_seconds(py, value.unbind())) .transpose()? .flatten(); + let custom_llm_provider = string("custom_llm_provider")?; + let shaping = self.shaping(py, &model, custom_llm_provider.as_deref(), arguments)?; Ok(MessagesCall { model, body, api_key: string("api_key")?, api_base: string("api_base")?, - custom_llm_provider: string("custom_llm_provider")?, - extra_headers: argument("extra_headers")? - .map(|value| from_py(&value)) - .transpose()?, + extra_headers: self.merged_headers(py, arguments)?, + provider_specific_header: self.provider_specific_header(py, arguments)?, + custom_llm_provider, timeout: optional_timeout(timeout), + shaping, }) } + fn merged_headers( + &self, + py: Python<'_>, + arguments: &Bound<'_, PyDict>, + ) -> PyResult>> { + let request = self.request.bind(py); + let mapping = |name: &str| -> PyResult>> { + lookup(arguments, request, name)? + .filter(|value| !value.is_none()) + .map(|value| from_py(&value)) + .transpose() + }; + Ok(merge_headers( + mapping("headers")?, + mapping("extra_headers")?, + )) + } + + fn provider_specific_header( + &self, + py: Python<'_>, + arguments: &Bound<'_, PyDict>, + ) -> PyResult> { + lookup(arguments, self.request.bind(py), "provider_specific_header")? + .filter(|value| !value.is_none()) + .map(|value| from_py(&value)) + .transpose() + } + + fn shaping( + &self, + py: Python<'_>, + model: &str, + custom_llm_provider: Option<&str>, + arguments: &Bound<'_, PyDict>, + ) -> PyResult { + let projected = py.import(ROUTE_HOST_MODULE)?.getattr("shaping")?.call1(( + model, + custom_llm_provider, + arguments, + ))?; + from_py(&projected) + } + fn provider(&self, py: Python<'_>) -> String { self.request .bind(py) @@ -112,7 +193,7 @@ impl MessagesRouteHost { return error; } let mapped = py - .import("litellm.rust_bridge.messages.route_host") + .import(ROUTE_HOST_MODULE) .and_then(|module| module.getattr("map_failure")) .and_then(|map| map.call1((error.value(py), self.request.bind(py), self.provider(py)))) .and_then(|mapped| { @@ -148,7 +229,7 @@ impl RouteHost for MessagesRouteHost { fn complete(&mut self, py: Python<'_>, response: MessagesOutput) -> PyResult> { match response { MessagesOutput::Message(message) => py - .import("litellm.rust_bridge.messages.route_host")? + .import(ROUTE_HOST_MODULE)? .getattr("response")? .call1((to_py(py, message.as_ref())?,)) .map(Bound::unbind), @@ -161,17 +242,12 @@ impl RouteHost for MessagesRouteHost { } fn classify(&self, py: Python<'_>, error: Error) -> PyResult { - let native = match error { - Error::Transport(TransportError::Http { status, body }) => { - let error = RustUpstreamError::new_err((status, body)); - error - .value(py) - .setattr("headers", Vec::<(String, String)>::new())?; - error - } - other => messages_error_to_pyerr(other), - }; - Ok(self.map_failure(py, native)) + if let Error::Secret(source) = &error + && let Some(original) = crate::secrets::python_error(py, source.source_error()) + { + return Ok(original); + } + Ok(self.map_failure(py, native_error(py, error)?)) } fn host_error(error: &PyErr) -> Error { @@ -184,3 +260,62 @@ impl RouteHost for MessagesRouteHost { visit.call(&self.request) } } + +#[cfg(test)] +mod tests { + use rstest::rstest; + use serde_json::json; + + use super::*; + + fn map(value: Value) -> Map { + serde_json::from_value(value).unwrap() + } + + #[rstest] + #[case::extra_over_forwarded( + Some(json!({"X-Priority": "forwarded", "X-Forwarded-Only": "keep"})), + Some(json!({"X-Priority": "extra", "X-Extra-Only": "also-keep"})), + Some(json!({"X-Priority": "extra", "X-Forwarded-Only": "keep", "X-Extra-Only": "also-keep"})), + )] + #[case::only_forwarded(Some(json!({"X-Forwarded": "yes"})), None, Some(json!({"X-Forwarded": "yes"})))] + #[case::only_extra_headers( + None, + Some(json!({"X-Custom-Header": "from-kwargs", "X-Auth-Token": "token123"})), + Some(json!({"X-Custom-Header": "from-kwargs", "X-Auth-Token": "token123"})), + )] + #[case::nothing(None, Some(json!({})), None)] + fn headers_merge_forwarded_then_extra( + #[case] forwarded: Option, + #[case] extra_headers: Option, + #[case] expected: Option, + ) { + assert_eq!( + merge_headers(forwarded.map(map), extra_headers.map(map)), + expected.map(map) + ); + } + + #[rstest] + #[case::rejected_request(Error::InvalidRequest("does not support top_k=5".into()), true)] + #[case::unresolvable_provider(Error::InvalidProvider("openai".into()), false)] + #[case::upstream_failure( + Error::Transport(TransportError::Http { status: 400, body: "bad".into() }), + false, + )] + fn only_request_rejections_carry_the_request_error_marker( + #[case] error: Error, + #[case] marked: bool, + ) { + Python::initialize(); + Python::attach(|py| { + let native = native_error(py, error).unwrap(); + let marker = native + .value(py) + .getattr_opt(REQUEST_ERROR_MARKER) + .unwrap() + .map(|value| value.extract::().unwrap()); + assert_eq!(marker.unwrap_or(false), marked); + }); + } +} diff --git a/litellm-rust/crates/python-bridge/src/routes/messages/mod.rs b/litellm-rust/crates/python-bridge/src/routes/messages/mod.rs index 804d883e9ae..fd474e6b2d4 100644 --- a/litellm-rust/crates/python-bridge/src/routes/messages/mod.rs +++ b/litellm-rust/crates/python-bridge/src/routes/messages/mod.rs @@ -39,11 +39,12 @@ fn run_messages( "the Rust Messages route does not serve this provider", )); } + let secrets = crate::secrets::source(py)?; run_legacy_call( py, SURFACE, PublicCall::capture(&request, &args, &kwargs)?, - crate::logger::LoggedMachine::new(messages_machine()), + crate::logger::LoggedMachine::new(messages_machine(secrets)), MessagesRouteHost::new(request.unbind()), asynchronous, ) diff --git a/litellm-rust/crates/types/Cargo.toml b/litellm-rust/crates/types/Cargo.toml index 6a2efa90ab4..0a0927386f0 100644 --- a/litellm-rust/crates/types/Cargo.toml +++ b/litellm-rust/crates/types/Cargo.toml @@ -8,3 +8,6 @@ repository.workspace = true [dependencies] serde.workspace = true serde_json.workspace = true + +[dev-dependencies] +rstest.workspace = true diff --git a/litellm-rust/crates/types/src/llms/anthropic_messages/anthropic_request.rs b/litellm-rust/crates/types/src/llms/anthropic_messages/anthropic_request.rs index 50eedf7ba09..2f7a75ba517 100644 --- a/litellm-rust/crates/types/src/llms/anthropic_messages/anthropic_request.rs +++ b/litellm-rust/crates/types/src/llms/anthropic_messages/anthropic_request.rs @@ -17,12 +17,48 @@ pub enum MessageContent { #[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)] pub struct ContentBlock { + #[serde(rename = "type", default, skip_serializing_if = "Option::is_none")] + pub block_type: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub text: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub thinking: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub signature: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub data: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub name: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tool_use_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub input: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub content: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub provider_specific_fields: Option, #[serde(skip_serializing_if = "Option::is_none")] pub cache_control: Option, #[serde(flatten)] pub extra: Map, } +impl ContentBlock { + pub fn text(text: impl Into) -> Self { + Self { + block_type: Some("text".to_string()), + text: Some(text.into()), + ..Self::default() + } + } + + pub fn is_type(&self, block_type: &str) -> bool { + self.block_type.as_deref() == Some(block_type) + } +} + #[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)] pub struct CacheControl { #[serde(rename = "type", skip_serializing_if = "Option::is_none")] @@ -85,6 +121,126 @@ pub struct AnthropicMessagesRequest { pub speed: Option, #[serde(skip_serializing_if = "Option::is_none")] pub inference_geo: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub reasoning_effort: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub compaction: Option, #[serde(flatten)] pub extra: Map, } + +impl AnthropicMessage { + pub fn blocks(&self) -> &[ContentBlock] { + match &self.content { + MessageContent::Blocks(blocks) => blocks, + MessageContent::Text(_) => &[], + } + } + + pub fn with_blocks(self, blocks: Vec) -> Self { + Self { + content: MessageContent::Blocks(blocks), + ..self + } + } +} + +#[cfg(test)] +mod tests { + use rstest::rstest; + use serde_json::json; + + use super::*; + + fn round_trip(value: &Value) -> Value { + let parsed: T = serde_json::from_value(value.clone()).unwrap(); + serde_json::to_value(parsed).unwrap() + } + + #[rstest] + #[case::text(json!({"type": "text", "text": "hi"}))] + #[case::text_with_citations_and_cache_control(json!({ + "type": "text", + "text": "hi", + "citations": [{"type": "char_location", "cited_text": "x"}], + "cache_control": {"type": "ephemeral", "ttl": "1h", "scope": "global", "future": 1} + }))] + #[case::image(json!({"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": "AA=="}}))] + #[case::thinking(json!({"type": "thinking", "thinking": "hmm", "signature": "sig"}))] + #[case::redacted_thinking(json!({"type": "redacted_thinking", "data": "opaque"}))] + #[case::tool_use(json!({"type": "tool_use", "id": "toolu_1", "name": "f", "input": {"q": [1, null]}}))] + #[case::tool_result_with_text(json!({"type": "tool_result", "tool_use_id": "toolu_1", "content": "ok", "is_error": false}))] + #[case::tool_result_with_blocks(json!({"type": "tool_result", "tool_use_id": "toolu_1", "content": [{"type": "text", "text": "ok"}]}))] + #[case::web_search_result_with_nulls(json!({ + "type": "web_search_tool_result", + "tool_use_id": "srvtoolu_1", + "content": [{"type": "web_search_result", "url": "u", "page_age": null, "encrypted_content": ""}] + }))] + #[case::provider_specific_fields(json!({"type": "tool_use", "id": "t", "name": "f", "input": {}, "provider_specific_fields": {"x": 1}}))] + #[case::untyped(json!({"unknown": {"nested": true}}))] + fn content_block_round_trips_unchanged(#[case] block: Value) { + assert_eq!(round_trip::(&block), block); + } + + #[test] + fn text_constructor_serializes_as_a_text_block() { + assert_eq!( + serde_json::to_value(ContentBlock::text("hello")).unwrap(), + json!({"type": "text", "text": "hello"}) + ); + } + + #[rstest] + #[case::same_type(json!({"type": "tool_use"}), "tool_use", true)] + #[case::other_type(json!({"type": "tool_result"}), "tool_use", false)] + #[case::prefix_of_type(json!({"type": "tool_use"}), "tool", false)] + #[case::no_type(json!({"text": "x"}), "text", false)] + fn is_type_matches_the_exact_block_type( + #[case] block: Value, + #[case] block_type: &str, + #[case] expected: bool, + ) { + let block: ContentBlock = serde_json::from_value(block).unwrap(); + assert_eq!(block.is_type(block_type), expected); + } + + #[rstest] + #[case::string_content(json!({"role": "user", "content": "hi"}), vec![])] + #[case::block_content( + json!({"role": "user", "content": [{"type": "text", "text": "a"}, {"type": "text", "text": "b"}]}), + vec![ContentBlock::text("a"), ContentBlock::text("b")], + )] + fn message_blocks_list_only_block_content( + #[case] message: Value, + #[case] expected: Vec, + ) { + let message: AnthropicMessage = serde_json::from_value(message).unwrap(); + assert_eq!(message.blocks(), expected.as_slice()); + } + + #[rstest] + #[case::replaces_string_content(json!({"role": "assistant", "content": "old", "name": "kept"}))] + #[case::replaces_block_content(json!({"role": "assistant", "content": [{"type": "text", "text": "old"}], "name": "kept"}))] + fn with_blocks_replaces_content_and_keeps_the_rest(#[case] message: Value) { + let message: AnthropicMessage = serde_json::from_value(message).unwrap(); + assert_eq!( + serde_json::to_value(message.with_blocks(vec![ContentBlock::text("new")])).unwrap(), + json!({"role": "assistant", "content": [{"type": "text", "text": "new"}], "name": "kept"}) + ); + } + + #[rstest] + #[case::minimal(json!({"model": "m", "messages": [{"role": "user", "content": "hi"}]}))] + #[case::reasoning_effort_compaction_and_unknown_fields(json!({ + "model": "m", + "messages": [{"role": "user", "content": [{"type": "text", "text": "hi"}]}], + "max_tokens": 8, + "reasoning_effort": "high", + "compaction": {"type": "auto"}, + "safeguards": [{"type": "dangerous_tool_use", "classifier_context": {"v": 1}}], + "metadata": {"user_id": "u"} + }))] + fn request_round_trips_unchanged(#[case] request: Value) { + assert_eq!(round_trip::(&request), request); + } +} diff --git a/litellm-rust/crates/types/src/llms/anthropic_messages/anthropic_response.rs b/litellm-rust/crates/types/src/llms/anthropic_messages/anthropic_response.rs index 0c3876aac59..0a2653f352f 100644 --- a/litellm-rust/crates/types/src/llms/anthropic_messages/anthropic_response.rs +++ b/litellm-rust/crates/types/src/llms/anthropic_messages/anthropic_response.rs @@ -9,8 +9,6 @@ pub struct AnthropicMessagesResponse { pub role: String, pub model: String, pub content: Vec, - // Anthropic always includes stop_reason / stop_sequence, null until the turn - // ends; serialize them even when None so callers see the same shape as Python. pub stop_reason: Option, pub stop_sequence: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -20,3 +18,61 @@ pub struct AnthropicMessagesResponse { #[serde(flatten)] pub extra: Map, } + +#[cfg(test)] +mod tests { + use rstest::rstest; + use serde_json::json; + + use super::*; + + fn response( + stop_reason: Option<&str>, + stop_sequence: Option<&str>, + usage: Option, + container: Option, + ) -> AnthropicMessagesResponse { + AnthropicMessagesResponse { + id: "msg_1".to_string(), + message_type: "message".to_string(), + role: "assistant".to_string(), + model: "claude".to_string(), + content: vec![], + stop_reason: stop_reason.map(str::to_string), + stop_sequence: stop_sequence.map(str::to_string), + usage, + container, + extra: Map::new(), + } + } + + #[rstest] + #[case::turn_in_progress(None, None, json!(null), json!(null))] + #[case::ended_on_end_turn(Some("end_turn"), None, json!("end_turn"), json!(null))] + #[case::ended_on_stop_sequence(Some("stop_sequence"), Some("###"), json!("stop_sequence"), json!("###"))] + fn stop_fields_are_always_serialized( + #[case] stop_reason: Option<&str>, + #[case] stop_sequence: Option<&str>, + #[case] expected_reason: Value, + #[case] expected_sequence: Value, + ) { + let body: Value = serde_json::to_value(response(stop_reason, stop_sequence, None, None)) + .expect("serializable"); + assert_eq!(body.get("stop_reason"), Some(&expected_reason)); + assert_eq!(body.get("stop_sequence"), Some(&expected_sequence)); + } + + #[rstest] + #[case::absent(None, None)] + #[case::present(Some(json!({"input_tokens": 1})), Some(json!({"id": "c_1"})))] + fn usage_and_container_are_omitted_only_when_none( + #[case] usage: Option, + #[case] container: Option, + ) { + let body: Value = + serde_json::to_value(response(None, None, usage.clone(), container.clone())) + .expect("serializable"); + assert_eq!(body.get("usage").cloned(), usage); + assert_eq!(body.get("container").cloned(), container); + } +} diff --git a/litellm-rust/crates/types/src/utils.rs b/litellm-rust/crates/types/src/utils.rs index 7f0c18f9f2c..5ca56ec9e49 100644 --- a/litellm-rust/crates/types/src/utils.rs +++ b/litellm-rust/crates/types/src/utils.rs @@ -3,6 +3,21 @@ use serde_json::{Map, Value}; use crate::llms::openai::{ChatCompletionThinkingBlock, ChatCompletionToolCallChunk}; +#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)] +pub struct ProviderSpecificHeader { + #[serde(default)] + pub custom_llm_provider: String, + #[serde(default)] + pub extra_headers: Map, +} + +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +#[serde(untagged)] +pub enum ProviderSpecificHeaders { + One(ProviderSpecificHeader), + Many(Vec), +} + /// OpenAI `usage`, including the `prompt_tokens_details` split LiteLLM's Python /// path reports so cost tracking sees the same numbers on either path. #[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)] diff --git a/litellm/rust_bridge/messages/route_host.py b/litellm/rust_bridge/messages/route_host.py index beef0f81eca..d49d7b75a6f 100644 --- a/litellm/rust_bridge/messages/route_host.py +++ b/litellm/rust_bridge/messages/route_host.py @@ -1,12 +1,50 @@ from __future__ import annotations -from collections.abc import Mapping -from typing import cast # noqa: TID251 # narrows the normalized native payload to the public TypedDict +from collections.abc import Mapping, Sequence +from dataclasses import asdict, dataclass +from typing import Final, cast # noqa: TID251 # narrows the normalized native payload to the public TypedDict +from pydantic import TypeAdapter, ValidationError + +import litellm +from litellm.litellm_core_utils.core_helpers import normalize_drop_params +from litellm.llms.anthropic.experimental_pass_through.utils import is_reasoning_auto_summary_enabled from litellm.rust_bridge import failures from litellm.rust_bridge.messages.entrypoints import LiteLLMMessagesRequest from litellm.types.llms.anthropic_messages.anthropic_response import AnthropicMessagesResponse +_DROP_PATHS: Final = TypeAdapter(list[object]) + + +@dataclass(frozen=True, slots=True) +class EffortTiers: + minimal: bool + low: bool + medium: bool + high: bool + xhigh: bool + max: bool + + +@dataclass(frozen=True, slots=True) +class ModelCapabilities: + supports_reasoning: bool + supports_adaptive_thinking: bool + thinking_always_on: bool + supports_legacy_thinking: bool + supports_output_config: bool + supports_sampling_params: bool + supports_speed: bool + effort_tiers: EffortTiers + + +@dataclass(frozen=True, slots=True) +class MessagesShaping: + capabilities: ModelCapabilities + drop_params: bool + reasoning_auto_summary: bool + additional_drop_params: Sequence[str] + def response(value: Mapping[str, object]) -> AnthropicMessagesResponse: return cast( # cast-ok: AnthropicMessagesResponse is a TypedDict over the normalized native payload @@ -20,4 +58,72 @@ def arguments(request: LiteLLMMessagesRequest) -> Mapping[str, object]: def map_failure(error: Exception, request: LiteLLMMessagesRequest, request_provider: str) -> Exception: + if getattr(error, "messages_request_error", False): + return litellm.BadRequestError( + message=str(error), + model=request.model.removeprefix(f"{request_provider}/"), + llm_provider=request_provider, + ) return failures.map_native_failure(error, request.model, request_provider, arguments(request), request.api_base) + + +def _resolved_provider(model: str, custom_llm_provider: str | None) -> tuple[str, str]: + try: + resolved_model, provider, _, _ = litellm.get_llm_provider(model=model, custom_llm_provider=custom_llm_provider) + except Exception: # noqa: BLE001 # an unroutable model still shapes as a bare Anthropic id + return model, custom_llm_provider or "anthropic" + return resolved_model, provider + + +def model_capabilities(model: str, custom_llm_provider: str | None) -> ModelCapabilities: + from litellm.llms.anthropic.chat.transformation import AnthropicConfig + from litellm.llms.anthropic.common_utils import AnthropicModelInfo + + resolved_model, provider = _resolved_provider(model, custom_llm_provider) + + def supports(flag: str) -> bool: + return AnthropicModelInfo._supports_model_capability(model, flag, provider) # pyright: ignore[reportPrivateUsage] # same probes the Python transform runs; forking them would drift + + def tier(level: str) -> bool: + return AnthropicConfig._supports_effort_level(model, level, provider) # pyright: ignore[reportPrivateUsage] # same probe the Python transform runs + + return ModelCapabilities( + supports_reasoning=supports("supports_reasoning"), + supports_adaptive_thinking=supports("supports_adaptive_thinking"), + thinking_always_on=supports("thinking_always_on"), + supports_legacy_thinking=supports("supports_legacy_thinking"), + supports_output_config=supports("supports_output_config"), + supports_sampling_params=AnthropicModelInfo._supports_sampling_params(resolved_model), # pyright: ignore[reportPrivateUsage] # same gate the handler applies + supports_speed=AnthropicConfig._model_supports_speed_param(resolved_model, provider), # pyright: ignore[reportPrivateUsage] # same gate the handler applies + effort_tiers=EffortTiers( + minimal=tier("minimal"), + low=tier("low"), + medium=tier("medium"), + high=tier("high"), + xhigh=tier("xhigh"), + max=tier("max"), + ), + ) + + +def _drop_params(kwargs: Mapping[str, object]) -> bool: + return bool(litellm.drop_params) or normalize_drop_params(kwargs.get("drop_params")) is True + + +def _additional_drop_params(kwargs: Mapping[str, object]) -> tuple[str, ...]: + try: + configured: Final = _DROP_PATHS.validate_python(kwargs.get("additional_drop_params")) + except ValidationError: + return () + return tuple(path for path in configured if isinstance(path, str)) + + +def shaping(model: str, custom_llm_provider: str | None, kwargs: Mapping[str, object]) -> dict[str, object]: + return asdict( + MessagesShaping( + capabilities=model_capabilities(model, custom_llm_provider), + drop_params=_drop_params(kwargs), + reasoning_auto_summary=is_reasoning_auto_summary_enabled(), + additional_drop_params=_additional_drop_params(kwargs), + ) + ) diff --git a/tests/test_litellm/rust_bridge/AGENTS.md b/tests/test_litellm/rust_bridge/AGENTS.md index 351bd582ce7..994d24112e8 100644 --- a/tests/test_litellm/rust_bridge/AGENTS.md +++ b/tests/test_litellm/rust_bridge/AGENTS.md @@ -5,3 +5,5 @@ Test what each side of the bridge does, not the rollout policy that picks a side Call each path directly with an explicit decision instead. The Python path is the implementation the dispatcher falls back to, e.g. `litellm.ocr.main.ocr`. The Rust path is the native binding, e.g. `NATIVE_OCR.load()` from `litellm/rust_bridge/ocr/entrypoints.py`, called with the request, args and kwargs that dispatch would hand it. When the native side reads a policy-derived setting such as `settings.secret_manager().native`, pin that field in the test instead of deriving it from the catalog. `ocr/test_secrets.py` shows the pattern Rollout policy itself, meaning which rule matches and what `LITELLM_RUST` changes, belongs in `test_catalog.py`, `test_configuration.py` and `test_dispatch.py`, tested against rules the test builds rather than the shipped `catalog.RULES` + +Before adding a test here, ask whether it checks something Rust cannot. A `route_host.py` module is the Python half of a native route: it projects Python-only state (the cost map, `litellm.*` settings, request kwargs) into the plain values the Rust side consumes, and maps native failures back onto public exceptions. Those projections are what belongs here, because a wrong key or an ignored provider prefix ships the wrong value to Rust and no Rust test sees it. `messages/test_route_host.py` shows the shape. Behavior that lives in Rust (a request transform given its inputs, header assembly, stream relay) is tested in the crate, and the route end to end is tested against a recording server in `tests/test_litellm_rust/`. A test that only re-checks a Python helper the route host happens to call is a duplicate of that helper's own test and should not be added diff --git a/tests/test_litellm/rust_bridge/messages/test_route_host.py b/tests/test_litellm/rust_bridge/messages/test_route_host.py new file mode 100644 index 00000000000..f47333a45d9 --- /dev/null +++ b/tests/test_litellm/rust_bridge/messages/test_route_host.py @@ -0,0 +1,112 @@ +from dataclasses import astuple +from typing import Final + +import pytest + +import litellm +from litellm.rust_bridge.messages import route_host + +pytestmark = pytest.mark.usefixtures("local_model_cost_map") + + +def _flag_model(monkeypatch: pytest.MonkeyPatch, name: str, **flags: bool) -> None: + monkeypatch.setitem( + litellm.model_cost, + name, + { + "litellm_provider": "anthropic", + "mode": "chat", + "input_cost_per_token": 0, + "output_cost_per_token": 0, + **flags, + }, + ) + + +def test_capabilities_come_from_the_model_map_under_the_callers_provider(monkeypatch: pytest.MonkeyPatch) -> None: + _flag_model( + monkeypatch, + "claude-test-adaptive", + supports_reasoning=True, + supports_adaptive_thinking=True, + supports_output_config=True, + supports_xhigh_reasoning_effort=True, + supports_sampling_params=False, + ) + + capabilities: Final = route_host.model_capabilities("anthropic/claude-test-adaptive", None) + + assert capabilities.supports_adaptive_thinking + assert capabilities.supports_output_config + assert not capabilities.supports_legacy_thinking + assert not capabilities.supports_sampling_params + assert capabilities.effort_tiers.xhigh + assert not capabilities.effort_tiers.max + + +def test_unmapped_model_keeps_sampling_params_and_no_reasoning_features() -> None: + capabilities: Final = route_host.model_capabilities("anthropic/not-a-real-model", None) + + assert capabilities.supports_sampling_params + assert not capabilities.supports_reasoning + assert not capabilities.supports_adaptive_thinking + assert not any(astuple(capabilities.effort_tiers)) + + +@pytest.mark.parametrize( + ("global_flag", "kwargs", "expected"), + [ + (False, {}, False), + (True, {}, True), + (False, {"drop_params": "true"}, True), + (False, {"drop_params": "nonsense"}, False), + (False, {"drop_params": False}, False), + ], +) +def test_drop_params_merges_the_global_flag_with_the_request( + monkeypatch: pytest.MonkeyPatch, global_flag: bool, kwargs: dict[str, object], expected: bool +) -> None: + monkeypatch.setattr(litellm, "drop_params", global_flag) + + assert route_host.shaping("anthropic/not-a-real-model", None, kwargs)["drop_params"] is expected + + +@pytest.mark.parametrize( + ("configured", "expected"), + [ + (["tools[*].input_examples", 3, "metadata.user_id"], ("tools[*].input_examples", "metadata.user_id")), + ("tools", ()), + (None, ()), + ], +) +def test_additional_drop_params_keep_only_string_paths(configured: object, expected: tuple[str, ...]) -> None: + shaping: Final = route_host.shaping("anthropic/not-a-real-model", None, {"additional_drop_params": configured}) + + assert shaping["additional_drop_params"] == expected + + +def test_native_request_rejections_map_to_the_public_400() -> None: + from types import MappingProxyType + + from litellm.rust_bridge.messages.entrypoints import LiteLLMMessagesRequest + + request: Final = LiteLLMMessagesRequest( + model="anthropic/claude-sonnet-5", + messages=(), + max_tokens=8, + stream=None, + api_key=None, + api_base=None, + custom_llm_provider=None, + kwargs=MappingProxyType({}), + ) + rejected: Final = ValueError("claude-sonnet-5 does not support top_k=5") + rejected.messages_request_error = True # pyright: ignore[reportAttributeAccessIssue] # marker the native host sets + + mapped: Final = route_host.map_failure(rejected, request, "anthropic") + + assert isinstance(mapped, litellm.BadRequestError) + assert mapped.status_code == 400 + assert "does not support top_k=5" in mapped.message + assert mapped.model == "claude-sonnet-5" + assert not isinstance(route_host.map_failure(ValueError("plain"), request, "anthropic"), litellm.BadRequestError) diff --git a/tests/test_litellm/rust_bridge/messages/test_secrets.py b/tests/test_litellm/rust_bridge/messages/test_secrets.py new file mode 100644 index 00000000000..cf37ed0830b --- /dev/null +++ b/tests/test_litellm/rust_bridge/messages/test_secrets.py @@ -0,0 +1,111 @@ +from __future__ import annotations + +from collections.abc import Awaitable, Mapping +from dataclasses import replace +from types import MappingProxyType +from typing import Final, Protocol, cast # noqa: TID251 # narrows the parametrized path to its protocol + +import httpx +import pytest + +import litellm +from litellm.integrations.custom_secret_manager import CustomSecretManager +from litellm.llms.anthropic.experimental_pass_through.messages.handler import anthropic_messages +from litellm.rust_bridge import settings +from litellm.rust_bridge.messages.entrypoints import NATIVE_AMESSAGES, NATIVE_MESSAGES, LiteLLMMessagesRequest +from litellm.types.secret_managers.main import KeyManagementSettings, KeyManagementSystem +from tests.test_litellm_rust.support.recording_server import ResponseSpec, recording_service +from tests.test_litellm_rust.support.requests import MESSAGES, MESSAGES_MODEL, MESSAGES_RESPONSE + +pytest.importorskip("litellm.rust_bridge._native") + +pytestmark = pytest.mark.usefixtures("local_model_cost_map") + + +class Messages(Protocol): + def __call__(self) -> Awaitable[object]: ... + + +class _ManagedSecrets(CustomSecretManager): + def __init__(self, values: Mapping[str, str]) -> None: + super().__init__(secret_manager_name="rust_bridge_messages_test") + self.values: Final = values + + async def async_read_secret( + self, + secret_name: str, + optional_params: dict[str, object] | None = None, + timeout: float | httpx.Timeout | None = None, + ) -> str | None: + raise AssertionError("get_secret reads custom managers synchronously") + + def sync_read_secret( + self, + secret_name: str, + optional_params: dict[str, object] | None = None, + timeout: float | httpx.Timeout | None = None, + ) -> str | None: + return self.values.get(secret_name) + + +def _native_request() -> LiteLLMMessagesRequest: + return LiteLLMMessagesRequest( + model=MESSAGES_MODEL, + messages=MESSAGES, + max_tokens=8, + stream=None, + api_key=None, + api_base=None, + custom_llm_provider=None, + kwargs=MappingProxyType({}), + ) + + +def _public_kwargs() -> dict[str, object]: + return {"model": MESSAGES_MODEL, "messages": [dict(message) for message in MESSAGES], "max_tokens": 8} + + +async def _python_messages() -> object: + return await anthropic_messages(**_public_kwargs()) + + +async def _rust_messages() -> object: + route: Final = NATIVE_MESSAGES.load() + assert route is not None + return route(_native_request(), (), _public_kwargs()) + + +async def _rust_amessages() -> object: + route: Final = NATIVE_AMESSAGES.load() + assert route is not None + return await route(_native_request(), (), _public_kwargs()) + + +@pytest.fixture( + params=(_python_messages, _rust_messages, _rust_amessages), ids=("python-async", "rust-sync", "rust-async") +) +def messages(request: pytest.FixtureRequest) -> Messages: + return cast(Messages, request.param) + + +async def test_secret_manager_supplies_the_anthropic_key_and_base( + monkeypatch: pytest.MonkeyPatch, messages: Messages +) -> None: + for name in ("ANTHROPIC_API_KEY", "ANTHROPIC_AUTH_TOKEN", "ANTHROPIC_API_BASE", "ANTHROPIC_BASE_URL"): + monkeypatch.delenv(name, raising=False) + with recording_service() as server: + server.default_response = ResponseSpec(body=MESSAGES_RESPONSE) + monkeypatch.setattr( + litellm, + "secret_manager_client", + _ManagedSecrets({"ANTHROPIC_API_KEY": "vault-key", "ANTHROPIC_BASE_URL": server.base_url}), + ) + monkeypatch.setattr(litellm, "_key_management_system", KeyManagementSystem.CUSTOM) + monkeypatch.setattr(litellm, "_key_management_settings", KeyManagementSettings(access_mode="read_only")) + configured: Final = settings.secret_manager + monkeypatch.setattr(settings, "secret_manager", lambda: replace(configured(), native=True)) + + await messages() + + assert len(server.requests) == 1 + assert server.requests[0].headers["x-api-key"] == "vault-key" diff --git a/tests/test_litellm_rust/messages/test_request_shaping.py b/tests/test_litellm_rust/messages/test_request_shaping.py new file mode 100644 index 00000000000..f885fda5f42 --- /dev/null +++ b/tests/test_litellm_rust/messages/test_request_shaping.py @@ -0,0 +1,237 @@ +"""The native Messages route shapes the wire request the way the Python handler does. + +Model capability expectations come from model_prices_and_context_window.json (Claude Sonnet 5 is an +adaptive-thinking model without sampling params; Claude Haiku 4.5 is a legacy-thinking model), read at +2026-09-24; the cost map is LiteLLM's own file. +""" + +from collections.abc import Iterator +from typing import Final + +import pytest + +import litellm +from litellm.rust_bridge import catalog +from litellm.rust_bridge.catalog import Route, RouteRule +from litellm.rust_bridge.configuration import Rollout +from tests.test_litellm_rust.support.isolation import rebound +from tests.test_litellm_rust.support.recording_server import RecordingServer, ResponseSpec +from tests.test_litellm_rust.support.requests import MESSAGES, MESSAGES_RESPONSE + +pytestmark = pytest.mark.requires_rust_extension + +ADAPTIVE_MODEL: Final = "anthropic/claude-sonnet-5" +LEGACY_THINKING_MODEL: Final = "anthropic/claude-haiku-4-5" + + +@pytest.fixture(autouse=True) +def opt_messages_into_rust() -> Iterator[None]: + with rebound(catalog, "RULES", (RouteRule(Route.MESSAGES, Rollout.RUST_OPT_IN), *catalog.RULES)): + yield + + +@pytest.fixture +def messages_server(recording_server: RecordingServer) -> RecordingServer: + recording_server.default_response = ResponseSpec(body=MESSAGES_RESPONSE) + return recording_server + + +def arguments(server: RecordingServer, **kwargs: object) -> dict[str, object]: + return { + "model": ADAPTIVE_MODEL, + "messages": [dict(message) for message in MESSAGES], + "max_tokens": 8192, + "api_key": "test-key", + "api_base": server.base_url, + **kwargs, + } + + +def sent(server: RecordingServer) -> tuple[dict[str, object], dict[str, str]]: + assert len(server.requests) == 1 + request: Final = server.requests[0] + assert not request.headers.get("user-agent", "").startswith("python-httpx") + assert isinstance(request.body, dict) + return request.body, request.headers + + +@pytest.mark.asyncio +async def test_reasoning_effort_becomes_adaptive_thinking_and_effort_on_the_wire( + messages_server: RecordingServer, +) -> None: + await litellm.anthropic.messages.acreate(**arguments(messages_server, reasoning_effort="high")) + + body, _ = sent(messages_server) + assert "reasoning_effort" not in body + assert body["thinking"] == {"type": "adaptive", "display": "summarized"} + assert body["output_config"] == {"effort": "high"} + + +@pytest.mark.asyncio +async def test_claude_code_adaptive_payload_is_downgraded_to_a_capped_budget_for_a_legacy_model( + messages_server: RecordingServer, +) -> None: + await litellm.anthropic.messages.acreate( + **arguments( + messages_server, + model=LEGACY_THINKING_MODEL, + max_tokens=3000, + thinking={"type": "adaptive"}, + output_config={"effort": "high"}, + temperature=0, + ) + ) + + body, _ = sent(messages_server) + assert body["thinking"] == {"type": "enabled", "budget_tokens": 2999} + assert "output_config" not in body + assert "temperature" not in body + + +@pytest.mark.asyncio +async def test_removed_sampling_params_are_dropped_under_drop_params(messages_server: RecordingServer) -> None: + await litellm.anthropic.messages.acreate( + **arguments(messages_server, temperature=0.2, top_p=0.9, top_k=5, drop_params=True) + ) + + body, _ = sent(messages_server) + assert not {"temperature", "top_p", "top_k"} & body.keys() + + +@pytest.mark.asyncio +async def test_removed_sampling_params_are_rejected_without_drop_params(messages_server: RecordingServer) -> None: + messages_server.expected_requests = 0 + + with pytest.raises(litellm.BadRequestError, match="does not support top_k=5"): + await litellm.anthropic.messages.acreate(**arguments(messages_server, top_k=5)) + + assert messages_server.requests == [] + + +@pytest.mark.asyncio +async def test_replayed_history_is_sanitized_before_it_reaches_the_provider( + messages_server: RecordingServer, +) -> None: + history: Final = [ + {"role": "user", "content": "run it"}, + { + "role": "assistant", + "content": [ + {"type": "text", "text": ""}, + { + "type": "tool_use", + "id": "functions.Bash:0", + "name": "Bash", + "input": {}, + "provider_specific_fields": {"x": 1}, + }, + ], + }, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions.Bash:0", "content": "ok"}]}, + ] + + await litellm.anthropic.messages.acreate(**arguments(messages_server, messages=history)) + + body, _ = sent(messages_server) + assert body["messages"] == [ + {"role": "user", "content": "run it"}, + {"role": "assistant", "content": [{"type": "tool_use", "id": "functions_Bash_0", "name": "Bash", "input": {}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "functions_Bash_0", "content": "ok"}]}, + ] + + +@pytest.mark.asyncio +async def test_feature_betas_merge_into_the_forwarded_beta_header(messages_server: RecordingServer) -> None: + await litellm.anthropic.messages.acreate( + **arguments( + messages_server, + output_format={"type": "json_schema", "schema": {"type": "object"}}, + extra_headers={"anthropic-beta": "web-search-2025-03-05"}, + ) + ) + + _, headers = sent(messages_server) + assert headers["anthropic-beta"] == "structured-outputs-2025-11-13,web-search-2025-03-05" + + +@pytest.mark.asyncio +async def test_oauth_token_authenticates_as_a_bearer_with_the_oauth_beta(messages_server: RecordingServer) -> None: + await litellm.anthropic.messages.acreate(**arguments(messages_server, api_key="sk-ant-oat01-token")) + + _, headers = sent(messages_server) + assert "x-api-key" not in headers + assert headers["authorization"] == "Bearer sk-ant-oat01-token" + assert headers["anthropic-beta"] == "oauth-2025-04-20" + assert headers["anthropic-dangerous-direct-browser-access"] == "true" + + +@pytest.mark.asyncio +async def test_metadata_is_reduced_to_the_fields_anthropic_accepts(messages_server: RecordingServer) -> None: + await litellm.anthropic.messages.acreate( + **arguments(messages_server, metadata={"user_id": "u-1", "trace_id": "internal"}) + ) + + body, _ = sent(messages_server) + assert body["metadata"] == {"user_id": "u-1"} + + +@pytest.mark.asyncio +async def test_additional_drop_params_remove_nested_fields_from_the_wire(messages_server: RecordingServer) -> None: + tools: Final = [{"name": "lookup", "input_schema": {"type": "object"}, "input_examples": [{"q": "x"}]}] + + await litellm.anthropic.messages.acreate( + **arguments(messages_server, tools=tools, additional_drop_params=["tools[*].input_examples"]) + ) + + body, _ = sent(messages_server) + assert body["tools"] == [{"name": "lookup", "input_schema": {"type": "object"}}] + + +@pytest.mark.asyncio +async def test_provider_specific_headers_scoped_to_anthropic_reach_the_wire(messages_server: RecordingServer) -> None: + await litellm.anthropic.messages.acreate( + **arguments( + messages_server, + provider_specific_header=[ + {"custom_llm_provider": "anthropic, azure_ai", "extra_headers": {"x-scoped": "yes"}}, + {"custom_llm_provider": "openai", "extra_headers": {"x-other": "no"}}, + ], + ) + ) + + _, headers = sent(messages_server) + assert headers["x-scoped"] == "yes" + assert "x-other" not in headers + + +@pytest.mark.asyncio +async def test_scoped_headers_override_extra_headers_which_override_forwarded_headers( + messages_server: RecordingServer, +) -> None: + await litellm.anthropic.messages.acreate( + **arguments( + messages_server, + headers={"x-priority": "forwarded", "x-forwarded-only": "kept"}, + extra_headers={"x-priority": "extra", "x-extra-only": "kept"}, + provider_specific_header={"custom_llm_provider": "anthropic", "extra_headers": {"x-priority": "scoped"}}, + ) + ) + + _, headers = sent(messages_server) + assert {name: headers.get(name) for name in ("x-priority", "x-forwarded-only", "x-extra-only")} == { + "x-priority": "scoped", + "x-forwarded-only": "kept", + "x-extra-only": "kept", + } + + +@pytest.mark.asyncio +async def test_non_string_metadata_user_id_is_rejected_before_the_provider_call( + messages_server: RecordingServer, +) -> None: + messages_server.expected_requests = 0 + + with pytest.raises(litellm.BadRequestError, match=r"metadata\.user_id must be a string"): + await litellm.anthropic.messages.acreate(**arguments(messages_server, metadata={"user_id": 123})) + + assert messages_server.requests == [] From f39a56b004c83f4242f7f95c48504235f1af2180 Mon Sep 17 00:00:00 2001 From: ahamedshaik16 Date: Fri, 25 Sep 2026 00:45:01 +0530 Subject: [PATCH 90/96] fix(prometheus): add model_group label to deployment request and rate limit metrics (#42966) * fix(prometheus): add model_group label to deployment request and rate limit metrics litellm_deployment_total_requests, litellm_deployment_success_responses, litellm_deployment_failure_responses, litellm_deployment_tpm_limit and litellm_deployment_rpm_limit had no way to identify which model_group a pooled deployment belongs to, only requested_model, litellm_model_name and model_id, none of which name the alias a model_name resolves through when it fans out to more than one deployment. model_group was already resolved onto enum_values for every request in async_log_success_event, so this is a label-list addition for the metrics built directly from that enum_values (the two request counters). The failure counter builds its own UserAPIKeyLabelValues locally and had a model_group variable already in scope that it never passed through, and the tpm/rpm limit gauges are set from a helper that took no model_group parameter at all even though its only caller already had it on enum_values. Both now thread the value through. * test(prometheus): expect model_group in deployment success/total request labels test_set_llm_deployment_success_metrics asserts the exact label set passed to litellm_deployment_success_responses.labels() and litellm_deployment_total_requests.labels(), which now includes model_group since it was added to those metrics' label list. * fix(prometheus): bound model_group on deployment failure metrics On a pre-routing reject (no deployment selected), model_group is caller-supplied via litellm_params.metadata and was passed through unbounded, letting an unrecognized value mint unlimited label series on litellm_deployment_failure_responses / litellm_deployment_total_requests. Bound it with the same _bounded_requested_model_label used for requested_model on this path. When a deployment is actually selected, model_group is router-resolved and passed through as-is. Also documents the model_group parameter on _set_deployment_tpm_rpm_limit_metrics and the bounding behavior on set_llm_deployment_failure_metrics. --------- Co-authored-by: ahamedshaik16 <24526479+ahamedshaik16@users.noreply.github.com> --- litellm/integrations/prometheus.py | 25 +++ litellm/types/integrations/prometheus.py | 3 + .../test_prometheus_logging_callbacks.py | 2 + .../integrations/test_prometheus_labels.py | 170 ++++++++++++++++++ 4 files changed, 200 insertions(+) diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 2fdcb8ef745..fb010ab5886 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -2778,6 +2778,13 @@ class PrometheusLogger(CustomLogger): - increment deployment failure responses metric - increment deployment total requests metric + Both counters also carry a model_group label. When a deployment was + actually selected, model_group is the router-resolved value and is + trusted as-is. On a pre-routing reject (no deployment selected), it + is caller-supplied via litellm_params.metadata and is bounded with + _bounded_requested_model_label the same way requested_model is, so an + unrecognized value cannot mint unbounded label series. + Args: request_kwargs: dict @@ -2844,6 +2851,7 @@ class PrometheusLogger(CustomLogger): label_api_base = api_base label_api_provider = llm_provider label_requested_model = model_group or litellm_model_name + label_model_group = model_group else: label_litellm_model_name = "" label_model_id = "" @@ -2852,6 +2860,7 @@ class PrometheusLogger(CustomLogger): label_requested_model = ( _bounded_requested_model_label(litellm_model_name or model_group, router_originated=True) or "" ) + label_model_group = _bounded_requested_model_label(model_group, router_originated=True) enum_values: Final = UserAPIKeyLabelValues( litellm_model_name=label_litellm_model_name, @@ -2861,6 +2870,7 @@ class PrometheusLogger(CustomLogger): exception_status=exception_status, exception_class=(self._get_exception_class_name(exception) if exception else None), requested_model=label_requested_model, + model_group=label_model_group, hashed_api_key=hashed_api_key, api_key_alias=api_key_alias, user_email=user_email, @@ -2912,9 +2922,21 @@ class PrometheusLogger(CustomLogger): model_id: str | None, api_base: str | None, llm_provider: str | None, + model_group: str | None, ): """ Set the deployment TPM and RPM limits metrics + + Args: + model_info: the deployment's static model_info config (id, tpm, rpm, etc.) + litellm_params: the deployment's litellm_params, as a tpm/rpm fallback source + litellm_model_name: the resolved deployment model name + model_id: the deployment's model_id + api_base: the deployment's api_base + llm_provider: the deployment's custom_llm_provider + model_group: the router-resolved model_group the deployment belongs to, + from the caller's already-resolved enum_values.model_group (trusted, + not caller-supplied at this call site) """ tpm: Final = model_info.get("tpm") or litellm_params.get("tpm") rpm: Final = model_info.get("rpm") or litellm_params.get("rpm") @@ -2927,6 +2949,7 @@ class PrometheusLogger(CustomLogger): model_id=model_id, api_base=api_base, api_provider=llm_provider, + model_group=model_group, ), ) self.litellm_deployment_tpm_limit.labels(**_labels).set(tpm) @@ -2939,6 +2962,7 @@ class PrometheusLogger(CustomLogger): model_id=model_id, api_base=api_base, api_provider=llm_provider, + model_group=model_group, ), ) self.litellm_deployment_rpm_limit.labels(**_labels).set(rpm) @@ -3058,6 +3082,7 @@ class PrometheusLogger(CustomLogger): model_id=model_id, api_base=api_base, llm_provider=llm_provider, + model_group=enum_values.model_group, ) remaining_requests: int | None = None diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index c929ee2ee79..239fc7f2779 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -664,6 +664,7 @@ class PrometheusMetricLabels: ] litellm_deployment_tpm_limit = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.v2_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.MODEL_ID.value, UserAPIKeyLabelNames.API_BASE.value, @@ -770,6 +771,7 @@ class PrometheusMetricLabels: # Add deployment metrics litellm_deployment_failure_responses = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.REQUESTED_MODEL.value, UserAPIKeyLabelNames.v2_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.MODEL_ID.value, @@ -786,6 +788,7 @@ class PrometheusMetricLabels: ] litellm_deployment_total_requests = [ + UserAPIKeyLabelNames.MODEL_GROUP.value, UserAPIKeyLabelNames.REQUESTED_MODEL.value, UserAPIKeyLabelNames.v2_LITELLM_MODEL_NAME.value, UserAPIKeyLabelNames.MODEL_ID.value, diff --git a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py b/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py index 58cde4c8103..92ff3d5813c 100644 --- a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py +++ b/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py @@ -1031,6 +1031,7 @@ def test_set_llm_deployment_success_metrics(prometheus_logger): api_base="https://api.openai.com", api_provider="openai", requested_model="my_custom_model_group", + model_group="my_custom_model_group", hashed_api_key=standard_logging_payload["metadata"]["user_api_key_hash"], api_key_alias=standard_logging_payload["metadata"]["user_api_key_alias"], team=standard_logging_payload["metadata"]["user_api_key_team_id"], @@ -1047,6 +1048,7 @@ def test_set_llm_deployment_success_metrics(prometheus_logger): api_base="https://api.openai.com", api_provider="openai", requested_model="my_custom_model_group", + model_group="my_custom_model_group", hashed_api_key=standard_logging_payload["metadata"]["user_api_key_hash"], api_key_alias=standard_logging_payload["metadata"]["user_api_key_alias"], team=standard_logging_payload["metadata"]["user_api_key_team_id"], diff --git a/tests/test_litellm/integrations/test_prometheus_labels.py b/tests/test_litellm/integrations/test_prometheus_labels.py index 200f8e65add..41d0c44ff89 100644 --- a/tests/test_litellm/integrations/test_prometheus_labels.py +++ b/tests/test_litellm/integrations/test_prometheus_labels.py @@ -787,6 +787,176 @@ async def test_failure_hook_prefers_request_data_provider_over_exception_provide ) == ["azure"] +def test_model_group_in_deployment_metrics(): + """ + Test that model_group label is present on the deployment-scoped metrics + needed to build model-group dashboards (request counts, success/failure + counts, tpm/rpm limits). These metrics previously only carried + requested_model, litellm_model_name and model_id, none of which identify + the model_group a pooled deployment belongs to. + """ + model_group_label = UserAPIKeyLabelNames.MODEL_GROUP.value + + metrics_with_model_group = [ + "litellm_deployment_total_requests", + "litellm_deployment_success_responses", + "litellm_deployment_failure_responses", + "litellm_deployment_tpm_limit", + "litellm_deployment_rpm_limit", + ] + + for metric_name in metrics_with_model_group: + labels = PrometheusMetricLabels.get_labels(metric_name) + assert ( + model_group_label in labels + ), f"Metric {metric_name} should contain model_group label" + print(f"✅ {metric_name} contains model_group label") + + +def test_model_group_value_flows_through_deployment_metrics_label_factory(): + """ + The label being in the allow-list is necessary but not sufficient: the + factory must also carry the value from the enum through to the emitted + label. This would fail if the label were dropped from a metric's list or + if the value plumbing regressed, which the allow-list assertion above + cannot catch on its own. + """ + from unittest.mock import MagicMock + + from litellm.integrations.prometheus import ( + PrometheusLogger, + UserAPIKeyLabelValues, + prometheus_label_factory, + ) + + prometheus_logger = MagicMock() + prometheus_logger._cached_metric_labels = {} + prometheus_logger.label_filters = {} + prometheus_logger.get_labels_for_metric = ( + PrometheusLogger.get_labels_for_metric.__get__(prometheus_logger) + ) + + enum_values = UserAPIKeyLabelValues( + model_group="example-model-group", + litellm_model_name="gpt-4o-mini", + requested_model="example-model-group", + status_code="200", + ) + + for metric_name in [ + "litellm_deployment_total_requests", + "litellm_deployment_success_responses", + "litellm_deployment_failure_responses", + "litellm_deployment_tpm_limit", + "litellm_deployment_rpm_limit", + ]: + labels = prometheus_label_factory( + supported_enum_labels=prometheus_logger.get_labels_for_metric( + metric_name=metric_name + ), + enum_values=enum_values, + ) + assert ( + labels.get("model_group") == "example-model-group" + ), f"{metric_name} should emit model_group=example-model-group, got {labels.get('model_group')!r}" + + +def test_deployment_failure_metrics_emit_model_group_from_standard_logging_payload(): + """ + End-to-end emit wiring for the failure path. + + The label-list and factory tests above prove the label exists and that + the factory carries a value handed to it, but neither drives the real + set_llm_deployment_failure_metrics code path, so deleting the production + model_group=model_group assignment there would still pass them. This + calls it directly with a standard_logging_object carrying model_group and + asserts the real litellm_deployment_failure_responses / _total_requests + Counter series actually carry it. + """ + from litellm.integrations.prometheus import PrometheusLogger + + _clear_prometheus_registry() + try: + logger = PrometheusLogger() + logger.set_llm_deployment_failure_metrics( + request_kwargs={ + "model": "gpt-4o-mini", + "litellm_params": {"metadata": {}}, + "standard_logging_object": { + "model_group": "example-model-group", + "model_id": "model-123", + "api_base": "https://api.openai.com", + "request_tags": [], + }, + "exception": Exception("boom"), + } + ) + + for metric in ( + logger.litellm_deployment_failure_responses, + logger.litellm_deployment_total_requests, + ): + index = metric._labelnames.index("model_group") + values = {sample_key[index] for sample_key in metric._metrics} + assert values == {"example-model-group"}, ( + f"expected model_group=example-model-group on {metric._name}, got {values}" + ) + finally: + _clear_prometheus_registry() + + +def test_deployment_tpm_rpm_limit_metrics_emit_model_group_from_enum_values(): + """ + End-to-end emit wiring for the tpm/rpm limit gauges. + + _set_deployment_tpm_rpm_limit_metrics used to build its own + UserAPIKeyLabelValues with no model_group parameter at all, dropping the + value even though its only caller (set_llm_deployment_success_metrics) + already had it on enum_values. This drives set_llm_deployment_success_metrics + directly with a deployment that has tpm/rpm configured and asserts the real + litellm_deployment_tpm_limit / litellm_deployment_rpm_limit Gauge series + carry model_group; it fails if that plumbing is removed. + """ + import datetime + + from litellm.integrations.prometheus import PrometheusLogger, UserAPIKeyLabelValues + + _clear_prometheus_registry() + try: + logger = PrometheusLogger() + now = datetime.datetime.now() + enum_values = UserAPIKeyLabelValues( + model_group="example-model-group", + litellm_model_name="gpt-4o-mini", + requested_model="example-model-group", + status_code="200", + ) + logger.set_llm_deployment_success_metrics( + request_kwargs={ + "model": "gpt-4o-mini", + "litellm_params": {"metadata": {"model_info": {"id": "model-123", "tpm": 1000, "rpm": 10}}}, + "standard_logging_object": { + "model_group": "example-model-group", + "model_id": "model-123", + "api_base": "https://api.openai.com", + "hidden_params": {"additional_headers": None, "litellm_overhead_time_ms": None}, + }, + }, + start_time=now, + end_time=now, + enum_values=enum_values, + ) + + for metric in (logger.litellm_deployment_tpm_limit, logger.litellm_deployment_rpm_limit): + index = metric._labelnames.index("model_group") + values = {sample_key[index] for sample_key in metric._metrics} + assert values == {"example-model-group"}, ( + f"expected model_group=example-model-group on {metric._name}, got {values}" + ) + finally: + _clear_prometheus_registry() + + if __name__ == "__main__": test_user_email_in_required_metrics() test_user_email_label_exists() From 7abaf4edb2458b46ac4014beecd0e778ab5a9c89 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 12:30:42 -0700 Subject: [PATCH 91/96] fix(cost-map): add the Vertex shutdown date to gemini-2.5-flash-native-audio (#43024) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 1 + model_prices_and_context_window.json | 1 + 2 files changed, 2 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 8f38866387a..dad89fc58a8 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -68565,6 +68565,7 @@ "source": "https://api.together.ai/v1/models" }, "vertex_ai/gemini-2.5-flash-native-audio": { + "deprecation_date": "2026-12-13", "input_cost_per_audio_token": 3e-06, "input_cost_per_token": 5e-07, "litellm_provider": "vertex_ai", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 8f38866387a..dad89fc58a8 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -68565,6 +68565,7 @@ "source": "https://api.together.ai/v1/models" }, "vertex_ai/gemini-2.5-flash-native-audio": { + "deprecation_date": "2026-12-13", "input_cost_per_audio_token": 3e-06, "input_cost_per_token": 5e-07, "litellm_provider": "vertex_ai", From c9e8a04139af587d3280ea5fb248744e3785f500 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 12:35:34 -0700 Subject: [PATCH 92/96] feat(vertex): native batch JSONL passthrough with cost tracking (#42810) * feat(vertex): native batch JSONL passthrough with cost tracking Add a per-request `passthrough=true` multipart field on `POST /v1/files` (and the same kwarg on `litellm.create_file`) that uploads a native Vertex AI batch JSONL to the deployment's GCS bucket unchanged, so rows using `googleSearch` and other Gemini-only features run as written and the output, `groundingMetadata` included, comes back untouched. Passthrough is sticky through the GCS object path (`litellm-vertex-files/passthrough/...`), so batch create and output retrieval inherit it without new state. Native output rows are costed from their `usageMetadata` with the deployment's model and model_info, in the polling and retrieve paths and for the existing global `disable_vertex_batch_output_transformation` flag, which billed $0 before. The proxy requires the target to resolve to vertex_ai deployments only, refuses `passthrough` with a non-batch purpose, a non-default `target_storage`, or pre-call guardrails, and validates native rows on `request` instead of the OpenAI batch keys. * refactor(vertex): keep native batch row pricing inside the Vertex adapter Moves native Vertex batch row detection, response parsing, and per-row pricing from litellm/batches/batch_utils.py into litellm/llms/vertex_ai/batches/transformation.py, so batch_utils only aggregates the rows it gets back. Adds tests/test_litellm/files to the misc unit shard so the new test directory is claimed by a shard. * fix(files): say what a passthrough batch upload takes when a row is not native The missing-key 400 listed bare key names, so an OpenAI-shaped row under passthrough=true read "Each line must be a JSON object with keys request". The batch line shape now carries its own hint, and the passthrough one says a passthrough upload takes native Vertex batch rows with a request key * fix(batches): bill native Vertex embedding batch rows on the native cost path A native Vertex output row whose response holds an embedding was validated as a generateContent response, so the documented tokenCount-only shape counted as a failed row. Price embedding rows from their own usage (promptTokenCount, else tokenCount) with the helper the transformed embeddings path already used, and drop the prompt-details helper nothing calls anymore. * fix(batches): keep modality batch rates on native Vertex embedding rows An embedding row that carries usageMetadata was billed from promptTokenCount alone, so its promptTokensDetails no longer reached the audio, image, and video batch rates the way it did before the native cost path. Run every row with usageMetadata through the Gemini usage parser and keep the flat tokenCount fallback for embedding rows without it. * fix(batches): price native Vertex batch rows by modelVersion under a wildcard deployment A `vertex_ai/*` deployment hands the batch cost path `*` as the deployment model, which no cost map resolves, so every native (passthrough or flag-on) row was billed at $0. A wildcard deployment model now defers to the row's own `modelVersion`, the way the transformed path already prices by the row's `model`. Also moves the native passthrough tests under tests/test_litellm, the tree codecov reads, and covers the raw upload chunking, the embedding output translation, the unpriceable-row path, and the flag-on dispatch. * fix(batches): keep explicit deployment prices for native Vertex rows without a modelVersion Under a wildcard deployment a native batch row that carries no modelVersion (an embedding row, or a generateContent row Vertex returned without one) was billed at $0 even when the deployment's model_info sets explicit batch prices, because the cost calculator was never called. The row now falls back to the wildcard name, which the cost calculator prices from the explicit model_info, and only a row with neither a modelVersion nor a deployment model is billed at $0 with the warning --------- Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- .github/workflows/test-unit.yml | 1 + litellm/batches/batch_utils.py | 131 +++--- litellm/files/main.py | 16 + .../llms/vertex_ai/batches/transformation.py | 147 +++++-- .../llms/vertex_ai/files/transformation.py | 127 ++++-- .../batch_file_validation.py | 35 +- .../openai_files_endpoints/files_endpoints.py | 109 ++++- litellm/router.py | 1 + litellm/router_utils/batch_utils.py | 6 +- tests/e2e/batches/test_batches_e2e.py | 112 +++++ .../llm_nonconversational.yaml | 3 + tests/e2e/coverage_registry/schema.py | 1 + tests/e2e/e2e_http.py | 7 + .../test_router_batch_utils.py | 1 + .../test_litellm/batches/test_batch_utils.py | 387 ++++++++++++++++++ tests/test_litellm/files/__init__.py | 0 tests/test_litellm/files/test_main.py | 71 ++++ .../vertex_ai/batches/test_transformation.py | 38 +- .../llms/vertex_ai/files/__init__.py | 0 .../vertex_ai/files/test_transformation.py | 310 ++++++++++++++ .../test_files_batch_file_validation.py | 38 +- .../test_files_endpoint.py | 238 +++++++++++ tests/test_litellm/test_router.py | 33 ++ tests/unit/batches/test_batch_utils.py | 17 +- ui/litellm-dashboard/src/lib/http/schema.d.ts | 15 + 25 files changed, 1677 insertions(+), 167 deletions(-) create mode 100644 tests/test_litellm/batches/test_batch_utils.py create mode 100644 tests/test_litellm/files/__init__.py create mode 100644 tests/test_litellm/files/test_main.py create mode 100644 tests/test_litellm/llms/vertex_ai/files/__init__.py create mode 100644 tests/test_litellm/llms/vertex_ai/files/test_transformation.py diff --git a/.github/workflows/test-unit.yml b/.github/workflows/test-unit.yml index 4580ad17a19..686bbc89467 100644 --- a/.github/workflows/test-unit.yml +++ b/.github/workflows/test-unit.yml @@ -108,6 +108,7 @@ jobs: tests/test_litellm/completion_extras tests/test_litellm/containers tests/test_litellm/endpoints + tests/test_litellm/files tests/test_litellm/images tests/test_litellm/interactions tests/test_litellm/messages diff --git a/litellm/batches/batch_utils.py b/litellm/batches/batch_utils.py index 7209ac6a1e7..819a279a43c 100644 --- a/litellm/batches/batch_utils.py +++ b/litellm/batches/batch_utils.py @@ -11,7 +11,10 @@ from litellm.litellm_core_utils.get_litellm_params import AWS_CREDENTIAL_KWARGS_ from litellm.litellm_core_utils.llm_cost_calc.utils import parse_prompt_tokens_details from litellm.llms.base_llm.ocr.transformation import OCRUsageInfo from litellm.llms.bedrock.batches.transformation import titan_embedding_usage_from_batch_output -from litellm.llms.vertex_ai.batches.transformation import vertex_prompt_tokens_details +from litellm.llms.vertex_ai.batches.transformation import ( + is_native_vertex_batch_output_row, + native_vertex_batch_row_stats, +) from litellm.types.llms.openai import Batch from litellm.types.utils import ModelInfo, Usage from litellm.utils import token_counter @@ -31,6 +34,20 @@ class BatchCostUsageResult: _COMPLETED_BATCH_STATUSES: Final = frozenset({"completed", "complete"}) + + +def _uses_native_vertex_output( + custom_llm_provider: str, + model_name: str | None, + first_row: Mapping[str, object] | None, +) -> bool: + if custom_llm_provider != "vertex_ai": + return False + if model_name and getattr(litellm, "disable_vertex_batch_output_transformation", False): + return True + return first_row is not None and is_native_vertex_batch_output_row(first_row) + + _TERMINAL_BATCH_STATUSES: Final = _COMPLETED_BATCH_STATUSES | frozenset({"failed", "cancelled", "expired"}) @@ -66,12 +83,9 @@ async def calculate_batch_cost_and_usage( deployment-specific pricing (e.g. input_cost_per_token_batches) is used instead of the global cost map. """ - if ( - custom_llm_provider == "vertex_ai" - and model_name - and getattr(litellm, "disable_vertex_batch_output_transformation", False) - ): - return calculate_vertex_ai_batch_cost_and_usage(file_content_dictionary, model_name) + first_row: Final = file_content_dictionary[0] if file_content_dictionary else None + if _uses_native_vertex_output(custom_llm_provider, model_name, first_row): + return calculate_vertex_ai_batch_cost_and_usage(file_content_dictionary, model_name, model_info=model_info) return _aggregate_batch_cost_usage_models( entries=file_content_dictionary, @@ -126,11 +140,11 @@ async def _handle_completed_batch( ) output_file_result: Final = ( - calculate_vertex_ai_batch_cost_and_usage(_get_file_content_as_dictionary(file_content), model_name) - if ( - custom_llm_provider == "vertex_ai" - and model_name - and getattr(litellm, "disable_vertex_batch_output_transformation", False) + calculate_vertex_ai_batch_cost_and_usage( + _iter_batch_output_entries(file_content), model_name, model_info=model_info + ) + if _uses_native_vertex_output( + custom_llm_provider, model_name, next(_iter_batch_output_entries(file_content), None) ) else _aggregate_batch_cost_usage_models( entries=_iter_batch_output_entries(file_content), @@ -332,69 +346,36 @@ def _aggregate_batch_cost_usage_models( def calculate_vertex_ai_batch_cost_and_usage( - vertex_ai_batch_responses: list[dict], + vertex_ai_batch_responses: Iterable[dict], model_name: str | None = None, + model_info: ModelInfo | None = None, ) -> BatchCostUsageResult: """ - Calculate both cost and usage from raw Vertex AI batch responses. - - Used only when ``litellm.disable_vertex_batch_output_transformation = True``. - In that case the GCS predictions.jsonl is returned as-is, with each line in - the native Vertex format: - - {"request": ..., "response": {"candidates": [...], "usageMetadata": {...}}} - - usageMetadata contains promptTokenCount, candidatesTokenCount, totalTokenCount. - - A row with no ``response`` is counted as failed - the same signal already - used to skip it from cost/usage aggregation, since Vertex batch prediction - output doesn't establish a distinct error shape in this (non-default) path. + Cost and usage of a native Vertex predictions.jsonl, one + `{"request": ..., "response": {"candidates": [...], "usageMetadata": {...}, "modelVersion": ...}}` + generateContent row or `{"request": ..., "response": {"embedding": {...}, "usageMetadata": {...}}}` + embedding row per line. `model_name` (the deployment model) prices every row, else each row's own + `modelVersion` does; a row without a usable response counts as failed. """ from litellm.cost_calculator import batch_cost_calculator + from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import VertexGeminiConfig - total_prompt_cost = 0.0 # rebind-ok: loop accumulator, matches total_tokens below - total_completion_cost = 0.0 # rebind-ok: loop accumulator, matches total_tokens below - total_tokens = 0 - prompt_tokens = 0 - completion_tokens = 0 - successful_requests = 0 # rebind-ok: loop accumulator, matches total_cost/total_tokens above - failed_requests = 0 # rebind-ok: loop accumulator, matches total_cost/total_tokens above - actual_model_name: Final = model_name or "gemini-2.0-flash-001" - - for response in vertex_ai_batch_responses: - response_body = response.get("response") - if response_body is None: - failed_requests += 1 - continue - successful_requests += 1 - - usage_metadata = response_body.get("usageMetadata", {}) - _prompt = usage_metadata.get("promptTokenCount", 0) or 0 - _completion = usage_metadata.get("candidatesTokenCount", 0) or 0 - _total = usage_metadata.get("totalTokenCount", 0) or (_prompt + _completion) - - line_usage = Usage( - prompt_tokens=_prompt, - completion_tokens=_completion, - total_tokens=_total, - prompt_tokens_details=vertex_prompt_tokens_details(usage_metadata), + row_stats: Final = tuple( + native_vertex_batch_row_stats( + row, + model_name, + model_info=model_info, + calculate_usage=VertexGeminiConfig._calculate_usage, + cost_calculator=batch_cost_calculator, ) - - try: - p_cost, c_cost = batch_cost_calculator( - usage=line_usage, - model=actual_model_name, - custom_llm_provider="vertex_ai", - ) - total_prompt_cost += p_cost - total_completion_cost += c_cost - except Exception as e: - verbose_logger.debug("vertex_ai batch cost calculation error for line: %s", str(e)) - - prompt_tokens += _prompt - completion_tokens += _completion - total_tokens += _total - + for row in vertex_ai_batch_responses + ) + priced: Final = tuple(stats for stats in row_stats if stats is not None) + total_prompt_cost: Final = sum(stats.prompt_cost for stats in priced) + total_completion_cost: Final = sum(stats.completion_cost for stats in priced) + prompt_tokens: Final = sum(stats.usage.prompt_tokens for stats in priced) + completion_tokens: Final = sum(stats.usage.completion_tokens for stats in priced) + total_tokens: Final = sum(stats.total_tokens for stats in priced) total_cost: Final = total_prompt_cost + total_completion_cost verbose_logger.info( "vertex_ai batch cost: cost=%s, prompt=%d, completion=%d, total=%d, successful=%d, failed=%d", @@ -402,8 +383,8 @@ def calculate_vertex_ai_batch_cost_and_usage( prompt_tokens, completion_tokens, total_tokens, - successful_requests, - failed_requests, + len(priced), + len(row_stats) - len(priced), ) return BatchCostUsageResult( @@ -413,9 +394,13 @@ def calculate_vertex_ai_batch_cost_and_usage( prompt_tokens=prompt_tokens, completion_tokens=completion_tokens, ), - models=[actual_model_name], - successful_requests=successful_requests, - failed_requests=failed_requests, + models=( + [model_name] + if model_name + else list(dict.fromkeys(stats.model for stats in priced if stats.model is not None)) + ), + successful_requests=len(priced), + failed_requests=len(row_stats) - len(priced), prompt_cost=total_prompt_cost, completion_cost=total_completion_cost, ) diff --git a/litellm/files/main.py b/litellm/files/main.py index e0804244ff7..72832aeccc9 100644 --- a/litellm/files/main.py +++ b/litellm/files/main.py @@ -176,6 +176,22 @@ def create_file( if logging_obj is None: raise ValueError("logging_obj is required") client: Final = kwargs.get("client") + if litellm_params_dict.get("passthrough") is True and ( + custom_llm_provider != "vertex_ai" or purpose != "batch" + ): + raise litellm.exceptions.BadRequestError( + message=( + "`passthrough=True` uploads the file bytes unchanged for a native Vertex AI batch, so it needs " + f"custom_llm_provider='vertex_ai' and purpose='batch', got '{custom_llm_provider}' and '{purpose}'." + ), + model="n/a", + llm_provider=custom_llm_provider or "n/a", + response=httpx.Response( + status_code=400, + content="passthrough needs a vertex_ai batch", + request=httpx.Request(method="create_file", url="https://github.com/BerriAI/litellm"), + ), + ) ### TIMEOUT LOGIC ### timeout = optional_params.timeout or kwargs.get("request_timeout", 600) or 600 diff --git a/litellm/llms/vertex_ai/batches/transformation.py b/litellm/llms/vertex_ai/batches/transformation.py index f5f1ab2068a..a7dbb058465 100644 --- a/litellm/llms/vertex_ai/batches/transformation.py +++ b/litellm/llms/vertex_ai/batches/transformation.py @@ -1,7 +1,11 @@ -from collections.abc import Mapping -from typing import Any, Final +from collections.abc import Callable, Mapping +from dataclasses import dataclass +from typing import Any, Final, Protocol from urllib.parse import unquote +from pydantic import TypeAdapter, ValidationError + +from litellm._logging import verbose_logger from litellm._uuid import uuid from litellm.llms.vertex_ai.common_utils import ( VertexAIError, @@ -9,35 +13,128 @@ from litellm.llms.vertex_ai.common_utils import ( ) from litellm.types.llms.openai import BatchJobStatus, CreateBatchRequest from litellm.types.llms.vertex_ai import * -from litellm.types.utils import LiteLLMBatch, PromptTokensDetailsWrapper +from litellm.types.llms.vertex_ai import GenerateContentResponseBody +from litellm.types.utils import LiteLLMBatch, ModelInfo, Usage + +_NATIVE_VERTEX_RESPONSE: Final = TypeAdapter(GenerateContentResponseBody) -def vertex_prompt_tokens_details( - usage_metadata: Mapping[str, object], -) -> PromptTokensDetailsWrapper | None: - raw_details: Final = usage_metadata.get("promptTokensDetails") - if not isinstance(raw_details, list): - return None +def _int_field(mapping: Mapping[str, object], key: str) -> int: + value: Final = mapping.get(key) + if isinstance(value, int): + return value + return int(value) if isinstance(value, str) and value.isdigit() else 0 - def _normalize(detail: object) -> tuple[str, int] | None: - if not isinstance(detail, Mapping): + +def vertex_embedding_prompt_token_count(vertex_response: Mapping[str, object]) -> int: + """ + Prompt tokens billed for one Vertex Gemini Embedding batch row. + + Live rows report usage under `usageMetadata`; the documented `tokenCount` is kept as + a fallback. + """ + usage_metadata: Final = vertex_response.get("usageMetadata") + if isinstance(usage_metadata, Mapping): + return _int_field(usage_metadata, "promptTokenCount") + return _int_field(vertex_response, "tokenCount") + + +def is_vertex_embedding_batch_output_response(response_body: Mapping[str, object]) -> bool: + return isinstance(response_body.get("embedding"), dict) + + +def is_native_vertex_batch_output_row(row: Mapping[str, object]) -> bool: + return isinstance(row.get("request"), dict) + + +class NativeVertexBatchCostCalculator(Protocol): + def __call__( + self, + usage: Usage, + model: str, + custom_llm_provider: str | None = None, + model_info: ModelInfo | None = None, + ) -> tuple[float, float]: ... + + +@dataclass(frozen=True, slots=True) +class NativeVertexBatchRowStats: + usage: Usage + total_tokens: int + model: str | None + prompt_cost: float + completion_cost: float + + +def _native_vertex_row_usage( + response_body: Mapping[str, object], + calculate_usage: Callable[[GenerateContentResponseBody], Usage], +) -> Usage | None: + if "usageMetadata" not in response_body: + if not is_vertex_embedding_batch_output_response(response_body): return None - modality: Final = detail.get("modality") - token_count: Final = detail.get("tokenCount") - if not isinstance(modality, str) or not isinstance(token_count, int): - return None - return modality.upper(), token_count - - parsed_details: Final = tuple(_normalize(detail) for detail in raw_details) - normalized: Final = tuple(detail for detail in parsed_details if detail is not None) - if len(normalized) != len(parsed_details): + prompt_tokens: Final = vertex_embedding_prompt_token_count(response_body) + return Usage(prompt_tokens=prompt_tokens, completion_tokens=0, total_tokens=prompt_tokens) + try: + completion_response: Final = _NATIVE_VERTEX_RESPONSE.validate_python(response_body) + except ValidationError as e: + verbose_logger.debug("vertex_ai batch row response is not a GenerateContentResponse: %s", str(e)) return None + return calculate_usage(completion_response) - return PromptTokensDetailsWrapper( - text_tokens=sum(token_count for modality, token_count in normalized if modality in ("TEXT", "DOCUMENT")), - audio_tokens=sum(token_count for modality, token_count in normalized if modality == "AUDIO"), - image_tokens=sum(token_count for modality, token_count in normalized if modality == "IMAGE"), - video_tokens=sum(token_count for modality, token_count in normalized if modality == "VIDEO"), + +def native_vertex_batch_row_stats( + row: Mapping[str, object], + model_name: str | None, + *, + model_info: ModelInfo | None, + calculate_usage: Callable[[GenerateContentResponseBody], Usage], + cost_calculator: NativeVertexBatchCostCalculator, +) -> NativeVertexBatchRowStats | None: + """ + Usage and cost of one native Vertex predictions.jsonl row, a + `{"request": ..., "response": {"candidates": [...], "usageMetadata": {...}, "modelVersion": ...}}` + generateContent object or a `{"request": ..., "response": {"embedding": {...}, "usageMetadata": {...}}}` + embedding object (an embedding row without `usageMetadata` is billed from its documented `tokenCount`). + `model_name` (the deployment model) prices the row unless it is a wildcard, else its own `modelVersion` + does, else the wildcard name so explicit deployment prices still apply; a row without a response, a + generateContent row without `response.usageMetadata`, and a row whose response fails validation are + None (failed). + """ + response_body: Final = row.get("response") + if not isinstance(response_body, dict): + return None + usage: Final = _native_vertex_row_usage(response_body, calculate_usage) + if usage is None: + return None + total_tokens: Final = usage.total_tokens or (usage.prompt_tokens + usage.completion_tokens) + model_version: Final = response_body.get("modelVersion") + deployment_model: Final = model_name if model_name and "*" not in model_name else None + model: Final = deployment_model or (model_version if isinstance(model_version, str) else model_name) + if model is None: + verbose_logger.warning( + "vertex_ai batch output row could not be costed, so it is billed at $0 and the rest of the batch " + "is still billed: the row has no modelVersion and the batch has no deployment model" + ) + return NativeVertexBatchRowStats( + usage=usage, total_tokens=total_tokens, model=None, prompt_cost=0.0, completion_cost=0.0 + ) + try: + prompt_cost, completion_cost = cost_calculator( + usage=usage, model=model, custom_llm_provider="vertex_ai", model_info=model_info + ) + except Exception as e: # noqa: BLE001 # one unpriceable row must not abort the batch's cost accounting + verbose_logger.warning( + "vertex_ai batch output row could not be costed, so it is billed at $0 and the rest of the batch " + "is still billed. model=%s error=%s", + model, + str(e), + ) + return NativeVertexBatchRowStats( + usage=usage, total_tokens=total_tokens, model=model, prompt_cost=0.0, completion_cost=0.0 + ) + return NativeVertexBatchRowStats( + usage=usage, total_tokens=total_tokens, model=model, prompt_cost=prompt_cost, completion_cost=completion_cost ) diff --git a/litellm/llms/vertex_ai/files/transformation.py b/litellm/llms/vertex_ai/files/transformation.py index 80d32289c94..789b36ef3d0 100644 --- a/litellm/llms/vertex_ai/files/transformation.py +++ b/litellm/llms/vertex_ai/files/transformation.py @@ -9,7 +9,7 @@ from collections.abc import AsyncGenerator, Callable, Iterable, Iterator, Mappin from contextlib import aclosing from dataclasses import dataclass from types import MappingProxyType -from typing import Any, Final, TypedDict +from typing import IO, Any, Final, TypedDict from urllib.parse import quote, unquote import httpx @@ -41,6 +41,7 @@ from litellm.llms.base_llm.files.transformation import ( BaseFileUploadStream, LiteLLMLoggingObj, ) +from litellm.llms.vertex_ai.batches.transformation import vertex_embedding_prompt_token_count from litellm.llms.vertex_ai.common_utils import ( _convert_vertex_datetime_to_openai_datetime, get_vertex_ai_fine_tuned_endpoint_id, @@ -56,6 +57,7 @@ from litellm.types.files import StreamingMediaUploadConfig from litellm.types.llms.openai import ( AllMessageValues, CreateFileRequest, + FileContent, FileTypes, HttpxBinaryResponseContent, OpenAICreateFileRequestOptionalParams, @@ -87,6 +89,8 @@ _EMBED_REQUEST_FIELD_BY_GEMINI_PARAM: Final = ( _VERTEX_BATCH_FANNED_OUT_KEY_PATTERN: Final = re.compile(r"(?P[^#]*)#(?P\d+)/(?P\d+)") _JSONL_NEWLINE: Final = b"\n" _BATCH_OUTPUT_FIRST_ROW_PEEK_LIMIT_BYTES: Final = 32 * 1024 * 1024 +_PASSTHROUGH_MANAGED_GCS_PREFIX: Final = f"{VERTEX_AI_MANAGED_GCS_PREFIX}passthrough/" +_RAW_UPLOAD_CHUNK_BYTES: Final = 1024 * 1024 class _GcsObjectMetadataJson(TypedDict, total=False): @@ -418,19 +422,6 @@ def _split_vertex_batch_key(vertex_output_row: Mapping[str, object]) -> tuple[st return unquote(match["custom_id"]), int(match["index"]), int(match["total"]) -def _embedding_prompt_token_count(vertex_response: _VertexEmbeddingResponse) -> int: - """ - Prompt tokens billed for one Vertex Gemini Embedding batch row. - - Live rows report usage under `usageMetadata`; the documented `tokenCount` is kept as - a fallback. - """ - usage_metadata = vertex_response.get("usageMetadata") - if isinstance(usage_metadata, Mapping): - return int(usage_metadata.get("promptTokenCount") or 0) - return int(vertex_response.get("tokenCount") or 0) - - def _vertex_embeddings_rows_to_openai_batch_output_row( custom_id: str, vertex_output_rows: tuple[_VertexEmbeddingBatchRow, ...], @@ -471,7 +462,7 @@ def _vertex_embeddings_rows_to_openai_batch_output_row( ) responses = tuple(row["response"] for row in vertex_output_rows) - token_count = sum(_embedding_prompt_token_count(response) for response in responses) + token_count = sum(vertex_embedding_prompt_token_count(response) for response in responses) body = EmbeddingResponse( model=model or "", data=[ @@ -528,6 +519,16 @@ def _model_from_managed_gcs_url(url: str) -> str | None: return match.group(1) if match else None +def is_passthrough_managed_gcs_url(url: str) -> bool: + decoded_url: Final = unquote(url) + managed_prefix_start: Final = decoded_url.find(VERTEX_AI_MANAGED_GCS_PREFIX) + return managed_prefix_start >= 0 and decoded_url.startswith(_PASSTHROUGH_MANAGED_GCS_PREFIX, managed_prefix_start) + + +def is_passthrough_batch_upload(create_file_data: Mapping[str, object], litellm_params: Mapping[str, object]) -> bool: + return create_file_data.get("purpose") == "batch" and litellm_params.get("passthrough") is True + + def _is_embeddings_batch_entry(openai_entry: Mapping[str, object]) -> bool: """ Whether an OpenAI batch JSONL line targets the embeddings endpoint. @@ -791,6 +792,58 @@ class _OpenAIToVertexBatchUploadStream(BaseFileUploadStream): return self._iter_vertex_jsonl_chunks() +def _read_chunk_as_bytes(handle: IO[bytes]) -> bytes: + chunk: Final[bytes | str] = handle.read(_RAW_UPLOAD_CHUNK_BYTES) + return chunk.encode("utf-8") if isinstance(chunk, str) else bytes(chunk) + + +def _iter_raw_file_chunks(file_content: FileTypes) -> Iterator[bytes]: + content: Final[FileContent | str] = file_content[1] if isinstance(file_content, tuple) else file_content + if isinstance(content, (bytes, bytearray)): + yield from ( + bytes(content[offset : offset + _RAW_UPLOAD_CHUNK_BYTES]) + for offset in range(0, len(content), _RAW_UPLOAD_CHUNK_BYTES) + ) + return + if isinstance(content, str): + yield content.encode("utf-8") + return + if isinstance(content, PathLike): + with open(str(content), "rb") as handle: + yield from iter(lambda: handle.read(_RAW_UPLOAD_CHUNK_BYTES), b"") + return + if not hasattr(content, "read"): + raise ValueError("Unsupported file content type") + seek: Final = getattr(content, "seek", None) + if seek is None: + raise ValueError( + "Batch upload file handle must be seekable; got a non-seekable " + "stream. Pass bytes, a path, or a seekable handle." + ) + seek(0) + yield from iter(lambda: _read_chunk_as_bytes(content), b"") + + +class _RawFileUploadStream(BaseFileUploadStream): + def __init__(self, file_content: FileTypes) -> None: + self._file_content = file_content + + def iter_bytes(self) -> Iterator[bytes]: + return _iter_raw_file_chunks(self._file_content) + + +def _managed_batch_object_name(raw_model: str, *, passthrough: bool) -> str: + endpoint_id: Final = get_vertex_ai_fine_tuned_endpoint_id(raw_model) + model_path: Final = ( + f"endpoints/{endpoint_id}" + if endpoint_id is not None + else (raw_model if "publishers/google/models" in raw_model else f"publishers/google/models/{raw_model}") + ) + safe_model_path: Final = sanitize_cloud_object_path(model_path, fallback="model") + prefix: Final = _PASSTHROUGH_MANAGED_GCS_PREFIX if passthrough else VERTEX_AI_MANAGED_GCS_PREFIX + return f"{prefix}{safe_model_path}/{uuid.uuid4()}" + + class VertexAIFilesConfig(VertexBase, BaseFilesConfig): """ Config for VertexAI Files @@ -848,23 +901,34 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): if deployment_model else openai_jsonl_content[0].get("body", {}).get("model", "") ) - endpoint_id: Final = get_vertex_ai_fine_tuned_endpoint_id(raw_model) - model_path: Final = ( - f"endpoints/{endpoint_id}" - if endpoint_id is not None - else (raw_model if "publishers/google/models" in raw_model else f"publishers/google/models/{raw_model}") - ) - safe_model_path: Final = sanitize_cloud_object_path(model_path, fallback="model") - object_name: Final = f"{VERTEX_AI_MANAGED_GCS_PREFIX}{safe_model_path}/{uuid.uuid4()}" - return object_name + return _managed_batch_object_name(raw_model, passthrough=False) - def get_object_name(self, file_data: FileTypes, purpose: str, deployment_model: str | None = None) -> str: + def _get_passthrough_gcs_object_name(self, deployment_model: str | None) -> str: + if not deployment_model: + raise VertexAIError( + status_code=400, + message=( + "Native Vertex batch passthrough uploads need the deployment model to name the GCS object, " + "since native rows carry no model: pass `target_model_names` (proxy) or `model` (SDK)." + ), + ) + return _managed_batch_object_name(deployment_model.removeprefix("vertex_ai/"), passthrough=True) + + def get_object_name( + self, + file_data: FileTypes, + purpose: str, + deployment_model: str | None = None, + passthrough: bool = False, + ) -> str: """ Get the object name for the request. Reads only the first JSONL entry (streamed) for batch files, so a large upload is never materialized just to derive the GCS object name. """ + if purpose == "batch" and passthrough: + return self._get_passthrough_gcs_object_name(deployment_model) if purpose == "batch": ## 1. If jsonl, derive the object name from the deployment model (or the first entry's) first_entry: Final = next(_iter_openai_jsonl_entries(file_data), None) @@ -922,6 +986,7 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): file_data, purpose, deployment_model=configured_model if isinstance(configured_model, str) else None, + passthrough=is_passthrough_batch_upload(data, litellm_params), ) if object_prefix: object_name = f"{object_prefix}/{object_name}" @@ -984,6 +1049,14 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): if file_data is None: raise ValueError("file is required") + if is_passthrough_batch_upload(create_file_data, litellm_params): + return { + "streaming_media_upload": StreamingMediaUploadConfig( + body_stream=_RawFileUploadStream(file_data), + content_type="application/json", + ) + } + _, content_type = extract_file_metadata(file_data) if FilesAPIUtils.is_batch_jsonl_request( create_file_data=create_file_data, @@ -1164,6 +1237,8 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): # transformation, e.g. if they consume raw `predictions.jsonl` directly. if getattr(litellm, "disable_vertex_batch_output_transformation", False): return HttpxBinaryResponseContent(response=raw_response) + if is_passthrough_managed_gcs_url(str(raw_response.request.url)): + return HttpxBinaryResponseContent(response=raw_response) # Try to transform batch output if it's a JSONL file content: Final = raw_response.content @@ -1209,7 +1284,7 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig): Everything else is passed through unchanged, including a row that fails to transform mid-stream. """ - if litellm.disable_vertex_batch_output_transformation: + if litellm.disable_vertex_batch_output_transformation or is_passthrough_managed_gcs_url(request_url): return FileContentStreamingResult(stream_iterator=stream_iterator, headers=headers) first_line, buffered = await _peek_first_jsonl_line( diff --git a/litellm/proxy/openai_files_endpoints/batch_file_validation.py b/litellm/proxy/openai_files_endpoints/batch_file_validation.py index a41bd36d510..fcd3be56ae7 100644 --- a/litellm/proxy/openai_files_endpoints/batch_file_validation.py +++ b/litellm/proxy/openai_files_endpoints/batch_file_validation.py @@ -8,10 +8,25 @@ from typing_extensions import assert_never from litellm.proxy._types import ProxyException -BATCH_LINE_REQUIRED_KEYS: Final = ("custom_id", "method", "url", "body") _MB: Final = 1024 * 1024 +@dataclass(frozen=True, slots=True) +class BatchLineShape: + required_keys: tuple[str, ...] + hint: str + + +BATCH_LINE_SHAPE: Final = BatchLineShape( + required_keys=("custom_id", "method", "url", "body"), + hint="Each line must be a JSON object with keys custom_id, method, url, body", +) +PASSTHROUGH_BATCH_LINE_SHAPE: Final = BatchLineShape( + required_keys=("request",), + hint="A passthrough upload takes native Vertex batch rows, so each line must be a JSON object with a request key", +) + + @dataclass(frozen=True, slots=True) class BatchFileTooLarge: size_bytes: int @@ -42,6 +57,7 @@ class BatchFileLineNotObject: class BatchFileMissingLineKey: line_number: int key: str + line_shape: BatchLineShape = BATCH_LINE_SHAPE BatchFileValidationFailure = ( @@ -70,20 +86,20 @@ def _iter_lines(file_source: bytes | BinaryIO) -> Iterator[bytes]: return iter(file_source) -def _check_line(line_number: int, raw_line: bytes) -> BatchFileValidationFailure | None: +def _check_line(line_number: int, raw_line: bytes, line_shape: BatchLineShape) -> BatchFileValidationFailure | None: try: parsed: Final = json.loads(raw_line) except (json.JSONDecodeError, UnicodeDecodeError): return BatchFileInvalidJsonLine(line_number=line_number) if not isinstance(parsed, dict): return BatchFileLineNotObject(line_number=line_number) - missing: Final = next((key for key in BATCH_LINE_REQUIRED_KEYS if key not in parsed), None) + missing: Final = next((key for key in line_shape.required_keys if key not in parsed), None) if missing is None: return None - return BatchFileMissingLineKey(line_number=line_number, key=missing) + return BatchFileMissingLineKey(line_number=line_number, key=missing, line_shape=line_shape) -def _scan_lines(file_source: bytes | BinaryIO) -> BatchFileValidationFailure | None: +def _scan_lines(file_source: bytes | BinaryIO, line_shape: BatchLineShape) -> BatchFileValidationFailure | None: content_lines: Final = ( (line_number, raw_line) for line_number, raw_line in enumerate(_iter_lines(file_source), start=1) @@ -96,7 +112,7 @@ def _scan_lines(file_source: bytes | BinaryIO) -> BatchFileValidationFailure | N ( failure for line_number, raw_line in chain((first_line,), content_lines) - for failure in (_check_line(line_number, raw_line),) + for failure in (_check_line(line_number, raw_line, line_shape),) if failure is not None ), None, @@ -107,6 +123,7 @@ def check_batch_file_upload( filename: str | None, file_source: bytes | BinaryIO, max_batch_file_size_mb: int | None, + line_shape: BatchLineShape = BATCH_LINE_SHAPE, ) -> BatchFileValidationFailure | None: if filename is None or not filename.lower().endswith(".jsonl"): return BatchFileWrongExtension(filename=filename or "") @@ -114,7 +131,7 @@ def check_batch_file_upload( size_bytes: Final = _file_size_bytes(file_source) if size_bytes > max_batch_file_size_mb * _MB: return BatchFileTooLarge(size_bytes=size_bytes, limit_mb=max_batch_file_size_mb) - scan_failure: Final = _scan_lines(file_source) + scan_failure: Final = _scan_lines(file_source, line_shape) if not isinstance(file_source, bytes): file_source.seek(0) return scan_failure @@ -169,11 +186,11 @@ def raise_batch_file_validation_failure(failure: BatchFileValidationFailure) -> param="file", code=400, ) - case BatchFileMissingLineKey(line_number=line_number, key=key): + case BatchFileMissingLineKey(line_number=line_number, key=key, line_shape=line_shape): raise ProxyException( message=( f"Missing required parameter: '{key}' (batch input file line {line_number}). " - f"Each line must be a JSON object with keys {', '.join(BATCH_LINE_REQUIRED_KEYS)}. " + f"{line_shape.hint}. " "The file was not forwarded to the provider." ), type="invalid_request_error", diff --git a/litellm/proxy/openai_files_endpoints/files_endpoints.py b/litellm/proxy/openai_files_endpoints/files_endpoints.py index ea2cad558c2..f5e62da5962 100644 --- a/litellm/proxy/openai_files_endpoints/files_endpoints.py +++ b/litellm/proxy/openai_files_endpoints/files_endpoints.py @@ -57,6 +57,8 @@ from litellm.proxy.common_utils.openai_error_payload import ( openai_error_type, ) from litellm.proxy.openai_files_endpoints.batch_file_validation import ( + BATCH_LINE_SHAPE, + PASSTHROUGH_BATCH_LINE_SHAPE, check_batch_file_upload, raise_batch_file_validation_failure, ) @@ -207,10 +209,91 @@ def get_files_provider_config( return None +def _deployment_provider(llm_router: Router, model_id: str, team_id: str | None) -> str | None: + credentials: Final = llm_router.get_deployment_credentials_with_provider(model_id=model_id, team_id=team_id) + return None if credentials is None else credentials.get("custom_llm_provider") + + +def _resolves_to_vertex_deployments_only(llm_router: Router | None, model_name: str, team_id: str | None) -> bool: + if llm_router is None or _deployment_provider(llm_router, model_name, team_id) != "vertex_ai": + return False + return all( + _deployment_provider(llm_router, str(deployment["model_info"]["id"]), team_id) == "vertex_ai" + for deployment in llm_router.get_model_list(model_name=model_name, team_id=team_id) or () + if "id" in deployment.get("model_info", {}) + ) + + +def _validate_passthrough_upload( + *, + purpose: str, + target_model_names: Sequence[str], + model: str | None, + target_storage: str | None, + llm_router: Router | None, + team_id: str | None, +) -> None: + if purpose != "batch": + raise ProxyException( + message=( + "`passthrough` uploads the file bytes unchanged for a native Vertex batch, " + f"so purpose must be 'batch', got '{purpose}'." + ), + type="invalid_request_error", + param="passthrough", + code=400, + ) + if target_storage and target_storage != "default": + raise ProxyException( + message=( + "`passthrough` writes the native batch file to the Vertex AI deployment's GCS bucket, " + f"so it cannot be combined with target_storage='{target_storage}'." + ), + type="invalid_request_error", + param="target_storage", + code=400, + ) + named_deployments: Final = ( + *(("target_model_names", name) for name in target_model_names), + *((("model", model),) if model else ()), + ) + if not named_deployments: + raise ProxyException( + message=( + "`passthrough` needs the Vertex AI deployment that will run the batch, " + "since native rows carry no model: pass `target_model_names` or `model`." + ), + type="invalid_request_error", + param="target_model_names", + code=400, + ) + offending: Final = next( + ( + (param, name) + for param, name in named_deployments + if not _resolves_to_vertex_deployments_only(llm_router, name, team_id) + ), + None, + ) + if offending is None: + return + param, name = offending + raise ProxyException( + message=( + f"`passthrough` is only supported for Vertex AI deployments; '{name}' does not resolve " + "to vertex_ai deployments only." + ), + type="invalid_request_error", + param=param, + code=400, + ) + + async def _scan_batch_upload( *, file_source: bytes | BinaryIO, purpose: str, + passthrough: bool, request_metadata: Mapping[str, object], user_api_key_dict: UserAPIKeyAuth, proxy_logging_obj: ProxyLogging, @@ -222,6 +305,17 @@ async def _scan_batch_upload( or not proxy_logging_obj.has_pre_call_guardrails(request_metadata) ): return None + if passthrough: + raise ProxyException( + message=( + "Batch guardrails cannot scan native Vertex batch rows, so a `passthrough` upload is refused " + "when the key, team, or request has pre-call guardrails configured. " + "The file was not forwarded to the provider." + ), + type="invalid_request_error", + param="passthrough", + code=400, + ) outcome: Final = await scan_batch_input_file( file_source=file_source, request_metadata=request_metadata, @@ -458,6 +552,7 @@ async def create_file( custom_llm_provider: str = Form(default="openai"), file: UploadFile = File(...), litellm_metadata: str | None = Form(default=None), + passthrough: bool = Form(default=False), user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), ): """ @@ -560,17 +655,28 @@ async def create_file( if blocked_extension_failure is not None: raise_upload_validation_failure(blocked_extension_failure) + if passthrough: + _validate_passthrough_upload( + purpose=purpose, + target_model_names=target_model_names_list, + model=model_param, + target_storage=target_storage, + llm_router=llm_router, + team_id=user_api_key_dict.team_id, + ) + if purpose == "batch": batch_file_failure: Final = await asyncio.to_thread( check_batch_file_upload, file.filename, file_source, _MAX_BATCH_FILE_SIZE_MB_ADAPTER.validate_python(general_settings.get("max_batch_file_size_mb")), + PASSTHROUGH_BATCH_LINE_SHAPE if passthrough else BATCH_LINE_SHAPE, ) if batch_file_failure is not None: raise_batch_file_validation_failure(batch_file_failure) - data = {} + data = {"passthrough": True} if passthrough else {} # Parse expires_after if provided expires_after: FileExpiresAfter | None = None @@ -673,6 +779,7 @@ async def create_file( scan_result: Final = await _scan_batch_upload( file_source=file_source, purpose=purpose, + passthrough=passthrough, request_metadata=request_metadata, user_api_key_dict=user_api_key_dict, proxy_logging_obj=proxy_logging_obj, diff --git a/litellm/router.py b/litellm/router.py index d328fbbb12f..8960cd92cd8 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -5980,6 +5980,7 @@ class Router: replace_model_in_jsonl_bool: Final = should_replace_model_in_jsonl( purpose=purpose, + passthrough=kwargs.get("passthrough") is True, ) if replace_model_in_jsonl_bool: file = replace_model_in_jsonl( diff --git a/litellm/router_utils/batch_utils.py b/litellm/router_utils/batch_utils.py index be20c358202..386a2135239 100644 --- a/litellm/router_utils/batch_utils.py +++ b/litellm/router_utils/batch_utils.py @@ -62,15 +62,15 @@ def parse_jsonl_with_embedded_newlines(content: str) -> list[dict]: def should_replace_model_in_jsonl( purpose: OpenAIFilesPurpose, + passthrough: bool = False, ) -> bool: """ Check if the model name should be replaced in the JSONL file for the deployment model name. Azure raises an error on create batch if the model name for deployment is not in the .jsonl. + A passthrough upload keeps the caller's bytes untouched, so its rows are never rewritten. """ - if purpose == "batch": - return True - return False + return purpose == "batch" and not passthrough def replace_model_in_jsonl(file_content: FileTypes, new_model_name: str) -> FileTypes: diff --git a/tests/e2e/batches/test_batches_e2e.py b/tests/e2e/batches/test_batches_e2e.py index 9bb6d05bec8..6e9cf45e787 100644 --- a/tests/e2e/batches/test_batches_e2e.py +++ b/tests/e2e/batches/test_batches_e2e.py @@ -61,6 +61,7 @@ from e2e_http import ( StreamingResponse, Success, UnknownApiError, + proxy_error, require_successful_call, unwrap, ) @@ -1752,3 +1753,114 @@ class TestBatchTerminalState: assert (cost_row.total_tokens or 0) > 0, ( f"batch cost row has no token usage: {cost_row.total_tokens!r}" ) + + +NATIVE_VERTEX_BATCH_ROWS: Final = b"".join( + json.dumps( + { + "request": { + "contents": [{"role": "user", "parts": [{"text": text}]}], + "tools": [{"googleSearch": {"excludeDomains": ["example.com"]}}], + } + } + ).encode() + + b"\n" + for text in ("What is the tallest building in the world?", "Who won the last FIFA World Cup?") +) +VERTEX_BATCH_PROVIDER: Final = next(p for p in PROVIDERS if p.name == "vertex_ai") + + +class TestVertexNativePassthrough: + """`passthrough=true` on POST /v1/files uploads native Vertex batch JSONL byte for + byte (no OpenAI-to-Vertex translation, so `googleSearch` tools and the grounding + metadata they produce survive), and a batch created from that file is accepted. + + Terminal-state assertions (native output rows with groundingMetadata, the spend + row) are deliberately not here: retrieving a non-terminal batch books a $0 spend + row that blocks the real-cost row, the same reason TestBatchTerminalState polls + the list endpoint only. Those are proven by the PR's live curl proof instead. + """ + + @pytest.mark.covers( + "llm.files.vertex.native_passthrough.nonstream.works", + "llm.batches.vertex.native_passthrough.nonstream.works", + exercised_on=["files", "batches"], + ) + def test_native_jsonl_round_trips_untouched_and_starts_a_batch( + self, client: BatchClient, resources: ResourceManager, batch_deployments: None + ) -> None: + key = resources.key() + file = unwrap( + client.upload_file( + content=NATIVE_VERTEX_BATCH_ROWS, + form=FileUploadForm( + purpose="batch", target_model_names=VERTEX_BATCH_PROVIDER.model, passthrough=True + ), + key=key, + ) + ) + resources.defer(lambda: cleanup_file(client, file.id, key=key)) + assert_file_object(file, provider="vertex_ai") + assert is_managed_id(file.id), f"passthrough upload must return a managed file id, got {file.id!r}" + assert file.bytes == len(NATIVE_VERTEX_BATCH_ROWS), ( + f"passthrough upload must report the caller's byte count, got {file.bytes}" + ) + + downloaded = client.proxy.transport.download( + f"/v1/files/{file.id}/content", headers=client.proxy.transport.bearer(key) + ) + assert downloaded.status_code == 200, ( + f"file content must be 200, got {downloaded.status_code}: {downloaded.body[:300]}" + ) + assert downloaded.body.encode() == NATIVE_VERTEX_BATCH_ROWS, ( + "passthrough file content must be the uploaded native rows byte for byte" + ) + + created = client.create_batch(body=BatchCreateBody(input_file_id=file.id), key=key) + require_successful_call(created) + batch = BatchObject.model_validate_json(created.body) + resources.defer(lambda: cleanup_batch(client, batch.id, key=key, delete_output_files=True)) + assert is_managed_id(batch.id), f"passthrough batch must be LiteLLM-managed, got {batch.id!r}" + assert batch.status in CREATED_BATCH_STATUSES, f"passthrough batch has non-transitional status {batch.status!r}" + assert batch.input_file_id == file.id + + @pytest.mark.covers("llm.files.vertex.native_passthrough_validation.nonstream.works", exercised_on=["files"]) + @pytest.mark.parametrize( + "content, form, expected_param", + [ + pytest.param( + NATIVE_VERTEX_BATCH_ROWS, + FileUploadForm(purpose="batch", passthrough=True), + "target_model_names", + id="no-target-model", + ), + pytest.param( + NATIVE_VERTEX_BATCH_ROWS, + FileUploadForm(purpose="batch", target_model_names=OPENAI_BATCH_MODEL, passthrough=True), + "target_model_names", + id="non-vertex-target-model", + ), + pytest.param( + render_jsonl(VERTEX_BATCH_PROVIDER.raw_model), + FileUploadForm(purpose="batch", target_model_names=VERTEX_BATCH_PROVIDER.model, passthrough=True), + "request", + id="openai-shaped-rows", + ), + ], + ) + def test_passthrough_upload_is_rejected_outside_a_native_vertex_batch( + self, + content: bytes, + form: FileUploadForm, + expected_param: str, + client: BatchClient, + resources: ResourceManager, + batch_deployments: None, + ) -> None: + key = resources.key() + result = client.upload_file(content=content, form=form, key=key) + assert isinstance(result, UnknownApiError), f"expected a 400, got {result!r}" + assert result.status_code == 400, f"expected 400, got {result.status_code}: {result.body[:300]}" + error = proxy_error(result.body) + assert error.param == expected_param, f"unexpected error param in {error!r}" + assert "passthrough" in error.message diff --git a/tests/e2e/coverage_registry/llm_nonconversational.yaml b/tests/e2e/coverage_registry/llm_nonconversational.yaml index 3e389acc2a9..7d334ed41ff 100644 --- a/tests/e2e/coverage_registry/llm_nonconversational.yaml +++ b/tests/e2e/coverage_registry/llm_nonconversational.yaml @@ -21,6 +21,7 @@ - {id: llm.batches.openai_provider_fallback.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py", rationale: "Provider-fallback raw-id scenario"} - {id: llm.batches.azure_openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:98", rationale: "Azure batches all scenarios"} - {id: llm.batches.vertex.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:98", rationale: "Vertex batches"} +- {id: llm.batches.vertex.native_passthrough.nonstream.works, module: llm, tier: P1, subject_endpoint: batches, route: vertex, capability: native_passthrough, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py / LIT-4790", rationale: "A batch created from a passthrough-uploaded native Vertex JSONL file is accepted and starts on the deployment named at upload"} - {id: llm.batches.bedrock.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:98", rationale: "Bedrock batches (encoded/unified only)"} - {id: llm.batches.bedrock.assume_role.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: bedrock_converse, capability: assume_role, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Bedrock batch create under STS assume-role credentials"} - {id: llm.batches.bedrock.govcloud_partition.nonstream.works, module: llm, tier: P0, subject_endpoint: batches, route: bedrock_converse, capability: govcloud_partition, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Bedrock batch create in the us-gov-west-1 partition"} @@ -46,6 +47,8 @@ - {id: llm.files.openai.passthrough.nonstream.works, module: llm, tier: P1, subject_endpoint: files, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_passthrough_e2e.py", rationale: "POST/DELETE /openai_passthrough/v1/files relay OpenAI's own file object; the dedicated prefix must not bind as a provider name on the /{provider}/v1/files route (GitHub issue #36086)"} - {id: llm.files.azure_openai.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:45", rationale: "Azure file upload managed backend"} - {id: llm.files.vertex.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:52", rationale: "Vertex file upload to GCS"} +- {id: llm.files.vertex.native_passthrough.nonstream.works, module: llm, tier: P1, subject_endpoint: files, route: vertex, capability: native_passthrough, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py / LIT-4790", rationale: "POST /v1/files with passthrough=true ships native Vertex batch JSONL (googleSearch tools and all) to GCS untouched and GET /v1/files/{id}/content returns the same bytes"} +- {id: llm.files.vertex.native_passthrough_validation.nonstream.works, module: llm, tier: P1, subject_endpoint: files, route: vertex, capability: input_validation, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py / LIT-4790", rationale: "passthrough=true without a Vertex target_model_names, or with OpenAI-shaped rows, is a 400 naming the offending field and nothing is uploaded"} - {id: llm.files.bedrock.upload.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "batches/capabilities.py:59", rationale: "Bedrock file upload to S3"} - {id: llm.files.bedrock.govcloud_partition.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: bedrock_converse, capability: govcloud_partition, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py", rationale: "Bedrock file upload to an S3 bucket in the us-gov-west-1 partition"} - {id: llm.files.bedrock.split_s3_credentials.nonstream.works, module: llm, tier: P0, subject_endpoint: files, route: bedrock_converse, capability: split_s3_credentials, streaming: nonstream, assertions: [works], source: "test_batches_e2e.py / LIT-8297", rationale: "Bedrock file upload, content and delete sign S3 with s3_access_key_id / s3_secret_access_key when they differ from the aws_* identity"} diff --git a/tests/e2e/coverage_registry/schema.py b/tests/e2e/coverage_registry/schema.py index e009b02b69c..e5626144fad 100644 --- a/tests/e2e/coverage_registry/schema.py +++ b/tests/e2e/coverage_registry/schema.py @@ -73,6 +73,7 @@ LlmCapability = Literal[ "long_context_1m", "mid_conversation_system", "multi_turn", + "native_passthrough", "pdf_input", "prompt_cache_1h", "prompt_cache_5m", diff --git a/tests/e2e/e2e_http.py b/tests/e2e/e2e_http.py index 022caddd42a..e5d50d05c87 100644 --- a/tests/e2e/e2e_http.py +++ b/tests/e2e/e2e_http.py @@ -69,6 +69,7 @@ class FileUploadForm(BaseModel): purpose: str = "batch" target_model_names: str | None = None custom_llm_provider: str | None = None + passthrough: bool | None = None # ---------- Result types ---------- @@ -376,12 +377,18 @@ class ProxyErrorDetail(BaseModel): message: str type: str code: str + param: str | None = None class _ProxyErrorBody(BaseModel): error: ProxyErrorDetail +def proxy_error(body: str) -> ProxyErrorDetail: + """The proxy's own error envelope (`{"error": {message, type, param, code}}`) parsed off a rejected call.""" + return _ProxyErrorBody.model_validate_json(body).error + + def relayed_provider_rate_limit(outcome: RateLimitedError) -> ProxyErrorDetail | None: """The provider's own 429 as the proxy relayed it, or None when the 429 is the proxy's own.""" if PROVIDER_RATE_LIMIT_MARKER not in outcome.body: diff --git a/tests/router_unit_tests/test_router_batch_utils.py b/tests/router_unit_tests/test_router_batch_utils.py index e274ac61a01..4336185a07f 100644 --- a/tests/router_unit_tests/test_router_batch_utils.py +++ b/tests/router_unit_tests/test_router_batch_utils.py @@ -141,6 +141,7 @@ def test_should_replace_model_in_jsonl(): from litellm.router_utils.batch_utils import should_replace_model_in_jsonl assert should_replace_model_in_jsonl(purpose="batch") is True + assert should_replace_model_in_jsonl(purpose="batch", passthrough=True) is False assert should_replace_model_in_jsonl(purpose="test") is False assert should_replace_model_in_jsonl(purpose="user_data") is False diff --git a/tests/test_litellm/batches/test_batch_utils.py b/tests/test_litellm/batches/test_batch_utils.py new file mode 100644 index 00000000000..0b2bfe9d266 --- /dev/null +++ b/tests/test_litellm/batches/test_batch_utils.py @@ -0,0 +1,387 @@ +import json + +import pytest + +import litellm +import litellm.batches.batch_utils as bu +from litellm.types.llms.openai import Batch + +GROUNDED_USAGE_METADATA = { + "promptTokenCount": 19, + "candidatesTokenCount": 59, + "thoughtsTokenCount": 406, + "toolUsePromptTokenCount": 73, + "totalTokenCount": 557, + "promptTokensDetails": [{"modality": "TEXT", "tokenCount": 19}], + "candidatesTokensDetails": [{"modality": "TEXT", "tokenCount": 59}], + "toolUsePromptTokensDetails": [{"modality": "TEXT", "tokenCount": 73}], + "trafficType": "ON_DEMAND", +} +PASSTHROUGH_OUTPUT_URI = ( + "gs://litellm-bucket/litellm-vertex-files/passthrough/publishers/google/models/gemini-2.5-flash/u/" + "predictions.jsonl" +) +UNGROUNDED_USAGE_METADATA = { + "promptTokenCount": 20, + "candidatesTokenCount": 48, + "thoughtsTokenCount": 195, + "toolUsePromptTokenCount": 73, + "totalTokenCount": 336, + "promptTokensDetails": [{"modality": "TEXT", "tokenCount": 20}], + "trafficType": "ON_DEMAND", +} + + +def _batch(output_file_id: str) -> Batch: + return Batch( + id="b", + completion_window="24h", + created_at=1, + endpoint="/v1/chat/completions", + input_file_id="f", + object="batch", + status="completed", + output_file_id=output_file_id, + ) + + +def _vertex_jsonl(rows: list[dict]) -> bytes: + return "\n".join(json.dumps(row) for row in rows).encode() + + +def _vertex_openai_row(custom_id: str, model: str, prompt_tokens: int, completion_tokens: int) -> dict: + return { + "id": f"batch_req_{custom_id}", + "custom_id": custom_id, + "response": { + "status_code": 200, + "request_id": custom_id, + "body": { + "id": f"chatcmpl-{custom_id}", + "object": "chat.completion", + "model": model, + "choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}], + "usage": { + "prompt_tokens": prompt_tokens, + "completion_tokens": completion_tokens, + "total_tokens": prompt_tokens + completion_tokens, + }, + }, + }, + "error": None, + } + + +def _native_vertex_row(usage_metadata: dict, *, grounded: bool, model_version: str | None = "gemini-2.5-flash"): + candidate = {"content": {"role": "model", "parts": [{"text": "ok"}]}, "finishReason": "STOP"} + grounding = {"groundingMetadata": {"webSearchQueries": ["q"]}} if grounded else {} + response = {"candidates": [{**candidate, **grounding}], "usageMetadata": usage_metadata} + return { + "request": {"contents": [{"role": "user", "parts": [{"text": "q"}]}], "tools": [{"googleSearch": {}}]}, + "status": "", + "response": {**response, **({"modelVersion": model_version} if model_version else {})}, + "processed_time": "2026-09-23T19:02:00.000+00:00", + } + + +def _capture_cost_calls(monkeypatch, prompt_cost=0.5, completion_cost=0.25) -> list: + import litellm.cost_calculator as cc + + calls: list = [] + + def _calc(**kw): + calls.append(kw) + return (prompt_cost, completion_cost) + + monkeypatch.setattr(cc, "batch_cost_calculator", _calc) + return calls + + +def test_vertex_native_cost_bills_embedding_rows(monkeypatch): + monkeypatch.setitem(litellm.model_cost, "vertex_ai/gemini-embedding-2", {"input_cost_per_token_batches": 1e-7}) + rows = [ + { + "key": "id_1", + "status": "", + "request": {"content": {"parts": [{"text": "hello world"}]}}, + "response": {"embedding": {"values": [0.1, 0.2]}, "usageMetadata": {"promptTokenCount": 2}}, + }, + { + "key": "id_2", + "status": "", + "request": {"content": {"parts": [{"text": "hello"}]}}, + "response": {"embedding": {"values": [0.3]}, "tokenCount": "3"}, + }, + {"key": "id_3", "status": "INVALID_ARGUMENT", "request": {"content": {"parts": [{"text": ""}]}}}, + ] + + result = bu.calculate_vertex_ai_batch_cost_and_usage(rows, "gemini-embedding-2") + + assert (result.successful_requests, result.failed_requests) == (2, 1) + assert (result.usage.prompt_tokens, result.usage.completion_tokens, result.usage.total_tokens) == (5, 0, 5) + assert result.cost == pytest.approx(5 * 1e-7) + assert result.models == ["gemini-embedding-2"] + + +@pytest.mark.asyncio +async def test_native_vertex_rows_route_to_vertex_cost_path_without_flag(monkeypatch): + monkeypatch.setattr(litellm, "disable_vertex_batch_output_transformation", False, raising=False) + monkeypatch.setattr( + bu, "_aggregate_batch_cost_usage_models", lambda **kw: pytest.fail("generic path should not run") + ) + calls = _capture_cost_calls(monkeypatch) + rows = [ + _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True), + _native_vertex_row(UNGROUNDED_USAGE_METADATA, grounded=False), + ] + + result = await bu.calculate_batch_cost_and_usage( + file_content_dictionary=rows, custom_llm_provider="vertex_ai", model_name="gemini-2.5-flash" + ) + + assert result.cost == pytest.approx(1.5) + assert (result.successful_requests, result.failed_requests) == (2, 0) + assert result.models == ["gemini-2.5-flash"] + assert {(call["model"], call["custom_llm_provider"]) for call in calls} == {("gemini-2.5-flash", "vertex_ai")} + + +@pytest.mark.asyncio +async def test_openai_shaped_vertex_rows_keep_the_generic_path_without_flag(monkeypatch): + monkeypatch.setattr(litellm, "disable_vertex_batch_output_transformation", False, raising=False) + monkeypatch.setattr( + bu, "calculate_vertex_ai_batch_cost_and_usage", lambda *a, **kw: pytest.fail("native path should not run") + ) + _capture_cost_calls(monkeypatch) + rows = [_vertex_openai_row("request-1", "gemini-2.5-flash", 10, 5)] + + result = await bu.calculate_batch_cost_and_usage( + file_content_dictionary=rows, custom_llm_provider="vertex_ai", model_name="gemini-2.5-flash" + ) + + assert result.successful_requests == 1 + + +@pytest.mark.asyncio +async def test_native_vertex_rows_on_another_provider_keep_the_generic_path(monkeypatch): + monkeypatch.setattr( + bu, "calculate_vertex_ai_batch_cost_and_usage", lambda *a, **kw: pytest.fail("native path should not run") + ) + _capture_cost_calls(monkeypatch) + + result = await bu.calculate_batch_cost_and_usage( + file_content_dictionary=[_native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True)], + custom_llm_provider="openai", + ) + + assert result.successful_requests == 0 + + +@pytest.mark.asyncio +async def test_handle_completed_batch_routes_native_rows_without_flag(monkeypatch): + monkeypatch.setattr(litellm, "disable_vertex_batch_output_transformation", False, raising=False) + raw_rows = [_native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True)] + + async def fake_fetch(batch, custom_llm_provider, litellm_params=None): + return _vertex_jsonl(raw_rows) + + monkeypatch.setattr(bu, "_fetch_batch_output_file_content", fake_fetch) + monkeypatch.setattr( + bu, "_aggregate_batch_cost_usage_models", lambda **kw: pytest.fail("generic path should not run") + ) + calls = _capture_cost_calls(monkeypatch, prompt_cost=0.7, completion_cost=0.3) + deployment_model_info = {"input_cost_per_token_batches": 1e-6, "output_cost_per_token_batches": 2e-6} + + result = await bu._handle_completed_batch( + _batch(PASSTHROUGH_OUTPUT_URI), + custom_llm_provider="vertex_ai", + model_name="gemini-2.5-flash", + model_info=deployment_model_info, + ) + + assert result.cost == pytest.approx(1.0) + assert result.usage.total_tokens == 557 + assert [call["model_info"] for call in calls] == [deployment_model_info] + + +def test_native_vertex_usage_is_billed_like_the_online_path(monkeypatch): + calls = _capture_cost_calls(monkeypatch) + grounded = _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True) + ungrounded = _native_vertex_row(UNGROUNDED_USAGE_METADATA, grounded=False) + + result = bu.calculate_vertex_ai_batch_cost_and_usage([grounded, ungrounded], "gemini-2.5-flash") + + grounded_usage, ungrounded_usage = (call["usage"] for call in calls) + assert grounded_usage.prompt_tokens == 19 + assert grounded_usage.completion_tokens == 59 + 406 + assert grounded_usage.completion_tokens_details.reasoning_tokens == 406 + assert ungrounded_usage.prompt_tokens == 20 + 73 + assert ungrounded_usage.completion_tokens == 48 + 195 + assert (result.usage.prompt_tokens, result.usage.completion_tokens, result.usage.total_tokens) == ( + 19 + 93, + 465 + 243, + 557 + 336, + ) + + +def test_native_vertex_rows_are_priced_by_model_version_without_a_model_name(monkeypatch): + calls = _capture_cost_calls(monkeypatch) + rows = [ + _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True, model_version="gemini-2.5-flash"), + _native_vertex_row(UNGROUNDED_USAGE_METADATA, grounded=False, model_version="gemini-2.5-pro"), + _native_vertex_row(UNGROUNDED_USAGE_METADATA, grounded=False, model_version=None), + ] + + result = bu.calculate_vertex_ai_batch_cost_and_usage(rows) + + assert [call["model"] for call in calls] == ["gemini-2.5-flash", "gemini-2.5-pro"] + assert result.models == ["gemini-2.5-flash", "gemini-2.5-pro"] + assert result.cost == pytest.approx(1.5) + assert result.successful_requests == 3 + assert result.usage.total_tokens == 557 + 336 + 336 + + +def test_native_vertex_rows_without_usage_metadata_count_as_failed(monkeypatch): + _capture_cost_calls(monkeypatch) + rows = [ + {"request": {"contents": []}, "status": "Error: bad request", "processed_time": "t"}, + {"request": {"contents": []}, "response": {"candidates": []}}, + _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True), + ] + + result = bu.calculate_vertex_ai_batch_cost_and_usage(rows, "gemini-2.5-flash") + + assert (result.successful_requests, result.failed_requests) == (1, 2) + assert result.usage.total_tokens == 557 + + +def test_native_vertex_batch_whose_rows_all_failed_still_names_the_deployment_model(monkeypatch): + calls = _capture_cost_calls(monkeypatch) + rows = [{"request": {"contents": []}, "status": "Error: quota exceeded", "processed_time": "t"}] * 2 + + result = bu.calculate_vertex_ai_batch_cost_and_usage(rows, "gemini-2.5-flash") + + assert result.models == ["gemini-2.5-flash"] + assert (result.successful_requests, result.failed_requests, result.cost) == (0, 2, 0.0) + assert calls == [] + + +def test_native_vertex_rows_are_priced_with_the_deployment_model_info(monkeypatch): + calls = _capture_cost_calls(monkeypatch) + deployment_model_info = {"input_cost_per_token_batches": 1e-6, "output_cost_per_token_batches": 2e-6} + + bu.calculate_vertex_ai_batch_cost_and_usage( + [_native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True)], + "gemini-2.5-flash", + model_info=deployment_model_info, + ) + + assert [call["model_info"] for call in calls] == [deployment_model_info] + + +@pytest.mark.asyncio +async def test_native_vertex_rows_keep_the_deployment_model_info_through_the_batch_entrypoint(monkeypatch): + calls = _capture_cost_calls(monkeypatch) + deployment_model_info = {"input_cost_per_token_batches": 1e-6} + + await bu.calculate_batch_cost_and_usage( + file_content_dictionary=[_native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True)], + custom_llm_provider="vertex_ai", + model_name="gemini-2.5-flash", + model_info=deployment_model_info, + ) + + assert [call["model_info"] for call in calls] == [deployment_model_info] + + +def test_native_vertex_rows_are_priced_by_the_deployment_model_over_model_version(monkeypatch): + calls = _capture_cost_calls(monkeypatch) + rows = [_native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True, model_version="gemini-2.5-pro")] + + result = bu.calculate_vertex_ai_batch_cost_and_usage(rows, "gemini-2.5-flash") + + assert [call["model"] for call in calls] == ["gemini-2.5-flash"] + assert result.models == ["gemini-2.5-flash"] + + +def test_native_vertex_rows_that_fail_response_validation_count_as_failed(monkeypatch): + calls = _capture_cost_calls(monkeypatch) + rows = [ + {"request": {"contents": []}, "response": {"candidates": "nope", "usageMetadata": GROUNDED_USAGE_METADATA}}, + _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True), + ] + + result = bu.calculate_vertex_ai_batch_cost_and_usage(rows, "gemini-2.5-flash") + + assert (result.successful_requests, result.failed_requests) == (1, 1) + assert result.usage.total_tokens == 557 + assert len(calls) == 1 + + +@pytest.mark.parametrize("wildcard_model", ["*", "vertex_ai/*"]) +def test_native_vertex_rows_under_a_wildcard_deployment_are_priced_by_model_version(monkeypatch, wildcard_model): + calls = _capture_cost_calls(monkeypatch) + rows = [ + _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True, model_version="gemini-2.5-flash"), + _native_vertex_row(UNGROUNDED_USAGE_METADATA, grounded=False, model_version=None), + ] + + result = bu.calculate_vertex_ai_batch_cost_and_usage(rows, wildcard_model) + + assert [call["model"] for call in calls] == ["gemini-2.5-flash", wildcard_model] + assert result.cost == pytest.approx(1.5) + assert (result.successful_requests, result.failed_requests) == (2, 0) + assert result.usage.total_tokens == 557 + 336 + + +def test_native_vertex_row_without_model_version_under_a_wildcard_deployment_bills_its_explicit_prices(): + deployment_model_info = {"input_cost_per_token_batches": 1e-6, "output_cost_per_token_batches": 2e-6} + with_version = _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True, model_version="gemini-2.5-flash") + without_version = _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True, model_version=None) + + twin = bu.calculate_vertex_ai_batch_cost_and_usage([with_version], "vertex_ai/*", model_info=deployment_model_info) + both = bu.calculate_vertex_ai_batch_cost_and_usage( + [with_version, without_version], "vertex_ai/*", model_info=deployment_model_info + ) + + assert twin.cost > 0 + assert both.cost == pytest.approx(2 * twin.cost) + assert (both.successful_requests, both.failed_requests) == (2, 0) + + +def test_native_vertex_row_the_cost_map_cannot_price_is_billed_at_zero_and_the_rest_still_bills(monkeypatch): + import litellm.cost_calculator as cc + + def _calc(**kw): + if kw["model"] == "gemini-unpriced": + raise ValueError("no pricing") + return (0.5, 0.25) + + monkeypatch.setattr(cc, "batch_cost_calculator", _calc) + rows = [ + _native_vertex_row(GROUNDED_USAGE_METADATA, grounded=True, model_version="gemini-unpriced"), + _native_vertex_row(UNGROUNDED_USAGE_METADATA, grounded=False, model_version="gemini-2.5-flash"), + ] + + result = bu.calculate_vertex_ai_batch_cost_and_usage(rows) + + assert result.cost == pytest.approx(0.75) + assert (result.successful_requests, result.failed_requests) == (2, 0) + assert result.usage.total_tokens == 557 + 336 + assert result.models == ["gemini-unpriced", "gemini-2.5-flash"] + + +@pytest.mark.asyncio +async def test_flag_sends_every_vertex_row_down_the_native_path_when_a_model_is_known(monkeypatch): + monkeypatch.setattr(litellm, "disable_vertex_batch_output_transformation", True, raising=False) + monkeypatch.setattr( + bu, "_aggregate_batch_cost_usage_models", lambda **kw: pytest.fail("generic path should not run") + ) + calls = _capture_cost_calls(monkeypatch) + rows = [_vertex_openai_row("request-1", "gemini-2.5-flash", 10, 5)] + + result = await bu.calculate_batch_cost_and_usage( + file_content_dictionary=rows, custom_llm_provider="vertex_ai", model_name="gemini-2.5-flash" + ) + + assert calls == [] + assert (result.successful_requests, result.failed_requests) == (0, 1) diff --git a/tests/test_litellm/files/__init__.py b/tests/test_litellm/files/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/files/test_main.py b/tests/test_litellm/files/test_main.py new file mode 100644 index 00000000000..2704bdfb6ff --- /dev/null +++ b/tests/test_litellm/files/test_main.py @@ -0,0 +1,71 @@ +from typing import Final +from urllib.parse import parse_qs, urlparse + +import httpx +import pytest + +import litellm +from litellm.llms.custom_httpx.http_handler import HTTPHandler + +NATIVE_VERTEX_ROWS: Final = ( + b'{"request": {"contents": [{"role": "user", "parts": [{"text": "Who won the 2024 Tour de France?"}]}],' + b' "tools": [{"googleSearch": {"excludeDomains": ["example.com"]}}]}}\n' + b'{"request": {"contents": [{"role": "user", "parts": [{"text": "What is the tallest building in Tokyo?"}]}],' + b' "tools": [{"googleSearch": {}}]}}\n' +) + + +@pytest.mark.parametrize( + "custom_llm_provider, purpose", + [("openai", "batch"), ("vertex_ai", "assistants")], + ids=["non-vertex-provider", "non-batch-purpose"], +) +def test_create_file_passthrough_is_rejected_outside_a_vertex_batch(custom_llm_provider, purpose): + with pytest.raises(litellm.BadRequestError) as exc_info: + litellm.create_file( + file=("batch.jsonl", b'{"request": {"contents": []}}\n', "application/jsonl"), + purpose=purpose, + custom_llm_provider=custom_llm_provider, + passthrough=True, + api_key="sk-test", + api_base="http://127.0.0.1:9", + ) + + assert "vertex_ai" in str(exc_info.value) + assert "batch" in str(exc_info.value) + + +def _gcs_upload_transport(uploads: list[httpx.Request]) -> httpx.MockTransport: + def respond(request: httpx.Request) -> httpx.Response: + uploads.append(request) + object_name: Final = parse_qs(urlparse(str(request.url)).query)["name"][0] + return httpx.Response( + 200, + json={ + "id": f"my-bucket/{object_name}/1758585600000000", + "name": object_name, + "size": str(len(request.read())), + "timeCreated": "2026-09-23T00:00:00.000Z", + }, + ) + + return httpx.MockTransport(respond) + + +def test_create_file_passthrough_kwarg_ships_native_rows_byte_for_byte_under_the_passthrough_prefix(): + uploads: Final[list[httpx.Request]] = [] + file_object = litellm.create_file( + file=("batch.jsonl", NATIVE_VERTEX_ROWS, "application/jsonl"), + purpose="batch", + custom_llm_provider="vertex_ai", + passthrough=True, + model="vertex_ai/gemini-2.5-flash", + gcs_bucket_name="my-bucket", + api_key="test-token", + client=HTTPHandler(client=httpx.Client(transport=_gcs_upload_transport(uploads))), + ) + (upload,) = uploads + object_name: Final = parse_qs(urlparse(str(upload.url)).query)["name"][0] + assert upload.read() == NATIVE_VERTEX_ROWS + assert object_name.startswith("litellm-vertex-files/passthrough/publishers/google/models/gemini-2.5-flash/") + assert file_object.id == f"gs://my-bucket/{object_name}" diff --git a/tests/test_litellm/llms/vertex_ai/batches/test_transformation.py b/tests/test_litellm/llms/vertex_ai/batches/test_transformation.py index e6126b02790..ae045edbec5 100644 --- a/tests/test_litellm/llms/vertex_ai/batches/test_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/batches/test_transformation.py @@ -19,7 +19,6 @@ import pytest from litellm.llms.vertex_ai.batches.transformation import ( # noqa: E402 VertexAIBatchTransformation, - vertex_prompt_tokens_details, ) from litellm.llms.vertex_ai.common_utils import ( # noqa: E402 VertexAIError, @@ -41,27 +40,6 @@ ENDPOINT_INPUT_FILE = ( ) -def test_vertex_prompt_tokens_details_rejects_malformed_details(): - assert vertex_prompt_tokens_details({"promptTokensDetails": [1]}) is None - assert vertex_prompt_tokens_details({"promptTokensDetails": [{"modality": "AUDIO"}]}) is None - assert ( - vertex_prompt_tokens_details( - { - "promptTokensDetails": [ - {"modality": "AUDIO", "tokenCount": 1}, - "malformed", - ] - } - ) - is None - ) - - -# =========================================================================== # -# transform_openai_batch_request_to_vertex_ai_batch_request -# =========================================================================== # - - def test_transform_openai_request_builds_full_vertex_job(): with patch( "litellm.llms.vertex_ai.batches.transformation.uuid.uuid4", @@ -477,3 +455,19 @@ def test_list_response_none_jobs_treated_as_empty(): out = T.transform_vertex_ai_batch_list_response_to_openai_list_response({"batchPredictionJobs": None}) assert out["data"] == [] assert out["first_id"] is None + + +PASSTHROUGH_INPUT_FILE = ( + "gs://litellm-testing-bucket/litellm-vertex-files/passthrough/publishers/google/models/gemini-2.5-flash/uuid-1" +) + + +def test_get_model_from_passthrough_gcs_file(): + assert T._get_model_from_gcs_file(PASSTHROUGH_INPUT_FILE) == "publishers/google/models/gemini-2.5-flash" + + +def test_get_gcs_uri_prefix_keeps_passthrough_segment_so_output_lands_beside_input(): + assert ( + T._get_gcs_uri_prefix_from_file(PASSTHROUGH_INPUT_FILE) + == "gs://litellm-testing-bucket/litellm-vertex-files/passthrough/publishers/google/models/gemini-2.5-flash" + ) diff --git a/tests/test_litellm/llms/vertex_ai/files/__init__.py b/tests/test_litellm/llms/vertex_ai/files/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/vertex_ai/files/test_transformation.py b/tests/test_litellm/llms/vertex_ai/files/test_transformation.py new file mode 100644 index 00000000000..6958576b0c8 --- /dev/null +++ b/tests/test_litellm/llms/vertex_ai/files/test_transformation.py @@ -0,0 +1,310 @@ +import io +import json +import urllib.parse +from pathlib import Path +from unittest.mock import MagicMock +from urllib.parse import parse_qs, urlparse + +import httpx +import pytest + +from litellm.llms.vertex_ai.common_utils import VertexAIError +from litellm.llms.vertex_ai.files.transformation import VertexAIFilesConfig, is_passthrough_managed_gcs_url + +NATIVE_VERTEX_ROW = json.dumps( + { + "request": { + "contents": [{"role": "user", "parts": [{"text": "What is the tallest building in the world?"}]}], + "tools": [{"googleSearch": {"excludeDomains": ["example.com"]}}], + } + } +).encode() +NATIVE_VERTEX_JSONL = NATIVE_VERTEX_ROW + b"\n" + NATIVE_VERTEX_ROW + b"\n" +OPENAI_BATCH_JSONL = ( + b'{"custom_id": "r1", "method": "POST", "url": "/v1/chat/completions",' + b' "body": {"model": "gemini-2.5-flash", "messages": [{"role": "user", "content": "hi"}]}}\n' +) +PASSTHROUGH_OBJECT = ( + "litellm-vertex-files/passthrough/publishers/google/models/gemini-2.5-flash/uuid-1/predictions.jsonl" +) +TRANSFORMED_OBJECT = "litellm-vertex-files/publishers/google/models/gemini-2.5-flash/uuid-1/predictions.jsonl" +UPLOAD_CHUNK_BYTES = 1024 * 1024 + + +@pytest.fixture +def config() -> VertexAIFilesConfig: + return VertexAIFilesConfig() + + +def _gcs_media_url(object_name: str) -> str: + return ( + f"https://storage.googleapis.com/storage/v1/b/my-bucket/o/{urllib.parse.quote(object_name, safe='')}?alt=media" + ) + + +def _native_output_jsonl() -> bytes: + return ( + json.dumps( + { + "request": json.loads(NATIVE_VERTEX_ROW)["request"], + "status": "", + "response": { + "candidates": [ + { + "content": {"role": "model", "parts": [{"text": "The Burj Khalifa."}]}, + "finishReason": "STOP", + "groundingMetadata": {"webSearchQueries": ["tallest building in the world"]}, + } + ], + "modelVersion": "gemini-2.5-flash", + "usageMetadata": {"promptTokenCount": 20, "candidatesTokenCount": 48, "totalTokenCount": 68}, + }, + "processed_time": "2026-09-23T19:02:00.000+00:00", + } + ).encode() + + b"\n" + ) + + +def _upload_chunks(config: VertexAIFilesConfig, file: object, litellm_params: dict) -> list[bytes]: + body = config.transform_create_file_request( + model="", + create_file_data={"file": file, "purpose": "batch"}, + optional_params={}, + litellm_params=litellm_params, + ) + return list(body["streaming_media_upload"]["body_stream"].iter_bytes()) + + +def _upload_body_bytes(config: VertexAIFilesConfig, file: object, litellm_params: dict) -> bytes: + return b"".join(_upload_chunks(config, file, litellm_params)) + + +class TestPassthroughBatchUpload: + """`passthrough=True` on a batch upload ships the caller's native Vertex JSONL + to GCS byte for byte, filed under a `passthrough/` object path so the batch + output that lands beside it is recognized and returned untouched as well.""" + + def _upload_url(self, config, litellm_params, file, purpose="batch") -> str: + return config.get_complete_file_url( + api_base=None, + api_key=None, + model="", + optional_params={}, + litellm_params=litellm_params, + data={"file": file, "purpose": purpose}, + ) + + def test_passthrough_object_is_filed_under_passthrough_prefix_named_by_deployment_model(self, config): + url = self._upload_url( + config, + {"gcs_bucket_name": "my-bucket", "model": "vertex_ai/gemini-2.5-flash", "passthrough": True}, + ("batch.jsonl", NATIVE_VERTEX_JSONL, "application/jsonl"), + ) + object_name = parse_qs(urlparse(url).query)["name"][0] + assert object_name.startswith("litellm-vertex-files/passthrough/publishers/google/models/gemini-2.5-flash/") + + def test_passthrough_upload_without_deployment_model_is_rejected(self, config): + with pytest.raises(VertexAIError) as exc_info: + self._upload_url( + config, + {"gcs_bucket_name": "my-bucket", "passthrough": True}, + ("batch.jsonl", NATIVE_VERTEX_JSONL, "application/jsonl"), + ) + assert exc_info.value.status_code == 400 + assert "target_model_names" in exc_info.value.message + + def test_passthrough_flag_does_not_ship_a_non_batch_upload_raw(self, config): + result = config.transform_create_file_request( + model="", + create_file_data={"file": ("notes.txt", b"plain text", "text/plain"), "purpose": "user_data"}, + optional_params={}, + litellm_params={"gcs_bucket_name": "my-bucket", "passthrough": True}, + ) + assert result == b"plain text" + + def test_passthrough_flag_is_ignored_for_non_batch_purposes(self, config): + url = self._upload_url( + config, + {"gcs_bucket_name": "my-bucket", "model": "vertex_ai/gemini-2.5-flash", "passthrough": True}, + ("notes.txt", b"plain text", "text/plain"), + purpose="user_data", + ) + object_name = parse_qs(urlparse(url).query)["name"][0] + assert object_name.startswith("litellm-vertex-files/uploads/") + assert "passthrough" not in object_name + + @pytest.mark.parametrize( + "file", + [ + ("batch.jsonl", NATIVE_VERTEX_JSONL, "application/jsonl"), + NATIVE_VERTEX_JSONL, + ("batch.jsonl", io.BytesIO(NATIVE_VERTEX_JSONL), "application/jsonl"), + ("batch.jsonl", NATIVE_VERTEX_JSONL.decode(), "application/jsonl"), + ], + ids=["bytes-tuple", "bare-bytes", "handle-tuple", "text-tuple"], + ) + def test_passthrough_upload_body_is_the_callers_bytes(self, config, file): + body = config.transform_create_file_request( + model="", + create_file_data={"file": file, "purpose": "batch"}, + optional_params={}, + litellm_params={"passthrough": True}, + ) + stream = body["streaming_media_upload"]["body_stream"] + assert b"".join(stream.iter_bytes()) == NATIVE_VERTEX_JSONL + assert b"".join(stream.iter_bytes()) == NATIVE_VERTEX_JSONL + assert body["streaming_media_upload"]["content_type"] == "application/json" + + def test_passthrough_upload_streams_a_large_handle_in_bounded_chunks(self, config): + content = NATIVE_VERTEX_ROW * (3 * UPLOAD_CHUNK_BYTES // len(NATIVE_VERTEX_ROW) + 1) + chunks = _upload_chunks( + config, ("batch.jsonl", io.BytesIO(content), "application/jsonl"), {"passthrough": True} + ) + assert len(chunks) >= 3 + assert max(len(chunk) for chunk in chunks) <= UPLOAD_CHUNK_BYTES + assert b"".join(chunks) == content + + def test_passthrough_upload_streams_a_path_in_bounded_chunks(self, config, tmp_path: Path): + content = NATIVE_VERTEX_ROW * (2 * UPLOAD_CHUNK_BYTES // len(NATIVE_VERTEX_ROW) + 1) + batch_path = tmp_path / "batch.jsonl" + batch_path.write_bytes(content) + chunks = _upload_chunks(config, ("batch.jsonl", batch_path, "application/jsonl"), {"passthrough": True}) + assert len(chunks) >= 2 + assert max(len(chunk) for chunk in chunks) <= UPLOAD_CHUNK_BYTES + assert b"".join(chunks) == content + + def test_passthrough_upload_rejects_a_non_seekable_handle(self, config): + class _Pipe: + def read(self, size=-1): + return b"" + + with pytest.raises(ValueError, match="seekable"): + _upload_body_bytes(config, ("batch.jsonl", _Pipe(), "application/jsonl"), {"passthrough": True}) + + def test_passthrough_upload_rejects_content_that_is_neither_bytes_path_nor_handle(self, config): + with pytest.raises(ValueError, match="Unsupported file content type"): + _upload_body_bytes(config, ("batch.jsonl", 42, "application/jsonl"), {"passthrough": True}) + + def test_openai_rows_are_translated_unless_passthrough_is_set(self, config): + file = ("batch.jsonl", OPENAI_BATCH_JSONL, "application/jsonl") + translated = _upload_body_bytes(config, file, {}) + untouched = _upload_body_bytes(config, file, {"passthrough": True}) + assert untouched == OPENAI_BATCH_JSONL + assert translated != OPENAI_BATCH_JSONL + assert b'"contents"' in translated + + def test_passthrough_output_content_is_returned_untouched(self, config): + raw_jsonl = _native_output_jsonl() + + def _download(object_name: str) -> bytes: + raw_response = httpx.Response( + status_code=200, + content=raw_jsonl, + headers={"content-type": "application/octet-stream"}, + request=httpx.Request("GET", _gcs_media_url(object_name)), + ) + result = config.transform_file_content_response( + raw_response=raw_response, logging_obj=MagicMock(), litellm_params={} + ) + return result.response.content + + assert _download(PASSTHROUGH_OBJECT) == raw_jsonl + assert _download(f"team-a/{PASSTHROUGH_OBJECT}") == raw_jsonl + transformed = _download(TRANSFORMED_OBJECT) + assert transformed != raw_jsonl + assert json.loads(transformed.splitlines()[0])["response"]["body"]["choices"] + nested = _download(f"litellm-vertex-files/{PASSTHROUGH_OBJECT}") + assert nested != raw_jsonl + assert json.loads(nested.splitlines()[0])["response"]["body"]["choices"] + + def test_output_of_an_upload_whose_model_smuggles_the_passthrough_segment_is_still_transformed(self, config): + smuggled_model = b"litellm-vertex-files/passthrough/publishers/google/models/gemini-2.5-flash" + upload_url = self._upload_url( + config, + {"gcs_bucket_name": "my-bucket"}, + ("batch.jsonl", OPENAI_BATCH_JSONL.replace(b"gemini-2.5-flash", smuggled_model), "application/jsonl"), + ) + object_name = parse_qs(urlparse(upload_url).query)["name"][0] + raw_jsonl = _native_output_jsonl() + raw_response = httpx.Response( + status_code=200, + content=raw_jsonl, + headers={"content-type": "application/octet-stream"}, + request=httpx.Request("GET", _gcs_media_url(f"{object_name}/predictions.jsonl")), + ) + + result = config.transform_file_content_response( + raw_response=raw_response, logging_obj=MagicMock(), litellm_params={} + ) + + assert object_name.startswith("litellm-vertex-files/litellm-vertex-files/passthrough/") + assert json.loads(result.response.content.splitlines()[0])["response"]["body"]["choices"] + + @pytest.mark.parametrize( + "url, expected", + [ + (f"gs://my-bucket/{PASSTHROUGH_OBJECT}", True), + (f"gs://my-bucket/team-a/{PASSTHROUGH_OBJECT}", True), + (f"gs://my-bucket/litellm-vertex-files/{PASSTHROUGH_OBJECT}", False), + (_gcs_media_url(f"team-a/sub/{PASSTHROUGH_OBJECT}"), True), + (_gcs_media_url(f"litellm-vertex-files/publishers/google/models/x/{PASSTHROUGH_OBJECT}"), False), + (_gcs_media_url(TRANSFORMED_OBJECT), False), + ], + ids=["gs", "gs-prefixed", "gs-smuggled", "https-prefixed", "https-model-path-smuggled", "https-transformed"], + ) + def test_passthrough_detection_anchors_on_the_first_managed_segment(self, url, expected): + assert is_passthrough_managed_gcs_url(url) is expected + + @pytest.mark.asyncio + async def test_passthrough_output_stream_is_returned_untouched(self, config): + stream_iterator = object() + headers = {"content-type": "application/octet-stream"} + result = await config.transform_file_content_stream( + stream_iterator=stream_iterator, + headers=headers, + request_url=_gcs_media_url(f"team-a/{PASSTHROUGH_OBJECT}"), + logging_obj=MagicMock(), + litellm_params={}, + ) + assert result.stream_iterator is stream_iterator + assert result.headers == headers + + +class TestEmbeddingOutputTranslation: + EMBEDDING_OBJECT = ( + "litellm-vertex-files/publishers/google/models/gemini-embedding-2/prediction-model-1/predictions.jsonl" + ) + + def _transform(self, config: VertexAIFilesConfig, rows: list[dict]) -> list[dict]: + raw_response = httpx.Response( + status_code=200, + content="\n".join(json.dumps(row) for row in rows).encode(), + headers={"content-type": "application/octet-stream"}, + request=httpx.Request("GET", _gcs_media_url(self.EMBEDDING_OBJECT)), + ) + result = config.transform_file_content_response( + raw_response=raw_response, logging_obj=MagicMock(), litellm_params={} + ) + return [json.loads(line) for line in result.response.content.decode().splitlines()] + + def test_embedding_rows_become_openai_batch_rows_billed_by_their_prompt_tokens(self, config): + live_row = { + "key": "request-1", + "request": {"content": {"parts": [{"text": "hello world"}]}}, + "response": {"embedding": {"values": [-0.015, 0.024]}, "usageMetadata": {"promptTokenCount": 2}}, + } + documented_row = { + "key": "request-2", + "request": {"content": {"parts": [{"text": "hello"}]}}, + "response": {"embedding": {"values": [0.5]}, "tokenCount": "3"}, + } + + live, documented = self._transform(config, [live_row, documented_row]) + + assert (live["custom_id"], live["error"], live["response"]["status_code"]) == ("request-1", None, 200) + assert live["response"]["body"]["model"] == "gemini-embedding-2" + assert live["response"]["body"]["data"] == [{"embedding": [-0.015, 0.024], "index": 0, "object": "embedding"}] + live_usage, documented_usage = (row["response"]["body"]["usage"] for row in (live, documented)) + assert (live_usage["prompt_tokens"], live_usage["total_tokens"]) == (2, 2) + assert (documented_usage["prompt_tokens"], documented_usage["total_tokens"]) == (3, 3) diff --git a/tests/test_litellm/proxy/openai_files_endpoint/test_files_batch_file_validation.py b/tests/test_litellm/proxy/openai_files_endpoint/test_files_batch_file_validation.py index f5542fc0446..3a73c39e177 100644 --- a/tests/test_litellm/proxy/openai_files_endpoint/test_files_batch_file_validation.py +++ b/tests/test_litellm/proxy/openai_files_endpoint/test_files_batch_file_validation.py @@ -4,7 +4,8 @@ import pytest from litellm.proxy._types import ProxyException from litellm.proxy.openai_files_endpoints.batch_file_validation import ( - BATCH_LINE_REQUIRED_KEYS, + BATCH_LINE_SHAPE, + PASSTHROUGH_BATCH_LINE_SHAPE, BatchFileEmpty, BatchFileInvalidJsonLine, BatchFileLineNotObject, @@ -96,7 +97,7 @@ def test_non_object_line_rejected(): assert check_batch_file_upload("batch.jsonl", content, None) == BatchFileLineNotObject(line_number=2) -@pytest.mark.parametrize("missing_key", BATCH_LINE_REQUIRED_KEYS) +@pytest.mark.parametrize("missing_key", BATCH_LINE_SHAPE.required_keys) def test_missing_required_key_rejected(missing_key): import json @@ -174,3 +175,36 @@ def test_failures_map_to_openai_shaped_proxy_exceptions(failure, expected_code, assert exc_info.value.param == expected_param for fragment in expected_fragments: assert fragment in exc_info.value.message + + +NATIVE_VERTEX_LINE = b'{"request": {"contents": [{"role": "user", "parts": [{"text": "hi"}]}]}}' + + +def test_passthrough_keys_accept_native_vertex_rows(): + content = NATIVE_VERTEX_LINE + b"\n" + NATIVE_VERTEX_LINE + b"\n" + assert check_batch_file_upload("batch.jsonl", content, None, PASSTHROUGH_BATCH_LINE_SHAPE) is None + + +def test_passthrough_keys_reject_openai_rows(): + content = NATIVE_VERTEX_LINE + b"\n" + VALID_LINE + b"\n" + assert check_batch_file_upload( + "batch.jsonl", content, None, PASSTHROUGH_BATCH_LINE_SHAPE + ) == BatchFileMissingLineKey(line_number=2, key="request", line_shape=PASSTHROUGH_BATCH_LINE_SHAPE) + + +def test_default_keys_still_reject_native_vertex_rows(): + assert check_batch_file_upload("batch.jsonl", NATIVE_VERTEX_LINE, None) == BatchFileMissingLineKey( + line_number=1, key="custom_id" + ) + + +def test_passthrough_missing_key_message_says_what_a_passthrough_upload_takes(): + with pytest.raises(ProxyException) as exc_info: + raise_batch_file_validation_failure( + BatchFileMissingLineKey(line_number=3, key="request", line_shape=PASSTHROUGH_BATCH_LINE_SHAPE) + ) + assert exc_info.value.param == "request" + assert "line 3" in exc_info.value.message + assert "passthrough upload takes native Vertex batch rows" in exc_info.value.message + assert "with a request key." in exc_info.value.message + assert "custom_id" not in exc_info.value.message diff --git a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py index 48699b47e7f..84cf4ea7c32 100644 --- a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py +++ b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py @@ -5669,3 +5669,241 @@ def test_model_routed_file_retrieve_allows_key_with_model_grant(mocker: MockerFi assert response.status_code == 200, response.text assert captured_kwargs["api_key"] == "mistral-key" assert captured_kwargs["custom_llm_provider"] == "mistral" + + +NATIVE_VERTEX_BATCH_LINE = ( + b'{"request": {"contents": [{"role": "user", "parts": [{"text": "What is the tallest building?"}]}],' + b' "tools": [{"googleSearch": {"excludeDomains": ["example.com"]}}]}}\n' +) + + +def _passthrough_router() -> Router: + return Router( + model_list=[ + { + "model_name": "vertex-batch", + "litellm_params": { + "model": "vertex_ai/gemini-2.5-flash", + "vertex_project": "proj", + "vertex_location": "us-central1", + }, + "model_info": {"id": "vertex-batch-id"}, + }, + { + "model_name": "gpt-3.5-turbo", + "litellm_params": {"model": "openai/gpt-3.5-turbo", "api_key": "openai_api_key"}, + "model_info": {"id": "gpt-3.5-turbo-id"}, + }, + ] + ) + + +def _setup_passthrough_upload_endpoint(monkeypatch, llm_router: Router) -> list: + """Like _setup_batch_upload_endpoint, but reads the forwarded file bytes while the spool is open.""" + from litellm.proxy.openai_files_endpoints import files_endpoints as fe + + forwarded_calls = _setup_batch_upload_endpoint(monkeypatch, llm_router) + + async def fake_route_create_file(**kwargs): + upload_source = kwargs["_create_file_request"]["file"][1] + upload_source.seek(0) + forwarded_calls.append({**kwargs, "file_bytes": upload_source.read()}) + return OpenAIFileObject( + id="dummy-id", + object="file", + bytes=0, + created_at=1234567890, + filename="batch.jsonl", + purpose="batch", + status="uploaded", + ) + + monkeypatch.setattr(fe, "route_create_file", fake_route_create_file) + return forwarded_calls + + +def _upload(content: bytes, form: dict): + return client.post( + "/v1/files", + files={"file": ("batch.jsonl", content, "application/jsonl")}, + data=form, + headers={"Authorization": "Bearer test-key"}, + ) + + +def test_create_file_passthrough_forwards_native_vertex_rows_untouched(monkeypatch): + forwarded_calls = _setup_passthrough_upload_endpoint(monkeypatch, _passthrough_router()) + content = NATIVE_VERTEX_BATCH_LINE * 2 + + try: + response = _upload(content, {"purpose": "batch", "target_model_names": "vertex-batch", "passthrough": "true"}) + finally: + _teardown_batch_upload_endpoint() + + assert response.status_code == 200, response.text + (call,) = forwarded_calls + assert call["_create_file_request"]["passthrough"] is True + assert call["file_bytes"] == content + assert call["target_model_names_list"] == ["vertex-batch"] + + +def test_create_file_passthrough_rejects_rows_without_a_request(monkeypatch): + forwarded_calls = _setup_passthrough_upload_endpoint(monkeypatch, _passthrough_router()) + + try: + response = _upload( + NATIVE_VERTEX_BATCH_LINE + VALID_BATCH_LINE, + {"purpose": "batch", "target_model_names": "vertex-batch", "passthrough": "true"}, + ) + finally: + _teardown_batch_upload_endpoint() + + assert response.status_code == 400, response.text + error = response.json()["error"] + assert error["param"] == "request" + assert "line 2" in error["message"] + assert forwarded_calls == [] + + +def test_create_file_without_passthrough_still_rejects_native_vertex_rows(monkeypatch): + forwarded_calls = _setup_passthrough_upload_endpoint(monkeypatch, _passthrough_router()) + + try: + response = _upload(NATIVE_VERTEX_BATCH_LINE, {"purpose": "batch", "target_model_names": "vertex-batch"}) + finally: + _teardown_batch_upload_endpoint() + + assert response.status_code == 400, response.text + assert response.json()["error"]["param"] == "custom_id" + assert forwarded_calls == [] + + +@pytest.mark.parametrize( + "form, expected_param, expected_fragment", + [ + ({"purpose": "batch", "passthrough": "true"}, "target_model_names", "target_model_names"), + ( + {"purpose": "batch", "target_model_names": "gpt-3.5-turbo", "passthrough": "true"}, + "target_model_names", + "'gpt-3.5-turbo'", + ), + ( + {"purpose": "batch", "target_model_names": "vertex-batch,gpt-3.5-turbo", "passthrough": "true"}, + "target_model_names", + "'gpt-3.5-turbo'", + ), + ({"purpose": "user_data", "target_model_names": "vertex-batch", "passthrough": "true"}, "passthrough", "batch"), + ( + {"purpose": "batch", "target_model_names": "vertex-batch", "passthrough": "true", "target_storage": "s3"}, + "target_storage", + "'s3'", + ), + ( + {"purpose": "batch", "model": "gpt-3.5-turbo", "passthrough": "true"}, + "model", + "'gpt-3.5-turbo'", + ), + ], + ids=[ + "no-model", + "non-vertex-model", + "mixed-models", + "non-batch-purpose", + "target-storage", + "non-vertex-model-param", + ], +) +def test_create_file_passthrough_rejected_outside_a_vertex_batch(monkeypatch, form, expected_param, expected_fragment): + forwarded_calls = _setup_passthrough_upload_endpoint(monkeypatch, _passthrough_router()) + + try: + response = _upload(NATIVE_VERTEX_BATCH_LINE, form) + finally: + _teardown_batch_upload_endpoint() + + assert response.status_code == 400, response.text + error = response.json()["error"] + assert error["type"] == "invalid_request_error" + assert error["param"] == expected_param + assert expected_fragment in error["message"] + assert forwarded_calls == [] + + +def test_create_file_passthrough_accepts_the_model_param_as_the_deployment(monkeypatch): + forwarded_calls = _setup_passthrough_upload_endpoint(monkeypatch, _passthrough_router()) + + try: + response = _upload( + NATIVE_VERTEX_BATCH_LINE, {"purpose": "batch", "model": "vertex-batch", "passthrough": "true"} + ) + finally: + _teardown_batch_upload_endpoint() + + assert response.status_code == 200, response.text + (call,) = forwarded_calls + assert call["model"] == "vertex-batch" + assert call["_create_file_request"]["passthrough"] is True + + +def test_create_file_passthrough_rejects_a_model_group_with_a_non_vertex_deployment(monkeypatch): + mixed_router = Router( + model_list=[ + { + "model_name": "vertex-batch", + "litellm_params": { + "model": "vertex_ai/gemini-2.5-flash", + "vertex_project": "proj", + "vertex_location": "us-central1", + }, + "model_info": {"id": "vertex-batch-id"}, + }, + { + "model_name": "vertex-batch", + "litellm_params": {"model": "openai/gpt-4.1-mini", "api_key": "openai_api_key"}, + "model_info": {"id": "vertex-batch-openai-id"}, + }, + ] + ) + forwarded_calls = _setup_passthrough_upload_endpoint(monkeypatch, mixed_router) + + try: + response = _upload( + NATIVE_VERTEX_BATCH_LINE, {"purpose": "batch", "target_model_names": "vertex-batch", "passthrough": "true"} + ) + finally: + _teardown_batch_upload_endpoint() + + assert response.status_code == 400, response.text + error = response.json()["error"] + assert error["param"] == "target_model_names" + assert "'vertex-batch'" in error["message"] + assert forwarded_calls == [] + + +def test_create_file_passthrough_fails_closed_when_guardrails_would_scan_the_batch(monkeypatch): + """Batch guardrails read OpenAI-shaped rows, so a passthrough upload on a guardrailed + key is refused rather than forwarded unscanned.""" + from litellm.integrations.custom_guardrail import CustomGuardrail + from litellm.proxy.utils import ProxyLogging + + class _Redactor(CustomGuardrail): + async def async_pre_call_hook(self, user_api_key_dict, cache, data, call_type): + return data + + forwarded_calls = _setup_passthrough_upload_endpoint(monkeypatch, _passthrough_router()) + monkeypatch.setattr(litellm, "callbacks", [_Redactor(guardrail_name="g", default_on=True)]) + ProxyLogging._callback_capabilities_cache.clear() + + try: + response = _upload( + NATIVE_VERTEX_BATCH_LINE, {"purpose": "batch", "target_model_names": "vertex-batch", "passthrough": "true"} + ) + finally: + _teardown_batch_upload_endpoint() + ProxyLogging._callback_capabilities_cache.clear() + + assert response.status_code == 400, response.text + error = response.json()["error"] + assert error["param"] == "passthrough" + assert "guardrails" in error["message"] + assert forwarded_calls == [] diff --git a/tests/test_litellm/test_router.py b/tests/test_litellm/test_router.py index 411377f29cf..f985b212b01 100644 --- a/tests/test_litellm/test_router.py +++ b/tests/test_litellm/test_router.py @@ -528,6 +528,39 @@ async def test_async_router_acreate_file_with_jsonl(): assert first_call_content == non_jsonl_content +@pytest.mark.asyncio +async def test_async_router_acreate_file_passthrough_keeps_the_file_and_forwards_the_flag(): + """A passthrough batch upload must reach the provider byte for byte: the router + neither rewrites body.model to the deployment model nor drops the flag.""" + from io import BytesIO + from unittest.mock import MagicMock, patch + + jsonl_content = b'{"custom_id": "r1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "vertex-batch"}}\n' + router = litellm.Router( + model_list=[ + { + "model_name": "vertex-batch", + "litellm_params": {"model": "vertex_ai/gemini-2.5-flash", "vertex_project": "p"}, + } + ], + ) + + with patch("litellm.acreate_file", return_value=MagicMock()) as mock_acreate_file: + await router.acreate_file( + model="vertex-batch", purpose="batch", file=BytesIO(jsonl_content), passthrough=True + ) + forwarded = mock_acreate_file.call_args.kwargs + assert forwarded["passthrough"] is True + forwarded["file"].seek(0) + assert forwarded["file"].read() == jsonl_content + + mock_acreate_file.reset_mock() + await router.acreate_file(model="vertex-batch", purpose="batch", file=BytesIO(jsonl_content)) + rewritten = mock_acreate_file.call_args.kwargs["file"] + rewritten.seek(0) + assert b'"gemini-2.5-flash"' in rewritten.read() + + @pytest.mark.asyncio async def test_async_router_acreate_file_does_not_fall_back_across_model_groups(): """A file created for batches only exists under the credentials of the model group diff --git a/tests/unit/batches/test_batch_utils.py b/tests/unit/batches/test_batch_utils.py index a4de8eee23c..d1572f4a7c9 100644 --- a/tests/unit/batches/test_batch_utils.py +++ b/tests/unit/batches/test_batch_utils.py @@ -640,7 +640,7 @@ async def test_calculate_vertex_disable_transform_path(monkeypatch): monkeypatch.setattr( bu, "calculate_vertex_ai_batch_cost_and_usage", - lambda content, model: bu.BatchCostUsageResult( + lambda content, model, model_info=None: bu.BatchCostUsageResult( cost=9.9, usage=Usage(prompt_tokens=1, completion_tokens=2, total_tokens=3), models=["gemini-2.0-flash-001"], @@ -671,7 +671,7 @@ async def test_calculate_vertex_disable_transform_needs_model_name(monkeypatch): monkeypatch.setattr( bu, "calculate_vertex_ai_batch_cost_and_usage", - lambda content, model: pytest.fail("raw vertex path should not run"), + lambda content, model, model_info=None: pytest.fail("raw vertex path should not run"), ) result = await bu.calculate_batch_cost_and_usage(file_content_dictionary=[], custom_llm_provider="vertex_ai") @@ -735,7 +735,11 @@ def test_vertex_batch_usage_preserves_modality_token_details(monkeypatch): ) responses = [ { + "key": "id_1", + "status": "", + "request": {"content": {"parts": [{"text": "hello"}, {"fileData": {"mimeType": "audio/wav"}}]}}, "response": { + "embedding": {"values": [0.1, 0.2]}, "usageMetadata": { "promptTokenCount": 84, "candidatesTokenCount": 0, @@ -744,13 +748,14 @@ def test_vertex_batch_usage_preserves_modality_token_details(monkeypatch): {"modality": "AUDIO", "tokenCount": 64}, {"modality": "TEXT", "tokenCount": 20}, ], - } - } + }, + }, } ] result = bu.calculate_vertex_ai_batch_cost_and_usage(responses, "gemini-embedding-2") + assert (result.successful_requests, result.usage.prompt_tokens) == (1, 84) assert result.prompt_cost == pytest.approx(64 * 3.25e-6 + 20 * 1e-7) @@ -1336,7 +1341,7 @@ async def test_handle_completed_batch_vertex_disable_transform_path(monkeypatch) monkeypatch.setattr(litellm, "disable_vertex_batch_output_transformation", True, raising=False) seen: dict = {} - def fake_vertex_calc(content, model): + def fake_vertex_calc(content, model, model_info=None): seen["content"] = content seen["model"] = model return bu.BatchCostUsageResult( @@ -1358,7 +1363,7 @@ async def test_handle_completed_batch_vertex_disable_transform_path(monkeypatch) assert result.cost == 7.7 assert result.usage.total_tokens == 3 assert result.models == ["gemini-x"] - assert seen["content"] == raw_rows + assert list(seen["content"]) == raw_rows assert seen["model"] == "gemini-x" diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 2938e65cfde..052cc693f8b 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -25526,6 +25526,11 @@ export interface components { file: string; /** Litellm Metadata */ litellm_metadata?: string | null; + /** + * Passthrough + * @default false + */ + passthrough: boolean; /** Purpose */ purpose: string; /** @@ -25550,6 +25555,11 @@ export interface components { file: string; /** Litellm Metadata */ litellm_metadata?: string | null; + /** + * Passthrough + * @default false + */ + passthrough: boolean; /** Purpose */ purpose: string; /** @@ -25574,6 +25584,11 @@ export interface components { file: string; /** Litellm Metadata */ litellm_metadata?: string | null; + /** + * Passthrough + * @default false + */ + passthrough: boolean; /** Purpose */ purpose: string; /** From 1edc4ba580091d3e3d344b63d3f92d08b9b21a4d Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:01:12 -0700 Subject: [PATCH 93/96] fix(logging): pass provider response headers to callbacks on every endpoint (#42824) * fix(logging): pass provider response headers to callbacks on every endpoint Custom callbacks only received kwargs["response_headers"] for chat completions. Responses, image generation and edit, speech, and transcription calls either never recorded the provider's headers or recorded them in one place and not the other. Every handler now records the provider's httpx headers on the response's hidden params as "headers" (raw) and "additional_headers" (processed, with LiteLLM's own entries winning on a clash), and the logging object derives model_call_details["response_headers"] from those hidden params before cost calculation on the non-stream and both streaming success paths, keeping a handler-set value authoritative. Binary speech responses expose their hidden params to the standard logging payload, and the sync OpenAI transcription request always fetches the raw response. * test(images): point the legacy image and speech fakes at the raw response surface Image generation now goes through the SDK's raw response so the provider headers can be read, and the speech binary response now carries hidden params. The unit fakes in the image generation, xinference, proxy provider, image edit, Vertex speech, and otel suites still pinned the old call surface and the old "no hidden params" assertion, so they read an uncalled mock or a fake response without headers. * test(images): drop the rewritten mock comments and the generated edit PNGs * test(images): move the llm-span test's image fake to the raw response surface --------- Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- litellm/litellm_core_utils/core_helpers.py | 27 +++ litellm/litellm_core_utils/litellm_logging.py | 20 +- litellm/llms/custom_httpx/llm_http_handler.py | 20 +- litellm/llms/openai/openai.py | 29 ++- litellm/llms/openai/transcriptions/handler.py | 33 +-- tests/image_gen_tests/test_image_edits.py | 4 + tests/image_gen_tests/test_xinference.py | 30 ++- .../test_litellm_proxy_provider.py | 16 +- tests/llm_translation/test_openai.py | 13 +- .../otel/test_otel_v2_sources_of_truth.py | 8 +- .../litellm_core_utils/test_core_helpers.py | 68 ++++++ .../test_litellm_logging.py | 87 ++++++++ .../custom_httpx/test_llm_http_handler.py | 203 +++++++++++++++++- tests/test_litellm/llms/openai/test_openai.py | 99 ++++++++- .../test_openai_transcriptions_handler.py | 71 ++++++ .../test_non_chat_routes_open_llm_spans.py | 14 +- ...t_openai_image_generation_extra_headers.py | 50 +++-- .../text_to_speech/test_transformation.py | 1 + 18 files changed, 714 insertions(+), 79 deletions(-) create mode 100644 tests/test_litellm/llms/openai/transcriptions/test_openai_transcriptions_handler.py diff --git a/litellm/litellm_core_utils/core_helpers.py b/litellm/litellm_core_utils/core_helpers.py index 3afa6a913b5..b095b4b12c6 100644 --- a/litellm/litellm_core_utils/core_helpers.py +++ b/litellm/litellm_core_utils/core_helpers.py @@ -765,3 +765,30 @@ def set_response_cost_in_hidden_params(response: _CarriesHiddenParams, cost: flo RESPONSE_COST_HEADER: cost, } hidden_params["additional_headers"] = merged + + +_HIDDEN_PARAMS_ADAPTER: Final = TypeAdapter(Mapping[str, object]) +_PROVIDER_HEADERS_ADAPTER: Final = TypeAdapter(Mapping[str, str]) + + +def set_provider_response_headers_in_hidden_params( + response: _CarriesHiddenParams, headers: httpx.Headers | Mapping[str, str] +) -> None: + hidden_params: Final = response._hidden_params # pyright: ignore[reportPrivateUsage] # no public accessor + existing_additional_headers: Final[object] = hidden_params.get("additional_headers") + raw_headers: Final[dict[str, str]] = dict(headers) # mutable-ok: stored as the plain-dict hidden param + additional_headers: Final[dict[str, object]] = { # mutable-ok: assigned into the plain-dict hidden params + **process_response_headers(raw_headers), + **(existing_additional_headers if isinstance(existing_additional_headers, Mapping) else _NO_HEADERS), + } + hidden_params["headers"] = raw_headers + hidden_params["additional_headers"] = additional_headers + + +def get_provider_response_headers_from_hidden_params(response: object) -> Mapping[str, str] | None: + hidden_params: Final[object] = getattr(response, "_hidden_params", None) + try: + validated: Final = _HIDDEN_PARAMS_ADAPTER.validate_python(hidden_params) + return _PROVIDER_HEADERS_ADAPTER.validate_python(validated.get("headers")) + except ValidationError: + return None diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index a6391a2ae27..3b9419d483b 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -72,6 +72,7 @@ from litellm.litellm_core_utils.classifier_logging import ( is_classifier_call, ) from litellm.litellm_core_utils.core_helpers import ( + get_provider_response_headers_from_hidden_params, is_expected_client_error, reconstruct_model_name, set_response_cost_in_hidden_params, @@ -2353,6 +2354,15 @@ class Logging(LiteLLMLoggingBaseClass): ) return logging_result + def _surface_response_headers_from_result(self, logging_result: object) -> None: + existing: Final[object] = self.model_call_details.get("response_headers") + if existing is not None: + return + headers: Final = get_provider_response_headers_from_hidden_params(logging_result) + if headers is None: + return + self.model_call_details["response_headers"] = headers + def _merge_hidden_params_from_response_into_metadata(self, logging_result: object) -> None: """ Copy response._hidden_params into litellm_params.metadata['hidden_params']. @@ -2386,6 +2396,7 @@ class Logging(LiteLLMLoggingBaseClass): build_logging_payload: bool = True, ): """Resolve hidden params, compute response cost, and emit the standard logging payload.""" + self._surface_response_headers_from_result(logging_result) hidden_params: Final = getattr(logging_result, "_hidden_params", {}) if hidden_params: if self.model_call_details.get("litellm_params") is not None: @@ -2788,6 +2799,7 @@ class Logging(LiteLLMLoggingBaseClass): if complete_streaming_response is not None: verbose_logger.debug("Logging Details LiteLLM-Success Call streaming complete") self.model_call_details["complete_streaming_response"] = complete_streaming_response + self._surface_response_headers_from_result(complete_streaming_response) self.model_call_details["response_cost"] = self._response_cost_calculator( result=complete_streaming_response ) @@ -3302,6 +3314,7 @@ class Logging(LiteLLMLoggingBaseClass): print_verbose("Async success callbacks: Got a complete streaming response") self.model_call_details["async_complete_streaming_response"] = complete_streaming_response + self._surface_response_headers_from_result(complete_streaming_response) try: if self.model_call_details.get("cache_hit", False) is True: @@ -6362,12 +6375,15 @@ def _extract_response_obj_and_hidden_params( original_exception: Exception | None, ) -> tuple[dict, dict | None]: """Extract response_obj and hidden_params from init_response_obj.""" - hidden_params: dict | None = None + hidden_params: dict | None = ( + getattr(init_response_obj, "_hidden_params", None) + if isinstance(init_response_obj, BaseModel | HttpxBinaryResponseContent) + else None + ) if init_response_obj is None: response_obj = {} elif isinstance(init_response_obj, BaseModel): response_obj = init_response_obj.model_dump() - hidden_params = getattr(init_response_obj, "_hidden_params", None) elif isinstance(init_response_obj, dict): response_obj = init_response_obj elif isinstance(init_response_obj, HttpxBinaryResponseContent): diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index dba0dee38fc..4f31742aaaa 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -45,6 +45,7 @@ from litellm.litellm_core_utils.audio_utils.subtitle_utils import ( SUBTITLE_RESPONSE_FORMATS, synthesize_subtitle_document, ) +from litellm.litellm_core_utils.core_helpers import set_provider_response_headers_in_hidden_params from litellm.litellm_core_utils.get_litellm_params import AWS_CREDENTIAL_KWARGS_KEYS from litellm.litellm_core_utils.llm_request_utils import serialize_multipart_form_fields from litellm.litellm_core_utils.realtime_errors import ( @@ -1461,6 +1462,7 @@ class BaseLLMHTTPHandler: transformed: Final = provider_config.transform_audio_transcription_response( raw_response=response, ) + set_provider_response_headers_in_hidden_params(transformed, response.headers) if not provider_config.supports_subtitle_synthesis: return transformed requested_format: Final = optional_params.get("response_format") @@ -6960,11 +6962,13 @@ class BaseLLMHTTPHandler: provider_config=image_edit_provider_config, ) - return image_edit_provider_config.transform_image_edit_response( + image_edit_response: Final = image_edit_provider_config.transform_image_edit_response( model=model, raw_response=response, logging_obj=logging_obj, ) + set_provider_response_headers_in_hidden_params(image_edit_response, response.headers) + return image_edit_response async def async_image_edit_handler( self, @@ -7059,11 +7063,13 @@ class BaseLLMHTTPHandler: provider_config=image_edit_provider_config, ) - return image_edit_provider_config.transform_image_edit_response( + image_edit_response: Final = image_edit_provider_config.transform_image_edit_response( model=model, raw_response=response, logging_obj=logging_obj, ) + set_provider_response_headers_in_hidden_params(image_edit_response, response.headers) + return image_edit_response def image_generation_handler( self, @@ -7186,6 +7192,7 @@ class BaseLLMHTTPHandler: litellm_params=dict(litellm_params), encoding=None, ) + set_provider_response_headers_in_hidden_params(model_response, response.headers) return model_response @@ -7293,6 +7300,7 @@ class BaseLLMHTTPHandler: litellm_params=dict(litellm_params), encoding=None, ) + set_provider_response_headers_in_hidden_params(model_response, response.headers) return model_response @@ -12077,11 +12085,13 @@ class BaseLLMHTTPHandler: provider_config=text_to_speech_provider_config, ) - return text_to_speech_provider_config.transform_text_to_speech_response( + speech_response: Final = text_to_speech_provider_config.transform_text_to_speech_response( model=model, raw_response=response, logging_obj=logging_obj, ) + set_provider_response_headers_in_hidden_params(speech_response, response.headers) + return speech_response async def async_text_to_speech_handler( self, @@ -12176,11 +12186,13 @@ class BaseLLMHTTPHandler: provider_config=text_to_speech_provider_config, ) - return text_to_speech_provider_config.transform_text_to_speech_response( + speech_response: Final = text_to_speech_provider_config.transform_text_to_speech_response( model=model, raw_response=response, logging_obj=logging_obj, ) + set_provider_response_headers_in_hidden_params(speech_response, response.headers) + return speech_response ######################################################### ########## SKILLS API HANDLERS ########################## diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index 63874ca9619..d6340d182ae 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -27,6 +27,7 @@ from litellm import LlmProviders from litellm._logging import verbose_logger from litellm.constants import DEFAULT_MAX_RETRIES from litellm.files.types import FileContentStreamingResult +from litellm.litellm_core_utils.core_helpers import set_provider_response_headers_in_hidden_params from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.logging_utils import speech_request_body, track_llm_api_timing from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator @@ -1404,7 +1405,6 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): organization: str | None = None, headers: dict | None = None, ): - response = None try: openai_aclient: Final = self._get_openai_client( is_async=True, @@ -1428,8 +1428,10 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): ) request_data: Final = {**data, "extra_headers": headers} if headers else data - response = await openai_aclient.images.generate(**request_data, timeout=timeout) - stringified_response: Final = response.model_dump() + raw_response: Final = await openai_aclient.images.with_raw_response.generate( + **request_data, timeout=timeout + ) + stringified_response: Final = raw_response.parse().model_dump() ## LOGGING logging_obj.post_call( input=prompt, @@ -1437,11 +1439,13 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): additional_args={"complete_input_dict": data}, original_response=stringified_response, ) - return convert_to_model_response_object( + image_response: Final[ImageResponse] = convert_to_model_response_object( response_object=stringified_response, model_response_object=model_response, response_type="image_generation", ) + set_provider_response_headers_in_hidden_params(image_response, raw_response.headers) + return image_response except Exception as e: ## LOGGING logging_obj.post_call( @@ -1512,9 +1516,9 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): ## COMPLETION CALL request_data: Final = {**data, "extra_headers": headers} if headers else data - _response: Final = openai_client.images.generate(**request_data, timeout=timeout) + raw_response: Final = openai_client.images.with_raw_response.generate(**request_data, timeout=timeout) - response: Final = _response.model_dump() + response: Final = raw_response.parse().model_dump() ## LOGGING logging_obj.post_call( input=prompt, @@ -1522,11 +1526,13 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): additional_args={"complete_input_dict": data}, original_response=response, ) - return convert_to_model_response_object( + image_response: Final[ImageResponse] = convert_to_model_response_object( response_object=response, model_response_object=model_response, response_type="image_generation", ) + set_provider_response_headers_in_hidden_params(image_response, raw_response.headers) + return image_response except OpenAIError as e: ## LOGGING logging_obj.post_call( @@ -1609,7 +1615,9 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): input=input, **optional_params, ) - return HttpxBinaryResponseContent(response=response.response) + speech_response: Final = HttpxBinaryResponseContent(response=response.response) + set_provider_response_headers_in_hidden_params(speech_response, response.response.headers) + return speech_response async def async_audio_speech( self, @@ -1655,8 +1663,9 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): input=input, **optional_params, ) - - return HttpxBinaryResponseContent(response=response.response) + speech_response: Final = HttpxBinaryResponseContent(response=response.response) + set_provider_response_headers_in_hidden_params(speech_response, response.response.headers) + return speech_response class OpenAIFilesAPI(BaseLLM): diff --git a/litellm/llms/openai/transcriptions/handler.py b/litellm/llms/openai/transcriptions/handler.py index 701b3d30362..014251db821 100644 --- a/litellm/llms/openai/transcriptions/handler.py +++ b/litellm/llms/openai/transcriptions/handler.py @@ -4,11 +4,10 @@ import httpx from openai import AsyncOpenAI, OpenAI from pydantic import BaseModel -import litellm - if TYPE_CHECKING: from aiohttp import ClientSession from litellm.litellm_core_utils.audio_utils.utils import get_audio_file_name +from litellm.litellm_core_utils.core_helpers import set_provider_response_headers_in_hidden_params from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.base_llm.audio_transcription.transformation import ( BaseAudioTranscriptionConfig, @@ -31,11 +30,6 @@ class OpenAIAudioTranscription(OpenAIChatCompletion): data: dict, timeout: float | httpx.Timeout, ): - """ - Helper to: - - call openai_aclient.audio.transcriptions.with_raw_response when litellm.return_response_headers is True - - call openai_aclient.audio.transcriptions.create by default - """ try: raw_response = await openai_aclient.audio.transcriptions.with_raw_response.create(**data, timeout=timeout) headers: Final = dict(raw_response.headers) @@ -51,20 +45,11 @@ class OpenAIAudioTranscription(OpenAIChatCompletion): data: dict, timeout: float | httpx.Timeout, ): - """ - Helper to: - - call openai_aclient.audio.transcriptions.with_raw_response when litellm.return_response_headers is True - - call openai_aclient.audio.transcriptions.create by default - """ try: - if litellm.return_response_headers is True: - raw_response = openai_client.audio.transcriptions.with_raw_response.create(**data, timeout=timeout) - headers: Final = dict(raw_response.headers) - response = raw_response.parse() - return headers, response - else: - response = openai_client.audio.transcriptions.create(**data, timeout=timeout) - return None, response + raw_response: Final = openai_client.audio.transcriptions.with_raw_response.create(**data, timeout=timeout) + headers: Final = dict(raw_response.headers) + response: Final = raw_response.parse() + return headers, response except Exception as e: raise e @@ -133,11 +118,12 @@ class OpenAIAudioTranscription(OpenAIChatCompletion): "complete_input_dict": data, }, ) - _, response = self.make_sync_openai_audio_transcriptions_request( + headers, response = self.make_sync_openai_audio_transcriptions_request( openai_client=openai_client, data=data, timeout=timeout, ) + logging_obj.model_call_details["response_headers"] = headers if isinstance(response, BaseModel): stringified_response = response.model_dump() @@ -158,6 +144,7 @@ class OpenAIAudioTranscription(OpenAIChatCompletion): hidden_params=hidden_params, response_type="audio_transcription", ) + set_provider_response_headers_in_hidden_params(final_response, headers) return final_response async def async_audio_transcriptions( @@ -217,12 +204,14 @@ class OpenAIAudioTranscription(OpenAIChatCompletion): actual_model: Final = data.get("model", "whisper-1") hidden_params: Final = {"model": actual_model, "custom_llm_provider": "openai"} - return convert_to_model_response_object( + final_response: Final[TranscriptionResponse] = convert_to_model_response_object( response_object=stringified_response, model_response_object=model_response, hidden_params=hidden_params, response_type="audio_transcription", ) + set_provider_response_headers_in_hidden_params(final_response, headers) + return final_response except Exception as e: ## LOGGING logging_obj.post_call( diff --git a/tests/image_gen_tests/test_image_edits.py b/tests/image_gen_tests/test_image_edits.py index 0c2f57066e8..36fd65ba71b 100644 --- a/tests/image_gen_tests/test_image_edits.py +++ b/tests/image_gen_tests/test_image_edits.py @@ -250,6 +250,7 @@ async def test_azure_image_edit_litellm_sdk(): self._json_data = json_data self.status_code = status_code self.text = json.dumps(json_data) + self.headers = {} def json(self): return self._json_data @@ -370,6 +371,7 @@ async def test_openai_image_edit_cost_tracking(): self._json_data = json_data self.status_code = status_code self.text = json.dumps(json_data) + self.headers = {} def json(self): return self._json_data @@ -460,6 +462,7 @@ async def test_azure_image_edit_cost_tracking(): self._json_data = json_data self.status_code = status_code self.text = json.dumps(json_data) + self.headers = {} def json(self): return self._json_data @@ -737,6 +740,7 @@ async def test_image_edit_array_handling(): self._json_data = json_data self.status_code = status_code self.text = json.dumps(json_data) + self.headers = {} def json(self): return self._json_data diff --git a/tests/image_gen_tests/test_xinference.py b/tests/image_gen_tests/test_xinference.py index 3dc4fee85da..76cae593e41 100644 --- a/tests/image_gen_tests/test_xinference.py +++ b/tests/image_gen_tests/test_xinference.py @@ -24,9 +24,14 @@ async def test_xinference_image_generation(): def model_dump(self): return mock_openai_response - # Create a mock client with the images.generate method + class MockRawResponse: + headers = {} + + def parse(self): + return MockResponse() + mock_client = AsyncMock() - mock_client.images.generate = AsyncMock(return_value=MockResponse()) + mock_client.images.with_raw_response.generate = AsyncMock(return_value=MockRawResponse()) # Capture the actual arguments sent to OpenAI client captured_args = None @@ -36,9 +41,9 @@ async def test_xinference_image_generation(): nonlocal captured_args, captured_kwargs captured_args = args captured_kwargs = kwargs - return MockResponse() + return MockRawResponse() - mock_client.images.generate.side_effect = capture_generate_call + mock_client.images.with_raw_response.generate.side_effect = capture_generate_call # Mock the _get_openai_client method to return our mock client with patch.object( @@ -65,7 +70,7 @@ async def test_xinference_image_generation(): assert response.data[0].url == "https://example.com/image.png" # Validate that the OpenAI client was called with correct parameters - mock_client.images.generate.assert_called_once() + mock_client.images.with_raw_response.generate.assert_called_once() assert captured_kwargs is not None assert ( captured_kwargs["model"] == "stabilityai/stable-diffusion-3.5-large" @@ -97,9 +102,14 @@ async def test_xinference_image_generation_with_response_format(): def model_dump(self): return mock_openai_response - # Create a mock client with the images.generate method + class MockRawResponse: + headers = {} + + def parse(self): + return MockResponse() + mock_client = AsyncMock() - mock_client.images.generate = AsyncMock(return_value=MockResponse()) + mock_client.images.with_raw_response.generate = AsyncMock(return_value=MockRawResponse()) # Capture the actual arguments sent to OpenAI client captured_args = None @@ -109,9 +119,9 @@ async def test_xinference_image_generation_with_response_format(): nonlocal captured_args, captured_kwargs captured_args = args captured_kwargs = kwargs - return MockResponse() + return MockRawResponse() - mock_client.images.generate.side_effect = capture_generate_call + mock_client.images.with_raw_response.generate.side_effect = capture_generate_call # Mock the _get_openai_client method to return our mock client with patch.object( @@ -141,7 +151,7 @@ async def test_xinference_image_generation_with_response_format(): assert response.data[0].b64_json is not None # Validate that the OpenAI client was called with correct parameters - mock_client.images.generate.assert_called_once() + mock_client.images.with_raw_response.generate.assert_called_once() assert captured_kwargs is not None assert ( captured_kwargs["model"] == "stabilityai/stable-diffusion-3.5-large" diff --git a/tests/llm_translation/test_litellm_proxy_provider.py b/tests/llm_translation/test_litellm_proxy_provider.py index a10fc55ecc5..a7a2848a514 100644 --- a/tests/llm_translation/test_litellm_proxy_provider.py +++ b/tests/llm_translation/test_litellm_proxy_provider.py @@ -210,11 +210,14 @@ async def test_litellm_gateway_image_generation_direct(is_async): "created": 1, "data": [{"url": "https://example.com/image.png"}], } + mock_raw_response = MagicMock() + mock_raw_response.parse.return_value = mock_openai_response + mock_raw_response.headers = {} if is_async: # Mock the AsyncOpenAI client that gets created inside _get_openai_client mock_async_client = AsyncMock() - mock_async_client.images.generate = AsyncMock(return_value=mock_openai_response) + mock_async_client.images.with_raw_response.generate = AsyncMock(return_value=mock_raw_response) with patch( "litellm.llms.openai.openai.AsyncOpenAI", return_value=mock_async_client @@ -234,14 +237,14 @@ async def test_litellm_gateway_image_generation_direct(is_async): assert constructor_kwargs["base_url"] == "http://my-proxy" # Verify the AsyncOpenAI client was called correctly - mock_async_client.images.generate.assert_awaited_once() - call_kwargs = mock_async_client.images.generate.call_args.kwargs + mock_async_client.images.with_raw_response.generate.assert_awaited_once() + call_kwargs = mock_async_client.images.with_raw_response.generate.call_args.kwargs assert call_kwargs["model"] == "dall-e-3" assert call_kwargs["prompt"] == "A beautiful sunset over mountains" else: # Mock the sync OpenAI client that gets created inside _get_openai_client mock_sync_client = MagicMock() - mock_sync_client.images.generate.return_value = mock_openai_response + mock_sync_client.images.with_raw_response.generate.return_value = mock_raw_response with patch( "litellm.llms.openai.openai.OpenAI", return_value=mock_sync_client @@ -260,8 +263,8 @@ async def test_litellm_gateway_image_generation_direct(is_async): assert constructor_kwargs["base_url"] == "http://my-proxy" # Verify the OpenAI client was called correctly - mock_sync_client.images.generate.assert_called_once() - call_kwargs = mock_sync_client.images.generate.call_args.kwargs + mock_sync_client.images.with_raw_response.generate.assert_called_once() + call_kwargs = mock_sync_client.images.with_raw_response.generate.call_args.kwargs assert call_kwargs["model"] == "dall-e-3" assert call_kwargs["prompt"] == "A beautiful sunset over mountains" @@ -285,6 +288,7 @@ async def test_litellm_gateway_from_sdk_image_edit(is_async): self._json_data = json_data self.status_code = status_code self.text = json.dumps(json_data) + self.headers = {} def json(self): return self._json_data diff --git a/tests/llm_translation/test_openai.py b/tests/llm_translation/test_openai.py index af4ba85d58e..0488c4c68e6 100644 --- a/tests/llm_translation/test_openai.py +++ b/tests/llm_translation/test_openai.py @@ -313,7 +313,7 @@ def test_openai_max_retries_0(mock_get_openai_client): def test_openai_image_generation_forwards_organization(mock_get_openai_client): """Ensure organization flows to OpenAI client for image generation.""" - class _DummyImages: + class _DummyRawImages: def generate(self, **kwargs): # type: ignore class _Resp: def model_dump(self_inner): # minimal OpenAI ImagesResponse shape @@ -327,7 +327,16 @@ def test_openai_image_generation_forwards_organization(mock_get_openai_client): }, } - return _Resp() + class _RawResp: + headers = {} + + def parse(self_inner): + return _Resp() + + return _RawResp() + + class _DummyImages: + with_raw_response = _DummyRawImages() class _DummyClient: def __init__(self): diff --git a/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py b/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py index 7e93d3d67a7..57f4557c6f7 100644 --- a/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py +++ b/tests/test_litellm/integrations/otel/test_otel_v2_sources_of_truth.py @@ -1207,14 +1207,18 @@ def test_speech_response_without_a_byte_count_produces_no_output() -> None: def test_speech_binary_response_is_logged_as_its_summary_not_dropped() -> None: import httpx + from litellm.litellm_core_utils.core_helpers import set_provider_response_headers_in_hidden_params from litellm.litellm_core_utils.litellm_logging import _extract_response_obj_and_hidden_params from litellm.types.llms.openai import HttpxBinaryResponseContent raw: Final = httpx.Response(200, headers={"content-type": "audio/mpeg"}, content=b"\x00" * 1234) - response_obj, hidden_params = _extract_response_obj_and_hidden_params(HttpxBinaryResponseContent(raw), None) + speech: Final = HttpxBinaryResponseContent(raw) + set_provider_response_headers_in_hidden_params(speech, raw.headers) + response_obj, hidden_params = _extract_response_obj_and_hidden_params(speech, None) assert response_obj == {"object": "binary", "content_type": "audio/mpeg", "num_bytes": 1234} - assert hidden_params is None + assert hidden_params is not None + assert hidden_params["headers"]["content-type"] == "audio/mpeg" def test_speech_binary_response_still_streaming_reports_the_bytes_downloaded_so_far() -> None: diff --git a/tests/test_litellm/litellm_core_utils/test_core_helpers.py b/tests/test_litellm/litellm_core_utils/test_core_helpers.py index 6eeea271127..2a6dd347d5f 100644 --- a/tests/test_litellm/litellm_core_utils/test_core_helpers.py +++ b/tests/test_litellm/litellm_core_utils/test_core_helpers.py @@ -2,22 +2,27 @@ import logging +import httpx import pytest from litellm.litellm_core_utils.core_helpers import ( _FINISH_REASON_MAP, + RESPONSE_COST_HEADER, bind_budget_reservation_to_callbacks, budget_reservation_from_metadata, drop_params_env_flag, drop_params_flag, get_or_create_metadata_bucket, + get_provider_response_headers_from_hidden_params, map_finish_reason, normalize_drop_params, reconstruct_model_name, redact_nested_match_and_regex_keys, + set_provider_response_headers_in_hidden_params, unbind_budget_reservation_from_callbacks, ) from litellm.proxy._types import UserAPIKeyAuth +from litellm.types.utils import ImageResponse, TranscriptionResponse class TestBudgetReservationBinding: @@ -489,3 +494,66 @@ class TestIsExpectedClientError: category=RateLimitErrorCategory.VENDOR_RATE_LIMIT, ) assert is_expected_client_error(vendor_limit) is False + + +class TestProviderResponseHeadersInHiddenParams: + def test_records_raw_headers_and_the_processed_additional_headers(self): + response = ImageResponse() + response._hidden_params = {"additional_headers": {RESPONSE_COST_HEADER: 0.04}} + + set_provider_response_headers_in_hidden_params( + response, httpx.Headers({"X-Request-Id": "req_img", "x-ratelimit-remaining-requests": "41"}) + ) + + assert response._hidden_params["headers"] == { + "x-request-id": "req_img", + "x-ratelimit-remaining-requests": "41", + } + additional_headers = response._hidden_params["additional_headers"] + assert additional_headers["llm_provider-x-request-id"] == "req_img" + assert additional_headers["x-ratelimit-remaining-requests"] == "41" + assert additional_headers[RESPONSE_COST_HEADER] == 0.04 + + def test_litellm_owned_additional_headers_win_over_provider_headers(self): + response = TranscriptionResponse(text="hi") + response._hidden_params = {"additional_headers": {"llm_provider-x-request-id": "kept"}} + + set_provider_response_headers_in_hidden_params(response, {"x-request-id": "provider"}) + + assert response._hidden_params["additional_headers"]["llm_provider-x-request-id"] == "kept" + assert response._hidden_params["headers"] == {"x-request-id": "provider"} + + def test_getter_returns_the_recorded_headers(self): + response = ImageResponse() + + set_provider_response_headers_in_hidden_params(response, {"x-request-id": "req_img"}) + + assert get_provider_response_headers_from_hidden_params(response) == {"x-request-id": "req_img"} + + @pytest.mark.parametrize( + "hidden_params", + [ + None, + "headers", + {"additional_headers": {}}, + {"headers": "x-request-id: req_img"}, + {"headers": {"x-request-id": 7}}, + ], + ) + def test_getter_returns_none_without_a_string_header_mapping(self, hidden_params): + response = ImageResponse() + response._hidden_params = hidden_params + + assert get_provider_response_headers_from_hidden_params(response) is None + + def test_getter_returns_none_for_an_object_without_hidden_params(self): + assert get_provider_response_headers_from_hidden_params(object()) is None + + def test_headers_never_leak_into_a_sibling_response(self): + recorded = TranscriptionResponse() + sibling = TranscriptionResponse() + + set_provider_response_headers_in_hidden_params(recorded, {"x-request-id": "req_stt"}) + + assert get_provider_response_headers_from_hidden_params(sibling) is None + assert "additional_headers" not in sibling._hidden_params diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index 23c01841b1b..bb8098e6e23 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -24,6 +24,7 @@ from litellm.cost_calculator import ocr_batch_cost from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.litellm_logging import Logging as LitellmLogging from litellm.litellm_core_utils.litellm_logging import ( + _extract_response_obj_and_hidden_params, _get_status_fields, set_callbacks, ) @@ -32,6 +33,7 @@ from litellm.proxy._types import UserAPIKeyAuth from litellm.types.llms.openai import ResponseAPIUsage, ResponseCompletedEvent, ResponsesAPIResponse from litellm.types.utils import ( CallTypes, + ImageResponse, LiteLLMRealtimeStreamLoggingObject, ModelResponse, TextCompletionResponse, @@ -8694,3 +8696,88 @@ async def test_async_failure_handler_delivers_failure_payload_to_custom_logger() assert "smoke-failure" in payload["error_str"] assert payload["model"] == "openai/gpt-5.6" assert events.empty() + + +def _image_logging_obj() -> LitellmLogging: + logging_obj = LitellmLogging( + model="gpt-image-2", + messages="a cat", + stream=False, + call_type="aimage_generation", + start_time=time.time(), + litellm_call_id="response-headers-test", + function_id="response-headers-test", + ) + logging_obj.model_call_details["litellm_params"] = {"metadata": {}} + logging_obj.optional_params = {} + return logging_obj + + +def _image_result_with_headers(request_id: str) -> ImageResponse: + result = ImageResponse(created=1, data=[]) + result._hidden_params = {"headers": {"x-request-id": request_id}} + return result + + +def test_process_hidden_params_surfaces_response_headers_from_the_result(): + logging_obj = _image_logging_obj() + + logging_obj._process_hidden_params_and_response_cost( + _image_result_with_headers("req_img"), datetime.datetime.now(), datetime.datetime.now() + ) + + assert logging_obj.model_call_details["response_headers"] == {"x-request-id": "req_img"} + + +def test_process_hidden_params_keeps_handler_set_response_headers(): + logging_obj = _image_logging_obj() + logging_obj.model_call_details["response_headers"] = {"x-request-id": "from-handler"} + + logging_obj._process_hidden_params_and_response_cost( + _image_result_with_headers("from-result"), datetime.datetime.now(), datetime.datetime.now() + ) + + assert logging_obj.model_call_details["response_headers"] == {"x-request-id": "from-handler"} + + +def _assembled_stream_result_with_headers() -> ModelResponse: + result = _assembled_stream_result() + result._hidden_params = {"headers": {"x-request-id": "req_stream"}} + return result + + +@pytest.mark.asyncio +async def test_async_streaming_success_passes_result_headers_to_callback_kwargs(): + releasing = CustomLogger() + releasing.async_log_success_event = AsyncMock() + patcher, logging_obj = _streaming_logging_obj_with_callbacks([releasing]) + + with patcher: + await logging_obj.async_success_handler(result=_assembled_stream_result_with_headers()) + + kwargs = releasing.async_log_success_event.await_args.kwargs["kwargs"] + assert kwargs["response_headers"] == {"x-request-id": "req_stream"} + + +def test_sync_streaming_success_passes_result_headers_to_callback_kwargs(): + releasing = CustomLogger() + releasing.log_success_event = MagicMock() + patcher, logging_obj = _streaming_logging_obj_with_callbacks([releasing]) + + with patcher: + logging_obj.success_handler(result=_assembled_stream_result_with_headers()) + + kwargs = releasing.log_success_event.call_args.kwargs["kwargs"] + assert kwargs["response_headers"] == {"x-request-id": "req_stream"} + + +def test_extract_response_obj_and_hidden_params_reads_binary_content_hidden_params(): + from litellm.types.llms.openai import HttpxBinaryResponseContent as LiteLLMBinaryResponseContent + + result = LiteLLMBinaryResponseContent(response=httpx.Response(status_code=200, content=b"audio bytes")) + result._hidden_params = {"headers": {"x-request-id": "req_tts"}} + + response_obj, hidden_params = _extract_response_obj_and_hidden_params(result, None) + + assert hidden_params == {"headers": {"x-request-id": "req_tts"}} + assert response_obj["object"] == "binary" diff --git a/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py b/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py index 68f37c8ffcc..75b6ce4c626 100644 --- a/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py +++ b/tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py @@ -26,6 +26,8 @@ from litellm.llms.base_llm.search.transformation import BaseSearchConfig, Search from litellm.llms.bedrock.base_aws_llm import SignsRequestsWithAWS from litellm.llms.brave.search.transformation import BraveSearchConfig from litellm.llms.base_llm.image_edit.transformation import BaseImageEditConfig +from litellm.llms.base_llm.image_generation.transformation import BaseImageGenerationConfig +from litellm.llms.base_llm.text_to_speech.transformation import BaseTextToSpeechConfig from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.llms.custom_httpx.llm_http_handler import ( BaseLLMHTTPHandler, @@ -41,7 +43,7 @@ from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_tran from litellm.llms.mistral.ocr.transformation import MistralOCRConfig from litellm.llms.openai.videos.transformation import OpenAIVideoConfig from litellm.llms.tinyfish.search.transformation import TinyfishSearchConfig -from litellm.types.llms.openai import ResponsesAPIResponse +from litellm.types.llms.openai import HttpxBinaryResponseContent, ResponsesAPIResponse from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import ImageObject, ImageResponse, ModelResponse, TranscriptionResponse from tests.test_litellm.llms.bedrock.event_loop_probe import EventLoopProbe @@ -4166,3 +4168,202 @@ async def test_chat_completion_agentic_followup_does_not_repeat_request_params_f assert followup_calls[0]["temperature"] == 0.2 assert followup_calls[0]["api_base"] == "https://a" assert followup_calls[0]["model"] == "openai/gpt-5" + + +_UPSTREAM_HEADERS: Final = {"x-request-id": "req_upstream", "x-ratelimit-remaining-requests": "41"} + + +def _assert_upstream_headers_recorded(response) -> None: + assert response._hidden_params["headers"]["x-request-id"] == "req_upstream" + assert response._hidden_params["additional_headers"]["llm_provider-x-request-id"] == "req_upstream" + assert response._hidden_params["additional_headers"]["x-ratelimit-remaining-requests"] == "41" + + +def _json_with_upstream_headers(payload: dict) -> httpx.MockTransport: + return httpx.MockTransport(lambda request: httpx.Response(200, json=payload, headers=_UPSTREAM_HEADERS)) + + +def _binary_with_upstream_headers() -> httpx.MockTransport: + return httpx.MockTransport( + lambda request: httpx.Response( + 200, content=b"audio-bytes", headers={**_UPSTREAM_HEADERS, "content-type": "audio/mpeg"} + ) + ) + + +def test_audio_transcriptions_records_upstream_response_headers(): + client = HTTPHandler(client=httpx.Client(transport=_json_with_upstream_headers({"text": "transcribed"}))) + + response = BaseLLMHTTPHandler().audio_transcriptions( + client=client, + atranscription=False, + **_json_transcription_call_kwargs(_JSONBodyAudioTranscriptionConfig()), + ) + + _assert_upstream_headers_recorded(response) + + +@pytest.mark.asyncio +async def test_async_audio_transcriptions_records_upstream_response_headers(): + client = AsyncHTTPHandler() + client.client = httpx.AsyncClient(transport=_json_with_upstream_headers({"text": "transcribed"})) + + response = await BaseLLMHTTPHandler().async_audio_transcriptions( + client=client, + **_json_transcription_call_kwargs(_JSONBodyAudioTranscriptionConfig()), + ) + + _assert_upstream_headers_recorded(response) + + +def _image_edit_call_kwargs() -> dict: + return { + "model": "edit-model", + "image": b"raw-image", + "prompt": "add a hat", + "image_edit_provider_config": _ImageEditRecordingConfig(), + "image_edit_optional_request_params": {}, + "custom_llm_provider": "openai", + "litellm_params": GenericLiteLLMParams(), + "logging_obj": Mock(), + "timeout": 10.0, + } + + +def test_image_edit_handler_records_upstream_response_headers(): + client = HTTPHandler() + client.client = httpx.Client(transport=_json_with_upstream_headers({"transformed_by": "sync"})) + + response = BaseLLMHTTPHandler().image_edit_handler(client=client, **_image_edit_call_kwargs()) + + _assert_upstream_headers_recorded(response) + + +@pytest.mark.asyncio +async def test_async_image_edit_handler_records_upstream_response_headers(): + client = AsyncHTTPHandler() + client.client = httpx.AsyncClient(transport=_json_with_upstream_headers({"transformed_by": "async"})) + + response = await BaseLLMHTTPHandler().async_image_edit_handler(client=client, **_image_edit_call_kwargs()) + + _assert_upstream_headers_recorded(response) + + +class _HeaderImageGenerationConfig(BaseImageGenerationConfig): + def get_supported_openai_params(self, model): + return [] + + def map_openai_params(self, non_default_params, optional_params, model, drop_params): + return optional_params + + def get_complete_url(self, api_base, api_key, model, optional_params, litellm_params, stream=None): + return "https://images.example/v1/generations" + + def transform_image_generation_request(self, model, prompt, optional_params, litellm_params, headers): + return {"prompt": prompt} + + def transform_image_generation_response( + self, + model, + raw_response, + model_response, + logging_obj, + request_data, + optional_params, + litellm_params, + encoding, + api_key=None, + json_mode=None, + ): + return ImageResponse(data=[ImageObject(b64_json=raw_response.json()["b64_json"])]) + + +def _image_generation_call_kwargs() -> dict: + return { + "model": "image-model", + "prompt": "a cat", + "image_generation_provider_config": _HeaderImageGenerationConfig(), + "image_generation_optional_request_params": {}, + "custom_llm_provider": "openai", + "litellm_params": {}, + "logging_obj": Mock(), + "timeout": 10.0, + } + + +def test_image_generation_handler_records_upstream_response_headers(): + client = HTTPHandler() + client.client = httpx.Client(transport=_json_with_upstream_headers({"b64_json": "abc"})) + + response = BaseLLMHTTPHandler().image_generation_handler(client=client, **_image_generation_call_kwargs()) + + assert response.data[0].b64_json == "abc" + _assert_upstream_headers_recorded(response) + + +@pytest.mark.asyncio +async def test_async_image_generation_handler_records_upstream_response_headers(): + client = AsyncHTTPHandler() + client.client = httpx.AsyncClient(transport=_json_with_upstream_headers({"b64_json": "abc"})) + + response = await BaseLLMHTTPHandler().async_image_generation_handler( + client=client, **_image_generation_call_kwargs() + ) + + assert response.data[0].b64_json == "abc" + _assert_upstream_headers_recorded(response) + + +class _HeaderTextToSpeechConfig(BaseTextToSpeechConfig): + def get_supported_openai_params(self, model): + return [] + + def map_openai_params(self, model, optional_params, voice=None, drop_params=False, kwargs=None): + return voice, optional_params + + def validate_environment(self, headers, model, api_key=None, api_base=None): + return {} + + def get_complete_url(self, model, api_base, litellm_params): + return "https://tts.example/v1/speech" + + def transform_text_to_speech_request(self, model, input, voice, optional_params, litellm_params, headers): + return {"dict_body": {"input": input}} + + def transform_text_to_speech_response(self, model, raw_response, logging_obj): + return HttpxBinaryResponseContent(response=raw_response) + + +def _text_to_speech_call_kwargs() -> dict: + return { + "model": "tts-model", + "input": "hello", + "voice": "alloy", + "text_to_speech_provider_config": _HeaderTextToSpeechConfig(), + "text_to_speech_optional_params": {}, + "custom_llm_provider": "openai", + "litellm_params": {}, + "logging_obj": Mock(), + "timeout": 10.0, + } + + +def test_text_to_speech_handler_records_upstream_response_headers(): + client = HTTPHandler() + client.client = httpx.Client(transport=_binary_with_upstream_headers()) + + response = BaseLLMHTTPHandler().text_to_speech_handler(client=client, **_text_to_speech_call_kwargs()) + + assert response.content == b"audio-bytes" + _assert_upstream_headers_recorded(response) + + +@pytest.mark.asyncio +async def test_async_text_to_speech_handler_records_upstream_response_headers(): + client = AsyncHTTPHandler() + client.client = httpx.AsyncClient(transport=_binary_with_upstream_headers()) + + response = await BaseLLMHTTPHandler().async_text_to_speech_handler(client=client, **_text_to_speech_call_kwargs()) + + assert response.content == b"audio-bytes" + _assert_upstream_headers_recorded(response) diff --git a/tests/test_litellm/llms/openai/test_openai.py b/tests/test_litellm/llms/openai/test_openai.py index 9539e13a802..2be691fa65b 100644 --- a/tests/test_litellm/llms/openai/test_openai.py +++ b/tests/test_litellm/llms/openai/test_openai.py @@ -1,13 +1,15 @@ import asyncio import json from typing import Final +from unittest.mock import Mock import httpx import pytest -from openai import AsyncOpenAI +from openai import AsyncOpenAI, OpenAI import litellm from litellm.llms.openai.openai import OpenAIChatCompletion +from litellm.types.utils import ImageResponse @pytest.mark.parametrize( @@ -253,3 +255,98 @@ async def test_acompletion_streams_tool_call_arguments_over_injected_transport() assert tool_call.function.name == "get_weather" assert json.loads(tool_call.function.arguments) == {"city": "Paris"} assert rebuilt.choices[0].finish_reason == "tool_calls" + + +_PROVIDER_HEADERS: Final = {"x-request-id": "req_openai", "x-ratelimit-remaining-requests": "41"} + + +def _image_generation_transport() -> httpx.MockTransport: + return httpx.MockTransport( + lambda request: httpx.Response( + 200, json={"created": 1, "data": [{"b64_json": "abc"}]}, headers=_PROVIDER_HEADERS + ) + ) + + +def _speech_transport() -> httpx.MockTransport: + return httpx.MockTransport( + lambda request: httpx.Response( + 200, content=b"audio-bytes", headers={**_PROVIDER_HEADERS, "content-type": "audio/mpeg"} + ) + ) + + +def _assert_provider_headers_recorded(response) -> None: + assert response._hidden_params["headers"]["x-request-id"] == "req_openai" + assert response._hidden_params["additional_headers"]["llm_provider-x-request-id"] == "req_openai" + assert response._hidden_params["additional_headers"]["x-ratelimit-remaining-requests"] == "41" + + +def _image_generation_kwargs() -> dict: + return { + "model": "gpt-image-2", + "prompt": "a cat", + "timeout": 10, + "optional_params": {}, + "logging_obj": Mock(), + "api_key": "transport-only", + "model_response": ImageResponse(), + } + + +def test_image_generation_records_provider_response_headers(): + with httpx.Client(transport=_image_generation_transport()) as http_client: + response = OpenAIChatCompletion().image_generation( + client=OpenAI(api_key="transport-only", http_client=http_client), **_image_generation_kwargs() + ) + + _assert_provider_headers_recorded(response) + + +@pytest.mark.asyncio +async def test_aimage_generation_records_provider_response_headers(): + async with httpx.AsyncClient(transport=_image_generation_transport()) as http_client: + response = await OpenAIChatCompletion().image_generation( + client=AsyncOpenAI(api_key="transport-only", http_client=http_client), + aimg_generation=True, + **_image_generation_kwargs(), + ) + + _assert_provider_headers_recorded(response) + + +def _audio_speech_kwargs() -> dict: + return { + "model": "gpt-4o-mini-tts", + "input": "hello", + "voice": "alloy", + "optional_params": {}, + "api_key": "transport-only", + "api_base": None, + "organization": None, + "project": None, + "max_retries": 0, + "timeout": 10, + "logging_obj": Mock(), + } + + +def test_audio_speech_records_provider_response_headers(): + with httpx.Client(transport=_speech_transport()) as http_client: + response = OpenAIChatCompletion().audio_speech( + client=OpenAI(api_key="transport-only", http_client=http_client), **_audio_speech_kwargs() + ) + + _assert_provider_headers_recorded(response) + + +@pytest.mark.asyncio +async def test_async_audio_speech_records_provider_response_headers(): + async with httpx.AsyncClient(transport=_speech_transport()) as http_client: + response = await OpenAIChatCompletion().audio_speech( + client=AsyncOpenAI(api_key="transport-only", http_client=http_client), + aspeech=True, + **_audio_speech_kwargs(), + ) + + _assert_provider_headers_recorded(response) diff --git a/tests/test_litellm/llms/openai/transcriptions/test_openai_transcriptions_handler.py b/tests/test_litellm/llms/openai/transcriptions/test_openai_transcriptions_handler.py new file mode 100644 index 00000000000..f2dbb71fea1 --- /dev/null +++ b/tests/test_litellm/llms/openai/transcriptions/test_openai_transcriptions_handler.py @@ -0,0 +1,71 @@ +from typing import Final +from unittest.mock import Mock + +import httpx +import pytest +from openai import AsyncOpenAI, OpenAI + +from litellm.llms.openai.transcriptions.handler import OpenAIAudioTranscription +from litellm.types.utils import TranscriptionResponse + +_PROVIDER_HEADERS: Final = {"x-request-id": "req_stt", "x-ratelimit-remaining-requests": "41"} + + +def _transcription_transport() -> httpx.MockTransport: + return httpx.MockTransport(lambda request: httpx.Response(200, json={"text": "hello"}, headers=_PROVIDER_HEADERS)) + + +def _logging_obj() -> Mock: + logging_obj = Mock() + logging_obj.model_call_details = {} + return logging_obj + + +def _call_kwargs(logging_obj: Mock) -> dict: + return { + "model": "gpt-4o-mini-transcribe", + "audio_file": ("audio.wav", b"riff-bytes", "audio/wav"), + "optional_params": {}, + "litellm_params": {}, + "model_response": TranscriptionResponse(), + "timeout": 10.0, + "max_retries": 0, + "logging_obj": logging_obj, + "api_key": "transport-only", + "api_base": None, + } + + +def _assert_headers_recorded(response: TranscriptionResponse, logging_obj: Mock) -> None: + assert response.text == "hello" + assert response._hidden_params["headers"]["x-request-id"] == "req_stt" + assert response._hidden_params["additional_headers"]["llm_provider-x-request-id"] == "req_stt" + assert response._hidden_params["additional_headers"]["x-ratelimit-remaining-requests"] == "41" + assert logging_obj.model_call_details["response_headers"]["x-request-id"] == "req_stt" + + +def test_audio_transcriptions_records_provider_response_headers(): + logging_obj = _logging_obj() + + with httpx.Client(transport=_transcription_transport()) as http_client: + response = OpenAIAudioTranscription().audio_transcriptions( + client=OpenAI(api_key="transport-only", http_client=http_client), + atranscription=False, + **_call_kwargs(logging_obj), + ) + + _assert_headers_recorded(response, logging_obj) + + +@pytest.mark.asyncio +async def test_async_audio_transcriptions_records_provider_response_headers(): + logging_obj = _logging_obj() + + async with httpx.AsyncClient(transport=_transcription_transport()) as http_client: + response = await OpenAIAudioTranscription().audio_transcriptions( + client=AsyncOpenAI(api_key="transport-only", http_client=http_client), + atranscription=True, + **_call_kwargs(logging_obj), + ) + + _assert_headers_recorded(response, logging_obj) diff --git a/tests/test_litellm/test_non_chat_routes_open_llm_spans.py b/tests/test_litellm/test_non_chat_routes_open_llm_spans.py index d62959ccd43..02c95c4bb2a 100644 --- a/tests/test_litellm/test_non_chat_routes_open_llm_spans.py +++ b/tests/test_litellm/test_non_chat_routes_open_llm_spans.py @@ -46,9 +46,9 @@ class _FakeSpeech: )() -class _FakeImages: +class _FakeRawImages: async def generate(self, **kwargs: Any) -> Any: - return type( + parsed: Final = type( "_Images", (), { @@ -58,6 +58,16 @@ class _FakeImages: } }, )() + return type( + "_RawImages", + (), + {"parse": lambda self: parsed, "headers": httpx.Headers({"x-request-id": "req-image"})}, + )() + + +class _FakeImages: + def __init__(self) -> None: + self.with_raw_response = _FakeRawImages() class _FakeModerations: diff --git a/tests/unit/llms/openai/image_generation/test_openai_image_generation_extra_headers.py b/tests/unit/llms/openai/image_generation/test_openai_image_generation_extra_headers.py index 55ef74abd7b..11df07f5fea 100644 --- a/tests/unit/llms/openai/image_generation/test_openai_image_generation_extra_headers.py +++ b/tests/unit/llms/openai/image_generation/test_openai_image_generation_extra_headers.py @@ -12,6 +12,14 @@ import pytest from litellm.llms.openai.openai import OpenAIChatCompletion +from litellm.types.utils import ImageResponse + + +def _raw_image_response(mock_image_data): + raw_response = MagicMock() + raw_response.parse.return_value = mock_image_data + raw_response.headers = {"x-request-id": "req-image"} + return raw_response @pytest.fixture @@ -41,7 +49,7 @@ class TestImageGenerationExtraHeaders: } mock_openai_client = MagicMock() - mock_openai_client.images.generate.return_value = mock_image_data + mock_openai_client.images.with_raw_response.generate.return_value = _raw_image_response(mock_image_data) mock_openai_client.api_key = "test-key" mock_openai_client._base_url._uri_reference = "https://api.openai.com" @@ -58,7 +66,7 @@ class TestImageGenerationExtraHeaders: client=mock_openai_client, ) - _, kwargs = mock_openai_client.images.generate.call_args + _, kwargs = mock_openai_client.images.with_raw_response.generate.call_args assert kwargs.get("extra_headers") == test_headers def test_sync_image_generation_without_headers( @@ -72,7 +80,7 @@ class TestImageGenerationExtraHeaders: } mock_openai_client = MagicMock() - mock_openai_client.images.generate.return_value = mock_image_data + mock_openai_client.images.with_raw_response.generate.return_value = _raw_image_response(mock_image_data) mock_openai_client.api_key = "test-key" mock_openai_client._base_url._uri_reference = "https://api.openai.com" @@ -86,7 +94,7 @@ class TestImageGenerationExtraHeaders: client=mock_openai_client, ) - _, kwargs = mock_openai_client.images.generate.call_args + _, kwargs = mock_openai_client.images.with_raw_response.generate.call_args assert "extra_headers" not in kwargs @pytest.mark.asyncio @@ -101,7 +109,9 @@ class TestImageGenerationExtraHeaders: } mock_openai_client = MagicMock() - mock_openai_client.images.generate = AsyncMock(return_value=mock_image_data) + mock_openai_client.images.with_raw_response.generate = AsyncMock( + return_value=_raw_image_response(mock_image_data) + ) mock_openai_client.api_key = "test-key" test_headers = {"cf-aig-authorization": "Bearer custom-token"} @@ -109,7 +119,7 @@ class TestImageGenerationExtraHeaders: await openai_chat_completions.aimage_generation( prompt="A white cat", data={"model": "dall-e-3", "prompt": "A white cat"}, - model_response=MagicMock(), + model_response=ImageResponse(), timeout=60.0, logging_obj=mock_logging_obj, api_key="test-key", @@ -117,7 +127,7 @@ class TestImageGenerationExtraHeaders: client=mock_openai_client, ) - _, kwargs = mock_openai_client.images.generate.call_args + _, kwargs = mock_openai_client.images.with_raw_response.generate.call_args assert kwargs.get("extra_headers") == test_headers @pytest.mark.asyncio @@ -132,20 +142,22 @@ class TestImageGenerationExtraHeaders: } mock_openai_client = MagicMock() - mock_openai_client.images.generate = AsyncMock(return_value=mock_image_data) + mock_openai_client.images.with_raw_response.generate = AsyncMock( + return_value=_raw_image_response(mock_image_data) + ) mock_openai_client.api_key = "test-key" await openai_chat_completions.aimage_generation( prompt="A white cat", data={"model": "dall-e-3", "prompt": "A white cat"}, - model_response=MagicMock(), + model_response=ImageResponse(), timeout=60.0, logging_obj=mock_logging_obj, api_key="test-key", client=mock_openai_client, ) - _, kwargs = mock_openai_client.images.generate.call_args + _, kwargs = mock_openai_client.images.with_raw_response.generate.call_args assert "extra_headers" not in kwargs @pytest.mark.parametrize("is_async", [False, True]) @@ -169,11 +181,13 @@ class TestImageGenerationExtraHeaders: test_headers = {"cf-aig-authorization": "Bearer custom-token"} if is_async: - mock_openai_client.images.generate = AsyncMock(return_value=mock_image_data) + mock_openai_client.images.with_raw_response.generate = AsyncMock( + return_value=_raw_image_response(mock_image_data) + ) await openai_chat_completions.aimage_generation( prompt="A white cat", data={"model": "dall-e-3", "prompt": "A white cat"}, - model_response=MagicMock(), + model_response=ImageResponse(), timeout=60.0, logging_obj=mock_logging_obj, api_key="test-key", @@ -181,7 +195,7 @@ class TestImageGenerationExtraHeaders: client=mock_openai_client, ) else: - mock_openai_client.images.generate.return_value = mock_image_data + mock_openai_client.images.with_raw_response.generate.return_value = _raw_image_response(mock_image_data) openai_chat_completions.image_generation( model="dall-e-3", prompt="A white cat", @@ -197,7 +211,7 @@ class TestImageGenerationExtraHeaders: "complete_input_dict" ] assert "extra_headers" not in logged_body - _, kwargs = mock_openai_client.images.generate.call_args + _, kwargs = mock_openai_client.images.with_raw_response.generate.call_args assert kwargs.get("extra_headers") == test_headers def test_sync_image_generation_forwards_headers_to_async( @@ -242,7 +256,9 @@ class TestImageGenerationEntryPointHeaders: } mock_openai_client = MagicMock() - mock_openai_client.images.generate = AsyncMock(return_value=mock_image_data) + mock_openai_client.images.with_raw_response.generate = AsyncMock( + return_value=_raw_image_response(mock_image_data) + ) mock_openai_client.api_key = "test-key" mock_openai_client._base_url._uri_reference = "https://api.openai.com" @@ -256,6 +272,6 @@ class TestImageGenerationEntryPointHeaders: api_key="test-key", ) - mock_openai_client.images.generate.assert_called_once() - _, kwargs = mock_openai_client.images.generate.call_args + mock_openai_client.images.with_raw_response.generate.assert_called_once() + _, kwargs = mock_openai_client.images.with_raw_response.generate.call_args assert kwargs.get("extra_headers") == test_headers diff --git a/tests/unit/llms/vertex_ai/text_to_speech/test_transformation.py b/tests/unit/llms/vertex_ai/text_to_speech/test_transformation.py index b5eec42b569..ee7bdebe745 100644 --- a/tests/unit/llms/vertex_ai/text_to_speech/test_transformation.py +++ b/tests/unit/llms/vertex_ai/text_to_speech/test_transformation.py @@ -526,6 +526,7 @@ class TestVertexAILyriaTextToSpeechConfig: ): mock_response = Mock(spec=httpx.Response) mock_response.status_code = 200 + mock_response.headers = {"content-type": "application/json"} mock_response.json.return_value = response_json with ( patch.object( # test-quality-ok: litellm.speech has no seam for Vertex token minting From 5a8ec1378611e6620f5cfe291f0e971ca403293e Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 15:04:03 -0500 Subject: [PATCH 94/96] feat(usage): search keys beyond the top-N usage subset (#42827) --- litellm/proxy/_types.py | 2 + .../internal_user_endpoints.py | 153 ++++++++++++++-- .../internal_user_endpoints.py | 12 +- .../endpointaudit/coverage_allowlist.txt | 1 + .../proxy/auth/test_route_checks.py | 1 + .../test_internal_user_endpoints.py | 166 ++++++++++++++++++ .../_components/components/UsagePageView.tsx | 16 +- .../components/KeyActivityPanel.test.tsx | 64 +++++++ .../UsagePage/components/KeyActivityPanel.tsx | 73 +++++++- .../src/components/networking.tsx | 30 ++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 65 +++++++ 11 files changed, 560 insertions(+), 23 deletions(-) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 4affa55f903..51a3eb03067 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -699,6 +699,7 @@ class LiteLLMRoutes(enum.Enum): "/user/list", "/user/daily/activity", "/user/daily/activity/aggregated", + "/user/daily/activity/aggregated/search", # team "/team/new", "/team/update", @@ -901,6 +902,7 @@ class LiteLLMRoutes(enum.Enum): "/model/delete", "/user/daily/activity", "/user/daily/activity/aggregated", + "/user/daily/activity/aggregated/search", # Endpoint restricts results to organizations the caller is ORG_ADMIN # of; a caller who administers none gets an empty result set. "/organization/daily/activity", diff --git a/litellm/proxy/management_endpoints/internal_user_endpoints.py b/litellm/proxy/management_endpoints/internal_user_endpoints.py index 59d8dd821d8..7b25348aa53 100644 --- a/litellm/proxy/management_endpoints/internal_user_endpoints.py +++ b/litellm/proxy/management_endpoints/internal_user_endpoints.py @@ -28,6 +28,7 @@ from typing_extensions import ReadOnly, TypedDict import litellm from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid +from litellm.constants import USAGE_TOP_API_KEYS_LIMIT from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler from litellm.proxy._types import * from litellm.proxy.auth.auth_checks import ( @@ -87,11 +88,13 @@ from litellm.repositories.verification_token_repository import ( VerificationTokenRepository, ) from litellm.types.proxy.management_endpoints.common_daily_activity import ( + DailySpendMetadata, SpendAnalyticsPaginatedResponse, ) from litellm.types.proxy.management_endpoints.internal_user_endpoints import ( BulkUpdateUserRequest, BulkUpdateUserResponse, + KeyActivitySearchWhere, UserListResponse, UserSearchWhere, UserUpdateResult, @@ -2991,6 +2994,27 @@ async def get_user_daily_activity( ) +def _resolve_user_daily_activity_entity_id( + user_api_key_dict: UserAPIKeyAuth, + user_id: str | None, +) -> str | None: + is_admin: Final = _user_has_admin_view(user_api_key_dict) + + if is_admin: + return user_id + + caller_user_id: Final = require_caller_user_id_for_non_admin(user_api_key_dict) + effective_user_id: Final = user_id if user_id is not None else caller_user_id + if effective_user_id != caller_user_id: + raise HTTPException( + status_code=status.HTTP_403_FORBIDDEN, + detail={ # mutable-ok: FastAPI detail payload shape + "error": "Non-admin users can only view their own spend data." + }, + ) + return effective_user_id + + @router.get( "/user/daily/activity/aggregated", tags=["Budget & Spend Tracking", "Internal User management"], @@ -3057,20 +3081,7 @@ async def get_user_daily_activity_aggregated( ) try: - is_admin: Final = _user_has_admin_view(user_api_key_dict) - - if is_admin: - entity_id = user_id # None means global view, otherwise filter by user - else: - caller_user_id: Final = require_caller_user_id_for_non_admin(user_api_key_dict) - if user_id is None: - user_id = caller_user_id - if user_id != caller_user_id: - raise HTTPException( - status_code=status.HTTP_403_FORBIDDEN, - detail={"error": "Non-admin users can only view their own spend data."}, - ) - entity_id = user_id + entity_id: Final = _resolve_user_daily_activity_entity_id(user_api_key_dict, user_id) return await get_daily_activity_aggregated( prisma_client=prisma_client, @@ -3094,3 +3105,117 @@ async def get_user_daily_activity_aggregated( status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, detail={"error": f"Failed to fetch analytics: {e}"}, ) + + +@router.get( + "/user/daily/activity/aggregated/search", + tags=["Budget & Spend Tracking", "Internal User management"], # mutable-ok: FastAPI route tags shape + dependencies=[Depends(user_api_key_auth)], # mutable-ok: FastAPI route dependencies shape + response_model=SpendAnalyticsPaginatedResponse, +) +@management_endpoint_wrapper +async def search_user_daily_activity_keys( + search: str = fastapi.Query( + ..., + min_length=1, + description="Matches keys whose hash equals the value, or whose key alias or user ID contains it (case-insensitive)", + ), + start_date: str | None = fastapi.Query( + default=None, + description="Start date in YYYY-MM-DD format", + ), + end_date: str | None = fastapi.Query( + default=None, + description="End date in YYYY-MM-DD format", + ), + user_id: str | None = fastapi.Query( + default=None, + description="Filter by specific user ID. Admins can filter by any user or omit for global view. Non-admins must provide their own user_id.", + ), + timezone: int | None = fastapi.Query( + default=None, + description="Timezone offset in minutes from UTC (e.g., 480 for PST). " + "Matches JavaScript's Date.getTimezoneOffset() convention.", + ), + include_current_utc_day: bool = fastapi.Query( + default=False, + description="When the range ends on the caller's current local day, extend it to " + "today's UTC bucket so spend written after the caller's local midnight (in UTC " + "terms) is included. Requires the timezone parameter. Historical ranges are " + "never extended.", + ), + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), # noqa: B008 # FastAPI dependency injection +) -> SpendAnalyticsPaginatedResponse: + """ + Search verification tokens by exact token hash or by a case-insensitive substring of + the key alias or owning user ID, then return the aggregated daily activity for the + matches. Lets the Usage page surface keys that fell outside the top-spend subset + the aggregated endpoint loads. + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException( + status_code=500, + detail={ # mutable-ok: FastAPI detail payload shape + "error": CommonProxyErrors.db_not_connected_error.value + }, + ) + + if start_date is None or end_date is None: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail={"error": "Please provide start_date and end_date"}, # mutable-ok: FastAPI detail payload shape + ) + + try: + entity_id: Final = _resolve_user_daily_activity_entity_id(user_api_key_dict, user_id) + + search_or: Final = ( + {"token": search}, # mutable-ok: prisma serializes where clauses, keep plain dicts + {"key_alias": {"contains": search, "mode": "insensitive"}}, # mutable-ok: prisma where clause leaf + {"user_id": {"contains": search, "mode": "insensitive"}}, # mutable-ok: prisma where clause leaf + ) + where: Final[KeyActivitySearchWhere] = ( + {"OR": search_or} # mutable-ok: prisma where clause root + if entity_id is None + else {"user_id": entity_id, "OR": search_or} # mutable-ok: prisma where clause root + ) + matched_keys: Final = await VerificationTokenRepository(prisma_client).table.find_many( + where=where, + take=USAGE_TOP_API_KEYS_LIMIT, + order={"spend": "desc"}, # mutable-ok: prisma serializes order, keep it a plain dict + ) + tokens: Final = [key.token for key in matched_keys] # mutable-ok: api_key filter union expects a list + + if not tokens: + return SpendAnalyticsPaginatedResponse( + results=[], # mutable-ok: response model field shape + metadata=DailySpendMetadata( + api_key_limit=USAGE_TOP_API_KEYS_LIMIT, + total_api_keys=0, + ), + ) + + return await get_daily_activity_aggregated( + prisma_client=prisma_client, + table_name="litellm_dailyuserspend", + entity_id_field="user_id", + entity_id=entity_id, + entity_metadata_field=None, + start_date=start_date, + end_date=end_date, + model=None, + api_key=tokens, + timezone_offset_minutes=timezone, + include_current_utc_day=include_current_utc_day, + ) + + except HTTPException: + raise + except Exception as e: + verbose_proxy_logger.exception("/user/daily/activity/aggregated/search: Exception occured - %s", e) + raise HTTPException( + status_code=status.HTTP_500_INTERNAL_SERVER_ERROR, + detail={"error": f"Failed to fetch analytics: {e}"}, # mutable-ok: FastAPI detail payload shape + ) diff --git a/litellm/types/proxy/management_endpoints/internal_user_endpoints.py b/litellm/types/proxy/management_endpoints/internal_user_endpoints.py index 43e3899d523..05cbd4507a2 100644 --- a/litellm/types/proxy/management_endpoints/internal_user_endpoints.py +++ b/litellm/types/proxy/management_endpoints/internal_user_endpoints.py @@ -2,7 +2,7 @@ from collections.abc import Mapping, Sequence from typing import Any, Final, Literal from pydantic import BaseModel, ConfigDict, Field, field_validator -from typing_extensions import ReadOnly, TypedDict +from typing_extensions import NotRequired, ReadOnly, TypedDict from litellm.proxy._types import ( LiteLLM_UserTableWithKeyCount, @@ -28,6 +28,16 @@ class UserSearchWhere(TypedDict): OR: ReadOnly[tuple[Mapping[Literal["user_id", "user_email"], InsensitiveContains], ...]] +class KeyActivitySearchWhere(TypedDict): + """Prisma filter behind `/user/daily/activity/aggregated/search`: exact token hash, or key alias + or user id containing the term, case-insensitive.""" + + user_id: NotRequired[ReadOnly[str]] + OR: ReadOnly[ + tuple[Mapping[Literal["token"], str] | Mapping[Literal["key_alias", "user_id"], InsensitiveContains], ...] + ] + + class UserListResponse(BaseModel): """ Response model for the user list endpoint diff --git a/terraform/provider/tools/endpointaudit/coverage_allowlist.txt b/terraform/provider/tools/endpointaudit/coverage_allowlist.txt index f8277c83a64..d0f9c31ecaf 100644 --- a/terraform/provider/tools/endpointaudit/coverage_allowlist.txt +++ b/terraform/provider/tools/endpointaudit/coverage_allowlist.txt @@ -32,6 +32,7 @@ GET /team/spend/by_user GET /team/spend/report GET /user/daily/activity GET /user/daily/activity/aggregated +GET /user/daily/activity/aggregated/search GET /user/spend/report # Admin UI helper endpoints; serve UI forms and caller-scoped views, not desired state diff --git a/tests/test_litellm/proxy/auth/test_route_checks.py b/tests/test_litellm/proxy/auth/test_route_checks.py index e5179387f82..571e066e947 100644 --- a/tests/test_litellm/proxy/auth/test_route_checks.py +++ b/tests/test_litellm/proxy/auth/test_route_checks.py @@ -3517,6 +3517,7 @@ def test_internal_user_still_blocked_from_another_users_info(): [ "/user/daily/activity", "/user/daily/activity/aggregated", + "/user/daily/activity/aggregated/search", ], ) @pytest.mark.parametrize( diff --git a/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py index c663e63414c..8260aec9326 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py @@ -2659,6 +2659,172 @@ async def test_get_user_daily_activity_aggregated_non_admin_cannot_view_other_us assert mock_get_daily_agg.call_args.kwargs["entity_id"] == "regular-user-123" +@pytest.mark.asyncio +async def test_search_user_daily_activity_keys_passes_matched_tokens_to_aggregation(monkeypatch): + """The search endpoint resolves matching verification tokens by hash, alias, or + user id, then aggregates daily spend for exactly those tokens. This is what lets + the Usage page find keys outside the top-spend subset the aggregated endpoint caps.""" + from types import SimpleNamespace + from unittest.mock import AsyncMock, MagicMock + + from litellm.constants import USAGE_TOP_API_KEYS_LIMIT + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + search_user_daily_activity_keys, + ) + + mock_prisma_client = MagicMock() + mock_prisma_client.db.litellm_verificationtoken.find_many = AsyncMock( + return_value=[SimpleNamespace(token="tok-a"), SimpleNamespace(token="tok-b")] + ) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client) + + mock_response = MagicMock() + mock_get_daily_agg = AsyncMock(return_value=mock_response) + monkeypatch.setattr( + "litellm.proxy.management_endpoints.internal_user_endpoints.get_daily_activity_aggregated", + mock_get_daily_agg, + ) + + admin_key_dict = UserAPIKeyAuth( + user_id="admin-user-001", + user_role=LitellmUserRoles.PROXY_ADMIN, + ) + + result = await search_user_daily_activity_keys( + search="gamma", + start_date="2025-02-01", + end_date="2025-02-28", + user_id=None, + timezone=480, + include_current_utc_day=False, + user_api_key_dict=admin_key_dict, + ) + + assert result is mock_response + + find_many_kwargs = mock_prisma_client.db.litellm_verificationtoken.find_many.call_args.kwargs + assert find_many_kwargs["take"] == USAGE_TOP_API_KEYS_LIMIT + assert find_many_kwargs["where"]["OR"] == ( + {"token": "gamma"}, + {"key_alias": {"contains": "gamma", "mode": "insensitive"}}, + {"user_id": {"contains": "gamma", "mode": "insensitive"}}, + ) + assert "user_id" not in find_many_kwargs["where"] + + mock_get_daily_agg.assert_called_once_with( + prisma_client=mock_prisma_client, + table_name="litellm_dailyuserspend", + entity_id_field="user_id", + entity_id=None, + entity_metadata_field=None, + start_date="2025-02-01", + end_date="2025-02-28", + model=None, + api_key=["tok-a", "tok-b"], + timezone_offset_minutes=480, + include_current_utc_day=False, + ) + + +@pytest.mark.asyncio +async def test_search_user_daily_activity_keys_no_match_returns_empty_without_aggregating(monkeypatch): + from unittest.mock import AsyncMock, MagicMock + + from litellm.constants import USAGE_TOP_API_KEYS_LIMIT + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + search_user_daily_activity_keys, + ) + + mock_prisma_client = MagicMock() + mock_prisma_client.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[]) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client) + + mock_get_daily_agg = AsyncMock() + monkeypatch.setattr( + "litellm.proxy.management_endpoints.internal_user_endpoints.get_daily_activity_aggregated", + mock_get_daily_agg, + ) + + admin_key_dict = UserAPIKeyAuth( + user_id="admin-user-001", + user_role=LitellmUserRoles.PROXY_ADMIN, + ) + + result = await search_user_daily_activity_keys( + search="nothing-matches", + start_date="2025-02-01", + end_date="2025-02-28", + user_id=None, + timezone=None, + include_current_utc_day=False, + user_api_key_dict=admin_key_dict, + ) + + assert result.results == [] + assert result.metadata.api_key_limit == USAGE_TOP_API_KEYS_LIMIT + assert result.metadata.total_api_keys == 0 + mock_get_daily_agg.assert_not_called() + + +@pytest.mark.asyncio +async def test_search_user_daily_activity_keys_non_admin_scoped_to_caller(monkeypatch): + """Same scoping contract as the aggregated route: a non-admin with no user_id + is scoped to their own rows, and any other user_id is a 403.""" + from types import SimpleNamespace + from unittest.mock import AsyncMock, MagicMock + + from fastapi import HTTPException + + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + search_user_daily_activity_keys, + ) + + mock_prisma_client = MagicMock() + mock_prisma_client.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[SimpleNamespace(token="tok-a")]) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client) + + non_admin_key_dict = UserAPIKeyAuth( + user_id="user-1", + user_role=LitellmUserRoles.INTERNAL_USER, + ) + + mock_response = MagicMock() + mock_get_daily_agg = AsyncMock(return_value=mock_response) + monkeypatch.setattr( + "litellm.proxy.management_endpoints.internal_user_endpoints.get_daily_activity_aggregated", + mock_get_daily_agg, + ) + + result = await search_user_daily_activity_keys( + search="gamma", + start_date="2025-02-01", + end_date="2025-02-28", + user_id=None, + timezone=None, + include_current_utc_day=False, + user_api_key_dict=non_admin_key_dict, + ) + + assert result is mock_response + assert mock_get_daily_agg.call_args.kwargs["entity_id"] == "user-1" + find_many_kwargs = mock_prisma_client.db.litellm_verificationtoken.find_many.call_args.kwargs + assert find_many_kwargs["where"]["user_id"] == "user-1" + + with pytest.raises(HTTPException) as exc_info: + await search_user_daily_activity_keys( + search="gamma", + start_date="2025-02-01", + end_date="2025-02-28", + user_id="user-2", + timezone=None, + include_current_utc_day=False, + user_api_key_dict=non_admin_key_dict, + ) + + assert exc_info.value.status_code == 403 + assert "Non-admin users can only view their own spend data" in str(exc_info.value.detail) + + @pytest.mark.asyncio async def test_delete_user_cleans_up_created_by_invitation_links(mocker): """ diff --git a/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/UsagePageView.tsx b/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/UsagePageView.tsx index 228d8acf146..e0171d97423 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/UsagePageView.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/UsagePageView.tsx @@ -39,6 +39,7 @@ import { tagListCall, userDailyActivityAggregatedCall, userDailyActivityCall, + userDailyActivityKeySearchCall, } from "@/components/networking"; import AdvancedDatePicker from "@/components/shared/advanced_date_picker"; import { ChartLoader } from "@/components/shared/chart_loader"; @@ -437,6 +438,15 @@ const UsagePage: React.FC = ({ teams, organizations }) => { [userSpendData, modelViewType, teams], ); const keyMetrics = useMemo(() => processActivityData(userSpendData, "api_keys", teams), [userSpendData, teams]); + const searchKeys = useCallback( + (q: string) => { + if (!accessToken || !startTime || !endTime) return Promise.resolve({}); + return userDailyActivityKeySearchCall(accessToken, startTime, endTime, q, effectiveUserId).then((data) => + processActivityData(data, "api_keys", teams), + ); + }, + [accessToken, startTime, endTime, effectiveUserId, teams], + ); const mcpServerMetrics = useMemo( () => processActivityData(userSpendData, "mcp_servers", teams), [userSpendData, teams], @@ -865,7 +875,11 @@ const UsagePage: React.FC = ({ teams, organizations }) => { - + diff --git a/ui/litellm-dashboard/src/components/UsagePage/components/KeyActivityPanel.test.tsx b/ui/litellm-dashboard/src/components/UsagePage/components/KeyActivityPanel.test.tsx index 830139143e9..f8a7d07633b 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/components/KeyActivityPanel.test.tsx +++ b/ui/litellm-dashboard/src/components/UsagePage/components/KeyActivityPanel.test.tsx @@ -78,4 +78,68 @@ describe("KeyActivityPanel", () => { render(); expect(screen.queryByRole("note")).not.toBeInTheDocument(); }); + + it("finds keys outside the loaded top-spend subset via server search", async () => { + const searchKeys = vi + .fn<(query: string) => Promise>>() + .mockResolvedValue({ "hash-gamma": activity("gamma-low-key", "gamma@example.com", "user-gamma") }); + render( + , + ); + + fireEvent.change(screen.getByLabelText("Search keys"), { target: { value: "gamma" } }); + + expect(await screen.findByText("hash-gamma")).toBeInTheDocument(); + expect(searchKeys).toHaveBeenCalledWith("gamma"); + expect(screen.getByText("Showing 1 of 3 keys")).toBeInTheDocument(); + }); + + it("never calls the server search when every key is already loaded", async () => { + const searchKeys = vi + .fn<(query: string) => Promise>>() + .mockResolvedValue({ "hash-gamma": activity("gamma-low-key", "gamma@example.com", "user-gamma") }); + render(); + + fireEvent.change(screen.getByLabelText("Search keys"), { target: { value: "gamma" } }); + + expect(await screen.findByText('No keys match "gamma" in this date range')).toBeInTheDocument(); + await new Promise((resolve) => setTimeout(resolve, 400)); + expect(searchKeys).not.toHaveBeenCalled(); + }); + + it("drops stale server results as soon as the search callback is rebuilt", async () => { + const searchKeysA = vi + .fn<(query: string) => Promise>>() + .mockResolvedValue({ "hash-gamma": activity("gamma-low-key", "gamma@example.com", "user-gamma") }); + const searchKeysB = vi + .fn<(query: string) => Promise>>() + .mockReturnValue(new Promise(() => {})); + const { rerender } = render( + , + ); + + fireEvent.change(screen.getByLabelText("Search keys"), { target: { value: "gamma" } }); + expect(await screen.findByText("hash-gamma")).toBeInTheDocument(); + + rerender( + , + ); + + expect(screen.getByRole("status")).toHaveTextContent("Searching all keys"); + expect(screen.queryByText("hash-gamma")).not.toBeInTheDocument(); + }); + + it("reports a failed server search but keeps the local matches", async () => { + const searchKeys = vi + .fn<(query: string) => Promise>>() + .mockRejectedValue(new Error("boom")); + render( + , + ); + + fireEvent.change(screen.getByLabelText("Search keys"), { target: { value: "alice" } }); + + expect(await screen.findByRole("alert")).toHaveTextContent("Key search failed"); + expect(screen.getByTestId("rendered-keys")).toHaveTextContent("hash-alice"); + }); }); diff --git a/ui/litellm-dashboard/src/components/UsagePage/components/KeyActivityPanel.tsx b/ui/litellm-dashboard/src/components/UsagePage/components/KeyActivityPanel.tsx index 8b2141f8528..3467ba61b22 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/components/KeyActivityPanel.tsx +++ b/ui/litellm-dashboard/src/components/UsagePage/components/KeyActivityPanel.tsx @@ -1,5 +1,5 @@ import { Search, X } from "lucide-react"; -import React, { useMemo, useState } from "react"; +import React, { useEffect, useMemo, useState } from "react"; import { ActivityMetrics } from "@/components/activity_metrics"; import type { ApiKeyTruncation } from "@/components/EntityUsageExport/exportBlockedReason"; @@ -12,18 +12,67 @@ interface KeyActivityPanelProps { keyMetrics: Record; hidePromptCachingMetrics?: boolean; apiKeyTruncation?: ApiKeyTruncation; + searchKeys?: SearchKeys; } +type SearchKeys = (query: string) => Promise>; + +type RemoteSearch = + | { status: "idle" } + | { status: "loading"; query: string; searchKeys: SearchKeys } + | { status: "done"; query: string; searchKeys: SearchKeys; keys: Record } + | { status: "error"; query: string; searchKeys: SearchKeys }; + +const REMOTE_SEARCH_DEBOUNCE_MS = 300; + const KeyActivityPanel: React.FC = ({ keyMetrics, hidePromptCachingMetrics = false, apiKeyTruncation, + searchKeys, }) => { const [query, setQuery] = useState(""); + const [remote, setRemote] = useState({ status: "idle" }); const filtered = useMemo(() => filterKeyActivity(keyMetrics, query), [keyMetrics, query]); + const trimmedQuery = query.trim(); + const remoteEnabled = searchKeys !== undefined && apiKeyTruncation !== undefined && trimmedQuery !== ""; + + useEffect(() => { + if (!remoteEnabled) return; + let cancelled = false; + const timer = setTimeout(() => { + setRemote({ status: "loading", query: trimmedQuery, searchKeys }); + searchKeys(trimmedQuery) + .then((keys) => { + if (!cancelled) setRemote({ status: "done", query: trimmedQuery, searchKeys, keys }); + }) + .catch(() => { + if (!cancelled) setRemote({ status: "error", query: trimmedQuery, searchKeys }); + }); + }, REMOTE_SEARCH_DEBOUNCE_MS); + return () => { + cancelled = true; + clearTimeout(timer); + }; + }, [remoteEnabled, trimmedQuery, searchKeys]); + + const remoteMatchesSearch = + "searchKeys" in remote && remote.searchKeys === searchKeys && remote.query === trimmedQuery; + const remoteCurrent = remoteEnabled && remoteMatchesSearch; + const remoteLoading = remoteEnabled && (remote.status === "loading" || !remoteCurrent); + const remoteFailed = remoteCurrent && remote.status === "error"; + + const extraRemoteKeys = useMemo(() => { + const remoteKeys = remoteCurrent && remote.status === "done" ? remote.keys : {}; + return Object.fromEntries(Object.entries(remoteKeys).filter(([hash]) => !(hash in keyMetrics))); + }, [remoteCurrent, remote, keyMetrics]); + const displayed = useMemo(() => ({ ...extraRemoteKeys, ...filtered }), [extraRemoteKeys, filtered]); + const totalKeys = Object.keys(keyMetrics).length; - const shownKeys = Object.keys(filtered).length; - const isFiltering = query.trim() !== ""; + const shownKeys = Object.keys(displayed).length; + const totalShown = totalKeys + Object.keys(extraRemoteKeys).length; + const isFiltering = trimmedQuery !== ""; + const noMatches = isFiltering && !remoteLoading && totalKeys > 0 && shownKeys === 0; return (
@@ -47,8 +96,18 @@ const KeyActivityPanel: React.FC = ({ )} - Showing {shownKeys.toLocaleString()} of {totalKeys.toLocaleString()} keys + Showing {shownKeys.toLocaleString()} of {totalShown.toLocaleString()} keys + {remoteLoading && ( + + Searching all keys... + + )} + {remoteFailed && ( + + Key search failed + + )} {apiKeyTruncation !== undefined && ( Only the {apiKeyTruncation.limit.toLocaleString()} highest-spend keys of{" "} @@ -56,12 +115,12 @@ const KeyActivityPanel: React.FC = ({ )}
- {isFiltering && totalKeys > 0 && shownKeys === 0 ? ( + {noMatches ? (

- No keys match "{query.trim()}" in this date range + No keys match "{trimmedQuery}" in this date range

) : ( - + )}
); diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index c2f4fa80634..e6923a26d4c 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -2556,6 +2556,36 @@ export const userDailyActivityAggregatedCall = async ( } }; +export const userDailyActivityKeySearchCall = async ( + accessToken: string, + startTime: Date, + endTime: Date, + ...options: [search: string, userId?: string | null] +) => { + const [search, userId = null] = options; + try { + const formatDate = (date: Date) => { + const year = date.getFullYear(); + const month = String(date.getMonth() + 1).padStart(2, "0"); + const day = String(date.getDate()).padStart(2, "0"); + return `${year}-${month}-${day}`; + }; + return await apiClient.get(`/user/daily/activity/aggregated/search`, { + accessToken, + query: { + start_date: formatDate(startTime), + end_date: formatDate(endTime), + timezone: new Date().getTimezoneOffset().toString(), + search, + user_id: userId || undefined, + }, + }); + } catch (error) { + console.error("Failed to search user daily activity keys:", error); + throw error; + } +}; + export const gatewayDailyActivityCall = async (accessToken: string, startTime: Date, endTime: Date) => { /** * Get gateway request counts (SGR) recorded by the proxy middleware. diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 052cc693f8b..79d008c1b48 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -17326,6 +17326,29 @@ export interface paths { patch?: never; trace?: never; }; + "/user/daily/activity/aggregated/search": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Search User Daily Activity Keys + * @description Search verification tokens by exact token hash or by a case-insensitive substring of + * the key alias or owning user ID, then return the aggregated daily activity for the + * matches. Lets the Usage page surface keys that fell outside the top-spend subset + * the aggregated endpoint loads. + */ + get: operations["search_user_daily_activity_keys_user_daily_activity_aggregated_search_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/user/delete": { parameters: { query?: never; @@ -68731,6 +68754,48 @@ export interface operations { }; }; }; + search_user_daily_activity_keys_user_daily_activity_aggregated_search_get: { + parameters: { + query: { + /** @description Matches keys whose hash equals the value, or whose key alias or user ID contains it (case-insensitive) */ + search: string; + /** @description Start date in YYYY-MM-DD format */ + start_date?: string | null; + /** @description End date in YYYY-MM-DD format */ + end_date?: string | null; + /** @description Filter by specific user ID. Admins can filter by any user or omit for global view. Non-admins must provide their own user_id. */ + user_id?: string | null; + /** @description Timezone offset in minutes from UTC (e.g., 480 for PST). Matches JavaScript's Date.getTimezoneOffset() convention. */ + timezone?: number | null; + /** @description When the range ends on the caller's current local day, extend it to today's UTC bucket so spend written after the caller's local midnight (in UTC terms) is included. Requires the timezone parameter. Historical ranges are never extended. */ + include_current_utc_day?: boolean; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["SpendAnalyticsPaginatedResponse"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; delete_user_user_delete_post: { parameters: { query?: never; From c2eb549ee685b442b74bc7c1cf9c4c1a7698a5ec Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 15:04:12 -0500 Subject: [PATCH 95/96] feat(usage): search team keys beyond the top-N in the Team usage view (#42857) --- litellm/proxy/_types.py | 5 + .../management_endpoints/team_endpoints.py | 109 +++++++++ .../management_endpoints/team_endpoints.py | 24 ++ .../endpointaudit/coverage_allowlist.txt | 1 + .../test_team_daily_activity_key_search.py | 128 +++++++++++ .../management/test_team_daily_activity.py | 15 +- .../proxy/auth/test_route_checks.py | 49 ++++ .../test_team_endpoints.py | 212 ++++++++++++++++++ .../components/EntityUsage/EntityUsage.tsx | 14 +- .../src/components/networking.tsx | 25 +++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 60 ++++- 11 files changed, 635 insertions(+), 7 deletions(-) create mode 100644 tests/integration/spend/test_team_daily_activity_key_search.py diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 51a3eb03067..73e3e0e6ee0 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -307,6 +307,7 @@ class KeyManagementRoutes(str, enum.Enum): # team usage routes TEAM_DAILY_ACTIVITY = "/team/daily/activity" TEAM_DAILY_ACTIVITY_AGGREGATED = "/team/daily/activity/aggregated" + TEAM_DAILY_ACTIVITY_AGGREGATED_SEARCH = "/team/daily/activity/aggregated/search" # team spend-log viewing SPEND_LOGS = "/spend/logs" @@ -673,6 +674,7 @@ class LiteLLMRoutes(enum.Enum): KeyManagementRoutes.TEAM_KEY_BULK_UPDATE.value, KeyManagementRoutes.TEAM_DAILY_ACTIVITY.value, KeyManagementRoutes.TEAM_DAILY_ACTIVITY_AGGREGATED.value, + KeyManagementRoutes.TEAM_DAILY_ACTIVITY_AGGREGATED_SEARCH.value, KeyManagementRoutes.SPEND_LOGS.value, KeyManagementRoutes.SPEND_LOGS_V2.value, KeyManagementRoutes.KEY_RESET_SPEND.value, @@ -717,6 +719,7 @@ class LiteLLMRoutes(enum.Enum): "/team/permissions_bulk_update", "/team/daily/activity", "/team/daily/activity/aggregated", + "/team/daily/activity/aggregated/search", "/team/spend/by_user", # gateway request counts (SGR); deployment-wide, admin-only "/gateway/daily/activity", @@ -887,6 +890,7 @@ class LiteLLMRoutes(enum.Enum): "/team/permissions_update", "/team/daily/activity", "/team/daily/activity/aggregated", + "/team/daily/activity/aggregated/search", "/team/spend/by_user", "/team/{team_id}/members/me", # POST/GET the team's logging callbacks, and DELETE one of them. Every @@ -986,6 +990,7 @@ class LiteLLMRoutes(enum.Enum): "/user/daily/activity", "/team/daily/activity", "/team/daily/activity/aggregated", + "/team/daily/activity/aggregated/search", "/tag/daily/activity", "/tag/list", "/audit", diff --git a/litellm/proxy/management_endpoints/team_endpoints.py b/litellm/proxy/management_endpoints/team_endpoints.py index 6cec3e714ec..493c83c730a 100644 --- a/litellm/proxy/management_endpoints/team_endpoints.py +++ b/litellm/proxy/management_endpoints/team_endpoints.py @@ -40,6 +40,7 @@ from typing_extensions import ReadOnly, TypedDict, assert_never import litellm from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid +from litellm.constants import USAGE_TOP_API_KEYS_LIMIT from litellm.integrations.prometheus import PrometheusLogger from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.proxy._types import ( @@ -196,6 +197,7 @@ from litellm.repositories.verification_token_repository import ( from litellm.router import Router from litellm.types.proxy.auth.auth_checks import UserNotFoundError from litellm.types.proxy.management_endpoints.common_daily_activity import ( + DailySpendMetadata, SpendAnalyticsPaginatedResponse, ) from litellm.types.proxy.management_endpoints.team_endpoints import ( @@ -204,7 +206,9 @@ from litellm.types.proxy.management_endpoints.team_endpoints import ( BulkUpdateTeamMemberPermissionsRequest, BulkUpdateTeamMemberPermissionsResponse, GetTeamMemberPermissionsResponse, + TeamIdSearchFilter, TeamIdSearchMatch, + TeamKeyActivitySearchWhere, TeamListItem, TeamListResponse, TeamMemberAddResult, @@ -6805,6 +6809,111 @@ async def get_team_daily_activity_aggregated( ) +def _team_key_search_where(*, search: str, scope: _TeamDailyActivityScope) -> TeamKeyActivitySearchWhere: + """Caller scoping lives inside the same Prisma where as the search term so `take` + never trims visible matches in favour of keys the caller is not allowed to see.""" + search_or: Final = ( + {"token": search}, # mutable-ok: prisma where clause leaf + {"key_alias": {"contains": search, "mode": "insensitive"}}, # mutable-ok: prisma where clause leaf + {"user_id": {"contains": search, "mode": "insensitive"}}, # mutable-ok: prisma where clause leaf + ) + own_keys: Final = tuple(scope.api_key_filter) if isinstance(scope.api_key_filter, list) else None + team_filter: Final[TeamIdSearchFilter | None] = ( + { # mutable-ok: prisma where clause leaf + "in": tuple(scope.team_ids), + "notIn": tuple(scope.exclude_team_ids), + } + if scope.team_ids is not None and scope.exclude_team_ids is not None + else {"in": tuple(scope.team_ids)} # mutable-ok: prisma where clause leaf + if scope.team_ids is not None + else {"notIn": tuple(scope.exclude_team_ids)} # mutable-ok: prisma where clause leaf + if scope.exclude_team_ids is not None + else None + ) + if team_filter is None and own_keys is None: + return {"OR": search_or} # mutable-ok: prisma where clause root + if team_filter is None and own_keys is not None: + return {"token": {"in": own_keys}, "OR": search_or} # mutable-ok: prisma where clause root + if team_filter is not None and own_keys is None: + return {"team_id": team_filter, "OR": search_or} # mutable-ok: prisma where clause root + assert team_filter is not None and own_keys is not None + return { # mutable-ok: prisma where clause root + "team_id": team_filter, + "token": {"in": own_keys}, # mutable-ok: prisma where clause leaf + "OR": search_or, + } + + +@router.get( + "/team/daily/activity/aggregated/search", + response_model=SpendAnalyticsPaginatedResponse, + tags=["team management"], # mutable-ok: FastAPI route tags shape +) +async def search_team_daily_activity_keys( + user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], + search: str = fastapi.Query( + ..., + min_length=1, + description="Exact token hash, or a case-insensitive substring of the key alias or owning user id", + ), + team_ids: str | None = None, + start_date: str | None = None, + end_date: str | None = None, + exclude_team_ids: str | None = None, + timezone: int | None = None, +) -> SpendAnalyticsPaginatedResponse: + """Aggregated daily team activity for the keys matching `search`, across every key the caller may + see rather than only the top USAGE_TOP_API_KEYS_LIMIT keys by spend.""" + from litellm.proxy.proxy_server import ( + prisma_client, + proxy_logging_obj, + user_api_key_cache, + ) + + if prisma_client is None: + raise _daily_activity_error(status_code=500, message=CommonProxyErrors.db_not_connected_error.value) + + range_error: Final = _aggregated_date_range_error(start_date, end_date) + if range_error is not None: + raise _daily_activity_error(status_code=400, message=range_error) + + scope: Final = await _resolve_team_daily_activity_scope( + team_ids=team_ids, + exclude_team_ids=exclude_team_ids, + api_key=None, + user_api_key_dict=user_api_key_dict, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + proxy_logging_obj=proxy_logging_obj, + ) + matched_keys: Final = await _tokens_db(prisma_client).find_many( + where=_team_key_search_where(search=search, scope=scope), + take=USAGE_TOP_API_KEYS_LIMIT, + order={"spend": "desc"}, # mutable-ok: prisma serializes order, keep it a plain dict + ) + tokens: Final = [key.token for key in matched_keys] # mutable-ok: get_daily_activity_aggregated takes list[str] + if not tokens: + return SpendAnalyticsPaginatedResponse( + results=[], # mutable-ok: response model field shape + metadata=DailySpendMetadata(api_key_limit=USAGE_TOP_API_KEYS_LIMIT, total_api_keys=0), + ) + + return await get_daily_activity_aggregated( + prisma_client=prisma_client, + table_name="litellm_dailyteamspend", + entity_id_field="team_id", + entity_id=scope.team_ids, + entity_metadata_field=scope.team_alias_metadata, + start_date=start_date, + end_date=end_date, + model=None, + api_key=tokens, + exclude_entity_ids=scope.exclude_team_ids, + timezone_offset_minutes=timezone, + include_entity_breakdown=True, + ) + + def _team_user_spend_sql(*, team_count: int, restrict_to_user: bool) -> str: team_placeholders: Final = ", ".join(f"${i}" for i in range(3, 3 + team_count)) user_clause: Final = f' AND sl."user" = ${3 + team_count}' if restrict_to_user else "" diff --git a/litellm/types/proxy/management_endpoints/team_endpoints.py b/litellm/types/proxy/management_endpoints/team_endpoints.py index 4524c47ec38..aac2703e918 100644 --- a/litellm/types/proxy/management_endpoints/team_endpoints.py +++ b/litellm/types/proxy/management_endpoints/team_endpoints.py @@ -1,6 +1,8 @@ +from collections.abc import Mapping, Sequence from typing import Any, Final, Literal from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator +from typing_extensions import NotRequired, ReadOnly, TypedDict from litellm.proxy._types import ( KeyManagementRoutes, @@ -11,10 +13,32 @@ from litellm.proxy._types import ( MemberDeleteRequest, ) from litellm.proxy.common_utils.timezone_utils import budget_duration_error +from litellm.types.proxy.management_endpoints.internal_user_endpoints import InsensitiveContains from litellm.types.proxy.management_endpoints.management_v1 import ResourceResponse TeamIdSearchMatch = Literal["exact", "prefix"] + +TeamIdSearchFilter = TypedDict( + "TeamIdSearchFilter", + { # mutable-ok: functional TypedDict field map + "in": NotRequired[ReadOnly[Sequence[str]]], + "notIn": NotRequired[ReadOnly[Sequence[str]]], + }, +) + + +class TeamKeyActivitySearchWhere(TypedDict): + """Prisma filter behind `/team/daily/activity/aggregated/search`: exact token hash, or key alias + or user id containing the term, case-insensitive, narrowed to the teams and keys the caller may see.""" + + team_id: NotRequired[ReadOnly[TeamIdSearchFilter]] + token: NotRequired[ReadOnly[Mapping[Literal["in"], Sequence[str]]]] + OR: ReadOnly[ + tuple[Mapping[Literal["token"], str] | Mapping[Literal["key_alias", "user_id"], InsensitiveContains], ...] + ] + + MAX_BULK_TEAM_MEMBER_DELETES: Final = 500 MAX_BULK_TEAM_MEMBER_BUDGET_UPDATES: Final = 500 diff --git a/terraform/provider/tools/endpointaudit/coverage_allowlist.txt b/terraform/provider/tools/endpointaudit/coverage_allowlist.txt index d0f9c31ecaf..b0f28a9c740 100644 --- a/terraform/provider/tools/endpointaudit/coverage_allowlist.txt +++ b/terraform/provider/tools/endpointaudit/coverage_allowlist.txt @@ -28,6 +28,7 @@ GET /tag/user-agent/per-user-analytics GET /tag/wau GET /team/daily/activity GET /team/daily/activity/aggregated +GET /team/daily/activity/aggregated/search GET /team/spend/by_user GET /team/spend/report GET /user/daily/activity diff --git a/tests/integration/spend/test_team_daily_activity_key_search.py b/tests/integration/spend/test_team_daily_activity_key_search.py new file mode 100644 index 00000000000..2b0395cf4ba --- /dev/null +++ b/tests/integration/spend/test_team_daily_activity_key_search.py @@ -0,0 +1,128 @@ +import uuid +from datetime import datetime, timedelta, timezone +from hashlib import sha256 +from typing import Final + +import pytest +from integration._support.client import Gateway, eventually, object_value +from integration._support.database import read_rows +from pydantic import JsonValue + +_SEARCH_PATH: Final = "/team/daily/activity/aggregated/search" + + +def _range_around_today() -> dict[str, str]: + today: Final = datetime.now(timezone.utc) + return { + "start_date": (today - timedelta(days=1)).strftime("%Y-%m-%d"), + "end_date": (today + timedelta(days=1)).strftime("%Y-%m-%d"), + "timezone": "0", + } + + +def _team_key_breakdown(body: dict[str, JsonValue], team: str) -> dict[str, JsonValue]: + results: Final = body["results"] + assert isinstance(results, list) and len(results) == 1, body + entities: Final = object_value(object_value(object_value(results[0])["breakdown"])["entities"]) + return object_value(object_value(entities[team])["api_key_breakdown"]) + + +def test_team_key_search_returns_only_the_matching_key_spend_by_alias_and_by_hash(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + model: Final = scenario.model(input_cost_per_token=0.001, output_cost_per_token=0.002) + team: Final = scenario.team(models=[model]) + needle_alias: Final = f"needle-{uuid.uuid4().hex}" + needle: Final = scenario.key(team_id=team, models=[model], key_alias=needle_alias) + other: Final = scenario.key(team_id=team, models=[model], key_alias=f"other-{uuid.uuid4().hex}") + needle_digest: Final = sha256(needle.encode()).hexdigest() + other_digest: Final = sha256(other.encode()).hexdigest() + for key in (needle, other): + reply: Final = gateway.chat(model, key=key, text=f"key search {uuid.uuid4().hex}") + assert object_value(reply["usage"])["total_tokens"] == 40, reply + daily: Final = eventually( + lambda: read_rows('SELECT api_key, spend FROM "LiteLLM_DailyTeamSpend" WHERE team_id=%s', (team,)), + lambda values: sorted(row["api_key"] for row in values) == sorted((needle_digest, other_digest)), + seconds=70, + ) + assert all(float(row["spend"]) == pytest.approx(0.06) for row in daily), daily + for search in (needle_alias.upper(), needle_digest): + response: Final = gateway.request( + "GET", _SEARCH_PATH, params={"team_ids": team, "search": search, **_range_around_today()} + ) + assert response.status_code == 200, response.text + body: Final = object_value(response.json()) + assert object_value(body["metadata"])["total_spend"] == pytest.approx(0.06), response.text + per_key: Final = _team_key_breakdown(body, team) + assert set(per_key) == {needle_digest}, response.text + assert object_value(object_value(per_key[needle_digest])["metrics"])["spend"] == pytest.approx(0.06) + + +def test_team_key_search_is_scoped_to_the_teams_the_caller_belongs_to(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + model: Final = scenario.model(input_cost_per_token=0.001, output_cost_per_token=0.002) + team: Final = scenario.team(models=[model]) + needle_alias: Final = f"needle-{uuid.uuid4().hex}" + needle: Final = scenario.key(team_id=team, models=[model], key_alias=needle_alias) + needle_digest: Final = sha256(needle.encode()).hexdigest() + reply: Final = gateway.chat(model, key=needle, text=f"key search {uuid.uuid4().hex}") + assert object_value(reply["usage"])["total_tokens"] == 40, reply + eventually( + lambda: read_rows('SELECT api_key FROM "LiteLLM_DailyTeamSpend" WHERE team_id=%s', (team,)), + lambda values: [row["api_key"] for row in values] == [needle_digest], + seconds=70, + ) + outsider: Final = scenario.user(user_role="internal_user") + outsider_team: Final = scenario.team(models=[model], members_with_roles=[{"user_id": outsider, "role": "user"}]) + outsider_key: Final = scenario.key(user_id=outsider, team_id=outsider_team, models=[model]) + params: Final = {"search": needle_alias, **_range_around_today()} + admin_view: Final = gateway.request("GET", _SEARCH_PATH, params={"team_ids": team, **params}) + assert admin_view.status_code == 200, admin_view.text + assert set(_team_key_breakdown(object_value(admin_view.json()), team)) == {needle_digest}, admin_view.text + own_teams_view: Final = gateway.request("GET", _SEARCH_PATH, params=params, key=outsider_key) + assert own_teams_view.status_code == 200, own_teams_view.text + own_teams_body: Final = object_value(own_teams_view.json()) + assert own_teams_body["results"] == [], own_teams_view.text + assert object_value(own_teams_body["metadata"])["total_api_keys"] == 0, own_teams_view.text + foreign_team_view: Final = gateway.request( + "GET", _SEARCH_PATH, params={"team_ids": team, **params}, key=outsider_key + ) + assert foreign_team_view.status_code == 404, foreign_team_view.text + + +def test_team_key_search_excludes_teams_inside_the_where(gateway: Gateway) -> None: + """The dashboard always sends exclude_team_ids; a matching key in an excluded + team with higher spend must not consume a take slot nor appear in the result.""" + with gateway.scenario() as scenario: + model: Final = scenario.model(input_cost_per_token=0.001, output_cost_per_token=0.002) + team_keep: Final = scenario.team(models=[model]) + team_drop: Final = scenario.team(models=[model]) + shared_alias: Final = f"needle-{uuid.uuid4().hex}" + keep: Final = scenario.key(team_id=team_keep, models=[model], key_alias=f"{shared_alias}-keep") + drop: Final = scenario.key(team_id=team_drop, models=[model], key_alias=f"{shared_alias}-drop") + keep_digest: Final = sha256(keep.encode()).hexdigest() + drop_digest: Final = sha256(drop.encode()).hexdigest() + for _ in range(2): + reply: Final = gateway.chat(model, key=drop, text=f"key search {uuid.uuid4().hex}") + assert object_value(reply["usage"])["total_tokens"] == 40, reply + reply = gateway.chat(model, key=keep, text=f"key search {uuid.uuid4().hex}") + assert object_value(reply["usage"])["total_tokens"] == 40, reply + eventually( + lambda: read_rows( + 'SELECT api_key, spend FROM "LiteLLM_DailyTeamSpend" WHERE team_id IN (%s, %s)', + (team_keep, team_drop), + ), + lambda values: sorted(row["api_key"] for row in values) == sorted((keep_digest, drop_digest)), + seconds=70, + ) + response: Final = gateway.request( + "GET", + _SEARCH_PATH, + params={"search": shared_alias, "exclude_team_ids": team_drop, **_range_around_today()}, + ) + assert response.status_code == 200, response.text + body: Final = object_value(response.json()) + results: Final = body["results"] + assert isinstance(results, list) and len(results) == 1, body + entities: Final = object_value(object_value(object_value(results[0])["breakdown"])["entities"]) + assert set(entities) == {team_keep}, response.text + assert set(_team_key_breakdown(body, team_keep)) == {keep_digest}, response.text diff --git a/tests/proxy_behavior/management/test_team_daily_activity.py b/tests/proxy_behavior/management/test_team_daily_activity.py index d84cc4c94af..9bbc8fdde29 100644 --- a/tests/proxy_behavior/management/test_team_daily_activity.py +++ b/tests/proxy_behavior/management/test_team_daily_activity.py @@ -5,8 +5,9 @@ from .actors import Actor pytestmark = pytest.mark.asyncio(loop_scope="session") -# GET /team/daily/activity and its /aggregated variant (same shared scope -# resolver, so the matrix must hold for both). A proxy admin (admin view) sees +# GET /team/daily/activity, its /aggregated variant, and the key-search +# variant (same shared scope resolver, so the matrix must hold for all +# three). A proxy admin (admin view) sees # activity for any team. A non-admin is scoped to user_info.teams: a bare query # defaults to its own teams (200), and an explicit team_ids filter naming a # team it does not belong to is 404 (the VERIA-43 fix). Org admins have no @@ -43,8 +44,12 @@ _DATES = "start_date=2024-01-01&end_date=2024-12-31" @pytest.mark.parametrize( "endpoint", - ("/team/daily/activity", "/team/daily/activity/aggregated"), - ids=("paginated", "aggregated"), + ( + "/team/daily/activity", + "/team/daily/activity/aggregated", + "/team/daily/activity/aggregated/search", + ), + ids=("paginated", "aggregated", "search"), ) @pytest.mark.parametrize( "actor,team,expected_status", @@ -54,7 +59,7 @@ _DATES = "start_date=2024-01-01&end_date=2024-12-31" async def test_team_daily_activity_matrix( actor: Actor, team: str, expected_status: int, endpoint: str, proxy_client, world ): - query = _DATES + query = _DATES + ("&search=x" if endpoint.endswith("/search") else "") if team == "alpha": query += f"&team_ids={world.team_alpha_id}" elif team == "beta": diff --git a/tests/test_litellm/proxy/auth/test_route_checks.py b/tests/test_litellm/proxy/auth/test_route_checks.py index 571e066e947..7bb79a115dd 100644 --- a/tests/test_litellm/proxy/auth/test_route_checks.py +++ b/tests/test_litellm/proxy/auth/test_route_checks.py @@ -3600,6 +3600,55 @@ def test_user_daily_activity_aggregated_not_covered_by_prefix_match(): ) +@pytest.mark.parametrize( + "route", + [ + "/team/daily/activity", + "/team/daily/activity/aggregated", + "/team/daily/activity/aggregated/search", + ], +) +@pytest.mark.parametrize( + "user_role", + [ + LitellmUserRoles.INTERNAL_USER.value, + LitellmUserRoles.INTERNAL_USER_VIEW_ONLY.value, + ], +) +def test_team_daily_activity_routes_reachable_by_non_admin(route, user_role): + """The Team Usage dashboard calls all three team daily-activity routes, and + each handler self-scopes to the caller's teams and own keys + (_resolve_team_daily_activity_scope). self_managed_routes is the only list + granting them to a non-admin, and check_route_access is exact-match, so each + sub-path needs its own entry: dropping one 401s the dashboard before the + handler ever runs. + """ + user_obj = LiteLLM_UserTable( + user_id="test_user", + user_email="test@example.com", + user_role=user_role, + ) + valid_token = UserAPIKeyAuth(user_id="test_user", user_role=user_role) + request = MagicMock(spec=Request) + request.query_params = {} + + def outcome() -> str: + try: + RouteChecks.non_proxy_admin_allowed_routes_check( + user_obj=user_obj, + _user_role=user_role, + route=route, + request=request, + valid_token=valid_token, + request_data={}, + ) + except Exception as exc: + return f"denied: {exc}" + return "allowed" + + assert outcome() == "allowed" + + @pytest.mark.parametrize( "user_role", [ diff --git a/tests/test_litellm/proxy/management_endpoints/test_team_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_team_endpoints.py index b066b3b80e6..38241926f8e 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_team_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_team_endpoints.py @@ -14646,6 +14646,218 @@ async def test_get_team_daily_activity_aggregated_rejects_bad_ranges( mock_aggregated.assert_not_called() +def _key_search_team_setup(mock_db_client, user_id: str, team_id: str): + mock_user_info = LiteLLM_UserTable( + user_id=user_id, + teams=[team_id], + max_budget=1000.0, + spend=0.0, + user_email="test@example.com", + user_role="internal_user", + ) + mock_team = MagicMock(spec=LiteLLM_TeamTable) + mock_team.team_id = team_id + mock_team.team_alias = "Test Team" + mock_team.members_with_roles = [Member(user_id=user_id, role="user")] + mock_team.model_dump.return_value = { + "team_id": team_id, + "team_alias": "Test Team", + "members_with_roles": [{"user_id": user_id, "role": "user"}], + } + mock_db_client.db.litellm_teamtable.find_many = AsyncMock(return_value=[mock_team]) + return mock_user_info + + +@pytest.mark.asyncio +async def test_search_team_daily_activity_keys_scopes_where_before_take(mock_db_client): + """A member's search must put the team and own-key scoping inside the same + Prisma where as the term, because `take` trims rows before Python sees them: + scoped outside the where, the top-N slice could be spent entirely on keys + the caller is not allowed to see.""" + from litellm.constants import USAGE_TOP_API_KEYS_LIMIT + from litellm.proxy.management_endpoints.team_endpoints import ( + search_team_daily_activity_keys, + ) + + user_id = "test_user_123" + team_id = "test_team_456" + user_api_key_dict = UserAPIKeyAuth(user_id=user_id, user_role=LitellmUserRoles.INTERNAL_USER) + mock_user_info = _key_search_team_setup(mock_db_client, user_id, team_id) + + user_key_1 = MagicMock() + user_key_1.token = "user_key_1" + matched = MagicMock() + matched.token = "user_key_1" + mock_db_client.db.litellm_verificationtoken.find_many = AsyncMock(side_effect=[[user_key_1], [matched]]) + + with patch( + "litellm.proxy.management_endpoints.team_endpoints.get_user_object", + new_callable=AsyncMock, + ) as mock_get_user_object: + mock_get_user_object.return_value = mock_user_info + + with patch( + "litellm.proxy.management_endpoints.team_endpoints.get_daily_activity_aggregated", + new_callable=AsyncMock, + ) as mock_aggregated: + mock_aggregated.return_value = MagicMock() + + await search_team_daily_activity_keys( + user_api_key_dict=user_api_key_dict, + search="Needle", + team_ids=team_id, + start_date="2024-01-01", + end_date="2024-01-31", + exclude_team_ids=None, + timezone=480, + ) + + token_calls = mock_db_client.db.litellm_verificationtoken.find_many.call_args_list + assert len(token_calls) == 2 + search_kwargs = token_calls[1][1] + assert search_kwargs["where"] == { + "team_id": {"in": (team_id,)}, + "token": {"in": ("user_key_1",)}, + "OR": ( + {"token": "Needle"}, + {"key_alias": {"contains": "Needle", "mode": "insensitive"}}, + {"user_id": {"contains": "Needle", "mode": "insensitive"}}, + ), + } + assert search_kwargs["take"] == USAGE_TOP_API_KEYS_LIMIT + assert search_kwargs["order"] == {"spend": "desc"} + + call_kwargs = mock_aggregated.call_args[1] + assert call_kwargs["api_key"] == ["user_key_1"] + assert call_kwargs["entity_id"] == [team_id] + assert call_kwargs["table_name"] == "litellm_dailyteamspend" + assert call_kwargs["include_entity_breakdown"] is True + assert call_kwargs["timezone_offset_minutes"] == 480 + assert call_kwargs["model"] is None + assert call_kwargs["entity_metadata_field"] == {team_id: {"team_alias": "Test Team"}} + + +@pytest.mark.asyncio +async def test_search_team_daily_activity_keys_admin_unscoped_where(mock_db_client): + """An admin's search has no caller scoping, so the where is the bare OR over + token, key alias and user id; every matched hash is passed through to the + aggregation.""" + from litellm.constants import USAGE_TOP_API_KEYS_LIMIT + from litellm.proxy.management_endpoints.team_endpoints import ( + search_team_daily_activity_keys, + ) + + match_1 = MagicMock() + match_1.token = "h1" + match_2 = MagicMock() + match_2.token = "h2" + mock_db_client.db.litellm_teamtable.find_many = AsyncMock(return_value=[]) + mock_db_client.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[match_1, match_2]) + + with patch( + "litellm.proxy.management_endpoints.team_endpoints.get_daily_activity_aggregated", + new_callable=AsyncMock, + ) as mock_aggregated: + mock_aggregated.return_value = MagicMock() + + await search_team_daily_activity_keys( + user_api_key_dict=UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN), + search="Needle", + team_ids=None, + start_date="2024-01-01", + end_date="2024-01-31", + exclude_team_ids=None, + timezone=None, + ) + + search_kwargs = mock_db_client.db.litellm_verificationtoken.find_many.call_args[1] + assert search_kwargs["where"] == { + "OR": ( + {"token": "Needle"}, + {"key_alias": {"contains": "Needle", "mode": "insensitive"}}, + {"user_id": {"contains": "Needle", "mode": "insensitive"}}, + ) + } + assert search_kwargs["take"] == USAGE_TOP_API_KEYS_LIMIT + assert mock_aggregated.call_args[1]["api_key"] == ["h1", "h2"] + + +@pytest.mark.asyncio +async def test_search_team_daily_activity_keys_no_match_returns_empty_without_aggregating( + mock_db_client, +): + """A term matching no key still owes the caller the standard metadata shape + (api_key_limit, total_api_keys), and the aggregated query must not run.""" + from litellm.constants import USAGE_TOP_API_KEYS_LIMIT + from litellm.proxy.management_endpoints.team_endpoints import ( + search_team_daily_activity_keys, + ) + + mock_db_client.db.litellm_teamtable.find_many = AsyncMock(return_value=[]) + mock_db_client.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[]) + + with patch( + "litellm.proxy.management_endpoints.team_endpoints.get_daily_activity_aggregated", + new_callable=AsyncMock, + ) as mock_aggregated: + result = await search_team_daily_activity_keys( + user_api_key_dict=UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN), + search="Needle", + team_ids=None, + start_date="2024-01-01", + end_date="2024-01-31", + exclude_team_ids=None, + timezone=None, + ) + + assert result.results == [] + assert result.metadata.total_api_keys == 0 + assert result.metadata.api_key_limit == USAGE_TOP_API_KEYS_LIMIT + mock_aggregated.assert_not_called() + + +@pytest.mark.asyncio +async def test_search_team_daily_activity_keys_excludes_teams_in_where(mock_db_client): + """The dashboard always sends exclude_team_ids=litellm-dashboard; if that + filter stayed out of the where, matching keys in excluded teams could fill + the take=N slice and push visible matches out.""" + from litellm.proxy.management_endpoints.team_endpoints import ( + search_team_daily_activity_keys, + ) + + matched = MagicMock() + matched.token = "h1" + mock_db_client.db.litellm_teamtable.find_many = AsyncMock(return_value=[]) + mock_db_client.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[matched]) + + with patch( + "litellm.proxy.management_endpoints.team_endpoints.get_daily_activity_aggregated", + new_callable=AsyncMock, + ) as mock_aggregated: + mock_aggregated.return_value = MagicMock() + + await search_team_daily_activity_keys( + user_api_key_dict=UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN), + search="Needle", + team_ids=None, + start_date="2024-01-01", + end_date="2024-01-31", + exclude_team_ids="litellm-dashboard", + timezone=None, + ) + + search_kwargs = mock_db_client.db.litellm_verificationtoken.find_many.call_args[1] + assert search_kwargs["where"] == { + "team_id": {"notIn": ("litellm-dashboard",)}, + "OR": ( + {"token": "Needle"}, + {"key_alias": {"contains": "Needle", "mode": "insensitive"}}, + {"user_id": {"contains": "Needle", "mode": "insensitive"}}, + ), + } + assert mock_aggregated.call_args[1]["exclude_entity_ids"] == ["litellm-dashboard"] + + def _wire_new_team_prisma(mock_db_client): mock_db_client.jsonify_team_object = lambda db_data: db_data mock_db_client.get_data = AsyncMock(return_value=None) diff --git a/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/EntityUsage/EntityUsage.tsx b/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/EntityUsage/EntityUsage.tsx index 27460b21108..16dc41c3ba8 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/EntityUsage/EntityUsage.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/usage/_components/components/EntityUsage/EntityUsage.tsx @@ -20,7 +20,7 @@ import type { ColumnDef } from "@tanstack/react-table"; import PaginationStatusAlerts from "@/components/shared/PaginationStatusAlerts"; import { Tabs, TabsContent, TabsList, TabsTrigger } from "@/components/ui/tabs"; import { Tooltip, TooltipContent, TooltipTrigger } from "@/components/ui/tooltip"; -import React, { type ReactNode, useMemo, useState } from "react"; +import React, { type ReactNode, useCallback, useMemo, useState } from "react"; import TeamMultiSelect from "@/components/common_components/team_multi_select"; import UserDropdown from "@/components/common_components/UserDropdown"; import { ActivityMetrics, processActivityData } from "@/components/activity_metrics"; @@ -34,6 +34,7 @@ import { tagDailyActivityCall, teamDailyActivityAggregatedCall, teamDailyActivityCall, + teamDailyActivityKeySearchCall, userDailyActivityCall, } from "@/components/networking"; import { Logo } from "@/components/molecules/logo/Logo"; @@ -182,6 +183,16 @@ const EntityUsage: React.FC = ({ const modelBreakdownKey = modelViewType === "groups" ? "model_groups" : "models"; const modelMetrics = processActivityData(spendData, modelBreakdownKey, teams || []); const keyMetrics = processActivityData(spendData, "api_keys", teams || []); + const searchTeamKeys = useCallback( + (query: string) => { + if (!accessToken || !startTime || !endTime) return Promise.resolve({}); + const teamIds = Array.isArray(entityFilterArg) ? entityFilterArg : null; + return teamDailyActivityKeySearchCall(accessToken, startTime, endTime, query, teamIds).then((data) => + processActivityData(data, "api_keys", teams || []), + ); + }, + [accessToken, startTime, endTime, entityFilterArg, teams], + ); const agentMetrics = showAgentBreakdown ? processActivityData(agentSpendData, "entities", teams || []) : {}; const getAllTags = () => { @@ -667,6 +678,7 @@ const EntityUsage: React.FC = ({ keyMetrics={keyMetrics} hidePromptCachingMetrics={entityType === "agent"} apiKeyTruncation={apiKeyTruncation} + searchKeys={entityType === "team" ? searchTeamKeys : undefined} /> ), }, diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index e6923a26d4c..edaf56a8a16 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -1467,6 +1467,31 @@ export const teamDailyActivityAggregatedCall = async ( } }; +export const teamDailyActivityKeySearchCall = async ( + accessToken: string, + startTime: Date, + endTime: Date, + ...options: [search: string, teamIds?: string[] | null] +) => { + const [search, teamIds = null] = options; + try { + return await apiClient.get(`/team/daily/activity/aggregated/search`, { + accessToken, + query: { + start_date: formatDate(startTime), + end_date: formatDate(endTime), + timezone: new Date().getTimezoneOffset().toString(), + search, + team_ids: teamIds && teamIds.length > 0 ? teamIds.join(",") : undefined, + exclude_team_ids: "litellm-dashboard", + }, + }); + } catch (error) { + console.error("Failed to search team daily activity keys:", error); + throw error; + } +}; + export type TeamUserSpendResponse = components["schemas"]["TeamUserSpendResponse"]; export const teamSpendByUserCall = async ( diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 79d008c1b48..a398b2db93c 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -15645,6 +15645,27 @@ export interface paths { patch?: never; trace?: never; }; + "/team/daily/activity/aggregated/search": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Search Team Daily Activity Keys + * @description Aggregated daily team activity for the keys matching `search`, across every key the caller may + * see rather than only the top USAGE_TOP_API_KEYS_LIMIT keys by spend. + */ + get: operations["search_team_daily_activity_keys_team_daily_activity_aggregated_search_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/team/delete": { parameters: { query?: never; @@ -31378,7 +31399,7 @@ export interface components { * @description Enum for key management routes * @enum {string} */ - KeyManagementRoutes: "/key/generate" | "/key/update" | "/key/delete" | "/key/regenerate" | "/key/service-account/generate" | "/key/{key_id}/regenerate" | "/key/block" | "/key/unblock" | "/key/bulk_update" | "/team/key/bulk_update" | "/key/{key_id}/reset_spend" | "/key/access_group_assignment" | "/auto_router/manage" | "/key/info" | "/key/health" | "/key/list" | "/key/aliases" | "/team/daily/activity" | "/team/daily/activity/aggregated" | "/spend/logs" | "/spend/logs/v2"; + KeyManagementRoutes: "/key/generate" | "/key/update" | "/key/delete" | "/key/regenerate" | "/key/service-account/generate" | "/key/{key_id}/regenerate" | "/key/block" | "/key/unblock" | "/key/bulk_update" | "/team/key/bulk_update" | "/key/{key_id}/reset_spend" | "/key/access_group_assignment" | "/auto_router/manage" | "/key/info" | "/key/health" | "/key/list" | "/key/aliases" | "/team/daily/activity" | "/team/daily/activity/aggregated" | "/team/daily/activity/aggregated/search" | "/spend/logs" | "/spend/logs/v2"; /** * KeyManagementSystem * @enum {string} @@ -66624,6 +66645,43 @@ export interface operations { }; }; }; + search_team_daily_activity_keys_team_daily_activity_aggregated_search_get: { + parameters: { + query: { + /** @description Exact token hash, or a case-insensitive substring of the key alias or owning user id */ + search: string; + team_ids?: string | null; + start_date?: string | null; + end_date?: string | null; + exclude_team_ids?: string | null; + timezone?: number | null; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["SpendAnalyticsPaginatedResponse"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; delete_team_team_delete_post: { parameters: { query?: never; From 1fbd1e9ce90580f801612d2016e9e2cc4ab553b5 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:08:27 -0700 Subject: [PATCH 96/96] fix(bedrock): route unmapped openai family model ids to converse (#42713) * fix(bedrock): route unmapped openai family model ids to converse Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(bedrock): rename the e2e openai family backend constant global.openai.gpt-6-sol has a cost-map row now, so the constant no longer names an unmapped model --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- litellm/llms/bedrock/common_utils.py | 2 ++ .../test_bedrock_provider_matrix_e2e.py | 19 ++++++++++++++++++- .../llms/bedrock/test_bedrock_common_utils.py | 19 ++++++++++++++++++- 3 files changed, 38 insertions(+), 2 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 9b52f531cbb..5f044897b2c 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -1218,6 +1218,8 @@ class BedrockModelInfo(BaseLLMModelInfo): alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model) if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models: return "converse" + if _OPENAI_FAMILY_MODEL_RE.search(base_model): + return "converse" return "invoke" @staticmethod diff --git a/tests/e2e/llm_translation/test_bedrock_provider_matrix_e2e.py b/tests/e2e/llm_translation/test_bedrock_provider_matrix_e2e.py index 5f0a931109c..21333d39849 100644 --- a/tests/e2e/llm_translation/test_bedrock_provider_matrix_e2e.py +++ b/tests/e2e/llm_translation/test_bedrock_provider_matrix_e2e.py @@ -8,7 +8,8 @@ caller can hand AWS support the request id behind a completion. Regional inference-profile ids are the deployment shape most Bedrock customers run; a v1.90.0 regression timed them out, and the Converse route keeps them covered in test_chat_completions_regression_e2e.py, so the invoke route carries its own -rows here. +rows here. The file also covers Bedrock-native OpenAI model ids taking the +default (Converse) route with max_tokens. """ from __future__ import annotations @@ -26,6 +27,7 @@ pytestmark = pytest.mark.e2e CONVERSE_REGIONAL_BACKEND = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0" INVOKE_REGIONAL_BACKEND = "bedrock/invoke/us.anthropic.claude-haiku-4-5-20251001-v1:0" +OPENAI_FAMILY_BACKEND = "bedrock/global.openai.gpt-6-sol" PROVIDER_HEADER_PREFIX = "llm_provider-" BEDROCK_REQUEST_ID_HEADER = "llm_provider-x-amzn-requestid" @@ -199,3 +201,18 @@ class TestBedrockInvokeRegionalModelIds: ) _assert_streamed_completion(result) + + +class TestBedrockOpenAIFamilyDefaultRoute: + @pytest.mark.covers("llm.chat_completions.bedrock_converse.basic.nonstream.works", exercised_on=[]) + def test_openai_family_model_id_completes_with_max_tokens( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + model = _register_bedrock_model( + client, resources, "e2e-bedrock-openai-family", OPENAI_FAMILY_BACKEND + ) + key = resources.key() + + response = unwrap(client.proxy.chat(key, ChatBody(model=model, messages=_prompt(), max_tokens=64))) + + _assert_completion(response) diff --git a/tests/test_litellm/llms/bedrock/test_bedrock_common_utils.py b/tests/test_litellm/llms/bedrock/test_bedrock_common_utils.py index 117814a41ff..e5118f90e44 100644 --- a/tests/test_litellm/llms/bedrock/test_bedrock_common_utils.py +++ b/tests/test_litellm/llms/bedrock/test_bedrock_common_utils.py @@ -346,7 +346,7 @@ def test_route_prefix_matched_as_path_segment_not_substring(): BedrockModelInfo.get_bedrock_route("bedrock_mantle/openai.gpt-5.5") != "mantle" ) assert ( - BedrockModelInfo.get_bedrock_route("bedrock_mantle/openai.gpt-5.4") == "invoke" + BedrockModelInfo.get_bedrock_route("bedrock_mantle/openai.gpt-5.4") == "converse" ) assert ( BedrockModelInfo._explicit_mantle_route("bedrock_mantle/openai.gpt-5.5") @@ -964,3 +964,20 @@ def test_s3_static_key_pair_is_none_without_a_full_pair(partial_s3_pair): from litellm.llms.bedrock.common_utils import s3_static_key_pair assert s3_static_key_pair({"aws_access_key_id": "bedrock-key", **partial_s3_pair}) is None + + +def test_unmapped_openai_family_model_routes_to_converse(): + """A Bedrock-native OpenAI model that is not in the cost map yet must not fall to the invoke route. + + The invoke ``openai`` provider is the imported-model path and sends ``max_tokens``, which Bedrock + rejects for these models; Converse maps it to ``inferenceConfig.maxTokens``. + """ + from typing import Final + + import litellm + + unmapped: Final = "bedrock/global.openai.gpt-99-unmapped" + assert unmapped.removeprefix("bedrock/") not in litellm.bedrock_converse_models + assert BedrockModelInfo.get_bedrock_route(unmapped) == "converse" + imported: Final = "bedrock/openai/arn:aws:bedrock:us-east-1:123456789012:imported-model/abc123" + assert BedrockModelInfo.get_bedrock_route(imported) == "openai"