From 043c12b11e7b9572db22ffe5de4f40a21f3961e0 Mon Sep 17 00:00:00 2001 From: "Jugal D. Bhatt" <55304795+jugaldb@users.noreply.github.com> Date: Wed, 6 Aug 2025 19:03:58 -0700 Subject: [PATCH 01/21] added token breakdown in ui (#13357) --- .../src/components/view_logs/SessionView.tsx | 75 ++++++++++++++++++- .../src/components/view_logs/index.tsx | 10 ++- 2 files changed, 80 insertions(+), 5 deletions(-) diff --git a/ui/litellm-dashboard/src/components/view_logs/SessionView.tsx b/ui/litellm-dashboard/src/components/view_logs/SessionView.tsx index 1bd8a51192f..2d05680040b 100644 --- a/ui/litellm-dashboard/src/components/view_logs/SessionView.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/SessionView.tsx @@ -9,6 +9,7 @@ import { ArrowLeftIcon } from "@heroicons/react/outline" import { Button } from "antd" import { copyToClipboard as utilCopyToClipboard } from "../../utils/dataUtils" import { CheckIcon, CopyIcon } from "lucide-react" +import { Tooltip } from "antd" interface SessionViewProps { sessionId: string @@ -24,6 +25,21 @@ export const SessionView: React.FC = ({ sessionId, logs, onBac // Calculate session metrics const totalCost = logs.reduce((sum, log) => sum + (log.spend || 0), 0) const totalTokens = logs.reduce((sum, log) => sum + (log.total_tokens || 0), 0) + + // Calculate cache token totals from metadata + const totalCacheReadTokens = logs.reduce((sum, log) => { + const cacheReadTokens = log.metadata?.additional_usage_values?.cache_read_input_tokens || 0 + return sum + cacheReadTokens + }, 0) + + const totalCacheCreationTokens = logs.reduce((sum, log) => { + const cacheCreationTokens = log.metadata?.additional_usage_values?.cache_creation_input_tokens || 0 + return sum + cacheCreationTokens + }, 0) + + // Calculate total tokens including cache tokens + const totalTokensWithCache = totalTokens + totalCacheReadTokens + totalCacheCreationTokens + const startTime = logs.length > 0 ? new Date(logs[0].startTime) : new Date() const endTime = logs.length > 0 ? new Date(logs[logs.length - 1].endTime) : new Date() const durationMs = endTime.getTime() - startTime.getTime() @@ -100,10 +116,61 @@ export const SessionView: React.FC = ({ sessionId, logs, onBac Total Cost ${formatNumberWithCommas(totalCost, 6)} - - Total Tokens - {totalTokens} - + +
Usage breakdown
+
+
+
Input usage:
+
+
+ input: + {formatNumberWithCommas(logs.reduce((sum, log) => sum + (log.prompt_tokens || 0), 0))} +
+ {totalCacheReadTokens > 0 && ( +
+ input_cached_tokens: + {formatNumberWithCommas(totalCacheReadTokens)} +
+ )} + {totalCacheCreationTokens > 0 && ( +
+ input_cache_creation_tokens: + {formatNumberWithCommas(totalCacheCreationTokens)} +
+ )} +
+
+
+
Output usage:
+
+
+ output: + {formatNumberWithCommas(logs.reduce((sum, log) => sum + (log.completion_tokens || 0), 0))} +
+
+
+
+
+ Total usage: + {formatNumberWithCommas(totalTokensWithCache)} +
+
+
+ + } + placement="top" + overlayStyle={{ minWidth: '300px' }} + > + +
+ Total Tokens + ⓘ +
+ {formatNumberWithCommas(totalTokensWithCache)} +
+
{/* Request Timeline */} Session Logs diff --git a/ui/litellm-dashboard/src/components/view_logs/index.tsx b/ui/litellm-dashboard/src/components/view_logs/index.tsx index ea97d54a6f1..d2041b389fd 100644 --- a/ui/litellm-dashboard/src/components/view_logs/index.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/index.tsx @@ -925,7 +925,15 @@ export function RequestViewer({ row }: { row: Row }) {
Tokens: - {row.original.total_tokens} ({row.original.prompt_tokens}+{row.original.completion_tokens}) + {row.original.total_tokens} ({row.original.prompt_tokens} prompt tokens + {row.original.completion_tokens} completion tokens) +
+
+ Cache Read Tokens: + {formatNumberWithCommas(row.original.metadata?.additional_usage_values?.cache_read_input_tokens || 0)} +
+
+ Cache Creation Tokens: + {formatNumberWithCommas(row.original.metadata?.additional_usage_values.cache_creation_input_tokens)}
Cost: From 9c5e9d7362524a6698f6430dcfa86884ec1ad7bb Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 00:08:18 -0700 Subject: [PATCH 02/21] add groq/openai/gpt-oss (#13363) --- ...odel_prices_and_context_window_backup.json | 30 +++++++++++++++++++ model_prices_and_context_window.json | 30 +++++++++++++++++++ 2 files changed, 60 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 9120f1f079f..8281cb424fd 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -5486,6 +5486,36 @@ "litellm_provider": "groq", "mode": "audio_transcription" }, + "groq/openai/gpt-oss-20b": { + "max_tokens": 32768, + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 5e-07, + "litellm_provider": "groq", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_web_search": true + }, + "groq/openai/gpt-oss-120b": { + "max_tokens": 32766, + "max_input_tokens": 131072, + "max_output_tokens": 32766, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 7.5e-07, + "litellm_provider": "groq", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_web_search": true + }, "cerebras/llama3.1-8b": { "max_tokens": 128000, "max_input_tokens": 128000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 9120f1f079f..8281cb424fd 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -5486,6 +5486,36 @@ "litellm_provider": "groq", "mode": "audio_transcription" }, + "groq/openai/gpt-oss-20b": { + "max_tokens": 32768, + "max_input_tokens": 131072, + "max_output_tokens": 32768, + "input_cost_per_token": 1e-07, + "output_cost_per_token": 5e-07, + "litellm_provider": "groq", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_web_search": true + }, + "groq/openai/gpt-oss-120b": { + "max_tokens": 32766, + "max_input_tokens": 131072, + "max_output_tokens": 32766, + "input_cost_per_token": 1.5e-07, + "output_cost_per_token": 7.5e-07, + "litellm_provider": "groq", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_reasoning": true, + "supports_tool_choice": true, + "supports_web_search": true + }, "cerebras/llama3.1-8b": { "max_tokens": 128000, "max_input_tokens": 128000, From dfada882f1f84cf15ea21435410d2d086049171e Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 00:11:10 -0700 Subject: [PATCH 03/21] vtx test fix gemini-2.5-flash-lite --- tests/local_testing/test_amazing_vertex_completion.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/local_testing/test_amazing_vertex_completion.py b/tests/local_testing/test_amazing_vertex_completion.py index d5a7271ae20..d1c0fb4e01c 100644 --- a/tests/local_testing/test_amazing_vertex_completion.py +++ b/tests/local_testing/test_amazing_vertex_completion.py @@ -518,7 +518,7 @@ async def test_gemini_pro_vision(provider, sync_mode): litellm.num_retries = 3 if sync_mode: resp = litellm.completion( - model="{}/gemini-2.5-flash-lite-preview-0514".format(provider), + model="{}/gemini-2.5-flash-lite".format(provider), messages=[ {"role": "system", "content": "Be a good bot"}, { @@ -537,7 +537,7 @@ async def test_gemini_pro_vision(provider, sync_mode): ) else: resp = await litellm.acompletion( - model="{}/gemini-2.5-flash-lite-preview-0514".format(provider), + model="{}/gemini-2.5-flash-lite".format(provider), messages=[ {"role": "system", "content": "Be a good bot"}, { @@ -605,7 +605,7 @@ def test_completion_function_plus_pdf(load_pdf): image_message = {"role": "user", "content": image_content} response = completion( - model="vertex_ai_beta/gemini-2.5-flash-lite-preview-0514", + model="vertex_ai_beta/gemini-2.5-flash-lite", messages=[image_message], stream=False, ) From 7d978f0ffc1484b073bc4e53e651f5879e8ca3dc Mon Sep 17 00:00:00 2001 From: tanjiro <56165694+NANDINI-star@users.noreply.github.com> Date: Thu, 7 Aug 2025 23:41:30 +0900 Subject: [PATCH 04/21] provider logos on usage page (#13372) --- .../src/components/entity_usage.tsx | 24 ++++++- .../src/components/new_usage.tsx | 64 ++++++++++--------- 2 files changed, 58 insertions(+), 30 deletions(-) diff --git a/ui/litellm-dashboard/src/components/entity_usage.tsx b/ui/litellm-dashboard/src/components/entity_usage.tsx index a481de0868d..7b910dab57b 100644 --- a/ui/litellm-dashboard/src/components/entity_usage.tsx +++ b/ui/litellm-dashboard/src/components/entity_usage.tsx @@ -29,6 +29,7 @@ import { tagDailyActivityCall, teamDailyActivityCall } from './networking'; import TopKeyView from "./top_key_view"; import { formatNumberWithCommas } from "@/utils/dataUtils"; import { valueFormatterSpend } from "./usage/utils/value_formatters"; +import { getProviderLogoAndName } from "./provider_info_helpers"; interface EntityMetrics { metrics: { @@ -688,7 +689,28 @@ const EntityUsage: React.FC = ({ {getProviderSpend().map((provider) => ( - {provider.provider} + +
+ {provider.provider && ( + {`${provider.provider} { + const target = e.target as HTMLImageElement; + const parent = target.parentElement; + if (parent) { + const fallbackDiv = document.createElement('div'); + fallbackDiv.className = 'w-4 h-4 rounded-full bg-gray-200 flex items-center justify-center text-xs'; + fallbackDiv.textContent = provider.provider?.charAt(0) || '-'; + parent.replaceChild(fallbackDiv, target); + } + }} + /> + )} + {provider.provider} +
+
${formatNumberWithCommas(provider.spend, 2)} diff --git a/ui/litellm-dashboard/src/components/new_usage.tsx b/ui/litellm-dashboard/src/components/new_usage.tsx index 3dda412e203..ec9d462fb92 100644 --- a/ui/litellm-dashboard/src/components/new_usage.tsx +++ b/ui/litellm-dashboard/src/components/new_usage.tsx @@ -34,20 +34,14 @@ import { import AdvancedDatePicker from "./shared/advanced_date_picker" import { AreaChart } from "@tremor/react" -import { userDailyActivityCall, tagListCall } from "./networking"; -import { Tag } from "./tag_management/types"; -import ViewUserSpend from "./view_user_spend"; -import TopKeyView from "./top_key_view"; -import { ActivityMetrics, processActivityData } from "./activity_metrics"; -import UserAgentActivity from "./user_agent_activity"; -import { - SpendMetrics, - DailyData, - ModelActivityData, - MetricWithMetadata, - KeyMetricWithMetadata, -} from "./usage/types"; -import EntityUsage from "./entity_usage"; +import { userDailyActivityCall, tagListCall } from "./networking" +import { Tag } from "./tag_management/types" +import ViewUserSpend from "./view_user_spend" +import TopKeyView from "./top_key_view" +import { ActivityMetrics, processActivityData } from "./activity_metrics" +import UserAgentActivity from "./user_agent_activity" +import { SpendMetrics, DailyData, ModelActivityData, MetricWithMetadata, KeyMetricWithMetadata } from "./usage/types" +import EntityUsage from "./entity_usage" import { old_admin_roles, v2_admin_role_names, @@ -62,6 +56,7 @@ import { formatNumberWithCommas } from "@/utils/dataUtils" import { valueFormatterSpend } from "./usage/utils/value_formatters" import CloudZeroExportModal from "./cloudzero_export_modal" import { ChartLoader } from "./shared/chart_loader" +import { getProviderLogoAndName } from "./provider_info_helpers" interface NewUsagePageProps { accessToken: string | null @@ -416,16 +411,8 @@ const NewUsagePage: React.FC = ({ accessToken, userRole, user {all_admin_roles.includes(userRole || "") ? Global Usage : Your Usage} Team Usage - {all_admin_roles.includes(userRole || "") ? ( - Tag Usage - ) : ( - <> - )} - {all_admin_roles.includes(userRole || "") ? ( - User Agent Activity - ) : ( - <> - )} + {all_admin_roles.includes(userRole || "") ? Tag Usage : <>} + {all_admin_roles.includes(userRole || "") ? User Agent Activity : <>} {/* Your Usage Panel */} @@ -658,7 +645,29 @@ const NewUsagePage: React.FC = ({ accessToken, userRole, user .filter((provider) => provider.spend > 0) .map((provider) => ( - {provider.provider} + +
+ {provider.provider && ( + {`${provider.provider} { + const target = e.target as HTMLImageElement + const parent = target.parentElement + if (parent) { + const fallbackDiv = document.createElement("div") + fallbackDiv.className = + "w-4 h-4 rounded-full bg-gray-200 flex items-center justify-center text-xs" + fallbackDiv.textContent = provider.provider?.charAt(0) || "-" + parent.replaceChild(fallbackDiv, target) + } + }} + /> + )} + {provider.provider} +
+
${formatNumberWithCommas(provider.spend, 2)} {provider.successful_requests.toLocaleString()} @@ -725,10 +734,7 @@ const NewUsagePage: React.FC = ({ accessToken, userRole, user {/* User Agent Activity Panel */} - +
From 30fc5b871ce578ba1d1dcf705427e197b4f9b040 Mon Sep 17 00:00:00 2001 From: Edward D'Amato <19622548+edwarddamato@users.noreply.github.com> Date: Thu, 7 Aug 2025 17:40:11 +0200 Subject: [PATCH 05/21] feat(integrations): allow setting of braintrust callback base url (#13368) * feat(integrations): allow setting of braintrust callback base url * chore(misc): remove extra additions due to merge --- .../docs/observability/braintrust.md | 4 ++ docs/my-website/docs/proxy/config_settings.md | 1 + litellm/integrations/braintrust_logging.py | 2 +- litellm/proxy/_types.py | 2 +- .../test_unit_tests_init_callbacks.py | 1 + .../integrations/test_braintrust_logging.py | 43 +++++++++++++++++++ 6 files changed, 51 insertions(+), 2 deletions(-) create mode 100644 tests/test_litellm/integrations/test_braintrust_logging.py diff --git a/docs/my-website/docs/observability/braintrust.md b/docs/my-website/docs/observability/braintrust.md index 79f3cf13be2..eb26680b18a 100644 --- a/docs/my-website/docs/observability/braintrust.md +++ b/docs/my-website/docs/observability/braintrust.md @@ -15,6 +15,7 @@ import os # set env os.environ["BRAINTRUST_API_KEY"] = "" +os.environ["BRAINTRUST_API_BASE"] = "https://api.braintrustdata.com/v1" os.environ['OPENAI_API_KEY']="" # set braintrust as a callback, litellm will send the data to braintrust @@ -35,6 +36,7 @@ response = litellm.completion( ```env BRAINTRUST_API_KEY="" +BRAINTRUST_API_BASE="https://api.braintrustdata.com/v1" ``` 2. Add braintrust to callbacks @@ -157,6 +159,8 @@ For more examples, [**Click Here**](../proxy/user_keys.md#chatcompletions) +You can use `BRAINTRUST_API_BASE` to point to your self-hosted Braintrust data plane. Read more about this [here](https://www.braintrust.dev/docs/guides/self-hosting). + ## Full API Spec Here's everything you can pass in metadata for a braintrust request diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md index c8aee990415..3b903935a04 100644 --- a/docs/my-website/docs/proxy/config_settings.md +++ b/docs/my-website/docs/proxy/config_settings.md @@ -369,6 +369,7 @@ router_settings: | BEDROCK_MAX_POLICY_SIZE | Maximum size for Bedrock policy. Default is 75 | BERRISPEND_ACCOUNT_ID | Account ID for BerriSpend service | BRAINTRUST_API_KEY | API key for Braintrust integration +| BRAINTRUST_API_BASE | Base URL for Braintrust API. Default is https://api.braintrustdata.com/v1 | CACHED_STREAMING_CHUNK_DELAY | Delay in seconds for cached streaming chunks. Default is 0.02 | CIRCLE_OIDC_TOKEN | OpenID Connect token for CircleCI | CIRCLE_OIDC_TOKEN_V2 | Version 2 of the OpenID Connect token for CircleCI diff --git a/litellm/integrations/braintrust_logging.py b/litellm/integrations/braintrust_logging.py index c68674f77ba..8149a6131e8 100644 --- a/litellm/integrations/braintrust_logging.py +++ b/litellm/integrations/braintrust_logging.py @@ -42,7 +42,7 @@ class BraintrustLogger(CustomLogger): ) -> None: super().__init__() self.validate_environment(api_key=api_key) - self.api_base = api_base or API_BASE + self.api_base = api_base or os.getenv("BRAINTRUST_API_BASE") or API_BASE self.default_project_id = None self.api_key: str = api_key or os.getenv("BRAINTRUST_API_KEY") # type: ignore self.headers = { diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index b69d27775c1..17257e09c5f 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2210,7 +2210,7 @@ class AllCallbacks(LiteLLMPydanticObjectBase): braintrust: CallbackOnUI = CallbackOnUI( litellm_callback_name="braintrust", - litellm_callback_params=["BRAINTRUST_API_KEY"], + litellm_callback_params=["BRAINTRUST_API_KEY","BRAINTRUST_API_BASE"], ui_callback_name="Braintrust", ) diff --git a/tests/logging_callback_tests/test_unit_tests_init_callbacks.py b/tests/logging_callback_tests/test_unit_tests_init_callbacks.py index 3c9d31890c7..a1760ca6371 100644 --- a/tests/logging_callback_tests/test_unit_tests_init_callbacks.py +++ b/tests/logging_callback_tests/test_unit_tests_init_callbacks.py @@ -33,6 +33,7 @@ expected_env_vars = { "LAGO_API_BASE": "mock_base", "LAGO_API_EVENT_CODE": "mock_event_code", "OPENMETER_API_KEY": "openmeter_api_key", + "BRAINTRUST_API_BASE": "braintrust_api_base", "BRAINTRUST_API_KEY": "braintrust_api_key", "GALILEO_API_KEY": "galileo_api_key", "LITERAL_API_KEY": "literal_api_key", diff --git a/tests/test_litellm/integrations/test_braintrust_logging.py b/tests/test_litellm/integrations/test_braintrust_logging.py new file mode 100644 index 00000000000..5ae40e82760 --- /dev/null +++ b/tests/test_litellm/integrations/test_braintrust_logging.py @@ -0,0 +1,43 @@ +import os +import unittest +from unittest.mock import patch + +from litellm.integrations.braintrust_logging import BraintrustLogger + +class TestBraintrustLogger(unittest.TestCase): + @patch.dict(os.environ, {"BRAINTRUST_API_KEY": "test-env-api-key"}) + @patch.dict(os.environ, {"BRAINTRUST_API_BASE": "https://test-env-api.com/v1"}) + def test_init_with_env_var(self): + """Test BraintrustLogger initialization with environment variable.""" + logger = BraintrustLogger() + self.assertEqual(logger.api_key, "test-env-api-key") + self.assertEqual(logger.api_base, "https://test-env-api.com/v1") + self.assertEqual(logger.headers["Authorization"], "Bearer test-env-api-key") + self.assertEqual(logger.headers["Content-Type"], "application/json") + + def test_init_with_explicit_params(self): + """Test BraintrustLogger initialization with explicit parameters.""" + logger = BraintrustLogger(api_key="explicit-key", api_base="https://custom-api.com/v1") + self.assertEqual(logger.api_key, "explicit-key") + self.assertEqual(logger.api_base, "https://custom-api.com/v1") + self.assertEqual(logger.headers["Authorization"], "Bearer explicit-key") + + @patch.dict(os.environ, {}, clear=True) + def test_init_missing_api_key(self): + """Test BraintrustLogger initialization fails without API key.""" + with self.assertRaises(Exception) as context: + BraintrustLogger() + self.assertIn("Missing keys=['BRAINTRUST_API_KEY']", str(context.exception)) + + def test_validate_environment_with_api_key(self): + """Test validate_environment method with valid API key.""" + logger = BraintrustLogger(api_key="test-key") + # Should not raise an exception + logger.validate_environment(api_key="test-key") + + def test_validate_environment_missing_api_key(self): + """Test validate_environment method with missing API key.""" + with patch.dict(os.environ, {}, clear=True): + with self.assertRaises(Exception) as context: + BraintrustLogger(api_key=None) + self.assertIn("Missing keys=['BRAINTRUST_API_KEY']", str(context.exception)) \ No newline at end of file From 96dca4eff81308d44b5425571839c04019eb3445 Mon Sep 17 00:00:00 2001 From: Anand Khinvasara Date: Thu, 7 Aug 2025 08:42:11 -0700 Subject: [PATCH 06/21] fix: 12152 - Redacted sensitive information logged in bedrock guardrails (#13356) --- .../guardrail_hooks/bedrock_guardrails.py | 49 +- .../test_bedrock_guardrails.py | 861 ++++++++++++++++++ 2 files changed, 909 insertions(+), 1 deletion(-) create mode 100644 tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py diff --git a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py index 384d958946b..15953f4229b 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py +++ b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py @@ -5,6 +5,7 @@ # +-------------------------------------------------------------+ # Thank you users! We ❤️ you! - Krrish & Ishaan +import copy import os import sys @@ -50,6 +51,51 @@ from litellm.types.utils import ( GUARDRAIL_NAME = "bedrock" +def _redact_pii_matches(response_json: dict) -> dict: + try: + # Create a deep copy to avoid modifying the original response + redacted_response = copy.deepcopy(response_json) + + # Get assessments from the response + assessments = redacted_response.get("assessments", []) + if not assessments: + return redacted_response + + for assessment in assessments: + # Redact PII entities in sensitive information policy + sensitive_info_policy = assessment.get("sensitiveInformationPolicy") + if sensitive_info_policy: + pii_entities = sensitive_info_policy.get("piiEntities", []) + for pii_entity in pii_entities: + if "match" in pii_entity: + pii_entity["match"] = "[REDACTED]" + + # Redact regex matches + regexes = sensitive_info_policy.get("regexes", []) + for regex_match in regexes: + if "match" in regex_match: + regex_match["match"] = "[REDACTED]" + + # Redact custom word matches in word policy + word_policy = assessment.get("wordPolicy") + if word_policy: + custom_words = word_policy.get("customWords", []) + for custom_word in custom_words: + if "match" in custom_word: + custom_word["match"] = "[REDACTED]" + + managed_words = word_policy.get("managedWordLists", []) + for managed_word in managed_words: + if "match" in managed_word: + managed_word["match"] = "[REDACTED]" + + return redacted_response + except Exception as e: + # We do not want to fail in any case so this is just a warning + verbose_proxy_logger.warning("Guardrail log redaction failed: %s", str(e)) + return response_json + + class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): def __init__( self, @@ -271,10 +317,11 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): data=prepared_request.body, # type: ignore headers=prepared_request.headers, # type: ignore ) - verbose_proxy_logger.debug("Bedrock AI response: %s", response.text) if response.status_code == 200: # check if the response was flagged _json_response = response.json() + redacted_response = _redact_pii_matches(_json_response) + verbose_proxy_logger.debug("Bedrock AI response : %s", redacted_response) bedrock_guardrail_response = BedrockGuardrailResponse(**_json_response) if self._should_raise_guardrail_blocked_exception( bedrock_guardrail_response diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py new file mode 100644 index 00000000000..9e44a3eb419 --- /dev/null +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_bedrock_guardrails.py @@ -0,0 +1,861 @@ +""" +Unit tests for Bedrock Guardrails +""" + +import os +import sys +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +sys.path.insert(0, os.path.abspath("../../../../../..")) + +from litellm.proxy._types import UserAPIKeyAuth +from litellm.proxy.guardrails.guardrail_hooks.bedrock_guardrails import ( + BedrockGuardrail, + _redact_pii_matches, +) + + +@pytest.mark.asyncio +async def test__redact_pii_matches_function(): + """Test the _redact_pii_matches function directly""" + + # Test case 1: Response with PII entities + response_with_pii = { + "action": "GUARDRAIL_INTERVENED", + "assessments": [ + { + "sensitiveInformationPolicy": { + "piiEntities": [ + {"type": "NAME", "match": "John Smith", "action": "BLOCKED"}, + { + "type": "US_SOCIAL_SECURITY_NUMBER", + "match": "324-12-3212", + "action": "BLOCKED", + }, + {"type": "PHONE", "match": "607-456-7890", "action": "BLOCKED"}, + ] + } + } + ], + "outputs": [{"text": "Input blocked by PII policy"}], + } + + # Call the redaction function + redacted_response = _redact_pii_matches(response_with_pii) + + # Verify that PII matches are redacted + pii_entities = redacted_response["assessments"][0]["sensitiveInformationPolicy"][ + "piiEntities" + ] + + assert pii_entities[0]["match"] == "[REDACTED]", "Name should be redacted" + assert pii_entities[1]["match"] == "[REDACTED]", "SSN should be redacted" + assert pii_entities[2]["match"] == "[REDACTED]", "Phone should be redacted" + + # Verify other fields remain unchanged + assert pii_entities[0]["type"] == "NAME" + assert pii_entities[1]["type"] == "US_SOCIAL_SECURITY_NUMBER" + assert pii_entities[2]["type"] == "PHONE" + assert redacted_response["action"] == "GUARDRAIL_INTERVENED" + assert redacted_response["outputs"][0]["text"] == "Input blocked by PII policy" + + print("PII redaction function test passed") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_no_pii(): + """Test _redact_pii_matches with response that has no PII""" + + response_no_pii = {"action": "NONE", "assessments": [], "outputs": []} + + # Call the redaction function + redacted_response = _redact_pii_matches(response_no_pii) + + # Should return the same response unchanged + assert redacted_response == response_no_pii + print("No PII redaction test passed") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_empty_assessments(): + """Test _redact_pii_matches with empty assessments""" + + response_empty_assessments = { + "action": "GUARDRAIL_INTERVENED", + "assessments": [{"sensitiveInformationPolicy": {"piiEntities": []}}], + "outputs": [{"text": "Some output"}], + } + + # Call the redaction function + redacted_response = _redact_pii_matches(response_empty_assessments) + + # Should return the same response unchanged + assert redacted_response == response_empty_assessments + print("Empty assessments redaction test passed") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_malformed_response(): + """Test _redact_pii_matches with malformed response (should not crash)""" + + # Test with completely malformed response + malformed_response = { + "action": "GUARDRAIL_INTERVENED", + "assessments": "not_a_list", # This should cause an exception + } + + # Should not crash and return original response + redacted_response = _redact_pii_matches(malformed_response) + assert redacted_response == malformed_response + + # Test with missing keys + missing_keys_response = { + "action": "GUARDRAIL_INTERVENED" + # Missing assessments key + } + + redacted_response = _redact_pii_matches(missing_keys_response) + assert redacted_response == missing_keys_response + + print("Malformed response redaction test passed") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_multiple_assessments(): + """Test _redact_pii_matches with multiple assessments containing PII""" + + response_multiple_assessments = { + "action": "GUARDRAIL_INTERVENED", + "assessments": [ + { + "sensitiveInformationPolicy": { + "piiEntities": [ + { + "type": "EMAIL", + "match": "john@example.com", + "action": "ANONYMIZED", + } + ] + } + }, + { + "sensitiveInformationPolicy": { + "piiEntities": [ + { + "type": "CREDIT_DEBIT_CARD_NUMBER", + "match": "1234-5678-9012-3456", + "action": "BLOCKED", + }, + { + "type": "ADDRESS", + "match": "123 Main St, Anytown USA", + "action": "ANONYMIZED", + }, + ] + } + }, + ], + "outputs": [{"text": "Multiple PII detected"}], + } + + # Call the redaction function + redacted_response = _redact_pii_matches(response_multiple_assessments) + + # Verify all PII in all assessments are redacted + assessment1_pii = redacted_response["assessments"][0]["sensitiveInformationPolicy"][ + "piiEntities" + ] + assessment2_pii = redacted_response["assessments"][1]["sensitiveInformationPolicy"][ + "piiEntities" + ] + + assert assessment1_pii[0]["match"] == "[REDACTED]", "Email should be redacted" + assert assessment2_pii[0]["match"] == "[REDACTED]", "Credit card should be redacted" + assert assessment2_pii[1]["match"] == "[REDACTED]", "Address should be redacted" + + # Verify types remain unchanged + assert assessment1_pii[0]["type"] == "EMAIL" + assert assessment2_pii[0]["type"] == "CREDIT_DEBIT_CARD_NUMBER" + assert assessment2_pii[1]["type"] == "ADDRESS" + + print("Multiple assessments redaction test passed") + + +@pytest.mark.asyncio +async def test_bedrock_guardrail_logging_uses_redacted_response(): + """Test that the Bedrock guardrail uses redacted response for logging""" + + # Create proper mock objects + mock_user_api_key_dict = UserAPIKeyAuth() + + guardrail = BedrockGuardrail( + guardrailIdentifier="test-guardrail", guardrailVersion="DRAFT" + ) + + # Mock the Bedrock API response with PII + mock_bedrock_response = MagicMock() + mock_bedrock_response.status_code = 200 + mock_bedrock_response.json.return_value = { + "action": "GUARDRAIL_INTERVENED", + "outputs": [{"text": "Hello, my phone number is {PHONE}"}], + "assessments": [ + { + "sensitiveInformationPolicy": { + "piiEntities": [ + { + "type": "PHONE", + "match": "+1 412 555 1212", # This should be redacted in logs + "action": "ANONYMIZED", + } + ] + } + } + ], + } + + request_data = { + "model": "gpt-4o", + "messages": [ + {"role": "user", "content": "Hello, my phone number is +1 412 555 1212"}, + ], + } + + # Mock AWS credentials to avoid credential loading issues in CI + mock_credentials = MagicMock() + mock_credentials.access_key = "test-access-key" + mock_credentials.secret_key = "test-secret-key" + mock_credentials.token = None + + # Mock AWS-related methods to ensure test runs without external dependencies + with patch.object( + guardrail.async_handler, "post", new_callable=AsyncMock + ) as mock_post, patch( + "litellm.proxy.guardrails.guardrail_hooks.bedrock_guardrails.verbose_proxy_logger.debug" + ) as mock_debug, patch.object( + guardrail, "_load_credentials", return_value=(mock_credentials, "us-east-1") + ) as mock_load_creds, patch.object( + guardrail, "_prepare_request", return_value=MagicMock() + ) as mock_prepare_request: + + mock_post.return_value = mock_bedrock_response + + # Call the method that should log the redacted response + await guardrail.make_bedrock_api_request( + source="INPUT", + messages=request_data.get("messages"), + request_data=request_data, + ) + + # Verify that debug logging was called + mock_debug.assert_called() + + # Get the logged response (second argument to debug call) + logged_calls = mock_debug.call_args_list + bedrock_response_log_call = None + + for call in logged_calls: + args, kwargs = call + if len(args) >= 2 and "Bedrock AI response" in str(args[0]): + bedrock_response_log_call = call + break + + assert ( + bedrock_response_log_call is not None + ), "Should have logged Bedrock AI response" + + # Extract the logged response data + logged_response = bedrock_response_log_call[0][ + 1 + ] # Second argument to debug call + + # Verify that the logged response has redacted PII + assert ( + logged_response["assessments"][0]["sensitiveInformationPolicy"][ + "piiEntities" + ][0]["match"] + == "[REDACTED]" + ) + + # Verify other fields are preserved + assert logged_response["action"] == "GUARDRAIL_INTERVENED" + assert ( + logged_response["assessments"][0]["sensitiveInformationPolicy"][ + "piiEntities" + ][0]["type"] + == "PHONE" + ) + + print("Bedrock guardrail logging redaction test passed") + + +@pytest.mark.asyncio +async def test_bedrock_guardrail_original_response_not_modified(): + """Test that the original response is not modified by redaction, only the logged version""" + + # Create proper mock objects + mock_user_api_key_dict = UserAPIKeyAuth() + + guardrail = BedrockGuardrail( + guardrailIdentifier="test-guardrail", guardrailVersion="DRAFT" + ) + + # Mock the Bedrock API response with PII + original_response_data = { + "action": "GUARDRAIL_INTERVENED", + "outputs": [{"text": "Hello, my phone number is {PHONE}"}], + "assessments": [ + { + "sensitiveInformationPolicy": { + "piiEntities": [ + { + "type": "PHONE", + "match": "+1 412 555 1212", # This should NOT be modified in original + "action": "ANONYMIZED", + } + ] + } + } + ], + } + + mock_bedrock_response = MagicMock() + mock_bedrock_response.status_code = 200 + mock_bedrock_response.json.return_value = original_response_data + + request_data = { + "model": "gpt-4o", + "messages": [ + {"role": "user", "content": "Hello, my phone number is +1 412 555 1212"}, + ], + } + + # Mock AWS credentials to avoid credential loading issues in CI + mock_credentials = MagicMock() + mock_credentials.access_key = "test-access-key" + mock_credentials.secret_key = "test-secret-key" + mock_credentials.token = None + + # Mock AWS-related methods to ensure test runs without external dependencies + with patch.object( + guardrail.async_handler, "post", new_callable=AsyncMock + ) as mock_post, patch.object( + guardrail, "_load_credentials", return_value=(mock_credentials, "us-east-1") + ) as mock_load_creds, patch.object( + guardrail, "_prepare_request", return_value=MagicMock() + ) as mock_prepare_request: + + mock_post.return_value = mock_bedrock_response + + # Call the method + result = await guardrail.make_bedrock_api_request( + source="INPUT", + messages=request_data.get("messages"), + request_data=request_data, + ) + + # Verify that the original response data was not modified + # (The json() method should return the original data) + original_data = mock_bedrock_response.json() + assert ( + original_data["assessments"][0]["sensitiveInformationPolicy"][ + "piiEntities" + ][0]["match"] + == "+1 412 555 1212" + ) + + # Verify that the returned BedrockGuardrailResponse contains original data + assert ( + result["assessments"][0]["sensitiveInformationPolicy"]["piiEntities"][0][ + "match" + ] + == "+1 412 555 1212" + ) + + print("Original response not modified test passed") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_preserves_non_pii_entities(): + """Test that _redact_pii_matches only affects PII-related entities and preserves other assessment data""" + + response_with_mixed_data = { + "action": "GUARDRAIL_INTERVENED", + "assessments": [ + { + "sensitiveInformationPolicy": { + "piiEntities": [ + { + "type": "EMAIL", + "match": "user@example.com", + "action": "ANONYMIZED", + "confidence": "HIGH", + } + ], + "regexes": [ + { + "name": "custom_pattern", + "match": "some_pattern_match", + "action": "BLOCKED", + } + ], + }, + "contentPolicy": { + "filters": [ + { + "type": "VIOLENCE", + "confidence": "MEDIUM", + "action": "BLOCKED", + } + ] + }, + "topicPolicy": { + "topics": [ + { + "name": "Restricted Topic", + "type": "DENY", + "action": "BLOCKED", + } + ] + }, + } + ], + "outputs": [{"text": "Content blocked"}], + } + + # Call the redaction function + redacted_response = _redact_pii_matches(response_with_mixed_data) + + # Verify that PII entity matches are redacted + pii_entities = redacted_response["assessments"][0]["sensitiveInformationPolicy"][ + "piiEntities" + ] + assert pii_entities[0]["match"] == "[REDACTED]", "PII match should be redacted" + assert pii_entities[0]["type"] == "EMAIL", "PII type should be preserved" + assert pii_entities[0]["action"] == "ANONYMIZED", "PII action should be preserved" + assert pii_entities[0]["confidence"] == "HIGH", "PII confidence should be preserved" + + # Verify that regex matches are also redacted (updated behavior) + regexes = redacted_response["assessments"][0]["sensitiveInformationPolicy"][ + "regexes" + ] + assert regexes[0]["match"] == "[REDACTED]", "Regex match should be redacted" + assert regexes[0]["name"] == "custom_pattern", "Regex name should be preserved" + assert regexes[0]["action"] == "BLOCKED", "Regex action should be preserved" + + # Verify that other policies are completely unchanged + content_policy = redacted_response["assessments"][0]["contentPolicy"] + assert content_policy["filters"][0]["type"] == "VIOLENCE" + assert content_policy["filters"][0]["confidence"] == "MEDIUM" + + topic_policy = redacted_response["assessments"][0]["topicPolicy"] + assert topic_policy["topics"][0]["name"] == "Restricted Topic" + + # Verify top-level fields are unchanged + assert redacted_response["action"] == "GUARDRAIL_INTERVENED" + assert redacted_response["outputs"][0]["text"] == "Content blocked" + + print("Preserves non-PII entities test passed") + + +@pytest.mark.asyncio +async def test_pii_redaction_matches_debug_output_format(): + """Test that demonstrates the exact behavior shown in your debug output""" + + # This matches the structure from your debug output + original_response = { + "action": "GUARDRAIL_INTERVENED", + "actionReason": "Guardrail blocked.", + "assessments": [ + { + "invocationMetrics": { + "guardrailCoverage": { + "textCharacters": {"guarded": 84, "total": 84} + }, + "guardrailProcessingLatency": 322, + "usage": { + "contentPolicyImageUnits": 0, + "contentPolicyUnits": 0, + "contextualGroundingPolicyUnits": 0, + "sensitiveInformationPolicyFreeUnits": 0, + "sensitiveInformationPolicyUnits": 1, + "topicPolicyUnits": 0, + "wordPolicyUnits": 0, + }, + }, + "sensitiveInformationPolicy": { + "piiEntities": [ + { + "action": "BLOCKED", + "detected": True, + "match": "John Smith", + "type": "NAME", + }, + { + "action": "BLOCKED", + "detected": True, + "match": "324-12-3212", + "type": "US_SOCIAL_SECURITY_NUMBER", + }, + { + "action": "BLOCKED", + "detected": True, + "match": "607-456-7890", + "type": "PHONE", + }, + ] + }, + } + ], + "blockedResponse": "Input blocked by PII policy", + "guardrailCoverage": {"textCharacters": {"guarded": 84, "total": 84}}, + "output": [{"text": "Input blocked by PII policy"}], + "outputs": [{"text": "Input blocked by PII policy"}], + "usage": { + "contentPolicyImageUnits": 0, + "contentPolicyUnits": 0, + "contextualGroundingPolicyUnits": 0, + "sensitiveInformationPolicyFreeUnits": 0, + "sensitiveInformationPolicyUnits": 1, + "topicPolicyUnits": 0, + "wordPolicyUnits": 0, + }, + } + + # Apply redaction + redacted_response = _redact_pii_matches(original_response) + + # Verify the redacted response matches your expected debug output + pii_entities = redacted_response["assessments"][0]["sensitiveInformationPolicy"][ + "piiEntities" + ] + + # All PII matches should be redacted + assert pii_entities[0]["match"] == "[REDACTED]", "NAME should be redacted" + assert pii_entities[1]["match"] == "[REDACTED]", "SSN should be redacted" + assert pii_entities[2]["match"] == "[REDACTED]", "PHONE should be redacted" + + # But all other fields should be preserved + assert pii_entities[0]["type"] == "NAME" + assert pii_entities[1]["type"] == "US_SOCIAL_SECURITY_NUMBER" + assert pii_entities[2]["type"] == "PHONE" + assert pii_entities[0]["action"] == "BLOCKED" + assert pii_entities[0]["detected"] == True + + # Verify that the original response is unchanged + original_pii_entities = original_response["assessments"][0][ + "sensitiveInformationPolicy" + ]["piiEntities"] + assert ( + original_pii_entities[0]["match"] == "John Smith" + ), "Original should be unchanged" + assert ( + original_pii_entities[1]["match"] == "324-12-3212" + ), "Original should be unchanged" + assert ( + original_pii_entities[2]["match"] == "607-456-7890" + ), "Original should be unchanged" + + # Verify all other metadata is preserved in redacted response + assert redacted_response["action"] == "GUARDRAIL_INTERVENED" + assert redacted_response["actionReason"] == "Guardrail blocked." + assert redacted_response["blockedResponse"] == "Input blocked by PII policy" + assert ( + redacted_response["assessments"][0]["invocationMetrics"][ + "guardrailProcessingLatency" + ] + == 322 + ) + + print("PII redaction matches debug output format test passed") + print( + f"Original PII values preserved: {[e['match'] for e in original_pii_entities]}" + ) + print(f"Redacted PII values: {[e['match'] for e in pii_entities]}") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_with_regex_matches(): + """Test redaction of regex matches in sensitive information policy""" + + response_with_regex = { + "action": "GUARDRAIL_INTERVENED", + "assessments": [ + { + "sensitiveInformationPolicy": { + "regexes": [ + { + "name": "SSN_PATTERN", + "match": "123-45-6789", + "action": "BLOCKED", + }, + { + "name": "CREDIT_CARD_PATTERN", + "match": "4111-1111-1111-1111", + "action": "ANONYMIZED", + }, + ] + } + } + ], + "outputs": [{"text": "Regex patterns detected"}], + } + + # Call the redaction function + redacted_response = _redact_pii_matches(response_with_regex) + + # Verify that regex matches are redacted + regexes = redacted_response["assessments"][0]["sensitiveInformationPolicy"][ + "regexes" + ] + + assert regexes[0]["match"] == "[REDACTED]", "SSN regex match should be redacted" + assert ( + regexes[1]["match"] == "[REDACTED]" + ), "Credit card regex match should be redacted" + + # Verify other fields are preserved + assert regexes[0]["name"] == "SSN_PATTERN", "Regex name should be preserved" + assert regexes[0]["action"] == "BLOCKED", "Regex action should be preserved" + assert regexes[1]["name"] == "CREDIT_CARD_PATTERN", "Regex name should be preserved" + assert regexes[1]["action"] == "ANONYMIZED", "Regex action should be preserved" + + # Verify original response is unchanged + original_regexes = response_with_regex["assessments"][0][ + "sensitiveInformationPolicy" + ]["regexes"] + assert original_regexes[0]["match"] == "123-45-6789", "Original should be unchanged" + assert ( + original_regexes[1]["match"] == "4111-1111-1111-1111" + ), "Original should be unchanged" + + print("Regex matches redaction test passed") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_with_custom_words(): + """Test redaction of custom word matches in word policy""" + + response_with_custom_words = { + "action": "GUARDRAIL_INTERVENED", + "assessments": [ + { + "wordPolicy": { + "customWords": [ + { + "match": "confidential_data", + "action": "BLOCKED", + }, + { + "match": "secret_information", + "action": "ANONYMIZED", + }, + ] + } + } + ], + "outputs": [{"text": "Custom words detected"}], + } + + # Call the redaction function + redacted_response = _redact_pii_matches(response_with_custom_words) + + # Verify that custom word matches are redacted + custom_words = redacted_response["assessments"][0]["wordPolicy"]["customWords"] + + assert ( + custom_words[0]["match"] == "[REDACTED]" + ), "First custom word match should be redacted" + assert ( + custom_words[1]["match"] == "[REDACTED]" + ), "Second custom word match should be redacted" + + # Verify other fields are preserved + assert ( + custom_words[0]["action"] == "BLOCKED" + ), "Custom word action should be preserved" + assert ( + custom_words[1]["action"] == "ANONYMIZED" + ), "Custom word action should be preserved" + + # Verify original response is unchanged + original_custom_words = response_with_custom_words["assessments"][0]["wordPolicy"][ + "customWords" + ] + assert ( + original_custom_words[0]["match"] == "confidential_data" + ), "Original should be unchanged" + assert ( + original_custom_words[1]["match"] == "secret_information" + ), "Original should be unchanged" + + print("Custom words redaction test passed") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_with_managed_words(): + """Test redaction of managed word matches in word policy""" + + response_with_managed_words = { + "action": "GUARDRAIL_INTERVENED", + "assessments": [ + { + "wordPolicy": { + "managedWordLists": [ + { + "match": "inappropriate_word", + "action": "BLOCKED", + "type": "PROFANITY", + }, + { + "match": "offensive_term", + "action": "ANONYMIZED", + "type": "HATE_SPEECH", + }, + ] + } + } + ], + "outputs": [{"text": "Managed words detected"}], + } + + # Call the redaction function + redacted_response = _redact_pii_matches(response_with_managed_words) + + # Verify that managed word matches are redacted + managed_words = redacted_response["assessments"][0]["wordPolicy"][ + "managedWordLists" + ] + + assert ( + managed_words[0]["match"] == "[REDACTED]" + ), "First managed word match should be redacted" + assert ( + managed_words[1]["match"] == "[REDACTED]" + ), "Second managed word match should be redacted" + + # Verify other fields are preserved + assert ( + managed_words[0]["action"] == "BLOCKED" + ), "Managed word action should be preserved" + assert ( + managed_words[0]["type"] == "PROFANITY" + ), "Managed word type should be preserved" + assert ( + managed_words[1]["action"] == "ANONYMIZED" + ), "Managed word action should be preserved" + assert ( + managed_words[1]["type"] == "HATE_SPEECH" + ), "Managed word type should be preserved" + + # Verify original response is unchanged + original_managed_words = response_with_managed_words["assessments"][0][ + "wordPolicy" + ]["managedWordLists"] + assert ( + original_managed_words[0]["match"] == "inappropriate_word" + ), "Original should be unchanged" + assert ( + original_managed_words[1]["match"] == "offensive_term" + ), "Original should be unchanged" + + print("Managed words redaction test passed") + + +@pytest.mark.asyncio +async def test__redact_pii_matches_comprehensive_coverage(): + """Test redaction across all supported policy types in a single response""" + + comprehensive_response = { + "action": "GUARDRAIL_INTERVENED", + "assessments": [ + { + "sensitiveInformationPolicy": { + "piiEntities": [ + { + "type": "EMAIL", + "match": "user@example.com", + "action": "ANONYMIZED", + } + ], + "regexes": [ + { + "name": "PHONE_PATTERN", + "match": "555-123-4567", + "action": "BLOCKED", + } + ], + }, + "wordPolicy": { + "customWords": [ + { + "match": "confidential", + "action": "BLOCKED", + } + ], + "managedWordLists": [ + { + "match": "inappropriate", + "action": "ANONYMIZED", + "type": "PROFANITY", + } + ], + }, + } + ], + "outputs": [{"text": "Multiple policy violations detected"}], + } + + # Call the redaction function + redacted_response = _redact_pii_matches(comprehensive_response) + + # Verify all match fields are redacted + assessment = redacted_response["assessments"][0] + + # PII entities + pii_entities = assessment["sensitiveInformationPolicy"]["piiEntities"] + assert ( + pii_entities[0]["match"] == "[REDACTED]" + ), "PII entity match should be redacted" + + # Regex matches + regexes = assessment["sensitiveInformationPolicy"]["regexes"] + assert regexes[0]["match"] == "[REDACTED]", "Regex match should be redacted" + + # Custom words + custom_words = assessment["wordPolicy"]["customWords"] + assert ( + custom_words[0]["match"] == "[REDACTED]" + ), "Custom word match should be redacted" + + # Managed words + managed_words = assessment["wordPolicy"]["managedWordLists"] + assert ( + managed_words[0]["match"] == "[REDACTED]" + ), "Managed word match should be redacted" + + # Verify all other fields are preserved + assert pii_entities[0]["type"] == "EMAIL" + assert regexes[0]["name"] == "PHONE_PATTERN" + assert managed_words[0]["type"] == "PROFANITY" + + # Verify original response is unchanged + original_assessment = comprehensive_response["assessments"][0] + assert ( + original_assessment["sensitiveInformationPolicy"]["piiEntities"][0]["match"] + == "user@example.com" + ) + assert ( + original_assessment["sensitiveInformationPolicy"]["regexes"][0]["match"] + == "555-123-4567" + ) + assert ( + original_assessment["wordPolicy"]["customWords"][0]["match"] == "confidential" + ) + assert ( + original_assessment["wordPolicy"]["managedWordLists"][0]["match"] + == "inappropriate" + ) + + print("Comprehensive coverage redaction test passed") From f58807ff6e43941deed9009941f33fc29e1ec11a Mon Sep 17 00:00:00 2001 From: unique-jakub Date: Thu, 7 Aug 2025 18:41:24 +0200 Subject: [PATCH 07/21] Add labels to migrations job template (#13343) * set labels on the migration job * update comment to retrigger the pipeline --- deploy/charts/litellm-helm/templates/migrations-job.yaml | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/deploy/charts/litellm-helm/templates/migrations-job.yaml b/deploy/charts/litellm-helm/templates/migrations-job.yaml index 143e62fceb3..cf10be0a76b 100644 --- a/deploy/charts/litellm-helm/templates/migrations-job.yaml +++ b/deploy/charts/litellm-helm/templates/migrations-job.yaml @@ -1,9 +1,11 @@ {{- if .Values.migrationJob.enabled }} -# This job runs the prisma migrations for the LiteLLM DB. +# This job runs the Prisma migrations for the LiteLLM DB. apiVersion: batch/v1 kind: Job metadata: name: {{ include "litellm.fullname" . }}-migrations + labels: + {{- include "litellm.labels" . | nindent 4 }} annotations: {{- if .Values.migrationJob.hooks.argocd.enabled }} argocd.argoproj.io/hook: PreSync @@ -18,6 +20,8 @@ metadata: spec: template: metadata: + labels: + {{- include "litellm.labels" . | nindent 8 }} annotations: {{- with .Values.migrationJob.annotations }} {{- toYaml . | nindent 8 }} From e8b4b2577450d502dab2b8f4b1d5264642ece67f Mon Sep 17 00:00:00 2001 From: breno-aumo <160534746+breno-aumo@users.noreply.github.com> Date: Thu, 7 Aug 2025 13:45:17 -0300 Subject: [PATCH 08/21] Update OCI docs (#13336) * add oci models to model_prices_and_context_window.json * remove unsupported and unavailable oci models from docs --- docs/my-website/docs/providers/oci.md | 9 -- model_prices_and_context_window.json | 131 ++++++++++++++++++++++++++ 2 files changed, 131 insertions(+), 9 deletions(-) diff --git a/docs/my-website/docs/providers/oci.md b/docs/my-website/docs/providers/oci.md index 36971376866..28beb71094a 100644 --- a/docs/my-website/docs/providers/oci.md +++ b/docs/my-website/docs/providers/oci.md @@ -6,20 +6,11 @@ LiteLLM supports the following models for OCI on-demand GenAI API. Check the [OCI Models List](https://docs.oracle.com/en-us/iaas/Content/generative-ai/pretrained-models.htm) to see if the model is available for your region. -- `cohere.command-a-03-2025` -- `cohere.command-r-08-2024` -- `cohere.command-plus-latest` (alias `cohere.command-r-plus-08-2024`) -- `cohere.command-r-16k` (deprecated) -- `cohere.command-r-plus` (deprecated) - - `meta.llama-4-maverick-17b-128e-instruct-fp8` - `meta.llama-4-scout-17b-16e-instruct` - `meta.llama-3.3-70b-instruct` - `meta.llama-3.2-90b-vision-instruct` -- `meta.llama-3.2-11b-vision-instruct` - `meta.llama-3.1-405b-instruct` -- `meta.llama-3.1-70b-instruct` -- `meta.llama-3-70b-instruct` - `xai.grok-4` - `xai.grok-3` diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 8281cb424fd..6ce46e5962f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -17693,5 +17693,136 @@ "supports_vision": false, "supports_system_messages": true, "supports_tool_choice": false + }, + "oci/meta.llama-4-maverick-17b-128e-instruct-fp8": { + "max_tokens": 512000, + "max_input_tokens": 512000, + "max_output_tokens": 4000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/meta.llama-4-scout-17b-16e-instruct": { + "max_tokens": 192000, + "max_input_tokens": 192000, + "max_output_tokens": 4000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/meta.llama-3.3-70b-instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 4000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/meta.llama-3.2-90b-vision-instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 4000, + "input_cost_per_token": 2.0e-06, + "output_cost_per_token": 2.0e-06, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/meta.llama-3.1-405b-instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 4000, + "input_cost_per_token": 1.068e-05, + "output_cost_per_token": 1.068e-05, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + + "oci/xai.grok-4": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 3.0e-06, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/xai.grok-3": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.0e-06, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/xai.grok-3-mini": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.0e-07, + "output_cost_per_token": 5.0e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/xai.grok-3-fast": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 5.0e-06, + "output_cost_per_token": 2.5e-05, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/xai.grok-3-mini-fast": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 6.0e-07, + "output_cost_per_token": 4.0e-06, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" } } From 4d941c914ed43aae326ab5a6a9e93e5ec166d0a7 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 10:59:53 -0700 Subject: [PATCH 09/21] [Feat] Responses API Session Handling - Multi media support (#13347) * rename ResponsesSessionHandler * use ResponsesSessionHandler * test session handler * refactor ResponsesSessionHandler * fix get_proxy_server_request_from_spend_log * use constant for LITELLM_TRUNCATED_PAYLOAD_FIELD * add _should_check_cold_storage_for_full_payload * add get_class_type_for_custom_logger_name * get_active_custom_logger_for_callback_name * add get_proxy_server_request_from_cold_storage to CustomLogger * add ColdStorageHandler * start using cold storage integration * add get_proxy_server_request_from_cold_storage * fixes from manual testing * s3 v2 fix getting region name * ChatCompletionImageUrlObject * use _get_configured_cold_storage_custom_logger * fixes for _should_check_cold_storage_for_full_payload * fix _download_object_from_s3 * test_s3_v2_with_cold_storage * add cold_storage_object_key to StandardLoggingMetadata * use get_proxy_server_request_from_cold_storage_with_object_key * add cold_storage_object_key to SpendLogsMetadata * add cold_storage_object_key * get_proxy_server_request_from_cold_storage_with_object_key * use get_proxy_server_request_from_cold_storage_with_object_key * test responses API * add get_proxy_server_request_from_cold_storage_with_object_key * session handler fixes * test session handler * fix ruff checks * _download_object_from_s3 * cleanup * test * lint fix * test_e2e_cold_storage_successful_retrieval * test_e2e_generate_cold_storage_object_key_successful * test_async_gcs_pub_sub_v1 * test fix * test fix * test fix * test_standard_logging_metadata_has_cold_storage_object_key_field * test_sanitize_request_body_for_spend_logs_payload_basic * test_transform_input_image_item_to_image_item_with_image_data --- cookbook/misc/test_responses_api.py | 53 +++ .../enterprise_callbacks/session_handler.py | 160 ---------- litellm/constants.py | 1 + litellm/integrations/custom_logger.py | 11 + litellm/integrations/s3_v2.py | 115 ++++++- .../custom_logger_registry.py | 12 + litellm/litellm_core_utils/litellm_logging.py | 59 ++++ .../logging_callback_manager.py | 30 +- litellm/proxy/_types.py | 1 + litellm/proxy/proxy_config.yaml | 13 +- .../spend_tracking/cold_storage_handler.py | 73 +++++ .../spend_tracking/spend_tracking_utils.py | 12 +- .../session_handler.py | 301 ++++++++++++++++++ .../transformation.py | 54 ++-- litellm/types/utils.py | 1 + .../test_amazing_s3_logs.py | 15 + .../test_gcs_pub_sub.py | 1 + .../test_otel_logging.py | 1 + .../test_litellm_logging.py | 70 ++++ .../test_spend_management_endpoints.py | 1 + .../test_spend_tracking_utils.py | 16 +- .../test_litellm_completion_responses.py | 20 +- .../test_session_handler.py | 127 +++++++- .../test_session_handler_with_cold_storage.py | 186 +++++++++++ 24 files changed, 1107 insertions(+), 226 deletions(-) create mode 100644 cookbook/misc/test_responses_api.py delete mode 100644 enterprise/litellm_enterprise/enterprise_callbacks/session_handler.py create mode 100644 litellm/proxy/spend_tracking/cold_storage_handler.py create mode 100644 litellm/responses/litellm_completion_transformation/session_handler.py rename tests/{enterprise/litellm_enterprise/enterprise_callbacks => test_litellm/responses/litellm_completion_transformation}/test_session_handler.py (58%) create mode 100644 tests/test_litellm/responses/litellm_completion_transformation/test_session_handler_with_cold_storage.py diff --git a/cookbook/misc/test_responses_api.py b/cookbook/misc/test_responses_api.py new file mode 100644 index 00000000000..5fd19c6f66f --- /dev/null +++ b/cookbook/misc/test_responses_api.py @@ -0,0 +1,53 @@ +import base64 +from openai import OpenAI +import time +client = OpenAI( + base_url="http://0.0.0.0:4001", + api_key="sk-1234" +) + +# Function to encode the image +def encode_image(image_path): + with open(image_path, "rb") as image_file: + return base64.b64encode(image_file.read()).decode("utf-8") + + +# Path to your image +image_path = "litellm/proxy/logo.jpg" + +# Getting the Base64 string +base64_image = encode_image(image_path) + + +response = client.responses.create( + model="bedrock/us.anthropic.claude-3-5-sonnet-20241022-v2:0", + input=[ + { + "role": "user", + "content": [ + { "type": "input_text", "text": "what color is the image"}, + { + "type": "input_image", + "image_url": f"data:image/jpeg;base64,{base64_image}", + }, + ], + } + ], +) + + + +print(response.output_text) +print("response1 id===", response.id) +print("sleeping for 20 seconds...") +time.sleep(20) +print("making follow up request for existing id") +response2 = client.responses.create( + model="bedrock/us.anthropic.claude-3-5-sonnet-20241022-v2:0", + previous_response_id=response.id, + input="ok, and what objects are in the image?" +) + +print(response2.output_text) + + diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/session_handler.py b/enterprise/litellm_enterprise/enterprise_callbacks/session_handler.py deleted file mode 100644 index 1a08a8f9101..00000000000 --- a/enterprise/litellm_enterprise/enterprise_callbacks/session_handler.py +++ /dev/null @@ -1,160 +0,0 @@ -import json -from typing import TYPE_CHECKING, Any, List, Optional, Union, cast - -from litellm._logging import verbose_proxy_logger -from litellm.proxy._types import SpendLogsPayload -from litellm.responses.utils import ResponsesAPIRequestUtils -from litellm.types.llms.openai import ( - AllMessageValues, - ChatCompletionResponseMessage, - GenericChatCompletionMessage, - ResponseInputParam, -) -from litellm.types.utils import ChatCompletionMessageToolCall, Message, ModelResponse - -if TYPE_CHECKING: - from litellm.responses.litellm_completion_transformation.transformation import ( - ChatCompletionSession, - ) -else: - ChatCompletionSession = Any - - -class _ENTERPRISE_ResponsesSessionHandler: - @staticmethod - async def get_chat_completion_message_history_for_previous_response_id( - previous_response_id: str, - ) -> ChatCompletionSession: - """ - Return the chat completion message history for a previous response id - """ - from litellm.responses.litellm_completion_transformation.transformation import ( - ChatCompletionSession, - LiteLLMCompletionResponsesConfig, - ) - - verbose_proxy_logger.debug( - "inside get_chat_completion_message_history_for_previous_response_id" - ) - all_spend_logs: List[ - SpendLogsPayload - ] = await _ENTERPRISE_ResponsesSessionHandler.get_all_spend_logs_for_previous_response_id( - previous_response_id - ) - verbose_proxy_logger.debug( - "found %s spend logs for this response id", len(all_spend_logs) - ) - - litellm_session_id: Optional[str] = None - if len(all_spend_logs) > 0: - litellm_session_id = all_spend_logs[0].get("session_id") - - chat_completion_message_history: List[ - Union[ - AllMessageValues, - GenericChatCompletionMessage, - ChatCompletionMessageToolCall, - ChatCompletionResponseMessage, - Message, - ] - ] = [] - for spend_log in all_spend_logs: - proxy_server_request: Union[str, dict] = ( - spend_log.get("proxy_server_request") or "{}" - ) - proxy_server_request_dict: Optional[dict] = None - response_input_param: Optional[Union[str, ResponseInputParam]] = None - if isinstance(proxy_server_request, dict): - proxy_server_request_dict = proxy_server_request - else: - proxy_server_request_dict = json.loads(proxy_server_request) - - ############################################################ - # Add Input messages for this Spend Log - ############################################################ - if proxy_server_request_dict: - _response_input_param = proxy_server_request_dict.get("input", None) - if isinstance(_response_input_param, str): - response_input_param = _response_input_param - elif isinstance(_response_input_param, dict): - response_input_param = cast( - ResponseInputParam, _response_input_param - ) - - if response_input_param: - chat_completion_messages = LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages( - input=response_input_param, - responses_api_request=proxy_server_request_dict or {}, - ) - chat_completion_message_history.extend(chat_completion_messages) - - ############################################################ - # Add Output messages for this Spend Log - ############################################################ - _response_output = spend_log.get("response", "{}") - if isinstance(_response_output, dict): - # transform `ChatCompletion Response` to `ResponsesAPIResponse` - model_response = ModelResponse(**_response_output) - for choice in model_response.choices: - if hasattr(choice, "message"): - chat_completion_message_history.append( - getattr(choice, "message") - ) - - verbose_proxy_logger.debug( - "chat_completion_message_history %s", - json.dumps(chat_completion_message_history, indent=4, default=str), - ) - return ChatCompletionSession( - messages=chat_completion_message_history, - litellm_session_id=litellm_session_id, - ) - - @staticmethod - async def get_all_spend_logs_for_previous_response_id( - previous_response_id: str, - ) -> List[SpendLogsPayload]: - """ - Get all spend logs for a previous response id - - - SQL query - - SELECT session_id FROM spend_logs WHERE response_id = previous_response_id, SELECT * FROM spend_logs WHERE session_id = session_id - """ - from litellm.proxy.proxy_server import prisma_client - - verbose_proxy_logger.debug("decoding response id=%s", previous_response_id) - - decoded_response_id = ( - ResponsesAPIRequestUtils._decode_responses_api_response_id( - previous_response_id - ) - ) - previous_response_id = decoded_response_id.get( - "response_id", previous_response_id - ) - if prisma_client is None: - return [] - - query = """ - WITH matching_session AS ( - SELECT session_id - FROM "LiteLLM_SpendLogs" - WHERE request_id = $1 - ) - SELECT * - FROM "LiteLLM_SpendLogs" - WHERE session_id IN (SELECT session_id FROM matching_session) - ORDER BY "endTime" ASC; - """ - - spend_logs = await prisma_client.db.query_raw(query, previous_response_id) - - verbose_proxy_logger.debug( - "Found the following spend logs for previous response id %s: %s", - previous_response_id, - json.dumps(spend_logs, indent=4, default=str), - ) - - return spend_logs diff --git a/litellm/constants.py b/litellm/constants.py index 27cea0eb040..c7404f10a78 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -766,6 +766,7 @@ MAXIMUM_TRACEBACK_LINES_TO_LOG = int(os.getenv("MAXIMUM_TRACEBACK_LINES_TO_LOG", X_LITELLM_DISABLE_CALLBACKS = "x-litellm-disable-callbacks" LITELLM_METADATA_FIELD = "litellm_metadata" OLD_LITELLM_METADATA_FIELD = "metadata" +LITELLM_TRUNCATED_PAYLOAD_FIELD = "litellm_truncated" ########################### LiteLLM Proxy Specific Constants ########################### ######################################################################################## diff --git a/litellm/integrations/custom_logger.py b/litellm/integrations/custom_logger.py index b5c7101dde9..ee7e771faa6 100644 --- a/litellm/integrations/custom_logger.py +++ b/litellm/integrations/custom_logger.py @@ -541,3 +541,14 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac model_call_details_copy["standard_logging_object"] = standard_logging_object_copy return model_call_details_copy + + + + async def get_proxy_server_request_from_cold_storage_with_object_key( + self, + object_key: str, + ) -> Optional[dict]: + """ + Get the proxy server request from cold storage using the object key directly. + """ + pass diff --git a/litellm/integrations/s3_v2.py b/litellm/integrations/s3_v2.py index 7df3e58b2da..efe18cb68ad 100644 --- a/litellm/integrations/s3_v2.py +++ b/litellm/integrations/s3_v2.py @@ -304,7 +304,10 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): data=prepped.body, headers=prepped.headers, ) - SigV4Auth(credentials, "s3", self.s3_region_name).add_auth(aws_request) + aws_region_name = self.get_aws_region_name_for_non_llm_api_calls( + aws_region_name=self.s3_region_name + ) + SigV4Auth(credentials, "s3", aws_region_name).add_auth(aws_request) # Prepare the signed headers signed_headers = dict(aws_request.headers.items()) @@ -444,7 +447,10 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): data=prepped.body, headers=prepped.headers, ) - SigV4Auth(credentials, "s3", self.s3_region_name).add_auth(aws_request) + aws_region_name = self.get_aws_region_name_for_non_llm_api_calls( + aws_region_name=self.s3_region_name + ) + SigV4Auth(credentials, "s3", aws_region_name).add_auth(aws_request) # Prepare the signed headers signed_headers = dict(aws_request.headers.items()) @@ -455,3 +461,108 @@ class S3Logger(CustomBatchLogger, BaseAWSLLM): response.raise_for_status() except Exception as e: verbose_logger.exception(f"Error uploading to s3: {str(e)}") + + + async def _download_object_from_s3(self, s3_object_key: str) -> Optional[dict]: + """ + Download and parse JSON object from S3. + + Args: + s3_object_key: The S3 object key to download + + Returns: + Optional[dict]: The parsed JSON object or None if not found/error + """ + try: + import hashlib + + import requests + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + except ImportError: + raise ImportError("Missing boto3 to call S3. Run 'pip install boto3'.") + + try: + from litellm.litellm_core_utils.asyncify import asyncify + + # Get AWS credentials + asyncified_get_credentials = asyncify(self.get_credentials) + credentials = await asyncified_get_credentials( + aws_access_key_id=self.s3_aws_access_key_id, + aws_secret_access_key=self.s3_aws_secret_access_key, + aws_session_token=self.s3_aws_session_token, + aws_region_name=self.s3_region_name, + aws_session_name=self.s3_aws_session_name, + aws_profile_name=self.s3_aws_profile_name, + aws_role_name=self.s3_aws_role_name, + aws_web_identity_token=self.s3_aws_web_identity_token, + aws_sts_endpoint=self.s3_aws_sts_endpoint, + ) + + verbose_logger.debug( + f"s3_v2 logger - downloading data from s3 - {s3_object_key}" + ) + + # Prepare the URL + url = f"https://{self.s3_bucket_name}.s3.{self.s3_region_name}.amazonaws.com/{s3_object_key}" + + if self.s3_endpoint_url: + url = self.s3_endpoint_url + "/" + s3_object_key + + # Prepare the request for GET operation + # For GET requests, we need x-amz-content-sha256 with hash of empty string + empty_string_hash = hashlib.sha256(b"").hexdigest() + headers = { + "x-amz-content-sha256": empty_string_hash, + } + req = requests.Request("GET", url, headers=headers) + prepped = req.prepare() + + # Sign the request + aws_request = AWSRequest( + method=prepped.method, + url=prepped.url, + headers=prepped.headers, + ) + SigV4Auth(credentials, "s3", self.s3_region_name).add_auth(aws_request) + + # Prepare the signed headers + signed_headers = dict(aws_request.headers.items()) + + # Make the request + response = await self.async_httpx_client.get(url, headers=signed_headers) + + if response.status_code != 200: + verbose_logger.exception("S3 object not found, saw response=", response.text) + return None + + # Parse JSON response + return response.json() + + except Exception as e: + verbose_logger.exception(f"Error downloading from S3: {str(e)}") + return None + + async def get_proxy_server_request_from_cold_storage_with_object_key( + self, + object_key: str, + ) -> Optional[dict]: + """ + Get the proxy server request from cold storage + + Allows fetching a dict of the proxy server request from s3 or GCS bucket. + + Args: + request_id: The unique request ID to search for + start_time: The start time of the request (datetime or ISO string) + + Returns: + Optional[dict]: The request data dictionary or None if not found + """ + try: + # Download and return the object from S3 + downloaded_object = await self._download_object_from_s3(object_key) + return downloaded_object + except Exception as e: + verbose_logger.exception(f"Error retrieving object {object_key} from cold storage: {str(e)}") + return None \ No newline at end of file diff --git a/litellm/litellm_core_utils/custom_logger_registry.py b/litellm/litellm_core_utils/custom_logger_registry.py index 9606b47b9b8..fd82ecdf2b2 100644 --- a/litellm/litellm_core_utils/custom_logger_registry.py +++ b/litellm/litellm_core_utils/custom_logger_registry.py @@ -10,6 +10,7 @@ Example: from typing import Union +from litellm import _custom_logger_compatible_callbacks_literal from litellm.integrations.agentops import AgentOps from litellm.integrations.anthropic_cache_control_hook import AnthropicCacheControlHook from litellm.integrations.argilla import ArgillaLogger @@ -150,3 +151,14 @@ class CustomLoggerRegistry: if callback_class == class_type: callback_strs.append(callback_str) return callback_strs + + + @classmethod + def get_class_type_for_custom_logger_name( + cls, + custom_logger_name: _custom_logger_compatible_callbacks_literal, + ) -> type: + """ + Get the class type for a given custom logger name + """ + return cls.CALLBACK_CLASS_STR_TO_CLASS_TYPE[custom_logger_name] diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index dfa941d8301..5462257c9b7 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -3830,6 +3830,8 @@ class StandardLoggingPayloadSetup: ] = None, usage_object: Optional[dict] = None, proxy_server_request: Optional[dict] = None, + start_time: Optional[dt_object] = None, + response_id: Optional[str] = None, ) -> StandardLoggingMetadata: """ Clean and filter the metadata dictionary to include only the specified keys in StandardLoggingMetadata. @@ -3881,6 +3883,7 @@ class StandardLoggingPayloadSetup: usage_object=usage_object, requester_custom_headers=None, user_api_key_request_route=None, + cold_storage_object_key=None, ) if isinstance(metadata, dict): # Filter the metadata dictionary to include only the specified keys @@ -3913,6 +3916,16 @@ class StandardLoggingPayloadSetup: proxy_server_request=proxy_server_request, ) + # Generate cold storage object key if cold storage is configured + if start_time is not None and response_id is not None: + cold_storage_object_key = StandardLoggingPayloadSetup._generate_cold_storage_object_key( + start_time=start_time, + response_id=response_id, + team_alias=clean_metadata.get("user_api_key_team_alias"), + ) + if cold_storage_object_key: + clean_metadata["cold_storage_object_key"] = cold_storage_object_key + return clean_metadata @staticmethod @@ -4071,6 +4084,49 @@ class StandardLoggingPayloadSetup: return api_base.rstrip("/") return api_base + @staticmethod + def _generate_cold_storage_object_key( + start_time: dt_object, + response_id: str, + team_alias: Optional[str] = None, + ) -> Optional[str]: + """ + Generate cold storage object key in the same format as S3Logger. + + Args: + start_time: The start time of the request + response_id: The response ID + team_alias: Optional team alias for team-based prefixing + + Returns: + Optional[str]: The generated object key or None if cold storage not configured + """ + # Generate object key in same format as S3Logger + from litellm.integrations.s3 import get_s3_object_key + from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler + + # Only generate object key if cold storage is configured + configured_cold_storage_logger = ColdStorageHandler._get_configured_cold_storage_custom_logger() + if configured_cold_storage_logger is None: + return None + + try: + # Generate file name in same format as litellm.utils.get_logging_id + s3_file_name = f"time-{start_time.strftime('%H-%M-%S-%f')}_{response_id}" + + + s3_object_key = get_s3_object_key( + s3_path="", # Use empty path as default + team_alias_prefix="", # Don't split by team alias for cold storage + start_time=start_time, + s3_file_name=s3_file_name, + ) + + return s3_object_key + except Exception: + # If any error occurs in generating the key, return None + return None + @staticmethod def get_error_information( original_exception: Optional[Exception], @@ -4322,6 +4378,8 @@ def get_standard_logging_object_payload( ), usage_object=usage.model_dump(), proxy_server_request=proxy_server_request, + start_time=start_time, + response_id=id, ) _request_body = proxy_server_request.get("body", {}) @@ -4469,6 +4527,7 @@ def get_standard_logging_metadata( usage_object=None, requester_custom_headers=None, user_api_key_request_route=None, + cold_storage_object_key=None, ) if isinstance(metadata, dict): # Filter the metadata dictionary to include only the specified keys diff --git a/litellm/litellm_core_utils/logging_callback_manager.py b/litellm/litellm_core_utils/logging_callback_manager.py index 44cb146f91a..9ec346c20a1 100644 --- a/litellm/litellm_core_utils/logging_callback_manager.py +++ b/litellm/litellm_core_utils/logging_callback_manager.py @@ -1,4 +1,4 @@ -from typing import Callable, List, Set, Type, Union +from typing import TYPE_CHECKING, Callable, List, Optional, Set, Type, Union import litellm from litellm._logging import verbose_logger @@ -6,6 +6,11 @@ from litellm.integrations.additional_logging_utils import AdditionalLoggingUtils from litellm.integrations.custom_logger import CustomLogger from litellm.types.utils import CallbacksByType +if TYPE_CHECKING: + from litellm import _custom_logger_compatible_callbacks_literal +else: + _custom_logger_compatible_callbacks_literal = str + class LoggingCallbackManager: """ @@ -343,3 +348,26 @@ class LoggingCallbackManager: elif callable(callback): return getattr(callback, "__name__", str(callback)) return str(callback) + + + def get_active_custom_logger_for_callback_name( + self, + callback_name: _custom_logger_compatible_callbacks_literal, + ) -> Optional[CustomLogger]: + """ + Get the active custom logger for a given callback name + """ + from litellm.litellm_core_utils.custom_logger_registry import ( + CustomLoggerRegistry, + ) + + # get the custom logger class type + custom_logger_class_type = CustomLoggerRegistry.get_class_type_for_custom_logger_name(callback_name) + + # get the active custom logger + custom_logger = self.get_custom_loggers_for_type(custom_logger_class_type) + + if len(custom_logger) == 0: + raise ValueError(f"No active custom logger found for callback name: {callback_name}") + + return custom_logger[0] diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 17257e09c5f..bb59f2e94b6 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2264,6 +2264,7 @@ class SpendLogsMetadata(TypedDict): error_information: Optional[StandardLoggingPayloadErrorInformation] usage_object: Optional[dict] model_map_information: Optional[StandardLoggingModelInformation] + cold_storage_object_key: Optional[str] # S3/GCS object key for cold storage retrieval class SpendLogsPayload(TypedDict): diff --git a/litellm/proxy/proxy_config.yaml b/litellm/proxy/proxy_config.yaml index c8d148f7086..aa63369ef2b 100644 --- a/litellm/proxy/proxy_config.yaml +++ b/litellm/proxy/proxy_config.yaml @@ -1,6 +1,15 @@ model_list: - - model_name: anthropic/* + - model_name: bedrock/* litellm_params: - model: anthropic/* + model: bedrock/* +litellm_settings: + callbacks: ["s3_v2"] + s3_callback_params: + s3_bucket_name: litellm-logs # AWS Bucket Name for S3 + s3_region_name: us-west-2 + +general_settings: + cold_storage_custom_logger: s3_v2 + store_prompts_in_cold_storage: true \ No newline at end of file diff --git a/litellm/proxy/spend_tracking/cold_storage_handler.py b/litellm/proxy/spend_tracking/cold_storage_handler.py new file mode 100644 index 00000000000..21e785425ad --- /dev/null +++ b/litellm/proxy/spend_tracking/cold_storage_handler.py @@ -0,0 +1,73 @@ +""" +This module is responsible for handling Getting/Setting the proxy server request from cold storage. + +It allows fetching a dict of the proxy server request from s3 or GCS bucket. +""" +from typing import Optional, cast + +import litellm +from litellm import _custom_logger_compatible_callbacks_literal +from litellm._logging import verbose_proxy_logger +from litellm.integrations.custom_logger import CustomLogger + + +class ColdStorageHandler: + """ + This class is responsible for handling Getting/Setting the proxy server request from cold storage. + + It allows fetching a dict of the proxy server request from s3 or GCS bucket. + """ + + async def get_proxy_server_request_from_cold_storage_with_object_key( + self, + object_key: str, + ) -> Optional[dict]: + """ + Get the proxy server request from cold storage using the object key directly. + + Args: + object_key: The S3/GCS object key to retrieve + + Returns: + Optional[dict]: The proxy server request dict or None if not found + """ + + # select the custom logger to use for cold storage + custom_logger_name: Optional[_custom_logger_compatible_callbacks_literal] = self._select_custom_logger_for_cold_storage() + + # if no custom logger name is configured, return None + if custom_logger_name is None: + return None + + # get the active/initialized custom logger + custom_logger: Optional[CustomLogger] = litellm.logging_callback_manager.get_active_custom_logger_for_callback_name(custom_logger_name) + + # if no custom logger is found, return None + if custom_logger is None: + return None + + proxy_server_request = await custom_logger.get_proxy_server_request_from_cold_storage_with_object_key( + object_key=object_key, + ) + + return proxy_server_request + + + + def _select_custom_logger_for_cold_storage( + self, + ) -> Optional[_custom_logger_compatible_callbacks_literal]: + cold_storage_custom_logger: Optional[_custom_logger_compatible_callbacks_literal] = ColdStorageHandler._get_configured_cold_storage_custom_logger() + + return cold_storage_custom_logger + + + @staticmethod + def _get_configured_cold_storage_custom_logger() -> Optional[_custom_logger_compatible_callbacks_literal]: + from litellm.proxy.proxy_server import general_settings + cold_storage_custom_logger: Optional[str] = general_settings.get("cold_storage_custom_logger") + if not cold_storage_custom_logger: + verbose_proxy_logger.debug("No cold storage custom logger found in general settings") + return None + + return cast(_custom_logger_compatible_callbacks_literal, cold_storage_custom_logger) \ No newline at end of file diff --git a/litellm/proxy/spend_tracking/spend_tracking_utils.py b/litellm/proxy/spend_tracking/spend_tracking_utils.py index ad5cad29e60..653426e8bd2 100644 --- a/litellm/proxy/spend_tracking/spend_tracking_utils.py +++ b/litellm/proxy/spend_tracking/spend_tracking_utils.py @@ -53,6 +53,7 @@ def _get_spend_logs_metadata( guardrail_information: Optional[StandardLoggingGuardrailInformation] = None, usage_object: Optional[dict] = None, model_map_information: Optional[StandardLoggingModelInformation] = None, + cold_storage_object_key: Optional[str] = None ) -> SpendLogsMetadata: if metadata is None: return SpendLogsMetadata( @@ -75,6 +76,7 @@ def _get_spend_logs_metadata( model_map_information=None, usage_object=None, guardrail_information=None, + cold_storage_object_key=cold_storage_object_key, ) verbose_proxy_logger.debug( "getting payload for SpendLogs, available keys in metadata: " @@ -98,6 +100,8 @@ def _get_spend_logs_metadata( clean_metadata["guardrail_information"] = guardrail_information clean_metadata["usage_object"] = usage_object clean_metadata["model_map_information"] = model_map_information + clean_metadata["cold_storage_object_key"] = cold_storage_object_key + return clean_metadata @@ -267,6 +271,11 @@ def get_logging_payload( # noqa: PLR0915 if standard_logging_payload is not None else None ), + cold_storage_object_key=( + standard_logging_payload["metadata"].get("cold_storage_object_key", None) + if standard_logging_payload is not None + else None + ), ) special_usage_fields = ["completion_tokens", "prompt_tokens", "total_tokens"] @@ -474,6 +483,7 @@ def _sanitize_request_body_for_spend_logs_payload( Recursively sanitize request body to prevent logging large base64 strings or other large values. Truncates strings longer than 1000 characters and handles nested dictionaries. """ + from litellm.constants import LITELLM_TRUNCATED_PAYLOAD_FIELD MAX_STRING_LENGTH = 1000 if visited is None: @@ -492,7 +502,7 @@ def _sanitize_request_body_for_spend_logs_payload( return [_sanitize_value(item) for item in value] elif isinstance(value, str): if len(value) > MAX_STRING_LENGTH: - return f"{value[:MAX_STRING_LENGTH]}... (truncated {len(value) - MAX_STRING_LENGTH} chars)" + return f"{value[:MAX_STRING_LENGTH]}... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} {len(value) - MAX_STRING_LENGTH} chars)" return value return value diff --git a/litellm/responses/litellm_completion_transformation/session_handler.py b/litellm/responses/litellm_completion_transformation/session_handler.py new file mode 100644 index 00000000000..28e76ca893f --- /dev/null +++ b/litellm/responses/litellm_completion_transformation/session_handler.py @@ -0,0 +1,301 @@ +import json +from typing import TYPE_CHECKING, Any, List, Optional, Union, cast + +from litellm._logging import verbose_proxy_logger +from litellm.proxy._types import SpendLogsPayload +from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler +from litellm.responses.utils import ResponsesAPIRequestUtils +from litellm.types.llms.openai import ( + AllMessageValues, + ChatCompletionResponseMessage, + GenericChatCompletionMessage, + ResponseInputParam, +) +from litellm.types.utils import ChatCompletionMessageToolCall, Message, ModelResponse + +if TYPE_CHECKING: + from litellm.responses.litellm_completion_transformation.transformation import ( + ChatCompletionSession, + ) +else: + ChatCompletionSession = Any + +######################################################## +# Cold Storage Handler +######################################################## +COLD_STORAGE_HANDLER = ColdStorageHandler() +######################################################## + +class ResponsesSessionHandler: + @staticmethod + async def get_chat_completion_message_history_for_previous_response_id( + previous_response_id: str, + ) -> ChatCompletionSession: + """ + Return the chat completion message history for a previous response id + """ + from litellm.responses.litellm_completion_transformation.transformation import ( + ChatCompletionSession, + ) + + verbose_proxy_logger.debug( + "inside get_chat_completion_message_history_for_previous_response_id" + ) + all_spend_logs: List[ + SpendLogsPayload + ] = await ResponsesSessionHandler.get_all_spend_logs_for_previous_response_id( + previous_response_id + ) + verbose_proxy_logger.debug( + "found %s spend logs for this response id", len(all_spend_logs) + ) + + litellm_session_id: Optional[str] = None + if len(all_spend_logs) > 0: + litellm_session_id = all_spend_logs[0].get("session_id") + + chat_completion_message_history: List[ + Union[ + AllMessageValues, + GenericChatCompletionMessage, + ChatCompletionMessageToolCall, + ChatCompletionResponseMessage, + Message, + ] + ] = [] + for spend_log in all_spend_logs: + chat_completion_message_history = await ResponsesSessionHandler.extend_chat_completion_message_with_spend_log_payload( + spend_log=spend_log, + chat_completion_message_history=chat_completion_message_history, + ) + + verbose_proxy_logger.debug( + "chat_completion_message_history %s", + json.dumps(chat_completion_message_history, indent=4, default=str), + ) + return ChatCompletionSession( + messages=chat_completion_message_history, + litellm_session_id=litellm_session_id, + ) + + @staticmethod + async def extend_chat_completion_message_with_spend_log_payload( + spend_log: SpendLogsPayload, + chat_completion_message_history: List[ + Union[ + AllMessageValues, + GenericChatCompletionMessage, + ChatCompletionMessageToolCall, + ChatCompletionResponseMessage, + Message, + ] + ] + ): + """ + Extend the chat completion message history with the spend log payload + """ + from litellm.responses.litellm_completion_transformation.transformation import ( + LiteLLMCompletionResponsesConfig, + ) + + proxy_server_request_dict = await ResponsesSessionHandler.get_proxy_server_request_from_spend_log( + spend_log=spend_log, + ) + response_input_param: Optional[Union[str, ResponseInputParam]] = None + _messages: Optional[Union[str, ResponseInputParam]] = None + + ############################################################ + # Add Input messages for this Spend Log + ############################################################ + if proxy_server_request_dict: + _response_input_param = proxy_server_request_dict.get("input", None) + _messages = proxy_server_request_dict.get("messages", None) + if isinstance(_response_input_param, str): + response_input_param = _response_input_param + elif isinstance(_response_input_param, dict): + response_input_param = cast( + ResponseInputParam, _response_input_param + ) + + if response_input_param: + chat_completion_messages = LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages( + input=response_input_param, + responses_api_request=proxy_server_request_dict or {}, + ) + chat_completion_message_history.extend(chat_completion_messages) + + ############################################################ + # Check if `messages` field is present in the proxy server request dict + ############################################################ + elif _messages: + # ensure all messages are /chat/completions/messages + # certain requests can be stored as Responses API format - this ensures they are transformed to /chat/completions/messages + chat_completion_messages = LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages( + input=_messages, + responses_api_request=proxy_server_request_dict or {}, + ) + chat_completion_message_history.extend(chat_completion_messages) + + ############################################################ + # Add Output messages for this Spend Log + ############################################################ + _response_output = spend_log.get("response", "{}") + if isinstance(_response_output, dict): + # transform `ChatCompletion Response` to `ResponsesAPIResponse` + model_response = ModelResponse(**_response_output) + for choice in model_response.choices: + if hasattr(choice, "message"): + chat_completion_message_history.append( + getattr(choice, "message") + ) + return chat_completion_message_history + + @staticmethod + async def get_proxy_server_request_from_spend_log( + spend_log: SpendLogsPayload, + ) -> Optional[dict]: + """ + Get the parsed proxy server request from the spend log + """ + proxy_server_request: Union[str, dict] = ( + spend_log.get("proxy_server_request") or "{}" + ) + proxy_server_request_dict: Optional[dict] = None + if isinstance(proxy_server_request, dict): + proxy_server_request_dict = proxy_server_request + else: + proxy_server_request_dict = json.loads(proxy_server_request) + + + ############################################################ + # Check if user has setup cold storage for session handling + ############################################################ + if ResponsesSessionHandler._should_check_cold_storage_for_full_payload(proxy_server_request_dict): + # Try to get cold storage object key from spend log metadata + _proxy_server_request_dict: Optional[dict] = None + cold_storage_object_key = ResponsesSessionHandler._get_cold_storage_object_key_from_spend_log(spend_log) + if cold_storage_object_key: + # Use the object key directly from metadata + _proxy_server_request_dict = await ResponsesSessionHandler.get_proxy_server_request_from_cold_storage_with_object_key( + object_key=cold_storage_object_key, + ) + if _proxy_server_request_dict: + proxy_server_request_dict = _proxy_server_request_dict + + return proxy_server_request_dict + + @staticmethod + def _get_cold_storage_object_key_from_spend_log(spend_log: SpendLogsPayload) -> Optional[str]: + """ + Extract the cold storage object key from spend log metadata. + + Args: + spend_log: The spend log payload containing metadata + + Returns: + Optional[str]: The cold storage object key if found, None otherwise + """ + try: + metadata_str = spend_log.get("metadata", "{}") + if isinstance(metadata_str, str): + metadata_dict = json.loads(metadata_str) + return metadata_dict.get("cold_storage_object_key") + elif isinstance(metadata_str, dict): + return metadata_str.get("cold_storage_object_key") + return None + except (json.JSONDecodeError, TypeError, AttributeError): + verbose_proxy_logger.debug("Failed to parse metadata from spend log to extract cold storage object key") + return None + + @staticmethod + async def get_proxy_server_request_from_cold_storage_with_object_key( + object_key: str, + ) -> Optional[dict]: + """ + Get the proxy server request from cold storage using the object key directly. + + Args: + object_key: The S3/GCS object key to retrieve + + Returns: + Optional[dict]: The proxy server request dict or None if not found + """ + verbose_proxy_logger.debug("inside get_proxy_server_request_from_cold_storage_with_object_key...") + + proxy_server_request_dict = await COLD_STORAGE_HANDLER.get_proxy_server_request_from_cold_storage_with_object_key( + object_key=object_key, + ) + + return proxy_server_request_dict + + @staticmethod + def _should_check_cold_storage_for_full_payload( + proxy_server_request_dict: Optional[dict], + ) -> bool: + """ + Only check cold storage when both are true + 1. `LITELLM_TRUNCATED_PAYLOAD_FIELD` is in the proxy server request dict + 2. `ColdStorageHandler._get_configured_cold_storage_custom_logger()` is not None + """ + from litellm.constants import LITELLM_TRUNCATED_PAYLOAD_FIELD + configured_cold_storage_custom_logger = ColdStorageHandler._get_configured_cold_storage_custom_logger() + if configured_cold_storage_custom_logger is None: + return False + if proxy_server_request_dict is None: + return True + if len(proxy_server_request_dict) == 0: + return True + if LITELLM_TRUNCATED_PAYLOAD_FIELD in proxy_server_request_dict: + return True + return False + + + + @staticmethod + async def get_all_spend_logs_for_previous_response_id( + previous_response_id: str, + ) -> List[SpendLogsPayload]: + """ + Get all spend logs for a previous response id + + + SQL query + + SELECT session_id FROM spend_logs WHERE response_id = previous_response_id, SELECT * FROM spend_logs WHERE session_id = session_id + """ + from litellm.proxy.proxy_server import prisma_client + + verbose_proxy_logger.debug("decoding response id=%s", previous_response_id) + + decoded_response_id = ( + ResponsesAPIRequestUtils._decode_responses_api_response_id( + previous_response_id + ) + ) + previous_response_id = decoded_response_id.get( + "response_id", previous_response_id + ) + if prisma_client is None: + return [] + + query = """ + WITH matching_session AS ( + SELECT session_id + FROM "LiteLLM_SpendLogs" + WHERE request_id = $1 + ) + SELECT * + FROM "LiteLLM_SpendLogs" + WHERE session_id IN (SELECT session_id FROM matching_session) + ORDER BY "endTime" ASC; + """ + + spend_logs = await prisma_client.db.query_raw(query, previous_response_id) + + verbose_proxy_logger.debug( + "Found the following spend logs for previous response id %s: %s", + previous_response_id, + json.dumps(spend_logs, indent=4, default=str), + ) + + return spend_logs diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index 13791666044..3706b2f7fc9 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -7,21 +7,14 @@ from typing import Any, Dict, List, Literal, Optional, Tuple, Union, cast from openai.types.responses.tool_param import FunctionToolParam from typing_extensions import TypedDict -from litellm._logging import verbose_logger - -try: - from litellm_enterprise.enterprise_callbacks.session_handler import ( - _ENTERPRISE_ResponsesSessionHandler, - ) -except Exception as e: - verbose_logger.debug( - f"[Non-Blocking] Unable to import _ENTERPRISE_ResponsesSessionHandler - LiteLLM Enterprise Feature - {str(e)}" - ) - _ENTERPRISE_ResponsesSessionHandler = None from litellm.caching import InMemoryCache from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj +from litellm.responses.litellm_completion_transformation.session_handler import ( + ResponsesSessionHandler, +) from litellm.types.llms.openai import ( AllMessageValues, + ChatCompletionImageObject, ChatCompletionImageUrlObject, ChatCompletionResponseMessage, ChatCompletionSystemMessage, @@ -75,13 +68,6 @@ class ChatCompletionSession(TypedDict, total=False): litellm_session_id: Optional[str] -class ChatCompletionImageItem(TypedDict): - """TypedDict for image items in chat completion content""" - - type: Literal["image"] - image_url: ChatCompletionImageUrlObject - - ########### End of Initialize Classes used for Responses API ########### @@ -210,20 +196,19 @@ class LiteLLMCompletionResponsesConfig: """ Async hook to get the chain of previous input and output pairs and return a list of Chat Completion messages """ - if _ENTERPRISE_ResponsesSessionHandler is not None: - chat_completion_session = ChatCompletionSession( - messages=[], litellm_session_id=None + chat_completion_session = ChatCompletionSession( + messages=[], litellm_session_id=None + ) + if previous_response_id: + chat_completion_session = await ResponsesSessionHandler.get_chat_completion_message_history_for_previous_response_id( + previous_response_id=previous_response_id ) - if previous_response_id: - chat_completion_session = await _ENTERPRISE_ResponsesSessionHandler.get_chat_completion_message_history_for_previous_response_id( - previous_response_id=previous_response_id - ) - _messages = litellm_completion_request.get("messages") or [] - session_messages = chat_completion_session.get("messages") or [] - litellm_completion_request["messages"] = session_messages + _messages - litellm_completion_request[ - "litellm_trace_id" - ] = chat_completion_session.get("litellm_session_id") + _messages = litellm_completion_request.get("messages") or [] + session_messages = chat_completion_session.get("messages") or [] + litellm_completion_request["messages"] = session_messages + _messages + litellm_completion_request[ + "litellm_trace_id" + ] = chat_completion_session.get("litellm_session_id") return litellm_completion_request @staticmethod @@ -485,7 +470,7 @@ class LiteLLMCompletionResponsesConfig: return new_item @staticmethod - def _transform_input_image_item_to_image_item(item: Dict[str, Any]) -> ChatCompletionImageItem: + def _transform_input_image_item_to_image_item(item: Dict[str, Any]) -> ChatCompletionImageObject: """ Transform a Responses API input_image item to a Chat Completion image item """ @@ -494,8 +479,8 @@ class LiteLLMCompletionResponsesConfig: detail=item.get("detail") or "auto" ) - return ChatCompletionImageItem( - type="image", + return ChatCompletionImageObject( + type="image_url", image_url=image_url_obj ) @@ -506,7 +491,6 @@ class LiteLLMCompletionResponsesConfig: """ Transform a Responses API content into a Chat Completion content """ - if isinstance(content, str): return content elif isinstance(content, list): diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 6606b0e0b89..75c7d28460b 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1910,6 +1910,7 @@ class StandardLoggingMetadata(StandardLoggingUserAPIKeyMetadata): vector_store_request_metadata: Optional[List[StandardLoggingVectorStoreRequest]] applied_guardrails: Optional[List[str]] usage_object: Optional[dict] + cold_storage_object_key: Optional[str] # S3/GCS object key for cold storage retrieval class StandardLoggingAdditionalHeaders(TypedDict, total=False): diff --git a/tests/logging_callback_tests/test_amazing_s3_logs.py b/tests/logging_callback_tests/test_amazing_s3_logs.py index 37666d72b79..c9d0987e86f 100644 --- a/tests/logging_callback_tests/test_amazing_s3_logs.py +++ b/tests/logging_callback_tests/test_amazing_s3_logs.py @@ -476,3 +476,18 @@ def test_s3_logging_r2(): # post, close log file and verify # Reset stdout to the original value print("Passed! Testing async s3 logging") + +from litellm.integrations.s3_v2 import S3Logger + +class TestS3Logger(S3Logger): + def __init__(self, *args, **kwargs): + self.recorded_requests = {} + self.logged_standard_logging_payload: Optional[StandardLoggingPayload] = None + super().__init__(*args, **kwargs) + + async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): + self.recorded_requests[response_obj["id"]] = start_time + print("recorded request", self.recorded_requests) + self.logged_standard_logging_payload = kwargs["standard_logging_object"] + return await super().async_log_success_event(kwargs, response_obj, start_time, end_time) + diff --git a/tests/logging_callback_tests/test_gcs_pub_sub.py b/tests/logging_callback_tests/test_gcs_pub_sub.py index f231d01d3fb..4172659e659 100644 --- a/tests/logging_callback_tests/test_gcs_pub_sub.py +++ b/tests/logging_callback_tests/test_gcs_pub_sub.py @@ -38,6 +38,7 @@ ignored_keys = [ "endTime", "metadata.model_map_information", "metadata.usage_object", + "metadata.cold_storage_object_key", ] diff --git a/tests/logging_callback_tests/test_otel_logging.py b/tests/logging_callback_tests/test_otel_logging.py index ff9d8300fef..aeb42bdaf79 100644 --- a/tests/logging_callback_tests/test_otel_logging.py +++ b/tests/logging_callback_tests/test_otel_logging.py @@ -279,6 +279,7 @@ def validate_redacted_message_span_attributes(span): "metadata.mcp_tool_call_metadata", "metadata.vector_store_request_metadata", "metadata.requester_custom_headers", + "metadata.cold_storage_object_key", ] _all_attributes = set( diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index e9b962f189f..317491d3172 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -394,3 +394,73 @@ def test_get_masked_values(): sensitive_object, unmasked_length=4, number_of_asterisks=4 ) assert masked_values["presidio_anonymizer_api_base"] is None + + +@pytest.mark.asyncio +async def test_e2e_generate_cold_storage_object_key_successful(): + """ + Test end-to-end generation of cold storage object key when cold storage is properly configured. + """ + from datetime import datetime, timezone + from unittest.mock import patch + + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + # Create test data + start_time = datetime(2025, 1, 15, 10, 30, 45, 123456, timezone.utc) + response_id = "chatcmpl-test-12345" + team_alias = "test-team" + + with patch("litellm.proxy.spend_tracking.cold_storage_handler.ColdStorageHandler._get_configured_cold_storage_custom_logger", return_value="s3"), \ + patch("litellm.integrations.s3.get_s3_object_key") as mock_get_s3_key: + + # Mock the S3 object key generation to return a predictable result + mock_get_s3_key.return_value = "2025-01-15/time-10-30-45-123456_chatcmpl-test-12345.json" + + # Call the function + result = StandardLoggingPayloadSetup._generate_cold_storage_object_key( + start_time=start_time, + response_id=response_id, + team_alias=team_alias + ) + + # Verify the S3 function was called with correct parameters + mock_get_s3_key.assert_called_once_with( + s3_path="", # Empty path as default + team_alias_prefix="", # No team alias prefix for cold storage + start_time=start_time, + s3_file_name="time-10-30-45-123456_chatcmpl-test-12345" + ) + + # Verify the result + assert result == "2025-01-15/time-10-30-45-123456_chatcmpl-test-12345.json" + assert result is not None + assert isinstance(result, str) + + +@pytest.mark.asyncio +async def test_e2e_generate_cold_storage_object_key_not_configured(): + """ + Test end-to-end generation of cold storage object key when cold storage is not configured. + """ + from datetime import datetime, timezone + from unittest.mock import patch + + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + # Create test data + start_time = datetime(2025, 1, 15, 10, 30, 45, 123456, timezone.utc) + response_id = "chatcmpl-test-67890" + team_alias = "another-team" + + with patch("litellm.proxy.spend_tracking.cold_storage_handler.ColdStorageHandler._get_configured_cold_storage_custom_logger", return_value=None): + + # Call the function + result = StandardLoggingPayloadSetup._generate_cold_storage_object_key( + start_time=start_time, + response_id=response_id, + team_alias=team_alias + ) + + # Verify the result is None when cold storage is not configured + assert result is None diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index 93be9717e96..5b1e784caa3 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -29,6 +29,7 @@ ignored_keys = [ "endTime", "metadata.model_map_information", "metadata.usage_object", + "metadata.cold_storage_object_key", ] diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py index 26f41b76d3b..eac631bdb0d 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py @@ -16,7 +16,7 @@ sys.path.insert( from unittest.mock import MagicMock, patch import litellm -from litellm.constants import REDACTED_BY_LITELM_STRING +from litellm.constants import LITELLM_TRUNCATED_PAYLOAD_FIELD, REDACTED_BY_LITELM_STRING from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.proxy.spend_tracking.spend_tracking_utils import ( _get_vector_store_request_for_spend_logs_payload, @@ -35,7 +35,7 @@ def test_sanitize_request_body_for_spend_logs_payload_long_string(): long_string = "a" * 2000 # Create a string longer than MAX_STRING_LENGTH request_body = {"text": long_string, "normal_text": "short text"} sanitized = _sanitize_request_body_for_spend_logs_payload(request_body) - assert len(sanitized["text"]) == 1000 + len("... (truncated 1000 chars)") + assert len(sanitized["text"]) == 1000 + len(f"... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} 1000 chars)") assert sanitized["normal_text"] == "short text" @@ -43,7 +43,7 @@ def test_sanitize_request_body_for_spend_logs_payload_nested_dict(): request_body = {"outer": {"inner": {"text": "a" * 2000, "normal": "short"}}} sanitized = _sanitize_request_body_for_spend_logs_payload(request_body) assert len(sanitized["outer"]["inner"]["text"]) == 1000 + len( - "... (truncated 1000 chars)" + f"... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} 1000 chars)" ) assert sanitized["outer"]["inner"]["normal"] == "short" @@ -54,11 +54,11 @@ def test_sanitize_request_body_for_spend_logs_payload_nested_list(): } sanitized = _sanitize_request_body_for_spend_logs_payload(request_body) assert len(sanitized["items"][0]["text"]) == 1000 + len( - "... (truncated 1000 chars)" + f"... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} 1000 chars)" ) assert sanitized["items"][1]["text"] == "short" assert len(sanitized["items"][2][0]["text"]) == 1000 + len( - "... (truncated 1000 chars)" + f"... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} 1000 chars)" ) @@ -81,14 +81,14 @@ def test_sanitize_request_body_for_spend_logs_payload_mixed_types(): "nested": {"list": ["short", "a" * 2000], "dict": {"key": "a" * 2000}}, } sanitized = _sanitize_request_body_for_spend_logs_payload(request_body) - assert len(sanitized["text"]) == 1000 + len("... (truncated 1000 chars)") + assert len(sanitized["text"]) == 1000 + len(f"... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} 1000 chars)") assert sanitized["number"] == 42 assert sanitized["nested"]["list"][0] == "short" assert len(sanitized["nested"]["list"][1]) == 1000 + len( - "... (truncated 1000 chars)" + f"... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} 1000 chars)" ) assert len(sanitized["nested"]["dict"]["key"]) == 1000 + len( - "... (truncated 1000 chars)" + f"... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} 1000 chars)" ) diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 00c55b9f60f..8e38011279d 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -134,9 +134,9 @@ class TestLiteLLMCompletionResponsesConfig: ) # Assert - expected = {"type": "image", "image_url": {"url": image_url, "detail": "high"}} + expected = {"type": "image_url", "image_url": {"url": image_url, "detail": "high"}} assert result == expected - assert result["type"] == "image" + assert result["type"] == "image_url" assert result["image_url"]["url"] == image_url assert result["image_url"]["detail"] == "high" @@ -154,9 +154,9 @@ class TestLiteLLMCompletionResponsesConfig: ) # Assert - expected = {"type": "image", "image_url": {"url": image_url, "detail": "high"}} + expected = {"type": "image_url", "image_url": {"url": image_url, "detail": "high"}} assert result == expected - assert result["type"] == "image" + assert result["type"] == "image_url" assert result["image_url"]["url"] == image_url assert result["image_url"]["detail"] == "high" @@ -174,9 +174,9 @@ class TestLiteLLMCompletionResponsesConfig: ) # Assert - expected = {"type": "image", "image_url": {"url": image_url, "detail": "auto"}} + expected = {"type": "image_url", "image_url": {"url": image_url, "detail": "auto"}} assert result == expected - assert result["type"] == "image" + assert result["type"] == "image_url" assert result["image_url"]["url"] == image_url assert result["image_url"]["detail"] == "auto" @@ -193,9 +193,9 @@ class TestLiteLLMCompletionResponsesConfig: ) # Assert - expected = {"type": "image", "image_url": {"url": "", "detail": "auto"}} + expected = {"type": "image_url", "image_url": {"url": "", "detail": "auto"}} assert result == expected - assert result["type"] == "image" + assert result["type"] == "image_url" assert result["image_url"]["url"] == "" assert result["image_url"]["detail"] == "auto" @@ -217,9 +217,9 @@ class TestLiteLLMCompletionResponsesConfig: ) # Assert - expected = {"type": "image", "image_url": {"url": "https://example.com/image.png", "detail": "auto"}} + expected = {"type": "image_url", "image_url": {"url": "https://example.com/image.png", "detail": "auto"}} assert result == expected - assert result["type"] == "image" + assert result["type"] == "image_url" assert result["image_url"]["url"] == "https://example.com/image.png" assert result["image_url"]["detail"] == "auto" assert "extra_field" not in result diff --git a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_session_handler.py b/tests/test_litellm/responses/litellm_completion_transformation/test_session_handler.py similarity index 58% rename from tests/enterprise/litellm_enterprise/enterprise_callbacks/test_session_handler.py rename to tests/test_litellm/responses/litellm_completion_transformation/test_session_handler.py index dbe163560b0..27edfef3eb5 100644 --- a/tests/enterprise/litellm_enterprise/enterprise_callbacks/test_session_handler.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_session_handler.py @@ -11,9 +11,9 @@ sys.path.insert( 0, os.path.abspath("../../..") ) # Adds the parent directory to the system path - -from enterprise.litellm_enterprise.enterprise_callbacks.session_handler import ( - _ENTERPRISE_ResponsesSessionHandler, +from litellm.responses.litellm_completion_transformation import session_handler +from litellm.responses.litellm_completion_transformation.session_handler import ( + ResponsesSessionHandler, ) @@ -111,7 +111,7 @@ async def test_get_chat_completion_message_history_for_previous_response_id(): # Mock the get_all_spend_logs_for_previous_response_id method with patch.object( - _ENTERPRISE_ResponsesSessionHandler, + ResponsesSessionHandler, "get_all_spend_logs_for_previous_response_id", new_callable=AsyncMock, ) as mock_get_spend_logs: @@ -119,7 +119,7 @@ async def test_get_chat_completion_message_history_for_previous_response_id(): # Test the function previous_response_id = "chatcmpl-935b8dad-fdc2-466e-a8ca-e26e5a8a21bb" - result = await _ENTERPRISE_ResponsesSessionHandler.get_chat_completion_message_history_for_previous_response_id( + result = await ResponsesSessionHandler.get_chat_completion_message_history_for_previous_response_id( previous_response_id ) @@ -166,17 +166,130 @@ async def test_get_chat_completion_message_history_empty_spend_logs(): Test get_chat_completion_message_history_for_previous_response_id with empty spend logs """ with patch.object( - _ENTERPRISE_ResponsesSessionHandler, + ResponsesSessionHandler, "get_all_spend_logs_for_previous_response_id", new_callable=AsyncMock, ) as mock_get_spend_logs: mock_get_spend_logs.return_value = [] previous_response_id = "non-existent-id" - result = await _ENTERPRISE_ResponsesSessionHandler.get_chat_completion_message_history_for_previous_response_id( + result = await ResponsesSessionHandler.get_chat_completion_message_history_for_previous_response_id( previous_response_id ) # Verify empty result structure assert result.get("messages") == [] assert result.get("litellm_session_id") is None + + +@pytest.mark.asyncio +async def test_e2e_cold_storage_successful_retrieval(): + """ + Test end-to-end cold storage functionality with successful retrieval of full proxy request from cold storage. + """ + # Mock spend logs with cold storage object key in metadata + mock_spend_logs = [ + { + "request_id": "chatcmpl-test-123", + "session_id": "session-456", + "metadata": '{"cold_storage_object_key": "s3://test-bucket/requests/session_456_req1.json"}', + "proxy_server_request": '{"litellm_truncated": true}', # Truncated payload + "response": { + "id": "chatcmpl-test-123", + "object": "chat.completion", + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "I am an AI assistant." + } + } + ] + } + } + ] + + # Full proxy request data from cold storage + full_proxy_request = { + "input": "Hello, who are you?", + "model": "gpt-4", + "messages": [{"role": "user", "content": "Hello, who are you?"}] + } + + with patch.object( + ResponsesSessionHandler, + "get_all_spend_logs_for_previous_response_id", + new_callable=AsyncMock, + ) as mock_get_spend_logs, \ + patch.object(session_handler, "COLD_STORAGE_HANDLER") as mock_cold_storage, \ + patch("litellm.proxy.spend_tracking.cold_storage_handler.ColdStorageHandler._get_configured_cold_storage_custom_logger", return_value="s3"): + + # Setup mocks + mock_get_spend_logs.return_value = mock_spend_logs + mock_cold_storage.get_proxy_server_request_from_cold_storage_with_object_key = AsyncMock(return_value=full_proxy_request) + + # Call the main function + result = await ResponsesSessionHandler.get_chat_completion_message_history_for_previous_response_id( + "chatcmpl-test-123" + ) + + # Verify cold storage was called with correct object key + mock_cold_storage.get_proxy_server_request_from_cold_storage_with_object_key.assert_called_once_with( + object_key="s3://test-bucket/requests/session_456_req1.json" + ) + + # Verify result structure + assert result.get("litellm_session_id") == "session-456" + assert len(result.get("messages", [])) >= 1 # At least the assistant response + + +@pytest.mark.asyncio +async def test_e2e_cold_storage_fallback_to_truncated_payload(): + """ + Test end-to-end cold storage functionality when object key is missing, falling back to truncated payload. + """ + # Mock spend logs without cold storage object key + mock_spend_logs = [ + { + "request_id": "chatcmpl-test-789", + "session_id": "session-999", + "metadata": '{"user_api_key": "test-key"}', # No cold storage object key + "proxy_server_request": '{"input": "Truncated message", "model": "gpt-4"}', # Regular payload + "response": { + "id": "chatcmpl-test-789", + "object": "chat.completion", + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "This is a response." + } + } + ] + } + } + ] + + with patch.object( + ResponsesSessionHandler, + "get_all_spend_logs_for_previous_response_id", + new_callable=AsyncMock, + ) as mock_get_spend_logs, \ + patch.object(session_handler, "COLD_STORAGE_HANDLER") as mock_cold_storage: + + # Setup mocks + mock_get_spend_logs.return_value = mock_spend_logs + + # Call the main function + result = await ResponsesSessionHandler.get_chat_completion_message_history_for_previous_response_id( + "chatcmpl-test-789" + ) + + # Verify cold storage was NOT called since no object key in metadata + mock_cold_storage.get_proxy_server_request_from_cold_storage_with_object_key.assert_not_called() + + # Verify result structure + assert result.get("litellm_session_id") == "session-999" + assert len(result.get("messages", [])) >= 1 # At least the assistant response diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_session_handler_with_cold_storage.py b/tests/test_litellm/responses/litellm_completion_transformation/test_session_handler_with_cold_storage.py new file mode 100644 index 00000000000..976152db353 --- /dev/null +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_session_handler_with_cold_storage.py @@ -0,0 +1,186 @@ +""" +Unit tests for cold storage object key integration. + +Tests for the changes to integrate cold storage handling across different components: +1. Add cold_storage_object_key field to StandardLoggingMetadata and SpendLogsMetadata +2. S3Logger generates object key when cold storage is enabled +3. Store object key in SpendLogsMetadata via spend_tracking_utils +4. Session handler uses object key from spend logs metadata +5. S3Logger supports retrieval using provided object key +""" + +import json +from datetime import datetime, timezone +from typing import Optional +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +from litellm.integrations.s3_v2 import S3Logger +from litellm.proxy._types import SpendLogsMetadata, SpendLogsPayload +from litellm.proxy.spend_tracking.cold_storage_handler import ColdStorageHandler +from litellm.proxy.spend_tracking.spend_tracking_utils import _get_spend_logs_metadata +from litellm.responses.litellm_completion_transformation.session_handler import ( + ResponsesSessionHandler, +) +from litellm.types.utils import StandardLoggingMetadata, StandardLoggingPayload + + +class TestColdStorageObjectKeyIntegration: + """Test suite for cold storage object key integration.""" + + def test_standard_logging_metadata_has_cold_storage_object_key_field(self): + """ + Test: Add cold_storage_object_key field to StandardLoggingMetadata. + + This test verifies that the StandardLoggingMetadata TypedDict has the + cold_storage_object_key field for storing S3/GCS object keys. + """ + from litellm.types.utils import StandardLoggingMetadata + + # Create a StandardLoggingMetadata instance with cold_storage_object_key + metadata = StandardLoggingMetadata( + user_api_key_hash="test_hash", + cold_storage_object_key="test/path/to/object.json" + ) + + # Verify the field can be set and accessed + assert metadata.get("cold_storage_object_key") == "test/path/to/object.json" + + assert "cold_storage_object_key" in StandardLoggingMetadata.__annotations__ + + def test_spend_logs_metadata_has_cold_storage_object_key_field(self): + """ + Test: Add cold_storage_object_key field to SpendLogsMetadata. + + This test verifies that the SpendLogsMetadata TypedDict has the + cold_storage_object_key field for storing S3/GCS object keys. + """ + # Create a SpendLogsMetadata instance with cold_storage_object_key + metadata = SpendLogsMetadata( + user_api_key="test_key", + cold_storage_object_key="test/path/to/object.json" + ) + + # Verify the field can be set and accessed + assert metadata.get("cold_storage_object_key") == "test/path/to/object.json" + + # Verify it's part of the SpendLogsMetadata annotations + assert "cold_storage_object_key" in SpendLogsMetadata.__annotations__ + + + def test_spend_tracking_utils_stores_object_key_in_metadata(self): + """ + Test: Store object key in SpendLogsMetadata via spend_tracking_utils. + + This test verifies that the _get_spend_logs_metadata function extracts + the cold_storage_object_key from StandardLoggingPayload and stores it + in SpendLogsMetadata. + """ + # Create test data + metadata = { + "user_api_key": "test_key", + "user_api_key_team_id": "test_team" + } + + + # Call the function + result = _get_spend_logs_metadata( + metadata=metadata, + cold_storage_object_key="test/path/to/object.json" + ) + + # Verify the object key is stored in the result + assert result.get("cold_storage_object_key") == "test/path/to/object.json" + + + def test_session_handler_extracts_object_key_from_spend_log(self): + """ + Test: Session handler extracts object key from spend logs metadata. + + This test verifies that the ResponsesSessionHandler can extract the + cold_storage_object_key from spend log metadata. + """ + # Create test spend log + spend_log = { + "request_id": "test_request_id", + "metadata": json.dumps({ + "cold_storage_object_key": "test/path/to/object.json", + "user_api_key": "test_key" + }) + } + + # Test the extraction method + object_key = ResponsesSessionHandler._get_cold_storage_object_key_from_spend_log(spend_log) + + assert object_key == "test/path/to/object.json" + + def test_session_handler_handles_dict_metadata_in_spend_log(self): + """ + Test: Session handler handles dict metadata in spend log. + + This test verifies that the method works when metadata is already a dict. + """ + # Create test spend log with dict metadata + spend_log = { + "request_id": "test_request_id", + "metadata": { + "cold_storage_object_key": "test/path/to/object.json", + "user_api_key": "test_key" + } + } + + # Test the extraction method + object_key = ResponsesSessionHandler._get_cold_storage_object_key_from_spend_log(spend_log) + + assert object_key == "test/path/to/object.json" + + + @pytest.mark.asyncio + async def test_cold_storage_handler_supports_object_key_retrieval(self): + """ + Test: ColdStorageHandler supports object key retrieval. + + This test verifies that the ColdStorageHandler has the new method + for retrieving objects using object keys directly. + """ + handler = ColdStorageHandler() + + # Mock the custom logger + mock_logger = AsyncMock() + mock_logger.get_proxy_server_request_from_cold_storage_with_object_key = AsyncMock( + return_value={"test": "data"} + ) + + with patch.object(handler, '_select_custom_logger_for_cold_storage', return_value="s3_v2"), \ + patch('litellm.logging_callback_manager.get_active_custom_logger_for_callback_name', return_value=mock_logger): + + result = await handler.get_proxy_server_request_from_cold_storage_with_object_key( + object_key="test/path/to/object.json" + ) + + assert result == {"test": "data"} + mock_logger.get_proxy_server_request_from_cold_storage_with_object_key.assert_called_once_with( + object_key="test/path/to/object.json" + ) + + @pytest.mark.asyncio + @patch('asyncio.create_task') # Mock asyncio.create_task to avoid event loop issues + async def test_s3_logger_supports_object_key_retrieval(self, mock_create_task): + """ + Test: S3Logger supports retrieval using provided object key. + + This test verifies that the S3Logger can retrieve objects using + the object key directly without generating it from request_id and start_time. + """ + # Create S3Logger instance + s3_logger = S3Logger(s3_bucket_name="test-bucket") + + # Mock the _download_object_from_s3 method + with patch.object(s3_logger, '_download_object_from_s3', return_value={"test": "data"}) as mock_download: + result = await s3_logger.get_proxy_server_request_from_cold_storage_with_object_key( + object_key="test/path/to/object.json" + ) + + assert result == {"test": "data"} + mock_download.assert_called_once_with("test/path/to/object.json") \ No newline at end of file From 9e0ba10f2399244681d7e4b185707b4043b04fb4 Mon Sep 17 00:00:00 2001 From: Low Jian Sheng <15527690+lowjiansheng@users.noreply.github.com> Date: Fri, 8 Aug 2025 02:36:16 +0800 Subject: [PATCH 10/21] Add GPT 5 models (#13377) * add gpt 5 modesl * update max tokens --- model_prices_and_context_window.json | 54 ++++++++++++++++++++++++++++ 1 file changed, 54 insertions(+) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6ce46e5962f..b4ff923c1e2 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -296,6 +296,60 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "gpt-5-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "gpt-5-mini-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "gpt-5-nano-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "watsonx/ibm/granite-3-8b-instruct": { "max_tokens": 8192, "max_input_tokens": 8192, From 729e1f530a141f5f275b2b8b5be115600997d9df Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 11:52:44 -0700 Subject: [PATCH 11/21] feat - add claude-opus-4-1 on cost map (#13384) --- ...odel_prices_and_context_window_backup.json | 211 ++++++++++++++++++ model_prices_and_context_window.json | 26 +++ 2 files changed, 237 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 8281cb424fd..5723e735597 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -296,6 +296,60 @@ "supports_system_messages": true, "supports_tool_choice": true }, + "gpt-5-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "gpt-5-mini-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "gpt-5-nano-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "watsonx/ibm/granite-3-8b-instruct": { "max_tokens": 8192, "max_input_tokens": 8192, @@ -5771,6 +5825,32 @@ "supports_reasoning": true, "supports_computer_use": true }, + "claude-opus-4-1": { + "max_tokens": 32000, + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 7.5e-05, + "search_context_cost_per_query": { + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01, + "search_context_size_high": 0.01 + }, + "cache_creation_input_token_cost": 1.875e-05, + "cache_read_input_token_cost": 1.5e-06, + "litellm_provider": "anthropic", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 159, + "supports_assistant_prefill": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_computer_use": true + }, "claude-opus-4-1-20250805": { "max_tokens": 32000, "max_input_tokens": 200000, @@ -17693,5 +17773,136 @@ "supports_vision": false, "supports_system_messages": true, "supports_tool_choice": false + }, + "oci/meta.llama-4-maverick-17b-128e-instruct-fp8": { + "max_tokens": 512000, + "max_input_tokens": 512000, + "max_output_tokens": 4000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/meta.llama-4-scout-17b-16e-instruct": { + "max_tokens": 192000, + "max_input_tokens": 192000, + "max_output_tokens": 4000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/meta.llama-3.3-70b-instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 4000, + "input_cost_per_token": 7.2e-07, + "output_cost_per_token": 7.2e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/meta.llama-3.2-90b-vision-instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 4000, + "input_cost_per_token": 2.0e-06, + "output_cost_per_token": 2.0e-06, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/meta.llama-3.1-405b-instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 4000, + "input_cost_per_token": 1.068e-05, + "output_cost_per_token": 1.068e-05, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + + "oci/xai.grok-4": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 3.0e-06, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/xai.grok-3": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.0e-06, + "output_cost_per_token": 1.5e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/xai.grok-3-mini": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 3.0e-07, + "output_cost_per_token": 5.0e-07, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/xai.grok-3-fast": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 5.0e-06, + "output_cost_per_token": 2.5e-05, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" + }, + "oci/xai.grok-3-mini-fast": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 6.0e-07, + "output_cost_per_token": 4.0e-06, + "litellm_provider": "oci", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": false, + "supports_tool_choice": false, + "source": "https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing" } } diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index b4ff923c1e2..5723e735597 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -5825,6 +5825,32 @@ "supports_reasoning": true, "supports_computer_use": true }, + "claude-opus-4-1": { + "max_tokens": 32000, + "max_input_tokens": 200000, + "max_output_tokens": 32000, + "input_cost_per_token": 1.5e-05, + "output_cost_per_token": 7.5e-05, + "search_context_cost_per_query": { + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01, + "search_context_size_high": 0.01 + }, + "cache_creation_input_token_cost": 1.875e-05, + "cache_read_input_token_cost": 1.5e-06, + "litellm_provider": "anthropic", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "tool_use_system_prompt_tokens": 159, + "supports_assistant_prefill": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_computer_use": true + }, "claude-opus-4-1-20250805": { "max_tokens": 32000, "max_input_tokens": 200000, From 087a1a622ce4f801c05774abf4f60461fd1565bc Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 11:58:37 -0700 Subject: [PATCH 12/21] =?UTF-8?q?feat:=20Add=20GPT-5=20model=20family=20wi?= =?UTF-8?q?th=20official=20OpenAI=20specifications=20(#13=E2=80=A6=20(#133?= =?UTF-8?q?86)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat: Add GPT-5 model family with official OpenAI specifications (#13378) * Add GPT-5 model family support Added four new GPT-5 models: - gpt-5: Flagship model for logic and multi-step tasks - gpt-5-mini: Cost-sensitive version for budget use cases - gpt-5-nano: Speed-optimized version for low latency - gpt-5-chat: Enterprise-focused version for advanced conversations * Update GPT-5 models with official OpenAI specifications - Add gpt-5-chat-latest with 400k context, 128k output tokens - Add gpt-5-2025-08-07 with enhanced reasoning capabilities - Add gpt-5-mini-2025-08-07 with cost-optimized pricing - Add gpt-5-nano-2025-08-07 with ultra-fast performance - Update existing gpt-5, gpt-5-mini, gpt-5-nano to match dated versions - All models now support reasoning tokens and 400k context window - Pricing updated per official OpenAI documentation * fix conflicts --------- Co-authored-by: Cole McIntosh <82463175+colesmcintosh@users.noreply.github.com> --- ...odel_prices_and_context_window_backup.json | 311 +++++++++++++++--- model_prices_and_context_window.json | 311 +++++++++++++++--- 2 files changed, 514 insertions(+), 108 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 5723e735597..ca20baf01ab 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -296,60 +296,6 @@ "supports_system_messages": true, "supports_tool_choice": true }, - "gpt-5-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 1.25e-06, - "output_cost_per_token": 1e-05, - "cache_read_input_token_cost": 1.25e-07, - "litellm_provider": "openai", - "mode": "chat", - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true - }, - "gpt-5-mini-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 2.5e-07, - "output_cost_per_token": 2e-06, - "cache_read_input_token_cost": 2.5e-08, - "litellm_provider": "openai", - "mode": "chat", - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true - }, - "gpt-5-nano-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 4e-07, - "cache_read_input_token_cost": 5e-09, - "litellm_provider": "openai", - "mode": "chat", - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true - }, "watsonx/ibm/granite-3-8b-instruct": { "max_tokens": 8192, "max_input_tokens": 8192, @@ -666,6 +612,263 @@ "search_context_size_high": 0.03 } }, + "gpt-5": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-mini": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-nano": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-chat": { + "max_tokens": 32768, + "max_input_tokens": 1047576, + "max_output_tokens": 32768, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "input_cost_per_token_batches": 2.5e-06, + "output_cost_per_token_batches": 1e-05, + "cache_read_input_token_cost": 1.25e-06, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true + }, + "gpt-5-chat-latest": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-mini-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-nano-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, "codex-mini-latest": { "max_tokens": 100000, "max_input_tokens": 200000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 5723e735597..ca20baf01ab 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -296,60 +296,6 @@ "supports_system_messages": true, "supports_tool_choice": true }, - "gpt-5-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 1.25e-06, - "output_cost_per_token": 1e-05, - "cache_read_input_token_cost": 1.25e-07, - "litellm_provider": "openai", - "mode": "chat", - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true - }, - "gpt-5-mini-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 2.5e-07, - "output_cost_per_token": 2e-06, - "cache_read_input_token_cost": 2.5e-08, - "litellm_provider": "openai", - "mode": "chat", - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true - }, - "gpt-5-nano-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 4e-07, - "cache_read_input_token_cost": 5e-09, - "litellm_provider": "openai", - "mode": "chat", - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true - }, "watsonx/ibm/granite-3-8b-instruct": { "max_tokens": 8192, "max_input_tokens": 8192, @@ -666,6 +612,263 @@ "search_context_size_high": 0.03 } }, + "gpt-5": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-mini": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-nano": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-chat": { + "max_tokens": 32768, + "max_input_tokens": 1047576, + "max_output_tokens": 32768, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "input_cost_per_token_batches": 2.5e-06, + "output_cost_per_token_batches": 1e-05, + "cache_read_input_token_cost": 1.25e-06, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true + }, + "gpt-5-chat-latest": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-mini-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-nano-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, "codex-mini-latest": { "max_tokens": 100000, "max_input_tokens": 200000, From 2e767c8faf26f73521c6fe7506d3a0296f65fd6d Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 12:50:37 -0700 Subject: [PATCH 13/21] [Feat] add azure/gpt-5 model family (#13385) * add azure/gpt-5 model family * add azure gpt-5 model family * fixes for gpt-5 * add azure/gpt-5 model family --- ...odel_prices_and_context_window_backup.json | 514 ++++++++++++++++++ model_prices_and_context_window.json | 514 ++++++++++++++++++ 2 files changed, 1028 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ca20baf01ab..51ec48bcbcc 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -2264,6 +2264,520 @@ "/v1/audio/speech" ] }, + "gpt-5": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-mini": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-mini-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-nano-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-nano": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-chat": { + "max_tokens": 32768, + "max_input_tokens": 1047576, + "max_output_tokens": 32768, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "input_cost_per_token_batches": 2.5e-06, + "output_cost_per_token_batches": 1e-05, + "cache_read_input_token_cost": 1.25e-06, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true + }, + "gpt-5-chat-latest": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-mini": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-mini-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-nano-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-nano": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-chat": { + "max_tokens": 32768, + "max_input_tokens": 1047576, + "max_output_tokens": 32768, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "input_cost_per_token_batches": 2.5e-06, + "output_cost_per_token_batches": 1e-05, + "cache_read_input_token_cost": 1.25e-06, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true + }, + "azure/gpt-5-chat-latest": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, "azure/gpt-4o-mini-tts": { "mode": "audio_speech", "input_cost_per_token": 2.5e-06, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ca20baf01ab..51ec48bcbcc 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -2264,6 +2264,520 @@ "/v1/audio/speech" ] }, + "gpt-5": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-mini": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-mini-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-nano-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-nano": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "gpt-5-chat": { + "max_tokens": 32768, + "max_input_tokens": 1047576, + "max_output_tokens": 32768, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "input_cost_per_token_batches": 2.5e-06, + "output_cost_per_token_batches": 1e-05, + "cache_read_input_token_cost": 1.25e-06, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true + }, + "gpt-5-chat-latest": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "openai", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-mini": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-mini-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.5e-07, + "output_cost_per_token": 2e-06, + "cache_read_input_token_cost": 2.5e-08, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-nano-2025-08-07": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-nano": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 4e-07, + "cache_read_input_token_cost": 5e-09, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, + "azure/gpt-5-chat": { + "max_tokens": 32768, + "max_input_tokens": 1047576, + "max_output_tokens": 32768, + "input_cost_per_token": 5e-06, + "output_cost_per_token": 2e-05, + "input_cost_per_token_batches": 2.5e-06, + "output_cost_per_token_batches": 1e-05, + "cache_read_input_token_cost": 1.25e-06, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true + }, + "azure/gpt-5-chat-latest": { + "max_tokens": 128000, + "max_input_tokens": 400000, + "max_output_tokens": 128000, + "input_cost_per_token": 1.25e-06, + "output_cost_per_token": 1e-05, + "cache_read_input_token_cost": 1.25e-07, + "litellm_provider": "azure", + "mode": "chat", + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_pdf_input": true, + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_native_streaming": true, + "supports_reasoning": true + }, "azure/gpt-4o-mini-tts": { "mode": "audio_speech", "input_cost_per_token": 2.5e-06, From 08ac2aeb6d9fe635fa5198ac813aa79715fbd9dd Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 13:13:05 -0700 Subject: [PATCH 14/21] Revert "Fix SSO Logout | Create Unified Login Page with SSO and Username/Password Options (#12703)" (#13387) This reverts commit a752d7acc9f9db145d0b1d49ddb53263b67d0b31. --- litellm/proxy/management_endpoints/ui_sso.py | 472 +++--------------- .../proxy/management_endpoints/test_ui_sso.py | 10 +- 2 files changed, 68 insertions(+), 414 deletions(-) diff --git a/litellm/proxy/management_endpoints/ui_sso.py b/litellm/proxy/management_endpoints/ui_sso.py index ee38eb6515d..451f110a894 100644 --- a/litellm/proxy/management_endpoints/ui_sso.py +++ b/litellm/proxy/management_endpoints/ui_sso.py @@ -51,6 +51,7 @@ from litellm.proxy.common_utils.admin_ui_utils import ( from litellm.proxy.common_utils.html_forms.jwt_display_template import ( jwt_display_template, ) +from litellm.proxy.common_utils.html_forms.ui_login import html_form from litellm.proxy.management_endpoints.internal_user_endpoints import new_user from litellm.proxy.management_endpoints.sso_helper_utils import ( check_is_admin_only_access, @@ -76,20 +77,16 @@ router = APIRouter() @router.get("/sso/key/generate", tags=["experimental"], include_in_schema=False) -async def serve_login_page( - request: Request, - source: Optional[str] = None, - key: Optional[str] = None, - error: Optional[str] = None, -): +async def google_login(request: Request, source: Optional[str] = None, key: Optional[str] = None): # noqa: PLR0915 """ Create Proxy API Keys using Google Workspace SSO. Requires setting PROXY_BASE_URL in .env PROXY_BASE_URL should be the your deployed proxy endpoint, e.g. PROXY_BASE_URL="https://litellm-production-7002.up.railway.app/" Example: - Serves a unified login page with options for both normal - username/password login and SSO. """ - from litellm.proxy.proxy_server import premium_user + from litellm.proxy.proxy_server import ( + premium_user, + user_custom_ui_sso_sign_in_handler, + ) microsoft_client_id = os.getenv("MICROSOFT_CLIENT_ID", None) google_client_id = os.getenv("GOOGLE_CLIENT_ID", None) @@ -102,334 +99,6 @@ async def serve_login_page( if is_disabled: return admin_ui_disabled() - ####### Check if user is a Enterprise / Premium User for SSO ####### - sso_available = False - if ( - microsoft_client_id is not None - or google_client_id is not None - or generic_client_id is not None - ): - if premium_user is True: - sso_available = True - - ####### Detect DB + MASTER KEY in .env ####### - missing_env_vars = show_missing_vars_in_env() - if missing_env_vars is not None: - return missing_env_vars - ######################################################### - # Construct Redirect URL - base_url_to_redirect_to: Optional[str] = None - base_url_to_redirect_to = os.getenv("PROXY_BASE_URL", "") - server_root_path = os.getenv("SERVER_ROOT_PATH", "") - if server_root_path != "": - base_url_to_redirect_to += server_root_path - ######################################################### - - # Build the unified login page HTML - error_message = "" - if error == "1": - error_message = """ -
- ⚠️ Invalid username or password. Please try again. -
- """ - - sso_button = "" - if sso_available: - sso_login_url = base_url_to_redirect_to - if sso_login_url.endswith("/"): - sso_login_url += "sso/login" - else: - sso_login_url += "/sso/login" - - sso_button = f""" - - """ - - if base_url_to_redirect_to.endswith("/"): - url_to_redirect_to = base_url_to_redirect_to + "login" - else: - url_to_redirect_to = base_url_to_redirect_to + "/login" - - unified_login_html = f""" - - - - - LiteLLM Login - - - - -
-
- -
-

Login

-

Access your LiteLLM Admin UI.

- - {error_message} - -
-
- - - - - - Default Credentials -
-

By default, Username is admin and Password is your set LiteLLM Proxy MASTER_KEY.

-

Need to set UI credentials or SSO? Check the documentation.

-
- - - - - - -
- - -
- - - {sso_button} -
- - - - """ - - from fastapi.responses import HTMLResponse - - return HTMLResponse(content=unified_login_html, status_code=200) - - -@router.get("/sso/login", tags=["experimental"], include_in_schema=False) -async def sso_login_redirect( - request: Request, source: Optional[str] = None, key: Optional[str] = None -): - """ - Handles SSO login redirect - this is what the "Login with SSO" button points to - """ - from litellm.proxy.proxy_server import ( - premium_user, - user_custom_ui_sso_sign_in_handler, - ) - - microsoft_client_id = os.getenv("MICROSOFT_CLIENT_ID", None) - google_client_id = os.getenv("GOOGLE_CLIENT_ID", None) - generic_client_id = os.getenv("GENERIC_CLIENT_ID", None) - ####### Check if user is a Enterprise / Premium User ####### if ( microsoft_client_id is not None @@ -444,12 +113,18 @@ async def sso_login_redirect( code=status.HTTP_403_FORBIDDEN, ) + ####### Detect DB + MASTER KEY in .env ####### + missing_env_vars = show_missing_vars_in_env() + if missing_env_vars is not None: + return missing_env_vars + ui_username = os.getenv("UI_USERNAME") + # get url from request - always use regular callback, but set state for CLI redirect_url = SSOAuthenticationHandler.get_redirect_url_for_sso( request=request, sso_callback_route="sso/callback", ) - + # Store CLI key in state for OAuth flow cli_state: Optional[str] = SSOAuthenticationHandler._get_cli_state( source=source, @@ -462,14 +137,11 @@ async def sso_login_redirect( from litellm_enterprise.proxy.auth.custom_sso_handler import ( EnterpriseCustomSSOHandler, ) - return await EnterpriseCustomSSOHandler.handle_custom_ui_sso_sign_in( request=request, ) except ImportError: - raise ValueError( - "Enterprise features are not available. Custom UI SSO sign-in requires LiteLLM Enterprise." - ) + raise ValueError("Enterprise features are not available. Custom UI SSO sign-in requires LiteLLM Enterprise.") # Check if we should use SSO handler if ( @@ -488,9 +160,16 @@ async def sso_login_redirect( generic_client_id=generic_client_id, state=cli_state, ) + elif ui_username is not None: + # No Google, Microsoft SSO + # Use UI Credentials set in .env + from fastapi.responses import HTMLResponse + + return HTMLResponse(content=html_form, status_code=200) else: - # No SSO configured, redirect back to login page - return RedirectResponse(url="/sso/key/generate", status_code=303) + from fastapi.responses import HTMLResponse + + return HTMLResponse(content=html_form, status_code=200) def generic_response_convertor( @@ -846,16 +525,15 @@ async def check_and_update_if_proxy_admin_id( async def auth_callback(request: Request, state: Optional[str] = None): # noqa: PLR0915 """Verify login""" verbose_proxy_logger.info(f"Starting SSO callback with state: {state}") - + # Check if this is a CLI login (state starts with our CLI prefix) from litellm.constants import LITELLM_CLI_SESSION_TOKEN_PREFIX - if state and state.startswith(f"{LITELLM_CLI_SESSION_TOKEN_PREFIX}:"): # Extract the key ID from the state key_id = state.split(":", 1)[1] verbose_proxy_logger.info(f"CLI SSO callback detected for key: {key_id}") return await cli_sso_callback(request, key=key_id) - + from litellm.proxy._types import LiteLLM_JWTAuth from litellm.proxy.auth.handle_jwt import JWTHandler from litellm.proxy.proxy_server import ( @@ -930,7 +608,7 @@ async def auth_callback(request: Request, state: Optional[str] = None): # noqa: status_code=401, detail="Result not returned by SSO provider.", ) - + return await SSOAuthenticationHandler.get_redirect_response_from_openid( result=result, request=request, @@ -940,26 +618,28 @@ async def auth_callback(request: Request, state: Optional[str] = None): # noqa: ) + + async def cli_sso_callback(request: Request, key: Optional[str] = None): """CLI SSO callback - generates the key with pre-specified ID""" verbose_proxy_logger.info(f"CLI SSO callback for key: {key}") - + from litellm.proxy.management_endpoints.key_management_endpoints import ( generate_key_helper_fn, ) from litellm.proxy.proxy_server import prisma_client - - if not key or not key.startswith("sk-"): + + if not key or not key.startswith('sk-'): raise HTTPException( status_code=400, - detail="Invalid key parameter. Must be a valid key ID starting with 'sk-'", + detail="Invalid key parameter. Must be a valid key ID starting with 'sk-'" ) - + if prisma_client is None: raise HTTPException( status_code=500, detail=CommonProxyErrors.db_not_connected_error.value ) - + # Generate a simple key for CLI usage with the pre-specified key ID try: await generate_key_helper_fn( @@ -973,57 +653,63 @@ async def cli_sso_callback(request: Request, key: Optional[str] = None): table_name="key", token=key, # Use the pre-specified key ID ) - + verbose_proxy_logger.info(f"Generated CLI key: {key}") - + # Return success page from fastapi.responses import HTMLResponse from litellm.proxy.common_utils.html_forms.cli_sso_success import ( render_cli_sso_success_page, ) - + html_content = render_cli_sso_success_page() return HTMLResponse(content=html_content, status_code=200) - + except Exception as e: verbose_proxy_logger.error(f"Error generating CLI key: {e}") - raise HTTPException(status_code=500, detail=f"Failed to generate key: {str(e)}") + raise HTTPException( + status_code=500, + detail=f"Failed to generate key: {str(e)}" + ) @router.get("/sso/cli/poll/{key_id}", tags=["experimental"], include_in_schema=False) async def cli_poll_key(key_id: str): """CLI polling endpoint - checks if key exists in DB""" from litellm.proxy.proxy_server import prisma_client - - if not key_id.startswith("sk-"): - raise HTTPException(status_code=400, detail="Invalid key ID format") - + + if not key_id.startswith('sk-'): + raise HTTPException( + status_code=400, + detail="Invalid key ID format" + ) + if prisma_client is None: raise HTTPException( status_code=500, detail=CommonProxyErrors.db_not_connected_error.value ) - + try: # Check if key exists in database from litellm.proxy.utils import hash_token - hashed_token = hash_token(key_id) - + key_obj = await prisma_client.db.litellm_verificationtoken.find_unique( where={"token": hashed_token} ) - + if key_obj: verbose_proxy_logger.info(f"CLI key found: {key_id}") return {"status": "ready", "key": key_id} else: return {"status": "pending"} - + except Exception as e: verbose_proxy_logger.error(f"Error polling for CLI key: {e}") raise HTTPException( - status_code=500, detail=f"Error checking key status: {str(e)}" + status_code=500, + detail=f"Error checking key status: {str(e)}" ) @@ -1125,7 +811,6 @@ class SSOAuthenticationHandler: """ Handler for SSO Authentication across all SSO providers """ - @staticmethod async def get_sso_login_redirect( redirect_url: str, @@ -1478,6 +1163,7 @@ class SSOAuthenticationHandler: _new_team_request.update(_default_team_params) team_request = NewTeamRequest(**_new_team_request) return team_request + @staticmethod def _get_cli_state(source: Optional[str], key: Optional[str]) -> Optional[str]: @@ -1490,15 +1176,13 @@ class SSOAuthenticationHandler: LITELLM_CLI_SESSION_TOKEN_PREFIX, LITELLM_CLI_SOURCE_IDENTIFIER, ) + return f"{LITELLM_CLI_SESSION_TOKEN_PREFIX}:{key}" if source == LITELLM_CLI_SOURCE_IDENTIFIER and key else None + + - return ( - f"{LITELLM_CLI_SESSION_TOKEN_PREFIX}:{key}" - if source == LITELLM_CLI_SOURCE_IDENTIFIER and key - else None - ) @staticmethod - async def get_redirect_response_from_openid( # noqa: PLR0915 + async def get_redirect_response_from_openid( # noqa: PLR0915 result: Union[OpenID, dict, CustomOpenID], request: Request, received_response: Optional[dict] = None, @@ -1518,18 +1202,14 @@ class SSOAuthenticationHandler: ) from litellm.proxy.utils import get_prisma_client_or_throw from litellm.types.proxy.ui_sso import ReturnedUITokenObject + prisma_client = get_prisma_client_or_throw("Prisma client is None, connect a database to your proxy") - prisma_client = get_prisma_client_or_throw( - "Prisma client is None, connect a database to your proxy" - ) # User is Authe'd in - generate key for the UI to access Proxy verbose_proxy_logger.info(f"SSO callback result: {result}") user_email: Optional[str] = getattr(result, "email", None) - user_id: Optional[str] = ( - getattr(result, "id", None) if result is not None else None - ) + user_id: Optional[str] = getattr(result, "id", None) if result is not None else None if user_email is not None and os.getenv("ALLOWED_EMAIL_DOMAINS") is not None: email_domain = user_email.split("@")[1] @@ -1714,8 +1394,7 @@ class SSOAuthenticationHandler: redirect_response = RedirectResponse(url=litellm_dashboard_ui, status_code=303) redirect_response.set_cookie(key="token", value=jwt_token) return redirect_response - - + class MicrosoftSSOHandler: """ Handles Microsoft SSO callback response and returns a CustomOpenID object @@ -2215,28 +1894,3 @@ async def debug_sso_callback(request: Request): ) return HTMLResponse(content=html_content) - - -@router.post("/sso/key/generate", tags=["experimental"], include_in_schema=False) -async def process_login(request: Request): - """ - Process username/password login from the unified login page - """ - try: - # Get form data - form_data = await request.form() - username = form_data.get("username") - password = form_data.get("password") - - if not username or not password: - return RedirectResponse(url="/sso/key/generate?error=1", status_code=303) - - # Import the actual login function from proxy_server - from litellm.proxy.proxy_server import login - - # Call the real login function that handles all the authentication properly - return await login(request) - - except Exception as e: - verbose_proxy_logger.error(f"Error processing login: {e}") - return RedirectResponse(url="/sso/key/generate?error=1", status_code=303) diff --git a/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py b/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py index 245f350be1b..f1565fd55fd 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py +++ b/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py @@ -938,10 +938,10 @@ class TestUISSO_FunctionsExistence: from litellm.proxy.management_endpoints.ui_sso import auth_callback assert callable(auth_callback) - def test_sso_login_redirect_exists(self): - """Test that sso_login_redirect function exists""" - from litellm.proxy.management_endpoints.ui_sso import sso_login_redirect - assert callable(sso_login_redirect) + def test_google_login_exists(self): + """Test that google_login function exists""" + from litellm.proxy.management_endpoints.ui_sso import google_login + assert callable(google_login) def test_sso_authentication_handler_exists(self): """Test that SSOAuthenticationHandler class exists with new methods""" @@ -1054,7 +1054,7 @@ class TestCustomUISSO: """Test that proper error is raised when enterprise module is not available""" from unittest.mock import MagicMock, patch - from litellm.proxy.management_endpoints.ui_sso import sso_login_redirect + from litellm.proxy.management_endpoints.ui_sso import google_login # Mock request mock_request = MagicMock() From b7ced315dd73288c97c3d42a841b614b2ff2ed9b Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 13:18:45 -0700 Subject: [PATCH 15/21] fix - publish new PIP / prisma migrations --- ...litellm_proxy_extras-0.2.16-py3-none-any.whl | Bin 0 -> 29772 bytes .../dist/litellm_proxy_extras-0.2.16.tar.gz | Bin 0 -> 15260 bytes litellm-proxy-extras/pyproject.toml | 4 ++-- poetry.lock | 7 +++---- pyproject.toml | 2 +- requirements.txt | 2 +- 6 files changed, 7 insertions(+), 8 deletions(-) create mode 100644 litellm-proxy-extras/dist/litellm_proxy_extras-0.2.16-py3-none-any.whl create mode 100644 litellm-proxy-extras/dist/litellm_proxy_extras-0.2.16.tar.gz diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.2.16-py3-none-any.whl b/litellm-proxy-extras/dist/litellm_proxy_extras-0.2.16-py3-none-any.whl new file mode 100644 index 0000000000000000000000000000000000000000..ce275d59451dfa33cded559350b4e94c88c45125 GIT binary patch literal 29772 zcmb5W1yG#b(lv^^ySwY)5Fog_JA=DR@Zc`NT@u{gf?IHRO$hF;A%Q!|d%pAE99B7eCgWkX(e!hWTWT0hd?PhLgXAg98arE>8 zntQsr7`w6pfz}SzZa^T5lNThI@^8*PmRBk-q# z^8f=_4g?88cq=ESN~22IPG;0Z`2H^;wrcA8$&gXLhl|*A5?7TLWeG}K5cpx(Ozs0q zOLMW`rAQ;ZOyd6rr@ghMi?N%vqk}6eJAj=Fz`@SV#?8*g#sM@jb~U%Nb}(oC_ph+H zI@|5)ZaA)Sqxnw($#;rS;*p^576s{>RXsIM>RPHPjyWAn4j`k(JrzT`{m zCT}-7r-al;6Tfz*%LZ_P}1J@Ow&z6Mk+e*StM{djVf zjmfl(A&ee}c=^Bvp>~H`8+mrKs@uAEJ8`hokT^KCfDhN-6c!WdZq8!AgvUxel=;z6 zVL`|EmTKvg8$NgBgRV#)IW69lCE^lW#%Q79CiQ*%M2BM2cd*ey+F^DtRrZPSAgq@V?+6?KQLdx^f5~3A!yOQn1 zM(9z>>bZ!MiIb(w-R@#YgU`Vq5VULVw);Q(Ej9a&?2Rok%uxYQ28bwGZl2Tbvtd&( zpH{jwhA@bPk8}+z@ivVye23SKjQP~&&W^}DJ`nVMX+84k^GeS&fTJ)ROZHQd)PKE< zpsgdI<;u7Qg&x%&h}&xg9l2F;A8?Q(zhsw9sV;uVQ-`;qH_Vc=QN zY|CJBzu8>2P`4o)6%|Kh>Ro)GX9e9?u01mcJwhpbU-u4a)0(IpaPU@n@~@CiV_kHR z08%W-c*NDw1ho@qTWcs=h}j$PBw_82yEEOu;_dVnVAtD+9xuO10HT3IXc~NbHf{_U zf|tdt6%iap4U5Z)!v$nC73qV=^tH2mMyh+&6=aZfB({&(KwRYLMP-2uL9PZT65Ny! zxrpr!t~*LsxN{P1^oyCusCEkqL)ANz&1P@n@lsx-3$e<`Ngr_}UwO2V`tpI^LgRj1 z&30n2A=}2856OXvNIe+R6v@T%*WT^{ipwayfe!WSEfw>qZykw=UhVV7iZ}9PyV>4Z zzDMGz(@}zd4WT(~8$!ebBOH$n3mFXsKaMmMHClq;gV}oc$^vZ3h;TMYc(0x~$z~~6 z7P^JYnJ$1dt(L1Cfvhh;dc&My@H)=eZ&kfz=BlZP-~JN6c16NMlF)J1yPJBl%;&>+ z42hpC)>Z(N*W+1hFHOjo({mv#wH|HD&!6Adnu!)^2Ni@I%{|2kgz&jpP07%c%}ac= zHxEx>kQ)>cQt9_62R0$6b$&Z-e2u+L7}OFOHSnbvs{tvwNf`b)4nkVJFk?ZBsitDE z0ib$Tm?UhTBAgp~Rxe(M^i!1{XE6s z`E$Aorsn8Je=xHG4cR^^QX1&yh-rFut6?=+<&MK}HG~6h^HAMdS7&#MN}|mG#T4^J z{9mih(q+vgE`nD$s-T(#(f z#qf(rE;tJ39LThG%3W<&-YK_)yA#et^?%t!B>eP22{yn}fppVma2+OsJ^q|$wIS=| znK*)n-HWEC9?q&>&j-7qKoBs5qe?XviL?o0y$V={xi zjz5d+EV!{tu0w}r8iXsUDJ^j^XT1!&T2GD?Xdbr1zB}Tq%aZWh!z<^M;4W=k>h)8} zic()6B{mslzPEzu#YQANRX>Qf64q1I@To zC1do5*=5~fQk<)KufZ1PV%2bwbPnS7m6F{Fs=2FW{g{c!B4*>`J3VS0Oko4t;T?=* z<`hNsN;?b6!#;)&x3Dt0VxjT_Fyw{xO`q+tTD`RSbaRwbh>$Uen?6zTC^1BS8oVBp z`ZASzQEETda-$*YdU{q+Yjyv=1JOqJ@r=$)jo8OsHJ_?wjH)V{ye#{4w(Ns|)oX$H z7$H_};5(GKzLNG0@RXRGhSQ3>uY%?_y0EHsx;deyNtMI-RL^BR(FPQW3IHbSf*NQN(_n2d>{KV@6FgP zrJOT7Qn_$HVdFm89VN|!7yQ_sf8#z9m2XyG*6qTQ34CkLWIojPjjntVhA#gCIyiPCUIx`Me93@V9+A*R_vX z+a)1A5%u(=C%~&?Y<*j}n{B!e18QPSIhiA0$@qgr-x2jN3V@XeJq@l3=q@fg!D{Gd zBYP-g_53jD?c(?p+kiZelM1(ImAW3qcI5~iV|-L4O=+%0hVSwvs!--xC~k3tU6$ZO zi`ywgxM>EyJ8V*Eqq>&;60yelv(-9R)SG?xRn!aQ-^CDs+Rn@xlt~BBGXDFF^04s& z*w_KTGHUE(4RkSgb~kr*a|OCN{{Lhcd%O;z4}b=y` z08L#S9e_5DCjW`I?(rhFApo?{^G9qk7KVeeq1;(F9J{2-fx7u5s|jPh_E7@VANLc@ zXB15QXz9kEpFA^|AiZ3wJUsQgG8*aN0*@d9yJHLv^}5H2#Nn+Nke#Whb{+AG&~`0O zlcWag5yMs;oV4ORRly5UKGPQwdJ((>A0@Xyi}DY7bUn!wn+m+h={f1INS1xMpIiR2 zF=1P8KN<~{>rBfakrh*6Yi0R`NsTlp_v6RG#1NLP@SQ~_M%x)>SB)y9^Kkyf^(|gRaSg0-P@VXW1Ahn8aeSEA3j|_4X#FJ4e>5cjmdF32 zpgGvs*|<2lxPc(QVrmO?bvHFNcXb6CJD342Kt;ja?EkRnpCVywY5AY9j*a=nAmMx8 zPYEkQ?Q_=oa7$P>>}jipAJC%q5_`K_sujC#FAe9wiYYR}r7(gAo?&97u8gedMd$Av zv}gItcb?Q1M<8Comc*G5Zbs+uu&U+lYDh%uw!++{onpQULyPQrruMfWaI*aZfrIVW&apRkFt!B6Z2^k@KSi38z~%_r zE1@1QQAs7@6lzj5m@zwByz?KW0Xi^*JKQ0q3e?}uvtU9484`5m41M#y_;{yL#drCv zc{e_dCaJrI5ydl8!7_*)7mdqX*!o;KV3!k0l2c3*6A(e~=s~4;v?RN{@p-B2XY;g- z+d$=jx$Pf&ULe&)n3AZ*c{lL5 z0laKHTpU0*b7Om;y}7*!sN*=9yVzU1y8g0t|K{gU`$jz?^0O*Qe)UtlO!8w@)-uJ^ zQVRt@7bQ>kF+s%2CO#D`wDn*wU7gH9I}Ehv99)e}|8caxg&sRz zhtfxe1~#2j(+XG{*no$Q9Z8|_9)~)lPQ+MMCi?Dke75-%u;l*UWn1=Zf({*(E2OFv zCm9=;BNq&_?x|z{R=3ZQ1_9Dj^zS@lFH_xF! zEs_UuTkggibz^oHoZ`PIqKNwbp)UkVBBp{s6`CFTzt9()ASn3Q_}ICBf$>l2_?7hk zq%YKe>Id$ixj!$ z6H^FHoR`6Q)=g$*r?1Ln^+*JNDu4iKX475IIkRxIX&)B#_4&2;jK3$)rX zg#8ypxMnjcHKaoGp(JF<>QIkO=(B*Cdgl+!B%Jnq$^^SHq_DD-j>s)Z&})-a=I>s` zzvKAiBbIQl8DogUVFUxd5jD?f4Kw@xqh3Pt-Sd#jRgNUQ;0?hfvC+MDKd}a^WmN?- zyKS3Acg*r_M?&o?2GOW|0Z2NxNItI4L#I4K^=+u?nk_)bKKV zLtTH_5dU}%+pxAun;B}|$y`Y%;-rI5GpW?tUk-++uLooHY2Zb(*9*SvkT8csYx7mSlfzQS z$i0_Kmh8<-w#s<@ksqeQ;}c4k!*qhXUS{YkMOO^}_$k*%-^?DSy%eP2pav+QGC;TA zE3PwKf5fzEhbfw?poV9I)1qd=g*0=qqs~c6%l(Vs{04t9|7mdl2JnyZU0T*2g8+ys zuRu%xzo85l8wUp;(9Ffr3Dj#Xf$k30AkFIVQ-b_duhuT+_T~<5ztd-KLBolDl=Wg< zPNr(p1^x5LVvQ}X3UVA$vPN^% zu`>GW+U&++skTq4qYS%c70`U)8wo>S14s#7Sx7rsn>w<(mE=*%VqQ%t6KX1*Z+im{V)%!1f(zqwNvhJ%c{ZlcCG|F=a`H*g z2(+sN*U-O5=p%Oteg%r`3#f=`{&j>L05(1j&VNPtKPvEzxv86@3(ytxX72iXu%QVP zHvIs!kaO=S30lHTwReK_)d&vM{KGERilR-a6h}gK)YC0BTZMtsjXKM7zL^_3Rhsb% zR**JNpD9MLEHi^GYa7C+$h^$kE#oEZHMjO6y@a)0sY%j1)?w);6i~6}4w+x4l zF?W)gqS2lnCLd` zFJ!65Bj=$*mbQQVLl)ybPLotwiy(|TfC69oY5+f^<9gVu+C^8>N@yl7A1xm&47Uwk z&44h-%n>8%_QH1ZT(R+t|5R}v-+G3tnWJm%+~T44u7$E=f#33O5jb-(ZYVuL&6M5$ zN#_KK+s0~G;gaX(kiAnWwSvm8b@`^Zo6(<>*R~8lZIAbg2I_?rGLi~k4Te?+L2xv`y_6%eFELEg^7+T70U*D&BG?4Vf>s7U=m>72Ap~( z91-}0>SgH$=B81%Z6oHsh}im0A6FZ`v3#QPDobZ-e%c`2;wc{aI8)V5|M_>~k#uf| z$bmqK0O=Eh{{|S)*no}qR{{HpfUUXLKR)FD7!aVnrTrh>W+QpVrVoG>cK(DFZb*NC z*jm9PNNgfctuQ1`;vzj7Mp?B+I68SPUp(%G4M>B2poLzqx4D)25A#{8G13CcpRx z3Q~cZFNUH=Zue4)GS1(=KP^tJW~S(H8ePRsICAN=S;u`TXgO(=vfVz(htXvrE=PI&%lz> znDEk=AW&wD?z_}iFA07yVCi4CX_%D!=Cobc9)DD!ZctvW={^p?;Pu`5G6)q&js;Ts<*{MsEVKJtk~2B) ztUuH?=I>ejut+HOv;8Q5)?YhFE&wlphn)?;@lP9aGj?iZr zW&|1eQJGzr8+d1ARXAt2D`$;=i-lhz;c`TYRa4c;%~{3i?jHX3Rt*fiXe)6bQT4rN zgspf47cTA}${~HfW^WMm!8t&l2>ACDG96Vf+?N(tahf}gOnrw< zV9<@BmW^@6(kqjanL8}SkwQcn*9MWdw_+s1JDAK0e1^u+aUccv+JrWBQC5~nVsoU^ z4dFP8k5VYLk^M{`A#}YB0UE?t-^R|~R&};a^;>77oVf&VeEFDFWYm1|uD$(_2|_fv z#of=`TY%PIOGYj>02?Pe7boW*G6aOAnYkzM9|Zq14_IUsK=SY((TL*Wj^wQIhiv>e z6XIAEnOP>sr=q9i!SfNQq5XxJ%9A;h(Tj>942nAo&Oi zgB`?#pVN0scVibb7tpi~MCJcIefLWc`Pl};KvPUoKKl8&IQ}AT!b6jjE!xJCc;>@+ z3EIWJyYqc0EJgYo7_n79UgYs~MN1WI`StSX?4tFn(t^wP9=Js2Hl7~9 zE$xJ`B4Tid>kV$ufb2&7dD`^ov#Pe$)B2_KU)S57;2GJ2tg8{`<>~a3O;?p(#EeTM z25-O(AHef%RoEg);bkl?^PQi4aDWV#x-Muo{r6+oOPUkf=@q+P5-qL|K0SZv)MG7^ zVl5D=AO4$j4K8-jivs|%q5q@c98GLM8ty+KTctYV0NPc4`-~2red4)yd$|nrAyrsf za`|k#kFR`)w@mE0qaz$2rW&``94kzCi^m%!;C8 zL;-ZDYh+Po<*k#Ehwpo%rHNE|1$^pMWIhz8`>*y~rh8N|m}F3gK4{COM4FFKs($2} zK1A{r5N7)tfG3u8Sh>V;Yny1m!a69Rn|d5n5>EJoq(5Ri3_AZYOBd@4BX4T4e9;F8 zf)pW|LffNujW+o)726HUOe2c|ps=@baNSpHteLmW6kqJGZZ#KNiHH?t5lBKj2UK|7 zNM6ydk4UguAgQq^mLl}?6(A}RMl!a#wpaV06FY}NjIN#FJhn4Dg)zaQXZ3L82DsLC znYzEVz^g1=CFO{*zQ8brD}6UqwXUTCFUXW%1-m&f?hDJ765OgHdmED(`h6%){p{Rs z=!;*gf>;jfA7g3U86Ofy(Efn`ul5J%Vh3oG13(kmpZ*Hy;BIFJ1Q|hVH?RNYIsT%? z|7HkPR0B$#|EcoI3^MXss!qQw)^x-QLzL_2TsWfc{A_1oak&+At4aBX1L2 z#YmAbd?n4Fw|LrC!XunO(RqWG8EE~F|NHG0(7z{p{~Ee;v4dm?7bnL*Qsg&>@lyx? zi?IJkmHg2R{vzOerBPW07I~vOxp>+CVZv`}{XcYqBDJxf zhO^^|F+>T4O8KCUc1T69sh-FTKQT#ptBCuOa+#gVANZ;6nygAjEs&+Xfumw$#w@4< zAsOa_CcZYwP<4-Vt6Efof*Nx2w<}`xVocV++v3ob;ktOfH?uT5-{KRj?I;6rA?kY8 zBW5qrlhfGD>XrGBcAt(|v~7#BzU>DU92r)R6;E?h$ZztDE75WG*EonndmmjWTrMG$ zme(ip%{P4us_0@UZ828~p0`F}=OLa9IsjDi4Uxdh^t>`c(k;pu@ecvddei$x8EtIg zIjRwraW0GOHA?95RL=g4eX!vGpM_|yvY=Rd{|kU~%`@HDu1T2H=)-0;{@3&8_%-Bo z+)2JmgD&wr&&X`TnwZL&XmbVcmjsG1gL9!i3s1Hl|K?=l)fGA;NDFrs;)m8fFBuPO z!;>55iEY+axtb>cY?b5hQj}4NWNHb-=_*kEtp0WvbAu)#oLuYx?ms#G^X$yd(b3i1 z;om*z*IAN{tE0p36NH?1b&8;bhTRK%8o@;Jf}mOIdS8y)?vfoiB74DwH(M+__rX!g ziUAD+srLD$9RZUgWynHNxaE&=cR;znNDu-2`!1=+}-!YSH*1B%C5>;nn%nN8XFXO4 zrlk+i8T92}=X+P|64C3>mpQ-jf&Cp^39?OfRS>8@2Qhywl|kULvvGjl|IqpW|4sdG zs-wcVoV=34<*y(Uv06Y*|?T(0jM`E@D6BstIWQwQS}2X<9&h z`Nx(2G@sTE7LGtuM>}_W5KI1VcKmN*aYunuWM&91SNS7~^_`%H@s37cYUmmxe<50E zTrHj<@4H#79NSNHN6Ct-2#SWumaKF#1rGrLYZx-KDT>b#9J{w!Va+CxH9DVQ zjr4g=ASdO^5ug~~{diA>DayW4b_pirmFrJP*s8xtW1?W**%NDUGd*!q5J|Kgi1)tW zyFmah8o_=9%oLTOGjsu2lyZ84NnZYCfBk&rE{6An93YaOx>!(>^Y!#bpZ;@C5>iYt z(`-zBBiVq&U`rt3acg>KPTZk){?TWRz8F+~z%JX8dioG}?CD1BAZO^CK03Ljqg`Wz ziS6=43I)SxAk@a^TupbFU=O%=h0X>>ro2M2@NX%EvJl!1*9BRhSku`_>+gfCo5EW3}ideI8cNYE-_BM_3-h$nkw9Rfm1wfhKB`qg76N z9-Xq0Z9$Wy!UeS+5sC&t5eaJLK|1#L)%;wDCt9|ud)#);t-E|vr!n&Mg!p?QB8k9n zG6dmc4q8Tkj}tFwti4EsR}^?SBm(ep7G1qemRwjt~Gq0=s`Jad3vqt8q$E z7Nn^aNgy79YV$%`0o$@+j*k~fBrwr8`o4wOQ|&!;tdE&E58o^6ZTB5&!w*VV*GmNQ z8X_xOn*QZR;5qP^e#n)zsGiVDO$D#ifI7E0hzyzjTgWUx<1Xp0r+`o z-QMXxdZeH3v!4tr;QTo%yaXx}C%*v+r-kqme$gF(G*!*;V zNuQ?!<3;)%ZMeHJa2E!eJbsx0sya?gx!g&^|7;6@PJ{uoy)_gD`88KKZM#|4{_WKE zKwFbDKf_wH>&$)E)8>VCns@siZ)c9B2}y^dl6^L*WIIzWjf(n+fj4#!3TcQ2)JHBD zu8)jdgs)AnyN&y40Ico`lF|x19IX!|U3}`2IDJH$@Tl?niG@s^mP~s2qiUF4sim4@ zmeFLlDEExBMLAg~MO{CjNAiZ#j^F-K-WiE!*1$j@&Vbg>694av|G%&G|0(Es0o(vS z&`B02=+NqC2?V(oYvX_BQop9Gzv|$xGXEb`U2MD(=#~=J!1>!Ka=2z##7<3br`r|m zd>nL;x!$DVAe^>CU9n$h#hg>l4XG%4A2{PlmZI9zY>={zq-kdG z$CRh#1lculro-TIbyd_gr+o{UTdQ=i2*kv7GO!dwwLUY-iY{bg zKX?n{j-!+HqvB|1;;ogEf#XF-QKIIJY-{+?Z6WW}5tDZ%lOB^TsxgV_fF7J;^23u-AuK`oWXNcPN{S5kVBU*Oe)+H zTEVtkc@sYvjq!%|JzX2T$R|$?`(_oP9-w|F@;x$q69EW!Jka9!3;h3`3XtS7wPkU0 zw6|lV8)Y4nmtj_|)lg;_W@V9Km1SlbSNUJ3e@`VO#71bKZ(xAb3B~{IPq~6-=k~@d zPA=B2_Qt3B_Kt)+7|%pM5c+_2+jg`E8@K)7f)b<^Y8@T*k z`7b~ALiH2d^=0C6=C#D>2r~GI68kg4Ir?=z)hD>mXmU_nSm$3!e&o4a3;Rxphj*6W|erP~%*j9-5Boj`vk~-)o4O z{ugsf!vJlTM0w*DoubVeysWRLxo|IYRj*mAjW4&XX~uGx*W7KFzH)ZXm~PXgU0OOV zb;c10#_x(kW(aKr8tIwmeosqtcyre@C;jyF91M0lY*bkovJs}(lpd(cw3~u4p&F^% zccQJ|bECR3Pu$>9F1t4plYyPaaLz`#g*~oLB4uC2Fm}oN?u)3Hm4$G})Bqzdx2%Oq zaXtMsb5i}b5BMo-yT#skj*ZIrC%rA79L_N+Tvf8B0~x@;0`}hC`i7xs*1nv&G1=ry zy3oeW_@!5o31DBmxT$GGXQ}lw3-Kz^u6gO20QlTBe6<wg}61@p97ZM60^0oI)#;R`boxECcWM!zq z>P1ifpgb2wF)eiDvd;RVW49yQT8sv`(E8;MSeqKT08$+I9OGQOTp|HyZz&Y!SWLX) z^tb8_Gz=Ok)b6LukMVG4j>r93FQLsszGyl*OKY>!&9K?AVnhw9^!CS-0xH(}FJc|% zZ?5}uOXu}981m>}z~9AP0Day&Dq`ejdo1$8A`2RLG+Gl=qyzn-md)+F2!PtbIH-cr9ZJcI& zDO2s$sQ3=dV1}X)lft2r5x0<`f6{%Q_*isOj*T-dC`BpLZUemtLXIJ#VOgmHSds#4pDykcrasdk$_h&Ph$-PD;WNWZzM$I6l#sN1g6Hpr&n&Jxq}Sn zBp9+z8}qnqkA8pgdh-eAM3aKI4u+ERjUUhavHuw&v=j#K@rK7y&W@H$e?@eoc6%J{ zM#CIc_sx$V`|?jGlIHlsa*BoZd&U&{j{lgXbnJ>32BCgwu=cMiwuC zp`2O$fE->{y}AVX9Ef!>FfeH9>SAjC+RvXw7?Te2>FP1VNCvTI%&Q(yyM~IAmzRT? zv0-$!N7`$9r{VFTFw+(4@ZwX-Ywn@)Tn!nSIp@swXq>#ZGBd+8iG`5jIdFFa#(L_W z?>Zsh%-^Mh-Fhf>fl5YF_nH5_pnK4T2R;{1k&7gE&NmSU)G|lHHE+Z>sDbXjJnw8a z$frH&d;&;Owjj5rOV+1niCVy$`#z6BR7*1W7%si=V?3kt34TUMW{8d}6Rtz;em(ATSsT0M zD5(W^@JGP2Hl$gY9i>hm#(qmF(+9o`JypKM95wfIOa!wT$A?yqU_gi<;H8-$At7MI#n(esz z8VrpUk=ts`&~b5*t$aa3gLf9=(Nl{7)Nv`n?8QiQ_-r)>0Kix?5(#z{Csj^87|iD#tf;zW_8>sG?lK=CCOW(Qq= zaapezS{_er(1f2e`HW)F{6+wO&;iUfPOa)#!H?FflegfNAu#I>|@}NmIIE! z{ZOx)rvl5vF88DX>Fcc$iuRf7N$aH$eO-~~xy>pJ)-$ctHc5xF*he%~28xz1*T{m= zXFG@Thc)~E|n8I`-I@jhRQUuhJpsnyEv4A ziu@1L;**)0#kJW{&p*=I_|S&lr`4J;UE?hiUNID~!p1xD8;i#ju7Z8o6;)qpuSG4mb)>Y% zh*m5t{0kW^)4opc(BzMftSrGKeM$0BfYPGyi|LQGGr*Vnr5G83LctUCZ8K#*PLlyU zCpH^cLSmta5@Ia`6Yfpj<~c0DBt4SXiHk6O69BK(RGZx7=-c3?ay+ZGT zLARZs!_5Gn;)!Se#?T-;q*l~MXVD$|(uIWyDTHJs@*?LfjM0b4j+a{VSyn+orwuK& z6dX^xp}uWw(Pds3(v~+QmKP=ZbgW7BqBPU(A=<+B(kS3^X)uI}6Nv9%mDqM9V=LtN zg8SU|y-=OsKK57%xsOMAr+|`u zX*L@SIpS=4TGpu@Dd=bN%`Qaf`O}9i3&QP@te_TkqJCQxMs$yur4dA?^b>(5!_%CG zv9u-bwLW7fX`zLZ42H+X6QbCli+v6p;2XF*$y@SC zKit$anS4)<8|1wK|1x-*0d@5%Nm$jgU;FS~G%c9xcFw&wvyTOMuU`75!g8DaMYFJx z;(Ucevz7zwmSsvRmq&rN+85sGAB`&w>&GO@pN*W`we}A6z4--EASc@utYXi#@`gGs z9J-b#?9L5SIeoRk(We=h&$<*d_BKX}9kK1ypId3q62l)~!GydZpVD#Wm9%Y|Qm^!! zeD~2abvTXqID-sW{J4|G#T$(Fpy0?GD(0kFo^N*bE>x1E+Nwl#+Gy0NT%AESlLPinBs%f1V=(~n)ixX>{o#+y;na!}5wipt`r%Kp%uSdA` zh00}SlldUkOi=0s_Ax81+lV8$!_4c_)1$bai^P$d-v z;i=>&@QR57KAU|t-X@MSJ$9o_g>QSqbZwxME5B>&bdlxtgHiX6_9MEI+ z<0#_7s+pw^e>&$UQ%;;LAf>hq}kM0zH;bR;kGn338oKOZFT*nvZ;n-oj>AB zxTpG9(Kzm2YLigF`6~`1buga%uE1$dNcSKSGA9V-;l?*38Lw5Dw}(Tt8SxWwo-awG$y?MkAtB1uMlG@}W5#$$) zGbHX5_>36eAt>~=A#CSl;@JO%*8B0>r^_-qzC+a?z3GW1Fz<|P=n#sr)g;c48KR?5 z2cY|SyXv^j=S7SOv|_rZ7pZB^utUt%peZz8v5k@5C*pAMuqEeMV_N(PC=e@7uP7MuAmcBR?0olv#!}E1W}EDQbpb zXyhs9dwYh2|Bd;KZH4Oy{f8t$=Mv}Z12kl~`qz-i0JLsGg396Lw~)b(Mi;fc!i?uy z;E-9HUbRMKxE=!#AMp-@<0g4{A7EfniBNX7j)epDec|zHc{g{M$nw)Q&)KFdx-MU@*yd8Vu3^c|*5I~tB-0)TUbmhaeD2H)u4=LS*FAh-LYLR`Pz z4|C9^t%=J8W_R#^-px0mWH&Pzpm`Jji2V!}E$#CB85@wLKGe!iVQLp%W@gS~c~Y6; z1~8A!5J4-5qwa zIk46QIUC2VVMw_i?BTSoXz{QbReJ`1`|%}Y^fHWvI|u3wM69rO&)F-ZeY}h}VPB4+ zU~0g-;}o47^k;eMpbO05A>VYbu_+Z*D|4v1An0^nsS|ybdXHlb(LTH3GMk-1K zWmC7LKCr(?GAEDrqmAwWGEBm;IlJDGKr2Se3{qD8qx{C^hW;m*5-3!LbmtT?V7#JO zwq=^x^vVHOsvsK3AKdZa- zxf9?z&v?q5fs@fxY=WW$5^rUhG&*tS9Q`CHrZbBJuKz8$)>;zg8G{p9U3iG}kN9RiG zZvO!80tf(c*TB(_ri%*gDRk+RpbnlAhZlZr%o1)ZOj5@6zG$1OKd zO61&4qK#&XacjGtzkDD{xO}7Hf-RjsQ;b$k-&qVpC%VUMcDYDkV56W*)x`g|`GmV0=-LSA-$P93oja_zqQ9J#keT7l7`{n)n8m6J zBsv=CstA0TY$C>zV41e2wyDx(Jd_^2>$@20lE|x8n#EW3myVnrwFIiY)jpTqapKK> zoFJVYX&cTUtKbBlQ62eA9y>A(=lgDa&%fn)s}V(I|2AbTXqS9x3h{t0mP)#4((vs-c!x&+c>ehwo>dC18Te3q}; z)td)DtmC138Zu8=^`R)>m(=xl6MTaZFSXQr+AzCNzSH%73~<&K8T+u!J8@4+XZHy^ z#e{|;?aR_g6m+YbVgMMj#3VQWLQJoBT7Qw zq-ggiaj9_PYj+@;`z4*@1U+?ztF&RVi)F&HFqu_&V_;?|re={NKU(hUC*0}Hw(4G? zctOI0`J92-HiU^gch<*A8GBwZA}-F~{@|4hK@{k?k zRTuaUX&GxB54zD7M%{Ah)xvnT+{C>|34-VjM2PfNYYLwX)G*&!Y^*Cx%>e;5>Rbtc zty~fKI@PPr8Rwg7unFM7DX7&#ziJ6J$U2_5%5Ey^WTwcS<}caNBeWo zV`eReJB)4PTz0zFafG_xdcZH*=H&$)DrFw(jd9Y*@Y9#+omNZo?XDnxm`7|xjFAFk zhlg0Mbc_A0+>EYe1QgWW2wl4yuMUPOL)fhp^k~~$&NH<{nwby!jzDF_Kfn7G8#-I; zBQe>qyD|Zx(I6~!{QEF`0R2`0CMv ztU5Ljr!!x7zdrRxo4(9FN=XrtFBc<5J&`oqgECHKLhoJdK|O;uz%>z5 zrloroU28T>;|=#X{=PE%W15-s*{cO4>4NUv+n8!d#6OB>|5Ur?*h=8AZxMJ-<%fVbihFLNJp8{x^{g`Q+{`-BRJc&a1R$J?k@ymr{f z4k=7d))B}5nBO}A0|%S1@0Q+j@p`^YYuN&h+v9XLH=*q5p(=_u^U{ZSr{$i) zXNUbnuLWv_N42ksalQlE;>f`9r&YpZK38*DASHcViwq@njOi`>8q2l|MsB4!C&`|O zrSh1Q5jo(6FgD%kG6vQrjZzznUmi~CaI&8DU7%xjeT%0?UfB`){hHT_Z(cA={L_2@ z3*X9#$UZY#A%Bi!gKEsw==qeuuAQKMdCY5sm^S{K?wF{&U2B9HWzQ-J(Y-fxmxLy1 zPIzdB(}2jNX(;AGxWNEUcl386i~BGcoLIAHurm?~2GaGeCiTlA@dV`0%OVlv>hQr3 z1G0X2YCV94emi~ND2Z;IPYp74Fmsm4M!{1r9%Cn*qn-n#BUQ|g-aUQ93QI1882 z@yq;{$yJz>MPJ_IuflB60?t+Cyt^loOroDS$*I8ZA6+#qdf8LTmpqrP4o3*mn_9)o z>z4^`?_uiGy*-5;bSQRe7T0=R%W0-LCk`){S=s5*Ox)K7Q36~mK(~ftztP90ipH5) zW-5r-Q90004A1yjhm10XHyP9M%uxoAm#5U-_aCl7EmF@Rn3geA2ya9_t=2L_4OyV} z*Tsf!)em{{vb4ugIbHds3?P?SEjO1~V>3(NB@^NDn&n8j#Z-nZ6{y2v3fra5gRN7?=UIl^#pi0~RN*?jSkwFtLvChu&xOTp{rzIcbS8-2FX%L?&IkGjtD ze~EU(YNuyUAwLq+FFv%uqRB;2lQnp`$D_t2hzm|8Vbg;&my@hH0yF(|cX`#XEa}gZ zgxmN{QIj5o*ob{MVhjWNh?-f1i2*ohKGQXTV4u7c^gZqf@c%HFI(bZYL%ViAkt9phUoCs zS8MysJJ&vXYea5h$nuE=<@fC_km!}1QfmGv)twi|MXl5|N?Qkk0FbE6SVFI<}$99V096 zf1g}h8C$S5G7(>=eQdTT`%E=t?TA^0VxL^t7^)~ZNyQp@Jt{q{>-}x^3;w-Tx4=3Lq?nZHIG7f+1nChqvej%1iT-MP4FPMGAgl>~|Att4hN+7W}}u zHB8^HrC@-b!8+JMFD$|7cZFdeQ#HDIYJe{*D`&jN&2oj2;YV-uDznXw@fW_z`HfR~ zQKI!b-=LeC>4mqUh(|n(1dtNcUf8Tt)w1p5-tQk2o%?Y-6%|3Hs<2uo_tpB=Qah)$ zI6oSxsajK7Yb%6@wU~J6v_5USn$ETO*;?{>xwkED5KKq10N*nl9}Rc9syDe8pT?Xw z-$QxpnoKsQ%$t(cJM=FMI;i1%@4?H;+ShMbL#EF zq+Y46btYJ-5E`{udkJlw7%Alt(qC0{s~arm3vr5%5aj&)yA}$bC1Oer67jj-{y(jp zcQD**--m?|(R+)i!Lm^!mgt>`7Ij0SMh&8eglIts5xqu>l?NLwA;Rib??Uw6Th!>@ zJ$fc>m^u4X*&7 z5TBq3pQWv<8}Dz!Qv7#R^6eBK;%6+?ny685IaA8;)O1k&s0YuBo#$jmK z^E|D%yS9?VZuHXlrL{b4(RKA~NycL(aU$D9GmXkR`oeUp9=hCB)A8lD*m7*HlR%?PWHEeyL`gK_ z^{A*)D`{faNo(3FnDY2}x+J`JbVS#mar8*#3@mgRzT9(|>+IZAVX-d|E&OGLP9xDJ zfvPH{sQHaABNbnMVWovU1K)BNV@Hzzi1OlF)}`KToId)xIuTgTx%bDLjGk#*$M^ZUy7dB1gpCEr?Z_Z8D}M*fih#KuC^>3%>oNJj9;E^pqi- z{R;i_Sw*d&w%l1R`zQMm^rSd7LEGN9_uPvS#31)*8A3WoBgx#8>iSLV+-{r{lFvt` z&EyEu&rfvR1;%``Aj__ECWN38y~FxAEU3(fl5`jDc!62mYHRYU7MC;!wOWFv{%z_K z>hHVMzAbL_{s(bjZK^14PKXy9F4B^LYc%Ihx|zR^ZSW9yE13LahaY~o)?KQLW#l{$ zW@Gp*+1iIf;g(4h3p_fttc!WETO!j0+withIhIojGo7RlR3)JiD(zw==0uH}&t!@h zRONJEIcX!RCLDpx=7lWiSK)KQy*8nG;M{PKwsUb&jKkujJfp$ zN=QJ?++6cH7J6G`<~lF66uaBM+U!@uuPaK(<0@jFUgQCC($R#!GHR@13p zhOBAxm>bVoev7j>(=UhZ#V*!ef;ig>mVrm>FA)RBxs~?XJInN3!|@q3(-UtcHakxq zZLV?SX}@HNE?lY-z7-N{y%wXDC=bRdgWoQ=QSK8^UU#{B)9fkjy&`CqwHi5*0#4Ba{6LZTCD$U3a@V7GqS{P<#L)ZTh=u*w zxK^5Svh3yNcL|!4K!nT^0g-d%c!D1R?+SrCw=={BYh8k=I^{e?%s1;A>t>F7AK1B~ zboK-BF+eOe%iK8l+Z?VT>Klylb7G*PsCb9CO_soqg$%a`k^3E+9V*UGw$7F{NRv)r z$*FYz$+7op`J_x1aWeC?DPpw)KN!izum%Q=EK$8!PxB0fDiEkAE##?mlRxSuC`GPz zJFw8;DLO6yu7piH>s41=MvV#6U}0Uy81Js4yyjhnI71`XJQZ4>^hq{rn+=I)q@eep zFan;VzNQ@6IR*x;gFO@bXjiL+Mxq#9s4A{^;mA|?Cj+hs`&bzL_57I$S+_zk?fGT1 z)aIK((ru1G{Co|B=d%iWqwxl|Qf0;jBiFzj@b&N^4A zk+{j51?tMKN#)@SXU}_7s^aBSx-sbQ=@@~$-44&~6b_yz)qW0UhZs957j*dgIMOJ` z88s_Lo?F@k=R%dRuyL>mC26yTyLtu_f@QgdMcv1EC*3+@8Ed~>KgMK#CsQn3yfYs-T>XP%})yv`Jz4UFR%Ce&h8G!vcz zJznsMJYOiYD(i8lvG;cU44+Sozf3MkS^tNLBekzDJfAoAj>*tnH^u7S!oK!=pEC5y zq%xg6gw#~!OJINr@zc%tD~gku1>%;eK}z>DDpT+YB0XYYG&G{=7Hq!a zbc%Rr*V4o8KOG`xW^*}Ig3x}>NS^L!UrJ!NuvXKlX-R>w<;H8o5s z{eGO3Pr9@Am_a3&u}0vilutXsN}8~DMk-!%GL>Op!Ms$Mf2gYWZI#Txi#|>eyZ`z85Ywq~W=8#8q z)~DQz2v$BZjk_!}bK!>+2Sh!5RQ^m+hn+}T(?***r1<>VW}2A^j+<4QUB}p*gK_CD zA_+RswXizN;oojpK;E|buu_*2VTwOpCj(ieV^Se9o#T>w^)@Wf`MY_x;gDpF`{ z#g>*o@_rSarl)S5pUu*+Eh2JVZA|#2Dh*DXB|UTDz)5r>Z7E>vW3`OAvSoIV;O)T0 zGM`%Z{CI&S#qiw;TZoX)oJWQ_PnFs0wsQ!esKz~f1N8ISzT4`}Us>X$@8t)@lQDuZ0|S#`%qKd%^A)EfNr#)Dg> z+?s|<1!EOW`A)>Tp6&^%C7yLT6zuj_oZ62m3f9)eMk49V1jyK2C@3ahoI#X*L95x) z=g9+{`B}1CcQ)w=!o1ly-KA63M5Z4~EPjhg>6;`#w@nVmvI?uSFc9&s) z+;6GiGMg^tYqy=HxDT(s`U29|%Y1_|u05 zIo56w@t%}|H*|(^l@2Ex&ZuemXM$bE(?)gi`}DZeJ#fp!?|I&zJ}GmFqJlxM~$Bv%+v%&yxG$9&0~+Vdmj{ZHuVGTbNMiI!vOA`U}7`PwII&sV5LQ(DuIz_o$i$Ih6 zuzPB7N}<;2D8ppfe0r1%ME*kVeqa%H~grel#a>~i9WnX`* zcxXq*pr+yJbqje4kIt$|KG2=lp-~~&WHw30KB6w|UOfS*-3%X6Y0U|#ifwSaZ^R=; z)80sAX@&Ot3hI^BJ2Vo1W08J-t$LPAdWG#6YPX^O zV%1}lX5SXF(OsM9V%8D1$iYq(P%=Vqe#NNkiEkx{eF0ZeQ1DyFskXjg*I0MpeT4L_ z;a#k5wiNpT4z1{S+f;f{J9`zy4bM6`_Z{FpF*c24QZtB0S|AS47>mYHAjB94Q-3V5 z9B5cYm%l+B(_{8h{I`6oKsKQG6wMY-j3L|4X;6UxKp`j^;G6_wfS=NX0ttZ9PBg(d zDaHiZhHf`UT}8ua>Hlc|HSHP@2q-v113kU+6F|SBI|HHs zEnR37S=3A#hHLXv&MqJUP_>07IK!OaSH)XE7@)KY4I|5p>2my z!Q4UxtOv@O(Dml5nAZPE5onr7)L}A4G?=I)fVDu?0=jmHALH78AYK6S0M!F%9s|Mu%=70y z4_ZV3EC31z&;|8>SMY!I1He4s1p#!Px!m6X|8ZRa7zG^3K}U&e{&m#f4(I@*fwKwd z=m?Z};jf;q{~!IwqyjJ-c%F^U{)AG>VwU~y(`{fd@Fe=r;0wBXm%Yo+oRt4 zp+{H!bAEr-tVtA!fB^CDfdXuQ{4%$-wKH^da&Y%BGOOl_cdQx!TrPb(ftw?>5H8Bvbr_ai~8QIRV5aI^3I>DM+aXQ%ZyuMu5onQ121FK|~B-v{FA z84S=O73{D&mnvs6b5X4nUyfYMzR==U( z7+^}8__Ep!wmX%u|Jl7Z=f{UoVb7!cRDHtU*wlolwG)^ z-Met_#$FqUQY32OB)*?Xm@ypKTPa_qLCvl9Q+q3NZe0fV4n_d8@MzrP2i)C zXWVjSM^3Z|{i5!lm~s%j1N(Wdcjx@h7}C+;%7aEz{!p@*=raZ7-jEPOVIpeu7jLZu z-8)Obc5y?d&;0T}BYgU*RD2v4Tq~ZC>y`sFkvaW*dAT9#y+y+@jH|ny%^nEN7n{L1 z=-M3Td>(wdIX`Zbg`$`UrAih`m|Zk;zhQfRecG#*mjw9vi1hX1W^V_cp3bcJ1Kf5K z0`uKMcN4hwFy{DpU{463k;U?pi(&ZC<)fvrmX@xM-nJOtc)U3TaScD#?Ar+qvy9TK z6eow{GLR>k2H`JbLJs%N)}NP;`w?1&f-DpVsH!G&lc%qb_G+O!diYph5I=~q973g} zmeft3P$WSyyJYaimO|OFo9!Ls4vMiTQdc+g2$G!MF6=w{-32HyA-lVrJotpDn^9U7 zs#g2${{2r!LqZ!dR~QdmK>%Mfesm1n4#0# z*ZG~NfC$K6pFe%mB zyQ^$z@1y^`xqkYfdIKsJTB_IwD6?|=TX3#$7@>;@v23G=dG|7OMiJWwH2vSRwfrVT zR=$Yvn;9zzrm>lQhnL-i&)tC#del1%fgx6vIB!ADLuxyL0h0K>Mcdw@_!oP0p+M5a z%u;^Lz%ss$GJ*9{yIo9;ner5^E9?}Sw5lOLS5yypY3xEjGroMWS~YoF4gC3v|CJaH z3%#I|4-#dZx5qyKTlEmLU=ZKm$_sA>Qdr9z}Qw>iku$5PT`*+n2{^Yo&40r)VgS^zk2Y#kDywbV&E#0MpJ zA%1;~qP3V=sNW(WY#kjfxi{1swbXr;@4-^Pjp$&rEt6)H0`@|7hTHPHK@rKjyVL#qS5M^}OJB_? zkuo$6oRLVFAq`PC3G{ylx*Z-GUE=;MRr{>T<+;R&wLst%rZaWd^~q1&3-W;3PfiwQ zV23O?&GK+}cl&vA?=nAIT9!1dVqn1#FIwggECv0@8os}bLBf+5W4N?LoL4gDiy5!5 zAk;GKzOCCyDfzXQa#~0yW)&@^M%Ot`M6{`%4XyR_&8Rv4nT?>RX)cVOaaO4ObpIt> zJdKjgHFls6EPB=p+iU~Si=7)GuK4mhQnoPecqs?ddNgtPD{dOZH@4_A>_3XW@8{uB z9@FEmsPP3NvjTLcwa6}f`oRI>{WtbRbjIcg!m>I)Otk!a*xw+Bw@iKs5)u{~cYUX} z*p9`&qZswzU^WccE)eAj;%cFb@%e*k+h#N^)m(GXm~P6ofH>kH^(&3|{?{b5@pNDP!_r1??K4Xl+=F^fv`0gPy8o~Xo%z;~Vhq|c) zrYA9wVd*|XVBqSlMfAR0Os`euv!Ug3P3O^CJwXH8xwT%rG=FoV&5j^jc|kZ&OA*|F z7Q~-26LaWD>SWEeGzoOIG5xG->lwyLD~|GNeo$VgKm(QDiHzC54!T zXR3bl?<(lH6P~2U<%*$9hZ>WkMW4%7js$aG^M68yl_b{-(hTY z8_TZz>#N%L+~R8T03-KBp*#Xdb!13~*@}iI6SoC`(UO zr_Oswh^yjn3T74}&gQ8)9|yJj#h9s5SCo4lKxu{awPL^d8|hYjhrIDe+xJR+aywv& zsu=g?!nM)mQz^~Hfb;UV;=i#F;Wy|$yz$BPlO$``-jf^DTY|&OykXJqQM6YS%Ed~X z;!yGG3Fg2c?RwepM$=#t^4~!*FVA-(Ser&|cqQK= z=-fH!?a(qr$EppN3;#9ja@41m;p7lQqI|bVZ^KPbvV}|2EfnIq z`6qETg0wo~+38nQ#ZQ0WgtOwK4DC=;dwZA}J_nVK5cMzED>2iR)IrDxokZ^4__+^T zFZ{ijq5 z`e)+57Z~G6py(Wt;z^cAd4ttRYhJBfynD0){@`Tm?zTbtM27yi@;x{(t5sv zN0R;Sa;Or*xIj_V_OImY#(3suzF*-PVr(DtYZGg4R2rdsGl+}J&V*wk4wox>h~h5^ zD^J1<7DL7f8qL+dR~g$_G}qfA71BwqWM-Zv*vw`gACyh6@8n6U?{;VG)jh{8J3T)n z4Jxee{ocdK^c0j|HvOdZXDbHkPGYJuthvT!nh9-ZBQpwkT}L>J?VT3-96+t|lLp_l ztNSL}oi%tn3 zaZXW_p!@s8D^ifuCa*bi=|Bs^if4v6z#!t%GB=ehf?_!yqs1gjY}6B3b}*|2RLm8< zJxzf!jv&eSopUsfk<)T_nVRjPq6o9Jy#oRgtkNLMSwND5a^82KJ zaAqtj{uGNxi&RY=thb)!%knrx}Z{okt>kjL3A zFj{FS@kRN#XzT-T$L}Y^Zh{WR$+pEe*$3G96g}7`c(cGCCvBHFzO&8h`Y#>{8f)vD zFp{$&PH1KEd+S^*P5gCD5#Q&AT-W7H&A;<0YP-nDRH_qJ{)l;O;v_;r>O@Ra@PfZN zNU60WVn^O|QV2g4=3^MHyQ&zc|7mfAmh{Bi*)#T`B(!-j>#^eTvYYz6&%nWO0?HJ0 zns^_u{P6TDy#;mHgEwEHpH;Gf*AvC6M`-Po?fb|Ee3j;2YjH!`?PM8X2F3 zj_$qM_}aYSytBO71NO~C1G|{@M^ge2MJoDO_ckD$Im_k)9SlB_^E>1|9v@$Do`p*G zfhH567}M>#QmeI=IQe*{)CU2q`w9YUz5#Ss* zAt94zL&$TAe}B6wy9v~Q!A+{8K=cX_EbYmoVF;~KSL)e-CR{1dK*l;m*C%)r2oCKqX5Z)=NB)cN*5S5vQ1UkVJWIi`(RMFrUQ%~f!@nM znsiS=>C?cA%`EV;ZrSm^?%(};{EB=tU{Qv+>Yg_>cAOVV$ z#lkxs!y+Eqk}EpMCHvr7AU_;fE87Po_kfF+H=(vGa7U)k0m4>`Q1@7M)TE#qmK0^* z1S%h3?K#IHwV`qV$kB2p*z}NcZOkiVp0QAev<{3=$ zYNsk13=Qz#x!Ker@Tkxub`0_(`u1zy6~25;zS!*1mw6 z767h%Ku98j?yQxHHBWYeujRfUhyGh@D4^lr+5K!ve$vrhm&_O{63IS55OM) zy4?XRmKey`&Kh`3O+BG1x0vM0c=9$}X)x1OOw7u5m^{8Q69f!Rn7T@IY1qoI5=q)K z-F8T99&m*OYsPRC3tBUS4AC_u*`m7i$NP*$DM+=XbWto7)@I7O#$d14s?G5Mkn8Ci zQF;vkx6ZKwYU}9Ug73ZQfHg&vU4&CU-ubD7p4O;%IEQSbdsVXxqOf9$1dc8&A3$56 zhW93t*UK+g0~~(6x3hKDD+Uby|K?QidGm7N2I$-aCMIzAfx7h#6jT^yg2C6Qf!Zdd zA4cK~-oF{2c2N9v@SYy_L5M4|lc0iKFx>iwQBx9t+6Sb2fwKQVsfl>`yyYhWWkbN& zGr$Wh9_)b}su`a*H_ru5A76x+MZPGxqK{Qq=yVC$KO4q8%)A+PoV^PL@&bcm@xU;~ z5l+h_4Ox!h(fxF%)}JCWg`VeA2Ss)du!-pQifF~Rx0rq4Lm?A*?c4)_>>hxe)mP(c zG8mCR2q)h%gR&f8u84GqZoH#z!1(9pHy}`418{nqjh6r<-h$ry;yk+@{~kalOlNxC zkZ~2z=qvpg=Qvag_g3lDVNZE_fI4OYPi2G@nTHfj$#LSO_twQky1)qN%pu#&DqE;0 z>C!{H+EFXw!eXd~FJ5oDyghpE0rqZ0O#%HPUC(C9gCKx#4+LAEJOaE&ysXE58+ z_#WtF;Q<(U1Ck`THxPR9w%y5crG)dobm~c`fJNdoP{|EJ-ilb?25Zm_T8Z>8>t6S& z%D9FquYM29B^EbXDd%dRyk8ZVs+@lq z0`@dTmn-a)m2Wcgng~(!r+I6SP9r+pk$6h_%+KeRC}~tln%dqL{*fYlxD~}Be6>&2 zEJqnt0@*8UkC%5A61B5YwEA;X_x5tj%!b)nu>%eYpF*W4q^hJ2^AiH6C2%O6(zyj7 zp9rHN^0COag99`#Cs5Fa;d>(kx3^1JnOQ%Re-3IM%1`&UK(wSH!t(f)Flo-JO>7q7 zE7n!QQL|a~uDJ;~$3ASXTmZ}Gf~OaOv_f0{U1OJljD|_${nr)^y^rxLph7Dq0bqXg z1>|P)dYQWfg1z3Xb!)=hX?O!y0B&4#!0SH@MFDQroX>!mLvd4Eoee#-wcavYudA5p z$@593(2%9I`WDMDpi-ZF2oRa=f=_QZQ6)Z#oh)xyrw@M6qEf6p87-p@?@wBi-B`a_ zcsniIm#DLsQq!rAj!J5GI4OvZ&?1*0QkKn49i)`2uq0ThXqKjh@+P$2XE{DjFK`96>E>=NAZ&79SighLUqG3_^~W9HvJYB2ck$wTU(d``BZbgX!+a;5 z16ELFU$k3 z_H)K6q{HrU$%MwnF9-Z#GgAAUW^JmMgseP!A>u`T`N$z^5DhQUbjOTpB^H;<-*?Lr z2gLFc=5N4Bz&{ZCxHCHmDL|2vLj4QOl6?@*%IE{|!q;K@F~|$_iY)`*euNpUqKxx= zk01AlV`Om|Q_}m&5$r0Aq`@5=XHTB_Dfa~J6AL6^bGbFMhv(=_L$RyqXiF&VQk2Z6 z8pK;xo)Kia8m*VlOheQ07nM8ZY00F>aTqg7W?dbfZHMf3mP(29q3keIGdb?afzoK= z{ka9hB{#5+iCytO=hkuX%X)KHuzxw0Nc=hn{>24XE^vA8>gL)i;ENcx%}6lV^(%B+ zJ`?tXBVpmu$TErhM#IT*Afse_`A*asWmdt>}D7A{8-}pD>X8MOHkea z(bH6}y%3KRks2nSD@W8!)VTF0^W*X76&NAEjqyl&=(NyU1sKMOBuE zek0X1ntf{b$wvH5drz}{_uv!p%Ub>A+h>dUxYW=Zpic7?^jHX-m(u~e21I91>%MMG z=LpQpEe=dM_V}b@P=o-6yL|<^@9m4>Ke@gMd9^LS#qtH}0ABSU+4Im7^KYz9!VQd~ zdw35_cD&NAnLy11(9HRd=g3>&Qt__)AJv15!040#E-*cL09ro%=f%Ztz|7u^f9r4U4vAXyl#-^HLHe>Ea-xCGbG|kQ{J34R<@V8)l3O}^4aJ)#dx#K^ z^;YQ09u?jnsRy5X4~*L&brJlk;D%mz7fyBl_M@oR4k%vZI65d^?Xy$3I|g^x>#IC6 zzz}lkpP#~fb*R$41U8c27j-q?76sYoFE8%fOvt8c1S;4@9xjSLs8=2~6>4+uE|;sMJdatZStijs6*Mro1xCyT}DM zq3frhe_|;M2MFhGKxF?AU-u6b<%^r^m)DUwa|IvRUkdi%OB(_P$^Mf+JAw3g;NAY8 zP3j&35d!z$bir7*&j&-o=kkR2n2vTp@)FRN_P)srY=)cxE|~x*mjUiFS^t4gBbU1yVFLhjeCa@lda z8QaQ8@^dI4pDWwGpzgl8Edc*TyTuvM@o(GLae?~JrN(TlHvY8YZ5);pCVU)s`!+4F zK?fK6FIe+B)IJ{Bo`K~L!~f=$ShA?!rH9{M`cXtzGtnAl0uO`o6PfFOPm#@7Q$GBm z$5L7i_!Ql%dQ&XKVGz9$rI><10H(|b#V|24eLpq%#m?JIE^zbmc9)W)_jp(?OO~QH z+p@ep2n%%++mwHccA55g6~I4zDaKFIn;=FSgCj1nOq^{SKiWhl2# z4qHw#^+iEiHO=9+s(c#l7r*cOBHYJovCvNsLn!gtS{Rt;F!iB|)^n&h2^R@rE<*YN zmU!e!;Hh}qJs~!8DxI(825j}dYYrNDxx7U2bor=?=2tHHT%L8%^SLKgQ3RjM#zSuu z^!|c{K3rygS1&fYA+PZHQQO5LmsXo-o0stB9YzP zR=J1Gy(reWbwn_+ac5!O$RzT=f7Q)r$!CPEfly(t5|QLxeLH>FDGG1M5~Sc4_~aoo z(d|Sj^{|ElO(`#*?EX>SO?n?jnfmFJM4?!eQItsHGWP0LDpFy*@LOY&?5ee$ok&Z| zvqZs9k}85pB*UJzuJn-9WUVh9=rLjhQ+7Z{0c4crS#9jLi9I*BC|soGgIxQg^Tut7 zMCm0MHtUrYr?CNj1)Ath_P00GlqqLn9DZK%K$dEG61Qk|_R(kMZ1;!CM;uB;-;DIg zq%*PQ-#V?WXg*}WO{s=yD6xjfwiUKIjaR#h!oq(_(|G`vp z2ZXH{g;KsBb>+C?pPq~V}n zDbaVsFZng3htc?kxB(y4+j_ggp{Ut8t72ICt=3y##ypjiDAAymlU|^qi)z+?erQ4I zFIF}E(+p{>dAu6r2TrG{#MdP$ZS~P5;|yFjS_CF_Dmxg<>79I-nopBR@K9kN*Rdli zbg;J8Kb&@%Qgt+pYLcCC)2M$VLf5dj98`fhJnEbkK0RGUTnbe`LWZ;XVrL?jjyV5X`43oHeuP!OSx&b)rAi(s`ClqFeEpREB_a~4 zST!w30;{lT8G%~y%Vu-zmY|IGKEB{kpIsfC9uC=DCN#vbV3IBK0hsuHMAkb5R1Z7o zF0rmy^T1j#jJjWe9}(98_FrzuM8{fhgOLbY{yU0c)@AfiT65zfj;3{C^ZeH1-_xNeu4h0T}IYkWDivx8b)#)t)YWaFKM)PbkA_lJ5*RBhXr@Uo(n?(SO;T z-|W&!M~w8S*lZksk{JPeV^~Cu7EHXNI?ZOtcSOw@&uwgD)N&&mu6{7}pMkonAf!opdkYZZEkR(o+3GudJPlQ%SodyCnfnyL%J<(?v@1|I{_C zllA7~xz&?em&4+6DLH_r!L}=@1>(%~#soNF#>4V~Y9bV8^T%@N^t6-&8*4d!E*OZU zx;U3wmP9uy{KtLXrN>X>7*380?DRcW`kT?Gh1o7YyYb-_T3#lsS1+je_UhPRw00wQ zpH}0~ebw%L2(!#OTG6oMR?&pZ=O*|JG&WJ_CA>9D38?uR3F8{#oe+**%pU@iI3&w? zixTlSWq*HDCM|a-D7MO~Iv8WA`{eQ`feYg`+@tBnYD{Cn*CY&=L?_Q*O~avR#Ei&n zWwF^0FWxSLi*i}xVCJy`{!bTw^&q?@sl7i0IAf(Q?7S#i%zQGFM9UVSBW;GMataC} z4+wfD6(cg#{m;%k09neDz@fgRP;B#8VDMpR>vmrhF9Nl$Y}YLn+>9ov340++jAa|& z(#2(sjM0Yy-u8HajDS?)jHG>pYL)dW?YE?0VePj03Z0fnYE@%;7(;6_S@Qi_I(Zem zxYi{C{-E_+G4~JCAnw=PJciZ&nRf)XpV~q(I6)&FKO&kn?d44tRitnImG!p#Pwo2H z7+HH-h@eu5s^k+;v*s&?AWL3p{1sTLpPamUxjv{{_UIuh64;`fczeGr@j}m7fWt+X z;%g&Qg`PZBe(`z}FwjXX%X8idi$HNgbWN~HqD4%_FeiKgp}(I!u&d0-hrZ-ow4j@r z9L3YgR7*cW>*|Ye6dGbS8g=~@O)gHAyr5LcNpo_Q{4{E281UqG zAu8=YX>$ioYeUA}!!6Q&9M^Xs`}u~B_CtpX7>)LP zzFcj(xl*(8Sj}MZp2s2K_9-Wmu)#=SyPdID97n`7g~grMvahFn=Dj(Vvw#|7etZ1N z&6;GIPr^b4&56oHI$+zDBfbI|UP$i{aco~Mwh_3-zQ9R<3Kzv_iX=CVvM&uE9CeCz zoA84ViP5>Ho5a&fieSrc>5G=Xf*5#@OKi=&Q@;|4*)-5W<1@~D3(KJbueF~n6DCQJ z*Z9eTP(H#(*7;Lp4$elmBIi54$AVVA9cis+E=;6S>Iu^X>QF1Xc@UBsJL<+G%oL_g zj|^l1<22r|w;XkReVTlcQ9DPVjo?N~!T3eUBu#lNs|{K3mikp2GZ<~d73mLr?O z0;dSMlg-?JB#qB=c3`7&(i4`whVj)i_n4d#&&!cdj-jm|x_N0^u%)Dv+$!sH=-Yf(v$jJ#`B9#p1Pn8-%CB0`7Aj%WjIlGr+;0sGHE%x8=j5i~u)(Au@)i<( zX`l2AsPH5Oa^sD)-+`R0=7vR|c8%h^VFs(80_x&;pYnEz%FCM7E&9}~Tx!~#{061{ zNfh@nzsM}hdBC0{RtC{e&!}U_9)D6>zxnaGMWp^}4n2uoIhr9sxv;s_jiG+*ccX`K zBJ3V!jJv!QC0vR<@ud(u*`sU(irGC!gv_kB!i(s2Ygajb&MsUn##j{x3T+R0>vglq zPi8F>E-hE92#<{Cv}7CQEM8PeOh!zW*}Utwr~K$pX5C9u`&$tY0)G{s^fA(n(U!z6 zoZNl`iLE`p1Va3TF~f@EdBY*Enc#58#pGKemt7^-y8VXX_<%OGwGb=@bf>lU#(_|5 zW*i&q(j`8`Q6eF#4NQBu)1w66)G65Hq?){VD#G8jB4?wcbaeJ6>0LuIeKBF3a0=3m zEfZs8!g=2j{50z)nuU-yQ}ChEI%Jh>O$rP0vM6|{XJBz7yvTQ3117OQzPY7n%-lR9 z*U@8A2T{{ojiUS(a*Vdwg_{4O&&g#4OF5qCT97DC93p^DH|?$Hrgpzd0&Trs-?_sCB8ZK|r#; zRO|H{HI(3|Nb{u3RtG$WCi4j=Q@$?j!yR8FVH}v=%af2cM{GV_C5P@vVD#HslvI74 z(ip4x_j?@UOuzGK`ko%6x{|P+!+vS+<!U+P}Qt27?|kf3TP!CCM$wC&xI8@js5R z`Oy9I!U>6oGbzuv%PVpBB~z~W8?5XjxbV7Juit(b(r6l)Go)a`e`D$w4s39l&S_7= z@`)Rjq2rt#v}6mrB7@1NuJ{ZS{Swv#xg?EgP}UFk@+Xpl`fI))7a_xy^ACT%fD~-F z=jV%G-omwK9PyAdqUBwX>>Y*qWOp4DKhab|5uY^A*a)QK*?PG{M6=AIV%`K<2sQQp z_7<9l8rh5Ij*~B5#?^JIoEbEJQK2UszFGNYIn8KOb25e#f_fGL!)25<;h&N7jX8rE z!!L)?nN-|*#_`VsxWt$`0M8v8%2$7h9>|49aELH%W!Un&&F(01o`Xd2Hw4|%iLL7Z997K{y3#3XYgW~1j`*k*R^z?h>Lny($~pD?4PNo z>rgd<#f^4gG!u0W*!!xHCBMpO6$WDoT-E&S_@Q9Dk>HoQOS(31PI6MJp11KOzDa}j zb3Q&1vVz*CS@>-VIxn(e63<-(!7q$N&e#5`*RN$E@}wlcv?+r7Az3>dK-*C7Vbdl?E} z%+6M3;*|K3ozymQI7>R?`_0}|?n*k%I*JYU-O~2qqRiI=@cOWB*V6FN-ED0wl-L48 z$B|&J?SpX0QL^Aik^1DVnhs$PqV^n==g)F1;88OOQtHO5p`0H$mV|rY$+Ka#(7zEy z)q^}VA+M7wKku!P6f<5Sb{=lG{GsP>6oARrR;QMQaHQ;Bua@iiJo%32edxi+k2P{& zPd2P)($fNSt94uFz1zDqeK{L`SXY8IIr8$S#h=`Inf|bUE@U5XeqcI3IW}dvarP%Mw4^`K^ zuW%QNolM%F>RHQ4g(J=7a?|j)lJM8tgKBnAPBu_-884=^l_>%e{Yc{&j9iSXo$*qQ zL$XhU;1v8p$b;ORgfmp;I6Cw>z1$9ib1V`;I_%^$aq5VQ{e%W$KX#V<65W#%qK6wH zDp!#lRKEp&Mvxl-cc)UU=3-_Es2BY{eVl7fU}ec8AIzq?X0R(LHt{N1n&N4OP2<6a z%|k=yUZi7#+5SpAu4tDWBD;r|5$@gDNVV95sxVCZC%)O%UK$TSU1EGIZYvO#>_NZe z^Aw`?XhJOY1M?RKOaRw}u9!o^wB`=WK$s$yr6IRnF^jobK%`ngq}dL7>nI@zO*;4Y z+Zc4nswsvNwm4@fT|^n~GIwnvpVIVsSeR8QHC&6NI1SNl7`=1kr(K#G#0oh%+s=Wc z%G8;;KMpr6`5B}?tJVHwl;ifv`s3LJ9;@N+(Eu-r&&7RR+>tQRYhh9MxF#qH*2KS>d2yCMyyj~DPhQE7EX!#T2Ttq2-?cii_-h&*Q!j9rC6*$do;w)pJpGm(HT z*Uwl$5ch4eaB^4eMlX}Hilw@Mq{FD6&%JNZ!OmFCsto5v)^*++75KITT5`GNqF2|- zXMybH+Um|H;|%Q8)Dye&a1bK3tf$KPt?%`QK(El(RZWy7v{-F>7P1^!hmbC%2%3 z?-pc^9@Bi|k_8qlJr)v%)i^h2U3xj4Z}V_{qrhFeTwsjrp6I@O+Q9N}EN)Ie2wDDe zq7{v`zMWk;T_XQ!f3T8d1Cl)^t~5tiN8>F_&y%;*fa-7S%>6gil^hp_rFbfA3vB9! zoO3IPc1PLO`J@%B%S&N<3_k?S=HbSB)0A&~Fx-l4F3K$REIUxdsO$MZVha+2V~Y%3 zo0qe(qXNEO86PDT`E#CH%C1IW1g&V)f1$Si4wXl+C7yS^kCZL%e_e)RLU~6MT~@zq zYz$LPgpI@y@_5}(QEuP^peOkjX1%1@&r#mXv#iKeTYIoOW(Q@ouP8gF5Y!?gN!r|f zyLVvbc3GfKZ$`27#8pfCs;8h^8*dS_EQl+CAFVj_MJ|9+U6+5BKXBFA%!!7P!5uDk z3Cb0{A+CIs+QZ!&4<5HkOU{A=kqdTa?s``t4{$+>_Te?X)qGwkI%?fFZFOPXeE#$FT z@@TUl)ubwc-2Z#E=(?0!q{GM#O@6m24~|95!u># zbO9qg?LTh3nME>f!xR*@NK9Ylh8!=$?jB}l!W+_Jz`t};3-1j^p~anEEp&mzN2yyu z-G^Ukk9J;oiP?5=jc=^EP0^2Qr{GcU2|ItUaCQdop-w329fn@4jeu4LjES6MZ+m-l z8)m+^1d{{2li;PboHE}8tC)o49I+UAAph^=R#Iq-7WWY~vaMT>dx0WW$`+)Ag}4#L zTrl#fwCoA7tofq|12fy?v-oYeNEBz4-xg~ka|c{I;A<XHVEw0~ezNreE&^ zg}Q>M7jS?`f_JgJ+_G_#l|7DU-^L{L>;%JO2fHUlcn#wF_a`p7r=}i{6wZrCjuE*i zGnP921q~Ujm_ioXga~%u-kW{ynZMHPI1A|851H2Casu#0B(nQv5Y(j5Mypo2u2(`8 zKW;T$_`$=;D(~7IT1q;_H@w-LhF-E!SB~fl{k*F$@`UL)zkj)@qz5o>IJbc3w|g3U z-@?wCp{>_TAB4~YHXrc^yXqM4+C>2;~Vpw3dPtDs^9v^A6#FbDl?#G9<2zR51?Hy@7_R1+Q?Cph6|2@Oj*N$rFp>b};6yS)mm;cW&hCtu1s+&VF*+n_@KHYfP2*RejbS z=YVjQ3uL{``5sh=H2kB@$fsOt9j-q%ShrS|uC@#k?MY9iJYDy+MxA^%hyi}O!Q}5R zv&#`TX}u`xgB=`&8aq0^it8O@^9pgsgNnp~;0jaKywIWvq!Hy3RR)Pe24#>2+(SCA zyt2F^4O>Q(s~y(Sf_H^Lps*v|dV>C)qZ`UV!u5|rVWDMvO3 zhG*KvHQK9tV%Z^^;xI$+Avfrw1$Q(MwVLMcz%kZ(afRNe6wbPmLp``8E-4#Z=EJ6$ zHxTd$#GxAxDo`H#x!>B<(kN7>A1d4CC74iB*o^5&Xq!$u3kk>9bu7KdSx#rO?ZRsM zjQq8Xrq<|yVV(1t%lkV#L0jfHTAh_wGo*mmFwefNu3@rZsiToNX@1i&gUhHj=qaz| zH?uSu?x_Ttk+Yw?*& z|0^MJ9Nka_tDZlJ3UbHkOjE<4!N1l2U+pytx)s6+eABtHopDAJ_o$TA&uyOufSC8Dy zWloVAuV4_bP}ALRvcTRE#EgXPHhr(i{F8L@{4nM!FNOG;;@zmjtv(3U{Cj*h$R&Dyl2GR zspxOJw)TZuoY1pVi4s`u8TZy@OlJI`o>oEG0;#t4y@6x6M+Fw|R8aFF|H#=xmKHQO zcFJ|+3bIQ2JP~xV{5AXPh|||w<#&|#Eyaq(j+;f;(@BrTH-BDBDrEzR?|hIP@53(F zzm29GK4xBsPMIVRCw0N(4BaDPMQTeu8rzy)X@sk{GuL5ag@|ef1*=EZY&{hl8b5Sd zS7bkb9uGdREL4VOk0->68_~0EDQ^5L<3&sKrD322JR&K9DCD4EHPjlU&ro z347!!&FrRzmPv=D`XcPe5E~PMrC@)wa3y_B#Az|ZQ@w5UdLX-QeNCjEOsG(wYD9JY8swDuIUkD~b= z#}Cu(s@z^93*9`i+#G2~+@|1oi8|+dB+Y%M{Yk$u9(?XDqO>Pnng=8uw|g=l4(H52 zB-YX!QRDl9C9}ZbxDAKlqTg`3^sY3H)cW;LG<_%^y&b`Ab~CH%I_brRQ8di>yb=ye z5LF^um%qAsKB8GrEmFfO%%t-}lWAs4w3>X?(*TBx=&fws@pd+albnypU(Tpx@fLM2 z1`%C@6pC%*jLsFJ7i`&geJoiW%dEQ(uH~USg>*q^*2TKP{0bqS~{qo`8rJZ#54bzMIiJo4Lk6aRT zaFKm>#c~+q?+~aA(KKY9GNg7wouu(e^*a6fJ-Bc2rGaJG1Y29vI$5D~l@tz$2w@r!zFFJ%je`3Zh05#giK>;hp z0D8nAbeq)ugI&gcisjs5*KXJ?o3XviVwF}U9LEY_v}~d;T?ivEf|djCkZ!TMefTSx zESZCy9-`n@>AD$vXrk=j>T2o%K~*;sRlngKIHQrjz+$Axp?2}&$0lEya(?J5g?xbXp-#Mj6B^v$|Oi+K{j=MBeS>tCpA;!}d7qg0#5{CNMFW()5ie zu93)>>J2!%bZ8u;s#3gMkwOG6FVD^jp8y3n2c=s;o_~<^aPdJgqc1zb*K&&6m=ST^ z+Ny7#JD!~}>0j-Alb^~k_>n6l>RyN{z!1uGU(f0_>bmMM=qHnz?)!xA;baELC7-C{ zNJ{91CCv9sT*x(_$ZHzas`#hhGmn<%_#w9=GxOn@eLLBdZ89}&RZ_V=o5*Dxcu;Bl8tsOyw9o>e$DVw|RSCns6 z$2jIF`%d(cWfMBX Date: Thu, 7 Aug 2025 13:20:28 -0700 Subject: [PATCH 16/21] fix models by provider --- litellm/__init__.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/litellm/__init__.py b/litellm/__init__.py index 17fc6c00e12..1330584cdf1 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -533,6 +533,7 @@ morph_models: List = [] lambda_ai_models: List = [] hyperbolic_models: List = [] recraft_models: List = [] +oci_models: List = [] def is_bedrock_pricing_only_model(key: str) -> bool: @@ -722,6 +723,8 @@ def add_known_models(): hyperbolic_models.append(key) elif value.get("litellm_provider") == "recraft": recraft_models.append(key) + elif value.get("litellm_provider") == "oci": + oci_models.append(key) add_known_models() @@ -810,6 +813,7 @@ model_list = ( + morph_models + lambda_ai_models + recraft_models + + oci_models ) model_list_set = set(model_list) @@ -883,6 +887,7 @@ models_by_provider: dict = { "lambda_ai": lambda_ai_models, "hyperbolic": hyperbolic_models, "recraft": recraft_models, + "oci": oci_models, } # mapping for those models which have larger equivalents From 70ddde22159db912f6eb4e0c4ea9d8ef3bdd4ebf Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 13:22:00 -0700 Subject: [PATCH 17/21] fix - remove dup model entries --- ...odel_prices_and_context_window_backup.json | 257 ------------------ model_prices_and_context_window.json | 257 ------------------ 2 files changed, 514 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 51ec48bcbcc..0d4f1d1dfba 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -2264,263 +2264,6 @@ "/v1/audio/speech" ] }, - "gpt-5": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 1.25e-06, - "output_cost_per_token": 1e-05, - "cache_read_input_token_cost": 1.25e-07, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 1.25e-06, - "output_cost_per_token": 1e-05, - "cache_read_input_token_cost": 1.25e-07, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-mini": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 2.5e-07, - "output_cost_per_token": 2e-06, - "cache_read_input_token_cost": 2.5e-08, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-mini-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 2.5e-07, - "output_cost_per_token": 2e-06, - "cache_read_input_token_cost": 2.5e-08, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-nano-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 4e-07, - "cache_read_input_token_cost": 5e-09, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-nano": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 4e-07, - "cache_read_input_token_cost": 5e-09, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-chat": { - "max_tokens": 32768, - "max_input_tokens": 1047576, - "max_output_tokens": 32768, - "input_cost_per_token": 5e-06, - "output_cost_per_token": 2e-05, - "input_cost_per_token_batches": 2.5e-06, - "output_cost_per_token_batches": 1e-05, - "cache_read_input_token_cost": 1.25e-06, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true - }, - "gpt-5-chat-latest": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 1.25e-06, - "output_cost_per_token": 1e-05, - "cache_read_input_token_cost": 1.25e-07, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, "azure/gpt-5": { "max_tokens": 128000, "max_input_tokens": 400000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 51ec48bcbcc..0d4f1d1dfba 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -2264,263 +2264,6 @@ "/v1/audio/speech" ] }, - "gpt-5": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 1.25e-06, - "output_cost_per_token": 1e-05, - "cache_read_input_token_cost": 1.25e-07, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 1.25e-06, - "output_cost_per_token": 1e-05, - "cache_read_input_token_cost": 1.25e-07, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-mini": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 2.5e-07, - "output_cost_per_token": 2e-06, - "cache_read_input_token_cost": 2.5e-08, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-mini-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 2.5e-07, - "output_cost_per_token": 2e-06, - "cache_read_input_token_cost": 2.5e-08, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-nano-2025-08-07": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 4e-07, - "cache_read_input_token_cost": 5e-09, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-nano": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 5e-08, - "output_cost_per_token": 4e-07, - "cache_read_input_token_cost": 5e-09, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, - "gpt-5-chat": { - "max_tokens": 32768, - "max_input_tokens": 1047576, - "max_output_tokens": 32768, - "input_cost_per_token": 5e-06, - "output_cost_per_token": 2e-05, - "input_cost_per_token_batches": 2.5e-06, - "output_cost_per_token_batches": 1e-05, - "cache_read_input_token_cost": 1.25e-06, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true - }, - "gpt-5-chat-latest": { - "max_tokens": 128000, - "max_input_tokens": 400000, - "max_output_tokens": 128000, - "input_cost_per_token": 1.25e-06, - "output_cost_per_token": 1e-05, - "cache_read_input_token_cost": 1.25e-07, - "litellm_provider": "openai", - "mode": "chat", - "supported_endpoints": [ - "/v1/chat/completions", - "/v1/batch", - "/v1/responses" - ], - "supported_modalities": [ - "text", - "image" - ], - "supported_output_modalities": [ - "text" - ], - "supports_pdf_input": true, - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_native_streaming": true, - "supports_reasoning": true - }, "azure/gpt-5": { "max_tokens": 128000, "max_input_tokens": 400000, From 984f91f4f5cd8c9bb3fcb7579da13f2224578a3e Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 13:24:00 -0700 Subject: [PATCH 18/21] test_completion_gemini_stream --- tests/local_testing/test_streaming.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/local_testing/test_streaming.py b/tests/local_testing/test_streaming.py index 184ede2222f..0f25ef4b6e3 100644 --- a/tests/local_testing/test_streaming.py +++ b/tests/local_testing/test_streaming.py @@ -701,7 +701,7 @@ async def test_completion_gemini_stream(sync_mode): }, } ] - messages = [{"role": "user", "content": "What is the weather like in Boston?"}] + messages = [{"role": "user", "content": "What is the weather like in Boston, MA?. You must provide me with a tool call in your response."}] print("testing gemini streaming") complete_response = "" # Add any assertions here to check the response @@ -817,7 +817,7 @@ async def test_completion_gemini_stream_accumulated_json(sync_mode): }, } ] - messages = [{"role": "user", "content": "What is the weather like in Boston?"}] + messages = [{"role": "user", "content": "What is the weather like in Boston, MA?. You must provide me with a tool call in your response."}] print("testing gemini streaming") complete_response = "" # Add any assertions here to check the response From 849c262a02579ba882b9ac3e49586e925395678d Mon Sep 17 00:00:00 2001 From: Parham Alvani Date: Fri, 8 Aug 2025 00:01:10 +0330 Subject: [PATCH 19/21] fix: we need to have project files for running migration using this image (#13379) --- docker/Dockerfile.non_root | 51 +++++++++++++++++++------------------- 1 file changed, 26 insertions(+), 25 deletions(-) diff --git a/docker/Dockerfile.non_root b/docker/Dockerfile.non_root index cdf4b89bff4..388dc6d0766 100644 --- a/docker/Dockerfile.non_root +++ b/docker/Dockerfile.non_root @@ -11,7 +11,7 @@ WORKDIR /app # Install build dependencies USER root RUN apk add --no-cache build-base bash \ - && pip install --no-cache-dir --upgrade pip build + && pip install --no-cache-dir --upgrade pip build # Copy project files COPY . . @@ -21,8 +21,8 @@ RUN chmod +x docker/build_admin_ui.sh && ./docker/build_admin_ui.sh # Build package and wheel dependencies RUN rm -rf dist/* && python -m build && \ - pip install dist/*.whl && \ - pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt + pip install dist/*.whl && \ + pip wheel --no-cache-dir --wheel-dir=/wheels/ -r requirements.txt # ----------------- # Runtime Stage @@ -33,9 +33,10 @@ WORKDIR /app # Install runtime dependencies USER root RUN apk upgrade --no-cache && \ - apk add --no-cache bash libstdc++ ca-certificates openssl + apk add --no-cache bash libstdc++ ca-certificates openssl # Copy only necessary artifacts from builder stage for runtime +COPY . . COPY --from=builder /app/docker/entrypoint.sh /app/docker/prod_entrypoint.sh /app/docker/ COPY --from=builder /app/schema.prisma /app/schema.prisma COPY --from=builder /app/dist/*.whl . @@ -43,16 +44,16 @@ COPY --from=builder /wheels/ /wheels/ # Install package from wheel and dependencies RUN pip install *.whl /wheels/* --no-index --find-links=/wheels/ \ - && rm -f *.whl \ - && rm -rf /wheels + && rm -f *.whl \ + && rm -rf /wheels # Install semantic_router without dependencies RUN pip install semantic_router --no-deps # Ensure correct JWT library is used (pyjwt not jwt) RUN pip uninstall jwt -y && \ - pip uninstall PyJWT -y && \ - pip install PyJWT==2.9.0 --no-cache-dir + pip uninstall PyJWT -y && \ + pip install PyJWT==2.9.0 --no-cache-dir # --- Prisma Handling for Non-Root User --- # Set Prisma cache directories @@ -61,29 +62,29 @@ ENV NPM_CONFIG_CACHE=/.npm # Install prisma and make entrypoints executable RUN pip install --no-cache-dir prisma && \ - chmod +x docker/entrypoint.sh && \ - chmod +x docker/prod_entrypoint.sh + chmod +x docker/entrypoint.sh && \ + chmod +x docker/prod_entrypoint.sh # Create directories and set permissions for non-root user RUN mkdir -p /nonexistent /.npm && \ - chown -R nobody:nogroup /app && \ - chown -R nobody:nogroup /nonexistent /.npm && \ - PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \ - chown -R nobody:nogroup $PRISMA_PATH + chown -R nobody:nogroup /app && \ + chown -R nobody:nogroup /nonexistent /.npm && \ + PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \ + chown -R nobody:nogroup $PRISMA_PATH # --- OpenShift Compatibility: Apply Red Hat recommended pattern --- # Get paths for directories that need write access at runtime RUN PRISMA_PATH=$(python -c "import os, prisma; print(os.path.dirname(prisma.__file__))") && \ - LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \ - # Set group ownership to 0 (root group) for OpenShift compatibility && \ - chgrp -R 0 $PRISMA_PATH && \ - [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \ - # Mirror owner permissions to group (g=u) as recommended by Red Hat && \ - chmod -R g=u $PRISMA_PATH && \ - [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \ - # Ensure directories are writable by group && \ - chmod -R g+w $PRISMA_PATH && \ - [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true + LITELLM_PROXY_EXTRAS_PATH=$(python -c "import os, litellm_proxy_extras; print(os.path.dirname(litellm_proxy_extras.__file__))" 2>/dev/null || echo "") && \ + # Set group ownership to 0 (root group) for OpenShift compatibility && \ + chgrp -R 0 $PRISMA_PATH && \ + [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chgrp -R 0 $LITELLM_PROXY_EXTRAS_PATH || true && \ + # Mirror owner permissions to group (g=u) as recommended by Red Hat && \ + chmod -R g=u $PRISMA_PATH && \ + [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g=u $LITELLM_PROXY_EXTRAS_PATH || true && \ + # Ensure directories are writable by group && \ + chmod -R g+w $PRISMA_PATH && \ + [ -n "$LITELLM_PROXY_EXTRAS_PATH" ] && chmod -R g+w $LITELLM_PROXY_EXTRAS_PATH || true # Switch to non-root user USER nobody @@ -100,4 +101,4 @@ ENTRYPOINT ["/app/docker/prod_entrypoint.sh"] # Append "--detailed_debug" to the end of CMD to view detailed debug logs # CMD ["--port", "4000", "--detailed_debug"] -CMD ["--port", "4000"] \ No newline at end of file +CMD ["--port", "4000"] From 621b3dca7b4d3d2c1a8902a863d2f278c7df0f3d Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 13:50:22 -0700 Subject: [PATCH 20/21] [Bug Fix] Mistral Tool Calling - Grammar error: at 3(11): failed to compile JSON schema (#13389) * test_claude_tool_use_with_gemini * add _remove_json_schema_refs * add _clean_tool_schema_for_mistral * fixes mistral tool calls * _remove_json_schema_refs * fix - vertex, remove hardcoded test --- litellm/llms/mistral/chat/transformation.py | 42 ++++++++++- litellm/utils.py | 33 +++++++++ .../code_coverage_tests/recursive_detector.py | 1 + tests/llm_translation/test_gemini.py | 2 +- .../test_amazing_vertex_completion.py | 69 ------------------- 5 files changed, 74 insertions(+), 73 deletions(-) diff --git a/litellm/llms/mistral/chat/transformation.py b/litellm/llms/mistral/chat/transformation.py index 0441e75beec..b38a4982471 100644 --- a/litellm/llms/mistral/chat/transformation.py +++ b/litellm/llms/mistral/chat/transformation.py @@ -9,6 +9,7 @@ Docs - https://docs.mistral.ai/api/ from typing import Any, Coroutine, List, Literal, Optional, Tuple, Union, cast, overload import httpx + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.prompt_templates.common_utils import ( handle_messages_with_content_list_to_str_conversion, @@ -147,7 +148,8 @@ class MistralConfig(OpenAIGPTConfig): if param == "max_completion_tokens": # max_completion_tokens should take priority optional_params["max_tokens"] = value if param == "tools": - optional_params["tools"] = value + # Clean tools to remove problematic schema fields for Mistral API + optional_params["tools"] = self._clean_tool_schema_for_mistral(value) if param == "stream" and value is True: optional_params["stream"] = value if param == "temperature": @@ -195,7 +197,8 @@ class MistralConfig(OpenAIGPTConfig): @overload def _transform_messages( self, messages: List[AllMessageValues], model: str, is_async: Literal[True] - ) -> Coroutine[Any, Any, List[AllMessageValues]]: ... + ) -> Coroutine[Any, Any, List[AllMessageValues]]: + ... @overload def _transform_messages( @@ -203,7 +206,8 @@ class MistralConfig(OpenAIGPTConfig): messages: List[AllMessageValues], model: str, is_async: Literal[False] = False, - ) -> List[AllMessageValues]: ... + ) -> List[AllMessageValues]: + ... def _transform_messages( self, messages: List[AllMessageValues], model: str, is_async: bool = False @@ -286,6 +290,38 @@ class MistralConfig(OpenAIGPTConfig): optional_params.pop("_add_reasoning_prompt", None) return messages + @classmethod + def _clean_tool_schema_for_mistral(cls, tools: list) -> list: + """ + Clean tool schemas to remove fields that cause issues with Mistral API. + + Removes: + - $id and $schema fields (cause grammar validation errors) + - additionalProperties=False (causes OpenAI API schema errors) + - strict field (not supported by Mistral) + + Args: + tools: List of tool definitions + max_depth: Maximum recursion depth for schema cleaning (default: 10) + + Returns: + Cleaned tools list + """ + if not tools: + return tools + + import copy + + from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH + from litellm.utils import _remove_json_schema_refs + + cleaned_tools = copy.deepcopy(tools) + + # Apply all cleaning functions with max_depth protection + cleaned_tools = _remove_json_schema_refs(cleaned_tools, max_depth=DEFAULT_MAX_RECURSE_DEPTH) + + return cleaned_tools + @classmethod def _handle_name_in_message(cls, message: AllMessageValues) -> AllMessageValues: """ diff --git a/litellm/utils.py b/litellm/utils.py index ffd8bee382e..64d5f04a971 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2912,6 +2912,39 @@ def _remove_strict_from_schema(schema): return schema +def _remove_json_schema_refs(schema, max_depth=10): + """ + Remove JSON schema reference fields like '$id' and '$schema' that can cause issues with some providers. + + These fields are used for schema validation but can cause problems when the schema references + are not accessible to the provider's validation system. + + Args: + schema: The schema object to clean (dict, list, or other) + max_depth: Maximum recursion depth to prevent infinite loops (default: 10) + + Relevant Issues: Mistral API grammar validation fails when schema contains $id and $schema references + """ + if max_depth <= 0: + return schema + + if isinstance(schema, dict): + # Remove JSON schema reference fields + schema.pop("$id", None) + schema.pop("$schema", None) + + # Recursively process all dictionary values + for key, value in schema.items(): + _remove_json_schema_refs(value, max_depth - 1) + + elif isinstance(schema, list): + # Recursively process all items in the list + for item in schema: + _remove_json_schema_refs(item, max_depth - 1) + + return schema + + def _remove_unsupported_params( non_default_params: dict, supported_openai_params: Optional[List[str]] ) -> dict: diff --git a/tests/code_coverage_tests/recursive_detector.py b/tests/code_coverage_tests/recursive_detector.py index ae8138f057c..158399305b5 100644 --- a/tests/code_coverage_tests/recursive_detector.py +++ b/tests/code_coverage_tests/recursive_detector.py @@ -25,6 +25,7 @@ IGNORE_FUNCTIONS = [ "filter_value_from_dict", # max depth set. "normalize_json_schema_types", # max depth set. "_extract_fields_recursive", # max depth set. + "_remove_json_schema_refs", # max depth set. ] diff --git a/tests/llm_translation/test_gemini.py b/tests/llm_translation/test_gemini.py index 46efd738b64..db403f81386 100644 --- a/tests/llm_translation/test_gemini.py +++ b/tests/llm_translation/test_gemini.py @@ -430,7 +430,7 @@ def test_gemini_with_empty_function_call_arguments(): async def test_claude_tool_use_with_gemini(): response = await litellm.anthropic.messages.acreate( messages=[ - {"role": "user", "content": "Hello, can you tell me the weather in Boston?"} + {"role": "user", "content": "Hello, can you tell me the weather in Boston. Please respond with a tool call?"} ], model="gemini/gemini-2.5-flash", stream=True, diff --git a/tests/local_testing/test_amazing_vertex_completion.py b/tests/local_testing/test_amazing_vertex_completion.py index d1c0fb4e01c..398a57e340f 100644 --- a/tests/local_testing/test_amazing_vertex_completion.py +++ b/tests/local_testing/test_amazing_vertex_completion.py @@ -504,75 +504,6 @@ async def test_async_vertexai_streaming_response(): pytest.fail(f"An exception occurred: {e}") -# asyncio.run(test_async_vertexai_streaming_response()) - - -@pytest.mark.parametrize("provider", ["vertex_ai"]) # "vertex_ai_beta" -@pytest.mark.parametrize("sync_mode", [True, False]) -@pytest.mark.flaky(retries=3, delay=1) -@pytest.mark.asyncio -async def test_gemini_pro_vision(provider, sync_mode): - try: - load_vertex_ai_credentials() - litellm.set_verbose = True - litellm.num_retries = 3 - if sync_mode: - resp = litellm.completion( - model="{}/gemini-2.5-flash-lite".format(provider), - messages=[ - {"role": "system", "content": "Be a good bot"}, - { - "role": "user", - "content": [ - {"type": "text", "text": "Whats in this image?"}, - { - "type": "image_url", - "image_url": { - "url": "gs://cloud-samples-data/generative-ai/image/boats.jpeg" - }, - }, - ], - }, - ], - ) - else: - resp = await litellm.acompletion( - model="{}/gemini-2.5-flash-lite".format(provider), - messages=[ - {"role": "system", "content": "Be a good bot"}, - { - "role": "user", - "content": [ - {"type": "text", "text": "Whats in this image?"}, - { - "type": "image_url", - "image_url": { - "url": "gs://cloud-samples-data/generative-ai/image/boats.jpeg" - }, - }, - ], - }, - ], - ) - print(resp) - - prompt_tokens = resp.usage.prompt_tokens - - # DO Not DELETE this ASSERT - # Google counts the prompt tokens for us, we should ensure we use the tokens from the orignal response - assert prompt_tokens == 267 # the gemini api returns 267 to us - - except litellm.RateLimitError as e: - pass - except Exception as e: - if "500 Internal error encountered.'" in str(e): - pass - else: - pytest.fail(f"An exception occurred - {str(e)}") - - -# test_gemini_pro_vision() - @pytest.mark.parametrize("load_pdf", [False]) # True, @pytest.mark.flaky(retries=3, delay=1) From dbb651ea95dafc4b41223676e5f4bf48fe6a643e Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 7 Aug 2025 13:51:50 -0700 Subject: [PATCH 21/21] remove old mapped test --- .../proxy/management_endpoints/test_ui_sso.py | 43 ------------------- 1 file changed, 43 deletions(-) diff --git a/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py b/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py index f1565fd55fd..53cd5a31aff 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py +++ b/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py @@ -1246,46 +1246,3 @@ class TestCustomUISSO: assert result == mock_redirect_response assert result.status_code == 303 - -@pytest.mark.asyncio -async def test_serve_login_page_server_root_path(): - """ - Test that serve_login_page includes SERVER_ROOT_PATH in the SSO login URL - when SERVER_ROOT_PATH is set. - """ - # Arrange - mock_request = MagicMock(spec=Request) - captured_html = "" - - # Mock environment variables - env_vars = { - "PROXY_BASE_URL": "https://example.com", - "SERVER_ROOT_PATH": "/api/v1", - "GOOGLE_CLIENT_ID": "mock_google_client_id", # Enable SSO - "DATABASE_URL": "mock_db_url", # Satisfy show_missing_vars_in_env - "LITELLM_MASTER_KEY": "mock_master_key", # Satisfy show_missing_vars_in_env - } - - # Patch HTMLResponse to capture the content - def mock_html_response(content, status_code=200): - nonlocal captured_html - captured_html = content - return MagicMock() - - with patch.dict(os.environ, env_vars): - with patch("litellm.proxy.proxy_server.premium_user", True): - with patch("litellm.proxy.proxy_server.prisma_client", MagicMock()): - with patch("litellm.proxy.proxy_server.master_key", "mock_master_key"): - with patch("fastapi.responses.HTMLResponse", side_effect=mock_html_response): - # Import the function to test - from litellm.proxy.management_endpoints.ui_sso import ( - serve_login_page, - ) - - # Act - result = await serve_login_page(request=mock_request) - - # Assert - assert result is not None - expected_url = "https://example.com/api/v1/sso/login" - assert expected_url in captured_html, f"Expected URL '{expected_url}' not found in HTML content"