From 2045feda2e5984855977dab44c3377f6b3ea9048 Mon Sep 17 00:00:00 2001 From: Josh Date: Sat, 21 Mar 2026 14:12:00 -0400 Subject: [PATCH 001/117] feat(prometheus): add org budget metric definitions and label names --- litellm/types/integrations/prometheus.py | 26 ++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index 0856d8a6f9b..52962f3e3bd 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -185,6 +185,8 @@ class UserAPIKeyLabelNames(Enum): USER_AGENT = "user_agent" CALLBACK_NAME = "callback_name" STREAM = "stream" + ORG_ID = "org_id" + ORG_ALIAS = "org_alias" DEFINED_PROMETHEUS_METRICS = Literal[ @@ -207,6 +209,9 @@ DEFINED_PROMETHEUS_METRICS = Literal[ "litellm_remaining_team_budget_metric", "litellm_team_max_budget_metric", "litellm_team_budget_remaining_hours_metric", + "litellm_remaining_org_budget_metric", + "litellm_org_max_budget_metric", + "litellm_org_budget_remaining_hours_metric", "litellm_remaining_api_key_budget_metric", "litellm_api_key_max_budget_metric", "litellm_api_key_budget_remaining_hours_metric", @@ -490,6 +495,21 @@ class PrometheusMetricLabels: UserAPIKeyLabelNames.TEAM_ALIAS.value, ] + litellm_remaining_org_budget_metric = [ + UserAPIKeyLabelNames.ORG_ID.value, + UserAPIKeyLabelNames.ORG_ALIAS.value, + ] + + litellm_org_max_budget_metric = [ + UserAPIKeyLabelNames.ORG_ID.value, + UserAPIKeyLabelNames.ORG_ALIAS.value, + ] + + litellm_org_budget_remaining_hours_metric = [ + UserAPIKeyLabelNames.ORG_ID.value, + UserAPIKeyLabelNames.ORG_ALIAS.value, + ] + litellm_remaining_api_key_budget_metric = [ UserAPIKeyLabelNames.API_KEY_HASH.value, UserAPIKeyLabelNames.API_KEY_ALIAS.value, @@ -721,6 +741,12 @@ class UserAPIKeyLabelValues(BaseModel): stream: Annotated[ Optional[str], Field(..., alias=UserAPIKeyLabelNames.STREAM.value) ] = None + org_id: Annotated[ + Optional[str], Field(..., alias=UserAPIKeyLabelNames.ORG_ID.value) + ] = None + org_alias: Annotated[ + Optional[str], Field(..., alias=UserAPIKeyLabelNames.ORG_ALIAS.value) + ] = None @field_validator("stream", mode="before") @classmethod From d39bcce4be204939cea3bd47743920b43eac1125 Mon Sep 17 00:00:00 2001 From: Josh Date: Sat, 21 Mar 2026 14:29:00 -0400 Subject: [PATCH 002/117] feat(prometheus): implement org budget gauge metrics and setters --- litellm/integrations/prometheus.py | 140 ++++++++++++++++++++++++++++- 1 file changed, 139 insertions(+), 1 deletion(-) diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 357e0229fc6..2f3b86a5960 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -180,6 +180,31 @@ class PrometheusLogger(CustomLogger): ), ) + # Remaining Budget for Org + self.litellm_remaining_org_budget_metric = self._gauge_factory( + "litellm_remaining_org_budget_metric", + "Remaining budget for org", + labelnames=self.get_labels_for_metric( + "litellm_remaining_org_budget_metric" + ), + ) + + # Max Budget for Org + self.litellm_org_max_budget_metric = self._gauge_factory( + "litellm_org_max_budget_metric", + "Maximum budget set for org", + labelnames=self.get_labels_for_metric("litellm_org_max_budget_metric"), + ) + + # Org Budget Reset At + self.litellm_org_budget_remaining_hours_metric = self._gauge_factory( + "litellm_org_budget_remaining_hours_metric", + "Remaining hours for org budget to be reset", + labelnames=self.get_labels_for_metric( + "litellm_org_budget_remaining_hours_metric" + ), + ) + # Remaining Budget for API Key self.litellm_remaining_api_key_budget_metric = self._gauge_factory( "litellm_remaining_api_key_budget_metric", @@ -922,6 +947,9 @@ class PrometheusLogger(CustomLogger): user_api_team_alias = standard_logging_payload["metadata"][ "user_api_key_team_alias" ] + user_api_key_org_id = standard_logging_payload["metadata"].get( + "user_api_key_org_id" + ) output_tokens = standard_logging_payload["completion_tokens"] tokens_used = standard_logging_payload["total_tokens"] response_cost = standard_logging_payload["response_cost"] @@ -1026,6 +1054,7 @@ class PrometheusLogger(CustomLogger): litellm_params=litellm_params, response_cost=response_cost, user_id=user_id, + user_api_key_org_id=user_api_key_org_id, ) # set proxy virtual key rpm/tpm metrics @@ -1181,6 +1210,7 @@ class PrometheusLogger(CustomLogger): litellm_params: dict, response_cost: float, user_id: Optional[str] = None, + user_api_key_org_id: Optional[str] = None, ): _metadata = litellm_params.get("metadata") or {} _team_spend = _metadata.get("user_api_key_team_spend", None) @@ -1213,12 +1243,16 @@ class PrometheusLogger(CustomLogger): user_max_budget=_user_max_budget, response_cost=response_cost, ), + self._set_org_budget_metrics_after_api_request( + org_id=user_api_key_org_id, + response_cost=response_cost, + ), return_exceptions=True, ) for i, r in enumerate(results): if isinstance(r, Exception): verbose_logger.debug( - f"[Non-Blocking] Prometheus: Budget metric lookup {['key', 'team', 'user'][i]} failed: {r}" + f"[Non-Blocking] Prometheus: Budget metric lookup {['key', 'team', 'user', 'org'][i]} failed: {r}" ) def _increment_top_level_request_and_spend_metrics( @@ -2744,6 +2778,110 @@ class PrometheusLogger(CustomLogger): ) ) + async def _set_org_budget_metrics_after_api_request( + self, + org_id: Optional[str], + response_cost: float, + ): + """ + Set org budget metrics after an LLM API request + + - Fetches org info from db + - Sets org budget metrics + """ + if not org_id: + return + + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + return + + try: + org_row = await prisma_client.db.litellm_organizationtable.find_unique( + where={"organization_id": org_id}, + include={"litellm_budget_table": True}, + ) + except Exception as e: + verbose_logger.debug( + f"[Non-Blocking] Prometheus: Error getting org info: {str(e)}" + ) + return + + if org_row is None: + return + + org_alias = org_row.organization_alias or "" + spend = org_row.spend or 0.0 + budget_table = org_row.litellm_budget_table + max_budget = budget_table.max_budget if budget_table else None + budget_reset_at = ( + getattr(budget_table, "budget_reset_at", None) if budget_table else None + ) + + self._set_org_budget_metrics( + org_id=org_id, + org_alias=org_alias, + spend=spend, + max_budget=max_budget, + budget_reset_at=budget_reset_at, + ) + + def _set_org_budget_metrics( + self, + org_id: str, + org_alias: str, + spend: float, + max_budget: Optional[float], + budget_reset_at: Optional[datetime], + ): + """ + Set org budget metrics for a single org + + - Remaining Budget + - Max Budget + - Budget Reset At + """ + enum_values = UserAPIKeyLabelValues( + org_id=org_id, + org_alias=org_alias, + ) + + _labels = prometheus_label_factory( + supported_enum_labels=self.get_labels_for_metric( + metric_name="litellm_remaining_org_budget_metric" + ), + enum_values=enum_values, + ) + self.litellm_remaining_org_budget_metric.labels(**_labels).set( + self._safe_get_remaining_budget( + max_budget=max_budget, + spend=spend, + ) + ) + + if max_budget is not None: + _labels = prometheus_label_factory( + supported_enum_labels=self.get_labels_for_metric( + metric_name="litellm_org_max_budget_metric" + ), + enum_values=enum_values, + ) + self.litellm_org_max_budget_metric.labels(**_labels).set(max_budget) + + if budget_reset_at is not None: + _labels = prometheus_label_factory( + supported_enum_labels=self.get_labels_for_metric( + metric_name="litellm_org_budget_remaining_hours_metric" + ), + enum_values=enum_values, + ) + self.litellm_org_budget_remaining_hours_metric.labels(**_labels).set( + self._get_remaining_hours_for_budget_reset( + budget_reset_at=budget_reset_at + ) + ) + def _set_key_budget_metrics(self, user_api_key_dict: UserAPIKeyAuth): """ Set virtual key budget metrics From e1481e5dc1492fdd7495ca9b3341a1fe6b957fe7 Mon Sep 17 00:00:00 2001 From: Josh Date: Sat, 21 Mar 2026 16:17:00 -0400 Subject: [PATCH 003/117] test(prometheus): add unit tests for org budget metrics --- .../test_prometheus_user_team_metrics.py | 149 +++++++++++++++++- 1 file changed, 148 insertions(+), 1 deletion(-) diff --git a/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py b/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py index cd76ba1e863..39ac488db7c 100644 --- a/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py +++ b/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py @@ -2,7 +2,7 @@ Unit tests for Prometheus user and team count metrics """ from datetime import datetime, timezone -from unittest.mock import MagicMock, patch +from unittest.mock import AsyncMock, MagicMock, patch import pytest from prometheus_client import REGISTRY @@ -523,3 +523,150 @@ async def test_set_user_budget_metrics_after_api_request_inf_when_genuinely_no_b assert actual_value == float("inf"), ( "remaining_user_budget_metric should be +Inf when user truly has no budget" ) + + +# --------------------------------------------------------------------------- +# Org budget metric tests +# --------------------------------------------------------------------------- + + +def test_org_budget_metrics_initialized(prometheus_logger): + """Test that the 3 org budget gauge metrics are initialized.""" + assert hasattr(prometheus_logger, "litellm_remaining_org_budget_metric") + assert hasattr(prometheus_logger, "litellm_org_max_budget_metric") + assert hasattr(prometheus_logger, "litellm_org_budget_remaining_hours_metric") + assert prometheus_logger.litellm_remaining_org_budget_metric is not None + assert prometheus_logger.litellm_org_max_budget_metric is not None + assert prometheus_logger.litellm_org_budget_remaining_hours_metric is not None + + +def test_set_org_budget_metrics_remaining_budget(prometheus_logger): + """_set_org_budget_metrics sets remaining budget gauge correctly.""" + prometheus_logger.litellm_remaining_org_budget_metric = MagicMock() + prometheus_logger.litellm_org_max_budget_metric = MagicMock() + prometheus_logger.litellm_org_budget_remaining_hours_metric = MagicMock() + + prometheus_logger._set_org_budget_metrics( + org_id="org-abc", + org_alias="my-org", + spend=200.0, + max_budget=500.0, + budget_reset_at=None, + ) + + set_call = prometheus_logger.litellm_remaining_org_budget_metric.labels().set + set_call.assert_called_once() + actual = set_call.call_args[0][0] + assert abs(actual - 300.0) < 0.01, f"Expected 300.0, got {actual}" + + +def test_set_org_budget_metrics_max_budget(prometheus_logger): + """_set_org_budget_metrics sets max budget gauge when max_budget is not None.""" + prometheus_logger.litellm_remaining_org_budget_metric = MagicMock() + prometheus_logger.litellm_org_max_budget_metric = MagicMock() + prometheus_logger.litellm_org_budget_remaining_hours_metric = MagicMock() + + prometheus_logger._set_org_budget_metrics( + org_id="org-abc", + org_alias="my-org", + spend=100.0, + max_budget=1000.0, + budget_reset_at=None, + ) + + prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_called_once_with( + 1000.0 + ) + + +def test_set_org_budget_metrics_no_max_budget(prometheus_logger): + """_set_org_budget_metrics does not set max budget gauge when max_budget is None.""" + prometheus_logger.litellm_remaining_org_budget_metric = MagicMock() + prometheus_logger.litellm_org_max_budget_metric = MagicMock() + prometheus_logger.litellm_org_budget_remaining_hours_metric = MagicMock() + + prometheus_logger._set_org_budget_metrics( + org_id="org-abc", + org_alias="my-org", + spend=50.0, + max_budget=None, + budget_reset_at=None, + ) + + prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_not_called() + + +def test_set_org_budget_metrics_remaining_hours(prometheus_logger): + """_set_org_budget_metrics sets remaining hours gauge when budget_reset_at is set.""" + prometheus_logger.litellm_remaining_org_budget_metric = MagicMock() + prometheus_logger.litellm_org_max_budget_metric = MagicMock() + prometheus_logger.litellm_org_budget_remaining_hours_metric = MagicMock() + + future_reset = datetime(2099, 1, 1, tzinfo=timezone.utc) + prometheus_logger._set_org_budget_metrics( + org_id="org-abc", + org_alias="my-org", + spend=10.0, + max_budget=500.0, + budget_reset_at=future_reset, + ) + + prometheus_logger.litellm_org_budget_remaining_hours_metric.labels().set.assert_called_once() + + +@pytest.mark.asyncio +async def test_set_org_budget_metrics_after_api_request(prometheus_logger): + """_set_org_budget_metrics_after_api_request fetches from DB and sets gauges.""" + import sys + + prometheus_logger.litellm_remaining_org_budget_metric = MagicMock() + prometheus_logger.litellm_org_max_budget_metric = MagicMock() + prometheus_logger.litellm_org_budget_remaining_hours_metric = MagicMock() + + budget_mock = MagicMock() + budget_mock.max_budget = 1000.0 + budget_mock.budget_reset_at = datetime(2099, 1, 1, tzinfo=timezone.utc) + + org_mock = MagicMock() + org_mock.organization_id = "org-xyz" + org_mock.organization_alias = "test-org" + org_mock.spend = 300.0 + org_mock.litellm_budget_table = budget_mock + org_mock.model_dump.return_value = {} + + mock_prisma = MagicMock() + mock_prisma.db.litellm_organizationtable.find_unique = AsyncMock( + return_value=org_mock + ) + + mock_proxy_server = MagicMock() + mock_proxy_server.prisma_client = mock_prisma + + with patch.dict(sys.modules, {"litellm.proxy.proxy_server": mock_proxy_server}): + await prometheus_logger._set_org_budget_metrics_after_api_request( + org_id="org-xyz", + response_cost=0.0, + ) + + prometheus_logger.litellm_remaining_org_budget_metric.labels().set.assert_called_once() + prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_called_once_with( + 1000.0 + ) + prometheus_logger.litellm_org_budget_remaining_hours_metric.labels().set.assert_called_once() + + +@pytest.mark.asyncio +async def test_set_org_budget_metrics_after_api_request_no_org_id(prometheus_logger): + """_set_org_budget_metrics_after_api_request is a no-op when org_id is None.""" + prometheus_logger.litellm_remaining_org_budget_metric = MagicMock() + prometheus_logger.litellm_org_max_budget_metric = MagicMock() + prometheus_logger.litellm_org_budget_remaining_hours_metric = MagicMock() + + await prometheus_logger._set_org_budget_metrics_after_api_request( + org_id=None, + response_cost=1.0, + ) + + prometheus_logger.litellm_remaining_org_budget_metric.labels().set.assert_not_called() + prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_not_called() + prometheus_logger.litellm_org_budget_remaining_hours_metric.labels().set.assert_not_called() From 7fcf99ffaf2b2d931ca341fa90ea772e31131e9d Mon Sep 17 00:00:00 2001 From: Josh Date: Mon, 23 Mar 2026 19:07:01 -0400 Subject: [PATCH 004/117] Add org budget metrics to failure path --- litellm/integrations/prometheus.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 2f3b86a5960..1c37a657165 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -1441,6 +1441,9 @@ class PrometheusLogger(CustomLogger): user_api_team_alias = standard_logging_payload["metadata"][ "user_api_key_team_alias" ] + user_api_key_org_id = standard_logging_payload["metadata"].get( + "user_api_key_org_id" + ) try: self.litellm_llm_api_failed_requests_metric.labels( @@ -1456,6 +1459,10 @@ class PrometheusLogger(CustomLogger): ), ).inc() self.set_llm_deployment_failure_metrics(kwargs) + await self._set_org_budget_metrics_after_api_request( + org_id=user_api_key_org_id, + response_cost=0, + ) except Exception as e: verbose_logger.exception( "prometheus Layer Error(): Exception occured - {}".format(str(e)) From 8a58281cbff0c217dbed25aaf6799dbed6ac0b6b Mon Sep 17 00:00:00 2001 From: Josh Date: Mon, 23 Mar 2026 19:33:57 -0400 Subject: [PATCH 005/117] Add org budget metrics initialization at startup --- litellm/integrations/prometheus.py | 71 ++++++++++++++++--- .../test_prometheus_user_team_metrics.py | 61 +++++++++++++--- 2 files changed, 112 insertions(+), 20 deletions(-) diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 1c37a657165..fd4ea5a67e7 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -2571,6 +2571,37 @@ class PrometheusLogger(CustomLogger): data_type="users", ) + async def _initialize_org_budget_metrics(self): + """ + Initialize org budget metrics by reusing the generic pagination logic. + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + verbose_logger.debug( + "Prometheus: skipping org metrics initialization, DB not initialized" + ) + return + + async def fetch_orgs( + page_size: int, page: int + ) -> Tuple[list, Optional[int]]: + skip = (page - 1) * page_size + orgs = await prisma_client.db.litellm_organizationtable.find_many( + skip=skip, + take=page_size, + order={"created_at": "desc"}, + include={"litellm_budget_table": True}, + ) + total_count = await prisma_client.db.litellm_organizationtable.count() + return orgs, total_count + + await self._initialize_budget_metrics( + data_fetch_function=fetch_orgs, + set_metrics_function=self._set_org_list_budget_metrics, + data_type="orgs", + ) + async def initialize_remaining_budget_metrics(self): """ Handler for initializing remaining budget metrics for all teams to avoid metric discrepancies. @@ -2605,10 +2636,11 @@ class PrometheusLogger(CustomLogger): """ Helper to initialize remaining budget metrics for all teams, API keys, and users. """ - verbose_logger.debug("Emitting key, team, user budget metrics....") + verbose_logger.debug("Emitting key, team, user, org budget metrics....") await self._initialize_team_budget_metrics() await self._initialize_api_key_budget_metrics() await self._initialize_user_budget_metrics() + await self._initialize_org_budget_metrics() await self._initialize_user_and_team_count_metrics() async def _initialize_user_and_team_count_metrics(self): @@ -2664,6 +2696,20 @@ class PrometheusLogger(CustomLogger): for user in users: self._set_user_budget_metrics(user) + async def _set_org_list_budget_metrics(self, orgs: list): + """Helper function to set budget metrics for a list of orgs""" + for org in orgs: + budget_table = getattr(org, "litellm_budget_table", None) + self._set_org_budget_metrics( + org_id=org.organization_id or "", + org_alias=org.organization_alias or "", + spend=org.spend or 0.0, + max_budget=budget_table.max_budget if budget_table else None, + budget_reset_at=getattr(budget_table, "budget_reset_at", None) + if budget_table + else None, + ) + async def _set_team_budget_metrics_after_api_request( self, user_api_team: Optional[str], @@ -2793,21 +2839,24 @@ class PrometheusLogger(CustomLogger): """ Set org budget metrics after an LLM API request - - Fetches org info from db + - Fetches org info via cache (get_org_object) - Sets org budget metrics """ if not org_id: return - from litellm.proxy.proxy_server import prisma_client + from litellm.proxy.auth.auth_checks import get_org_object + from litellm.proxy.proxy_server import prisma_client, user_api_key_cache if prisma_client is None: return try: - org_row = await prisma_client.db.litellm_organizationtable.find_unique( - where={"organization_id": org_id}, - include={"litellm_budget_table": True}, + org_info = await get_org_object( + org_id=org_id, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + include_budget_table=True, ) except Exception as e: verbose_logger.debug( @@ -2815,12 +2864,12 @@ class PrometheusLogger(CustomLogger): ) return - if org_row is None: + if org_info is None: return - org_alias = org_row.organization_alias or "" - spend = org_row.spend or 0.0 - budget_table = org_row.litellm_budget_table + org_alias = org_info.organization_alias or "" + _total_org_spend = (org_info.spend or 0.0) + response_cost + budget_table = org_info.litellm_budget_table max_budget = budget_table.max_budget if budget_table else None budget_reset_at = ( getattr(budget_table, "budget_reset_at", None) if budget_table else None @@ -2829,7 +2878,7 @@ class PrometheusLogger(CustomLogger): self._set_org_budget_metrics( org_id=org_id, org_alias=org_alias, - spend=spend, + spend=_total_org_spend, max_budget=max_budget, budget_reset_at=budget_reset_at, ) diff --git a/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py b/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py index 39ac488db7c..9bcf08fdd71 100644 --- a/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py +++ b/tests/test_litellm/integrations/test_prometheus_user_team_metrics.py @@ -616,7 +616,7 @@ def test_set_org_budget_metrics_remaining_hours(prometheus_logger): @pytest.mark.asyncio async def test_set_org_budget_metrics_after_api_request(prometheus_logger): - """_set_org_budget_metrics_after_api_request fetches from DB and sets gauges.""" + """_set_org_budget_metrics_after_api_request uses cache helper and accounts for response_cost.""" import sys prometheus_logger.litellm_remaining_org_budget_metric = MagicMock() @@ -632,23 +632,29 @@ async def test_set_org_budget_metrics_after_api_request(prometheus_logger): org_mock.organization_alias = "test-org" org_mock.spend = 300.0 org_mock.litellm_budget_table = budget_mock - org_mock.model_dump.return_value = {} mock_prisma = MagicMock() - mock_prisma.db.litellm_organizationtable.find_unique = AsyncMock( - return_value=org_mock - ) - mock_proxy_server = MagicMock() mock_proxy_server.prisma_client = mock_prisma + mock_proxy_server.user_api_key_cache = MagicMock() - with patch.dict(sys.modules, {"litellm.proxy.proxy_server": mock_proxy_server}): + with ( + patch.dict(sys.modules, {"litellm.proxy.proxy_server": mock_proxy_server}), + patch( + "litellm.proxy.auth.auth_checks.get_org_object", + AsyncMock(return_value=org_mock), + ), + ): await prometheus_logger._set_org_budget_metrics_after_api_request( org_id="org-xyz", - response_cost=0.0, + response_cost=50.0, ) - prometheus_logger.litellm_remaining_org_budget_metric.labels().set.assert_called_once() + # remaining budget should reflect spend + response_cost (300 + 50 = 350, remaining = 1000 - 350 = 650) + remaining_call = prometheus_logger.litellm_remaining_org_budget_metric.labels().set.call_args + assert remaining_call is not None + assert remaining_call[0][0] == pytest.approx(650.0) + prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_called_once_with( 1000.0 ) @@ -670,3 +676,40 @@ async def test_set_org_budget_metrics_after_api_request_no_org_id(prometheus_log prometheus_logger.litellm_remaining_org_budget_metric.labels().set.assert_not_called() prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_not_called() prometheus_logger.litellm_org_budget_remaining_hours_metric.labels().set.assert_not_called() + + +@pytest.mark.asyncio +async def test_initialize_org_budget_metrics(prometheus_logger): + """_initialize_org_budget_metrics fetches all orgs and sets gauges for each.""" + import sys + + prometheus_logger.litellm_remaining_org_budget_metric = MagicMock() + prometheus_logger.litellm_org_max_budget_metric = MagicMock() + prometheus_logger.litellm_org_budget_remaining_hours_metric = MagicMock() + + budget_mock = MagicMock() + budget_mock.max_budget = 500.0 + budget_mock.budget_reset_at = None + + org_mock = MagicMock() + org_mock.organization_id = "org-init" + org_mock.organization_alias = "init-org" + org_mock.spend = 100.0 + org_mock.litellm_budget_table = budget_mock + + mock_prisma = MagicMock() + mock_prisma.db.litellm_organizationtable.find_many = AsyncMock( + return_value=[org_mock] + ) + mock_prisma.db.litellm_organizationtable.count = AsyncMock(return_value=1) + + mock_proxy_server = MagicMock() + mock_proxy_server.prisma_client = mock_prisma + + with patch.dict(sys.modules, {"litellm.proxy.proxy_server": mock_proxy_server}): + await prometheus_logger._initialize_org_budget_metrics() + + prometheus_logger.litellm_remaining_org_budget_metric.labels().set.assert_called_once() + prometheus_logger.litellm_org_max_budget_metric.labels().set.assert_called_once_with( + 500.0 + ) From efd73a44f4979a9414f8a3650c53c464f736a4aa Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Thu, 26 Mar 2026 16:12:00 -0700 Subject: [PATCH 006/117] fix(index.md): test commit --- docs/my-website/blog/security_update_march_2026/index.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/my-website/blog/security_update_march_2026/index.md b/docs/my-website/blog/security_update_march_2026/index.md index 9f812806d08..240ab8e8f91 100644 --- a/docs/my-website/blog/security_update_march_2026/index.md +++ b/docs/my-website/blog/security_update_march_2026/index.md @@ -16,7 +16,7 @@ import TabItem from '@theme/TabItem'; > **Status:** Active investigation > **Last updated:** March 26, 2026 -> **Update (March 26):** Added `checkmarx[.]zone` to [Indicators of compromise](#indicators-of-compromise-iocs). +> **Update (March 26):** Added `checkmarx[.]zone` to [Indicators of compromise](#indicators-of-compromise-iocs) > **Update (March 25):** Added community-contributed scripts for scanning GitHub Actions and GitLab CI pipelines for the compromised versions. See [How to check if you are affected](#how-to-check-if-you-are-affected). s/o [@Zach Fury](https://www.linkedin.com/in/fryware/) for these scripts. From cb66672017e4e091cd05622b00e430425afdfddc Mon Sep 17 00:00:00 2001 From: Nicholas Gigliotti Date: Mon, 16 Mar 2026 18:21:16 -0400 Subject: [PATCH 007/117] Replace hardcoded Bedrock native structured output model set with cost JSON lookup Move the source of truth for which Bedrock models support native structured outputs (outputConfig.textFormat) from a hardcoded substring set (BEDROCK_NATIVE_STRUCTURED_OUTPUT_MODELS) to the cost JSON via a new "supports_native_structured_output" flag. This makes it possible to add support for new models (including Claude Sonnet 4.6, which was missing) by updating the JSON alone, with no code changes needed. --- .../bedrock/chat/converse_transformation.py | 366 ++---- ...odel_prices_and_context_window_backup.json | 176 +-- model_prices_and_context_window.json | 176 +-- .../chat/test_converse_transformation.py | 1017 ++++++----------- 4 files changed, 655 insertions(+), 1080 deletions(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index dd8b1b0a69f..78549b160fa 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -91,34 +91,6 @@ UNSUPPORTED_BEDROCK_CONVERSE_BETA_PATTERNS = [ "compact-2026-01-12", # The compact beta feature is not currently supported on the Converse and ConverseStream APIs ] -# Models that support Bedrock's native structured outputs API (outputConfig.textFormat) -# Uses substring matching against the Bedrock model ID -# Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/structured-output.html -BEDROCK_NATIVE_STRUCTURED_OUTPUT_MODELS = { - # Anthropic Claude 4.5+ - "claude-haiku-4-5", - "claude-sonnet-4-5", - "claude-opus-4-5", - "claude-opus-4-6", - # Qwen3 - "qwen3", - # DeepSeek - "deepseek-v3.1", - # Gemma 3 - "gemma-3", - # MiniMax - "minimax-m2", - # Mistral (magistral-small excluded: broken constrained decoding on Bedrock) - "ministral", - "mistral-large-3", - "voxtral", - # Moonshot - "kimi-k2", - # NVIDIA - "nemotron-nano", - # OpenAI (gpt-oss excluded: broken constrained decoding, works via tool-call fallback) -} - class AmazonConverseConfig(BaseConfig): """ @@ -188,8 +160,7 @@ class AmazonConverseConfig(BaseConfig): if isinstance(content, list): has_guarded_text = any( - isinstance(item, dict) and item.get("type") == "guarded_text" - for item in content + isinstance(item, dict) and item.get("type") == "guarded_text" for item in content ) if has_guarded_text: continue # Skip this message if it already has guarded_text @@ -350,13 +321,9 @@ class AmazonConverseConfig(BaseConfig): # Check if the model is a Nova 2 model (matches nova-2-lite, nova-2-pro, etc.) # Also check for nova-2/ spec prefix for imported models - return model_without_region.startswith( - "amazon.nova-2-" - ) or model_without_region.startswith("nova-2/") + return model_without_region.startswith("amazon.nova-2-") or model_without_region.startswith("nova-2/") - def _map_web_search_options( - self, web_search_options: dict, model: str - ) -> Optional[BedrockToolBlock]: + def _map_web_search_options(self, web_search_options: dict, model: str) -> Optional[BedrockToolBlock]: """ Map web_search_options to Nova grounding systemTool. @@ -385,9 +352,7 @@ class AmazonConverseConfig(BaseConfig): # (unlike Anthropic), so we just enable grounding with no options return BedrockToolBlock(systemTool={"name": "nova_grounding"}) - def _transform_reasoning_effort_to_reasoning_config( - self, reasoning_effort: str - ) -> dict: + def _transform_reasoning_effort_to_reasoning_config(self, reasoning_effort: str) -> dict: """ Transform reasoning_effort parameter to Nova 2 reasoningConfig structure. @@ -432,9 +397,7 @@ class AmazonConverseConfig(BaseConfig): } } - def _handle_reasoning_effort_parameter( - self, model: str, reasoning_effort: str, optional_params: dict - ) -> None: + def _handle_reasoning_effort_parameter(self, model: str, reasoning_effort: str, optional_params: dict) -> None: """ Handle the reasoning_effort parameter based on the model type. @@ -471,9 +434,7 @@ class AmazonConverseConfig(BaseConfig): optional_params["reasoning_effort"] = reasoning_effort elif self._is_nova_2_model(model): # Nova 2 models: transform to reasoningConfig - reasoning_config = self._transform_reasoning_effort_to_reasoning_config( - reasoning_effort - ) + reasoning_config = self._transform_reasoning_effort_to_reasoning_config(reasoning_effort) optional_params.update(reasoning_config) else: # Anthropic and other models: convert to thinking parameter @@ -493,8 +454,7 @@ class AmazonConverseConfig(BaseConfig): budget = thinking.get("budget_tokens") if isinstance(budget, int) and budget < BEDROCK_MIN_THINKING_BUDGET_TOKENS: verbose_logger.debug( - "Bedrock requires thinking.budget_tokens >= %d, got %d. " - "Clamping to minimum.", + "Bedrock requires thinking.budget_tokens >= %d, got %d. Clamping to minimum.", BEDROCK_MIN_THINKING_BUDGET_TOKENS, budget, ) @@ -518,9 +478,7 @@ class AmazonConverseConfig(BaseConfig): "parallel_tool_calls", ] - if ( - "arn" in model - ): # we can't infer the model from the arn, so just add all params + if "arn" in model: # we can't infer the model from the arn, so just add all params supported_params.append("tools") supported_params.append("tool_choice") supported_params.append("thinking") @@ -542,9 +500,7 @@ class AmazonConverseConfig(BaseConfig): or base_model.startswith("meta.llama3-3") or base_model.startswith("meta.llama4") or base_model.startswith("amazon.nova") - or supports_function_calling( - model=model, custom_llm_provider=self.custom_llm_provider - ) + or supports_function_calling(model=model, custom_llm_provider=self.custom_llm_provider) ): supported_params.append("tools") @@ -554,9 +510,7 @@ class AmazonConverseConfig(BaseConfig): if litellm.utils.supports_tool_choice( model=model, custom_llm_provider=self.custom_llm_provider - ) or litellm.utils.supports_tool_choice( - model=base_model, custom_llm_provider=self.custom_llm_provider - ): + ) or litellm.utils.supports_tool_choice(model=base_model, custom_llm_provider=self.custom_llm_provider): # only anthropic and mistral support tool choice config. otherwise (E.g. cohere) will fail the call - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ToolChoice.html supported_params.append("tool_choice") @@ -575,9 +529,7 @@ class AmazonConverseConfig(BaseConfig): model=model, custom_llm_provider=self.custom_llm_provider, ) - or supports_reasoning( - model=base_model, custom_llm_provider=self.custom_llm_provider - ) + or supports_reasoning(model=base_model, custom_llm_provider=self.custom_llm_provider) ): supported_params.append("thinking") supported_params.append("reasoning_effort") @@ -602,9 +554,7 @@ class AmazonConverseConfig(BaseConfig): return ToolChoiceValuesBlock(auto={}) elif isinstance(tool_choice, dict): # only supported for anthropic + mistral models - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ToolChoice.html - specific_tool = SpecificToolChoiceBlock( - name=tool_choice.get("function", {}).get("name", "") - ) + specific_tool = SpecificToolChoiceBlock(name=tool_choice.get("function", {}).get("name", "")) return ToolChoiceValuesBlock(tool=specific_tool) else: raise litellm.utils.UnsupportedParamsError( @@ -624,15 +574,9 @@ class AmazonConverseConfig(BaseConfig): return ["mp4", "mov", "mkv", "webm", "flv", "mpeg", "mpg", "wmv", "3gp"] def get_all_supported_content_types(self) -> List[str]: - return ( - self.get_supported_image_types() - + self.get_supported_document_types() - + self.get_supported_video_types() - ) + return self.get_supported_image_types() + self.get_supported_document_types() + self.get_supported_video_types() - def is_computer_use_tool_used( - self, tools: Optional[List[OpenAIChatCompletionToolParam]], model: str - ) -> bool: + def is_computer_use_tool_used(self, tools: Optional[List[OpenAIChatCompletionToolParam]], model: str) -> bool: """Check if computer use tools are being used in the request.""" if tools is None: return False @@ -645,9 +589,7 @@ class AmazonConverseConfig(BaseConfig): return True return False - def _transform_computer_use_tools( - self, computer_use_tools: List[OpenAIChatCompletionToolParam] - ) -> List[dict]: + def _transform_computer_use_tools(self, computer_use_tools: List[OpenAIChatCompletionToolParam]) -> List[dict]: """Transform computer use tools to Bedrock format.""" transformed_tools: List[dict] = [] @@ -689,9 +631,7 @@ class AmazonConverseConfig(BaseConfig): def _separate_computer_use_tools( self, tools: List[OpenAIChatCompletionToolParam], model: str - ) -> Tuple[ - List[OpenAIChatCompletionToolParam], List[OpenAIChatCompletionToolParam] - ]: + ) -> Tuple[List[OpenAIChatCompletionToolParam], List[OpenAIChatCompletionToolParam]]: """ Separate computer use tools from regular function tools. @@ -764,10 +704,27 @@ class AmazonConverseConfig(BaseConfig): @staticmethod def _supports_native_structured_outputs(model: str) -> bool: - """Check if the Bedrock model supports native structured outputs (outputConfig.textFormat).""" - return any( - substring in model for substring in BEDROCK_NATIVE_STRUCTURED_OUTPUT_MODELS - ) + """Check if the Bedrock model supports native structured outputs (outputConfig.textFormat). + + Looks up the ``supports_native_structured_output`` flag in + ``litellm.model_cost`` (set in the cost JSON). + Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/structured-output.html + """ + from litellm.llms.bedrock.common_utils import get_bedrock_base_model + + base_model = get_bedrock_base_model(model) + + # Try direct lookup + info = litellm.model_cost.get(base_model) + + # Try without version suffix (e.g. "model-v1:0" -> "model-v1") + if info is None and ":" in base_model: + info = litellm.model_cost.get(base_model.rsplit(":", 1)[0]) + + if info is not None: + return info.get("supports_native_structured_output", False) is True + + return False @staticmethod def _add_additional_properties_to_schema(schema: dict) -> dict: @@ -790,25 +747,18 @@ class AmazonConverseConfig(BaseConfig): # Recurse into nested schemas if "properties" in result and isinstance(result["properties"], dict): result["properties"] = { - k: AmazonConverseConfig._add_additional_properties_to_schema(v) - for k, v in result["properties"].items() + k: AmazonConverseConfig._add_additional_properties_to_schema(v) for k, v in result["properties"].items() } if "items" in result and isinstance(result["items"], dict): - result["items"] = AmazonConverseConfig._add_additional_properties_to_schema( - result["items"] - ) + result["items"] = AmazonConverseConfig._add_additional_properties_to_schema(result["items"]) for defs_key in ("$defs", "definitions"): if defs_key in result and isinstance(result[defs_key], dict): result[defs_key] = { - k: AmazonConverseConfig._add_additional_properties_to_schema(v) - for k, v in result[defs_key].items() + k: AmazonConverseConfig._add_additional_properties_to_schema(v) for k, v in result[defs_key].items() } for key in ("anyOf", "allOf", "oneOf"): if key in result and isinstance(result[key], list): - result[key] = [ - AmazonConverseConfig._add_additional_properties_to_schema(item) - for item in result[key] - ] + result[key] = [AmazonConverseConfig._add_additional_properties_to_schema(item) for item in result[key]] return result @@ -838,9 +788,7 @@ class AmazonConverseConfig(BaseConfig): } """ if json_schema is not None: - json_schema = AmazonConverseConfig._add_additional_properties_to_schema( - json_schema - ) + json_schema = AmazonConverseConfig._add_additional_properties_to_schema(json_schema) schema_str = json.dumps(json_schema) if json_schema is not None else "{}" json_schema_def: JsonSchemaDefinition = {"schema": schema_str} if name is not None: @@ -862,14 +810,9 @@ class AmazonConverseConfig(BaseConfig): non_default_params: dict, optional_params: dict, ): - optional_params = self._add_tools_to_optional_params( - optional_params=optional_params, tools=tools - ) + optional_params = self._add_tools_to_optional_params(optional_params=optional_params, tools=tools) - if ( - "meta.llama3-3-70b-instruct-v1:0" in model - and non_default_params.get("stream", False) is True - ): + if "meta.llama3-3-70b-instruct-v1:0" in model and non_default_params.get("stream", False) is True: optional_params["fake_stream"] = True def map_openai_params( @@ -913,7 +856,9 @@ class AmazonConverseConfig(BaseConfig): ) if param == "tool_choice": _tool_choice_value = self.map_tool_choice_values( - model=model, tool_choice=value, drop_params=drop_params # type: ignore + model=model, + tool_choice=value, + drop_params=drop_params, # type: ignore ) if _tool_choice_value is not None: optional_params["tool_choice"] = _tool_choice_value @@ -1024,14 +969,10 @@ class AmazonConverseConfig(BaseConfig): json_schema=json_schema, description=description, ) - optional_params = self._add_tools_to_optional_params( - optional_params=optional_params, tools=[_tool] - ) + optional_params = self._add_tools_to_optional_params(optional_params=optional_params, tools=[_tool]) if ( - litellm.utils.supports_tool_choice( - model=model, custom_llm_provider=self.custom_llm_provider - ) + litellm.utils.supports_tool_choice(model=model, custom_llm_provider=self.custom_llm_provider) and not is_thinking_enabled ): optional_params["tool_choice"] = ToolChoiceValuesBlock( @@ -1043,9 +984,7 @@ class AmazonConverseConfig(BaseConfig): optional_params["json_mode"] = True return optional_params - def update_optional_params_with_thinking_tokens( - self, non_default_params: dict, optional_params: dict - ): + def update_optional_params_with_thinking_tokens(self, non_default_params: dict, optional_params: dict): """ Handles scenario where max tokens is not specified. For anthropic models (anthropic api/bedrock/vertex ai), this requires having the max tokens being set and being greater than the thinking token budget. @@ -1063,13 +1002,9 @@ class AmazonConverseConfig(BaseConfig): is_thinking_enabled = self.is_thinking_enabled(optional_params) is_max_tokens_in_request = self.is_max_tokens_in_request(non_default_params) if is_thinking_enabled and not is_max_tokens_in_request: - thinking_token_budget = cast(dict, optional_params["thinking"]).get( - "budget_tokens", None - ) + thinking_token_budget = cast(dict, optional_params["thinking"]).get("budget_tokens", None) if thinking_token_budget is not None: - optional_params["maxTokens"] = ( - thinking_token_budget + DEFAULT_MAX_TOKENS - ) + optional_params["maxTokens"] = thinking_token_budget + DEFAULT_MAX_TOKENS @overload def _get_cache_point_block( @@ -1135,23 +1070,15 @@ class AmazonConverseConfig(BaseConfig): if message["role"] == "system": system_prompt_indices.append(idx) if isinstance(message["content"], str) and message["content"]: - system_content_blocks.append( - SystemContentBlock(text=message["content"]) - ) - cache_block = self._get_cache_point_block( - message, block_type="system", model=model - ) + system_content_blocks.append(SystemContentBlock(text=message["content"])) + cache_block = self._get_cache_point_block(message, block_type="system", model=model) if cache_block: system_content_blocks.append(cache_block) elif isinstance(message["content"], list): for m in message["content"]: if m.get("type") == "text" and m.get("text"): - system_content_blocks.append( - SystemContentBlock(text=m["text"]) - ) - cache_block = self._get_cache_point_block( - m, block_type="system", model=model - ) + system_content_blocks.append(SystemContentBlock(text=m["text"])) + cache_block = self._get_cache_point_block(m, block_type="system", model=model) if cache_block: system_content_blocks.append(cache_block) if len(system_prompt_indices) > 0: @@ -1189,16 +1116,10 @@ class AmazonConverseConfig(BaseConfig): # Exceptions should not be stored in optional_params (this is a defensive fix) cleaned_params = filter_exceptions_from_params(optional_params) inference_params = safe_deep_copy(cleaned_params) - supported_converse_params = list( - AmazonConverseConfig.__annotations__.keys() - ) + ["top_k"] + supported_converse_params = list(AmazonConverseConfig.__annotations__.keys()) + ["top_k"] supported_tool_call_params = ["tools", "tool_choice"] supported_config_params = list(self.get_config_blocks().keys()) - total_supported_params = ( - supported_converse_params - + supported_tool_call_params - + supported_config_params - ) + total_supported_params = supported_converse_params + supported_tool_call_params + supported_config_params inference_params.pop("json_mode", None) # used for handling json_schema # Anthropic-only key. Bedrock expects `outputConfig` (camelCase) and # will reject `output_config` if it leaks through pass-through routes. @@ -1209,25 +1130,15 @@ class AmazonConverseConfig(BaseConfig): if request_metadata is not None: self._validate_request_metadata(request_metadata) - output_config: Optional[OutputConfigBlock] = inference_params.pop( - "outputConfig", None - ) - inference_params.pop( - "output_config", None - ) # Bedrock Converse doesn't support it + output_config: Optional[OutputConfigBlock] = inference_params.pop("outputConfig", None) + inference_params.pop("output_config", None) # Bedrock Converse doesn't support it # keep supported params in 'inference_params', and set all model-specific params in 'additional_request_params' - additional_request_params = { - k: v for k, v in inference_params.items() if k not in total_supported_params - } - inference_params = { - k: v for k, v in inference_params.items() if k in total_supported_params - } + additional_request_params = {k: v for k, v in inference_params.items() if k not in total_supported_params} + inference_params = {k: v for k, v in inference_params.items() if k in total_supported_params} # Handle parallel_tool_calls configuration - parallel_tool_use_config = additional_request_params.pop( - "_parallel_tool_use_config", None - ) + parallel_tool_use_config = additional_request_params.pop("_parallel_tool_use_config", None) if parallel_tool_use_config is not None and is_claude_4_5_on_bedrock(model): for key, value in parallel_tool_use_config.items(): if ( @@ -1242,9 +1153,7 @@ class AmazonConverseConfig(BaseConfig): additional_request_params.pop("parallel_tool_calls", None) # Only set the topK value in for models that support it - additional_request_params.update( - self._handle_top_k_value(model, inference_params) - ) + additional_request_params.update(self._handle_top_k_value(model, inference_params)) # Filter out internal/MCP-related parameters that shouldn't be sent to the API # These are LiteLLM internal parameters, not API parameters @@ -1253,9 +1162,7 @@ class AmazonConverseConfig(BaseConfig): # Filter out non-serializable objects (exceptions, callables, logging objects, etc.) # from additional_request_params to prevent JSON serialization errors # This filters: Exception objects, callable objects (functions), Logging objects, etc. - additional_request_params = filter_exceptions_from_params( - additional_request_params - ) + additional_request_params = filter_exceptions_from_params(additional_request_params) return ( inference_params, @@ -1302,9 +1209,7 @@ class AmazonConverseConfig(BaseConfig): # Only separate tools if computer use tools are actually present if filtered_tools and self.is_computer_use_tool_used(filtered_tools, model): # Separate computer use tools from regular function tools - computer_use_tools, regular_tools = self._separate_computer_use_tools( - filtered_tools, model - ) + computer_use_tools, regular_tools = self._separate_computer_use_tools(filtered_tools, model) # Process regular function tools using existing logic bedrock_tools = _bedrock_tools_pt(regular_tools) @@ -1365,9 +1270,7 @@ class AmazonConverseConfig(BaseConfig): anthropic_beta_list.append(computer_use_header) # Transform computer use tools to proper Bedrock format - transformed_computer_tools = self._transform_computer_use_tools( - computer_use_tools - ) + transformed_computer_tools = self._transform_computer_use_tools(computer_use_tools) additional_request_params["tools"] = transformed_computer_tools else: # No computer use tools, process all tools as regular tools @@ -1396,15 +1299,9 @@ class AmazonConverseConfig(BaseConfig): """ Bedrock doesn't support tool calling without `tools=` param specified. """ - if ( - "tools" not in optional_params - and messages is not None - and has_tool_call_blocks(messages) - ): + if "tools" not in optional_params and messages is not None and has_tool_call_blocks(messages): if litellm.modify_params: - optional_params["tools"] = add_dummy_tool( - custom_llm_provider="bedrock_converse" - ) + optional_params["tools"] = add_dummy_tool(custom_llm_provider="bedrock_converse") else: raise litellm.UnsupportedParamsError( message="Bedrock doesn't support tool calling without `tools=` param specified. Pass `tools=` param OR set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add dummy tool to the request.", @@ -1458,9 +1355,7 @@ class AmazonConverseConfig(BaseConfig): bedrock_tool_config: Optional[ToolConfigBlock] = None if len(bedrock_tools) > 0: - tool_choice_values: ToolChoiceValuesBlock = inference_params.pop( - "tool_choice", None - ) + tool_choice_values: ToolChoiceValuesBlock = inference_params.pop("tool_choice", None) bedrock_tool_config = ToolConfigBlock( tools=bedrock_tools, ) @@ -1470,9 +1365,7 @@ class AmazonConverseConfig(BaseConfig): data: CommonRequestObject = { "additionalModelRequestFields": additional_request_params, "system": system_content_blocks, - "inferenceConfig": self._transform_inference_params( - inference_params=inference_params - ), + "inferenceConfig": self._transform_inference_params(inference_params=inference_params), } # Handle all config blocks @@ -1502,14 +1395,10 @@ class AmazonConverseConfig(BaseConfig): litellm_params: dict, headers: Optional[dict] = None, ) -> RequestObject: - messages, system_content_blocks = self._transform_system_message( - messages, model=model - ) + messages, system_content_blocks = self._transform_system_message(messages, model=model) # Convert last user message to guarded_text if guardrailConfig is present - messages = self._convert_consecutive_user_messages_to_guarded_text( - messages, optional_params - ) + messages = self._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) ## TRANSFORMATION ## _data: CommonRequestObject = self._transform_request_helper( @@ -1520,13 +1409,11 @@ class AmazonConverseConfig(BaseConfig): headers=headers, ) - bedrock_messages = ( - await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model=model, - llm_provider="bedrock_converse", - user_continue_message=litellm_params.pop("user_continue_message", None), - ) + bedrock_messages = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( + messages=messages, + model=model, + llm_provider="bedrock_converse", + user_continue_message=litellm_params.pop("user_continue_message", None), ) data: RequestObject = {"messages": bedrock_messages, **_data} @@ -1560,14 +1447,10 @@ class AmazonConverseConfig(BaseConfig): litellm_params: dict, headers: Optional[dict] = None, ) -> RequestObject: - messages, system_content_blocks = self._transform_system_message( - messages, model=model - ) + messages, system_content_blocks = self._transform_system_message(messages, model=model) # Convert last user message to guarded_text if guardrailConfig is present - messages = self._convert_consecutive_user_messages_to_guarded_text( - messages, optional_params - ) + messages = self._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) _data: CommonRequestObject = self._transform_request_helper( model=model, @@ -1616,9 +1499,7 @@ class AmazonConverseConfig(BaseConfig): encoding=encoding, ) - def _transform_reasoning_content( - self, reasoning_content_blocks: List[BedrockConverseReasoningContentBlock] - ) -> str: + def _transform_reasoning_content(self, reasoning_content_blocks: List[BedrockConverseReasoningContentBlock]) -> str: """ Extract the reasoning text from the reasoning content blocks @@ -1634,9 +1515,7 @@ class AmazonConverseConfig(BaseConfig): self, thinking_blocks: List[BedrockConverseReasoningContentBlock] ) -> List[Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]]: """Return a consistent format for thinking blocks between Anthropic and Bedrock.""" - thinking_blocks_list: List[ - Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock] - ] = [] + thinking_blocks_list: List[Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]] = [] for block in thinking_blocks: if "reasoningText" in block: _thinking_block = ChatCompletionThinkingBlock(type="thinking") @@ -1672,21 +1551,11 @@ class AmazonConverseConfig(BaseConfig): cache_creation_input_tokens = usage["cacheWriteInputTokens"] input_tokens += cache_creation_input_tokens - prompt_tokens_details = PromptTokensDetailsWrapper( - cached_tokens=cache_read_input_tokens - ) - reasoning_tokens = ( - token_counter(text=reasoning_content, count_response_tokens=True) - if reasoning_content - else 0 - ) + prompt_tokens_details = PromptTokensDetailsWrapper(cached_tokens=cache_read_input_tokens) + reasoning_tokens = token_counter(text=reasoning_content, count_response_tokens=True) if reasoning_content else 0 completion_tokens_details = CompletionTokensDetailsWrapper( reasoning_tokens=reasoning_tokens, - text_tokens=( - output_tokens - reasoning_tokens - if reasoning_tokens > 0 - else output_tokens - ), + text_tokens=(output_tokens - reasoning_tokens if reasoning_tokens > 0 else output_tokens), ) openai_usage = Usage( prompt_tokens=input_tokens, @@ -1701,9 +1570,7 @@ class AmazonConverseConfig(BaseConfig): def get_tool_call_names( self, - tools: Optional[ - Union[List[ToolBlock], List[OpenAIChatCompletionToolParam]] - ] = None, + tools: Optional[Union[List[ToolBlock], List[OpenAIChatCompletionToolParam]]] = None, ) -> List[str]: if tools is None: return [] @@ -1742,13 +1609,8 @@ class AmazonConverseConfig(BaseConfig): try: tool_call_names = self.get_tool_call_names(tools) json_content = json.loads(message.content) - if ( - json_content.get("type") == "function" - and json_content.get("name") in tool_call_names - ): - tool_calls = [ - ChatCompletionMessageToolCall(function=Function(**json_content)) - ] + if json_content.get("type") == "function" and json_content.get("name") in tool_call_names: + tool_calls = [ChatCompletionMessageToolCall(function=Function(**json_content))] message.tool_calls = tool_calls message.content = None @@ -1777,9 +1639,7 @@ class AmazonConverseConfig(BaseConfig): """ content_str = "" tools: List[ChatCompletionToolCallChunk] = [] - reasoningContentBlocks: Optional[ - List[BedrockConverseReasoningContentBlock] - ] = None + reasoningContentBlocks: Optional[List[BedrockConverseReasoningContentBlock]] = None citationsContentBlocks: Optional[List[CitationsContentBlock]] = None for idx, content in enumerate(content_blocks): """ @@ -1796,9 +1656,7 @@ class AmazonConverseConfig(BaseConfig): if "toolUse" in content: ## check tool name was formatted by litellm _response_tool_name = content["toolUse"]["name"] - response_tool_name = get_bedrock_tool_name( - response_tool_name=_response_tool_name - ) + response_tool_name = get_bedrock_tool_name(response_tool_name=_response_tool_name) _function_chunk = ChatCompletionToolCallFunctionChunk( name=response_tool_name, arguments=json.dumps(content["toolUse"]["input"]), @@ -1849,11 +1707,7 @@ class AmazonConverseConfig(BaseConfig): """ try: response_data = json.loads(json_str) - if ( - isinstance(response_data, dict) - and "properties" in response_data - and len(response_data) == 1 - ): + if isinstance(response_data, dict) and "properties" in response_data and len(response_data) == 1: response_data = response_data["properties"] return json.dumps(response_data) except json.JSONDecodeError: @@ -1877,11 +1731,7 @@ class AmazonConverseConfig(BaseConfig): if not json_mode or not tools: return tools if tools else None - json_tool_indices = [ - i - for i, t in enumerate(tools) - if t["function"].get("name") == RESPONSE_FORMAT_TOOL_NAME - ] + json_tool_indices = [i for i, t in enumerate(tools) if t["function"].get("name") == RESPONSE_FORMAT_TOOL_NAME] if not json_tool_indices: # No json_tool_call found, return tools unchanged @@ -1889,14 +1739,10 @@ class AmazonConverseConfig(BaseConfig): if len(json_tool_indices) == len(tools): # All tools are json_tool_call — convert first one to content - verbose_logger.debug( - "Processing JSON tool call response for response_format" - ) + verbose_logger.debug("Processing JSON tool call response for response_format") json_mode_content_str: Optional[str] = tools[0]["function"].get("arguments") if json_mode_content_str is not None: - json_mode_content_str = AmazonConverseConfig._unwrap_bedrock_properties( - json_mode_content_str - ) + json_mode_content_str = AmazonConverseConfig._unwrap_bedrock_properties(json_mode_content_str) chat_completion_message["content"] = json_mode_content_str return None @@ -1906,13 +1752,9 @@ class AmazonConverseConfig(BaseConfig): first_idx = json_tool_indices[0] json_mode_args = tools[first_idx]["function"].get("arguments") if json_mode_args is not None: - json_mode_args = AmazonConverseConfig._unwrap_bedrock_properties( - json_mode_args - ) + json_mode_args = AmazonConverseConfig._unwrap_bedrock_properties(json_mode_args) existing = chat_completion_message.get("content") or "" - chat_completion_message["content"] = ( - existing + json_mode_args if existing else json_mode_args - ) + chat_completion_message["content"] = existing + json_mode_args if existing else json_mode_args real_tools = [t for i, t in enumerate(tools) if i not in json_tool_indices] return real_tools if real_tools else None @@ -1990,9 +1832,7 @@ class AmazonConverseConfig(BaseConfig): chat_completion_message: ChatCompletionResponseMessage = {"role": "assistant"} content_str = "" tools: List[ChatCompletionToolCallChunk] = [] - reasoningContentBlocks: Optional[ - List[BedrockConverseReasoningContentBlock] - ] = None + reasoningContentBlocks: Optional[List[BedrockConverseReasoningContentBlock]] = None citationsContentBlocks: Optional[List[CitationsContentBlock]] = None if message is not None: @@ -2011,17 +1851,11 @@ class AmazonConverseConfig(BaseConfig): provider_specific_fields["citationsContent"] = citationsContentBlocks if provider_specific_fields: - chat_completion_message[ - "provider_specific_fields" - ] = provider_specific_fields + chat_completion_message["provider_specific_fields"] = provider_specific_fields if reasoningContentBlocks is not None: - chat_completion_message[ - "reasoning_content" - ] = self._transform_reasoning_content(reasoningContentBlocks) - chat_completion_message[ - "thinking_blocks" - ] = self._transform_thinking_blocks(reasoningContentBlocks) + chat_completion_message["reasoning_content"] = self._transform_reasoning_content(reasoningContentBlocks) + chat_completion_message["thinking_blocks"] = self._transform_thinking_blocks(reasoningContentBlocks) chat_completion_message["content"] = content_str filtered_tools = self._filter_json_mode_tools( json_mode=json_mode, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c53ee943c58..13df25c1425 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -722,7 +722,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "anthropic.claude-haiku-4-5@20251001": { "cache_creation_input_token_cost": 1.25e-06, @@ -745,7 +746,8 @@ "supports_tool_choice": true, "supports_vision": true, "tool_use_system_prompt_tokens": 346, - "supports_native_streaming": true + "supports_native_streaming": true, + "supports_native_structured_output": true }, "anthropic.claude-3-5-sonnet-20240620-v1:0": { "input_cost_per_token": 3e-06, @@ -967,7 +969,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, @@ -997,7 +1000,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "global.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, @@ -1027,7 +1031,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "us.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, @@ -1057,7 +1062,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "eu.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, @@ -1087,7 +1093,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "au.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, @@ -1117,7 +1124,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, @@ -1147,7 +1155,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "global.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, @@ -1177,7 +1186,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "us.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, @@ -1207,7 +1217,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "eu.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, @@ -1237,7 +1248,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "au.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, @@ -1267,7 +1279,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "anthropic.claude-sonnet-4-20250514-v1:0": { "cache_creation_input_token_cost": 3.75e-06, @@ -1327,7 +1340,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "anthropic.claude-v1": { "input_cost_per_token": 8e-06, @@ -1577,7 +1591,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "apac.anthropic.claude-3-sonnet-20240229-v1:0": { "input_cost_per_token": 3e-06, @@ -1665,7 +1680,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "azure/ada": { "input_cost_per_token": 1e-07, @@ -12187,7 +12203,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "eu.anthropic.claude-3-5-sonnet-20240620-v1:0": { "input_cost_per_token": 3e-06, @@ -12401,7 +12418,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "eu.meta.llama3-2-1b-instruct-v1:0": { "input_cost_per_token": 1.3e-07, @@ -14604,17 +14622,14 @@ "uses_embed_content": true }, "vertex_ai/gemini-embedding-2-preview": { - "input_cost_per_audio_per_second": 0.00016, - "input_cost_per_image": 0.00012, - "input_cost_per_token": 2e-07, - "input_cost_per_video_per_second": 0.00079, + "input_cost_per_token": 1.5e-07, "litellm_provider": "vertex_ai", "max_input_tokens": 8192, "max_tokens": 8192, "mode": "embedding", "output_cost_per_token": 0, "output_vector_size": 3072, - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing", + "source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal", "supports_multimodal": true, "uses_embed_content": true }, @@ -14631,18 +14646,6 @@ "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing", "uses_embed_content": true }, - "vertex_ai/gemini-embedding-2-preview": { - "input_cost_per_token": 1.5e-07, - "litellm_provider": "vertex_ai", - "max_input_tokens": 8192, - "max_tokens": 8192, - "mode": "embedding", - "output_cost_per_token": 0, - "output_vector_size": 3072, - "source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal", - "supports_multimodal": true, - "uses_embed_content": true - }, "gemini/gemini-embedding-001": { "input_cost_per_token": 1.5e-07, "litellm_provider": "gemini", @@ -16713,7 +16716,8 @@ "mode": "chat", "output_cost_per_token": 2.9e-07, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "google.gemma-3-27b-it": { "input_cost_per_token": 2.3e-07, @@ -16724,7 +16728,8 @@ "mode": "chat", "output_cost_per_token": 3.8e-07, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "google.gemma-3-4b-it": { "input_cost_per_token": 4e-08, @@ -16735,7 +16740,8 @@ "mode": "chat", "output_cost_per_token": 8e-08, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "google_pse/search": { "input_cost_per_query": 0.005, @@ -16770,7 +16776,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "global.anthropic.claude-sonnet-4-20250514-v1:0": { "cache_creation_input_token_cost": 3.75e-06, @@ -16822,7 +16829,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "global.amazon.nova-2-lite-v1:0": { "cache_read_input_token_cost": 7.5e-08, @@ -20551,7 +20559,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "jp.anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.375e-06, @@ -20573,7 +20582,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "lambda_ai/deepseek-llama3.3-70b": { "input_cost_per_token": 2e-07, @@ -21268,7 +21278,8 @@ "max_tokens": 8192, "mode": "chat", "output_cost_per_token": 1.2e-06, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "minimax.minimax-m2.1": { "input_cost_per_token": 3e-07, @@ -21281,7 +21292,8 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_native_structured_output": true }, "minimax/speech-02-hd": { "input_cost_per_character": 0.0001, @@ -21424,7 +21436,8 @@ "mode": "chat", "output_cost_per_token": 2e-07, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.ministral-3-3b-instruct": { "input_cost_per_token": 1e-07, @@ -21435,7 +21448,8 @@ "mode": "chat", "output_cost_per_token": 1e-07, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.ministral-3-8b-instruct": { "input_cost_per_token": 1.5e-07, @@ -21446,7 +21460,8 @@ "mode": "chat", "output_cost_per_token": 1.5e-07, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.mistral-7b-instruct-v0:2": { "input_cost_per_token": 1.5e-07, @@ -21488,7 +21503,8 @@ "mode": "chat", "output_cost_per_token": 1.5e-06, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.mistral-small-2402-v1:0": { "input_cost_per_token": 1e-06, @@ -21519,7 +21535,8 @@ "mode": "chat", "output_cost_per_token": 4e-08, "supports_audio_input": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.voxtral-small-24b-2507": { "input_cost_per_token": 1e-07, @@ -21530,7 +21547,8 @@ "mode": "chat", "output_cost_per_token": 3e-07, "supports_audio_input": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral/codestral-2405": { "input_cost_per_token": 1e-06, @@ -22217,7 +22235,8 @@ "mode": "chat", "output_cost_per_token": 2.5e-06, "supports_reasoning": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "moonshotai.kimi-k2.5": { "input_cost_per_token": 6e-07, @@ -22231,7 +22250,8 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": true, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_native_structured_output": true }, "moonshot/kimi-k2-0711-preview": { "cache_read_input_token_cost": 1.5e-07, @@ -23069,7 +23089,8 @@ "mode": "chat", "output_cost_per_token": 6e-07, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "nvidia.nemotron-nano-9b-v2": { "input_cost_per_token": 6e-08, @@ -23079,7 +23100,8 @@ "max_tokens": 8192, "mode": "chat", "output_cost_per_token": 2.3e-07, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "nvidia.nemotron-nano-3-30b": { "input_cost_per_token": 6e-08, @@ -23092,7 +23114,8 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_native_structured_output": true }, "o1": { "cache_read_input_token_cost": 7.5e-06, @@ -26176,7 +26199,8 @@ "output_cost_per_token": 1.8e-06, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "qwen.qwen3-235b-a22b-2507-v1:0": { "input_cost_per_token": 2.2e-07, @@ -26188,7 +26212,8 @@ "output_cost_per_token": 8.8e-07, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "qwen.qwen3-coder-30b-a3b-v1:0": { "input_cost_per_token": 1.5e-07, @@ -26200,7 +26225,8 @@ "output_cost_per_token": 6e-07, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "qwen.qwen3-32b-v1:0": { "input_cost_per_token": 1.5e-07, @@ -26212,7 +26238,8 @@ "output_cost_per_token": 6e-07, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "qwen.qwen3-next-80b-a3b": { "input_cost_per_token": 1.5e-07, @@ -26223,7 +26250,8 @@ "mode": "chat", "output_cost_per_token": 1.2e-06, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "qwen.qwen3-vl-235b-a22b": { "input_cost_per_token": 5.3e-07, @@ -26235,7 +26263,8 @@ "output_cost_per_token": 2.66e-06, "supports_function_calling": true, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "qwen.qwen3-coder-next": { "input_cost_per_token": 5e-07, @@ -26248,7 +26277,8 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_native_structured_output": true }, "recraft/recraftv2": { "litellm_provider": "recraft", @@ -28260,7 +28290,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "us.anthropic.claude-3-5-sonnet-20240620-v1:0": { "input_cost_per_token": 3e-06, @@ -28418,7 +28449,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "au.anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.375e-06, @@ -28439,7 +28471,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "us.anthropic.claude-opus-4-20250514-v1:0": { "cache_creation_input_token_cost": 1.875e-05, @@ -28491,7 +28524,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "global.anthropic.claude-opus-4-5-20251101-v1:0": { "cache_creation_input_token_cost": 6.25e-06, @@ -28517,7 +28551,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "eu.anthropic.claude-opus-4-5-20251101-v1:0": { "cache_creation_input_token_cost": 6.25e-06, @@ -28543,7 +28578,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "us.anthropic.claude-sonnet-4-20250514-v1:0": { "cache_creation_input_token_cost": 3.75e-06, @@ -31147,7 +31183,9 @@ "mode": "chat", "output_cost_per_token": 3.2e-06, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#glm-models", - "supported_regions": ["global"], + "supported_regions": [ + "global" + ], "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c53ee943c58..13df25c1425 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -722,7 +722,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "anthropic.claude-haiku-4-5@20251001": { "cache_creation_input_token_cost": 1.25e-06, @@ -745,7 +746,8 @@ "supports_tool_choice": true, "supports_vision": true, "tool_use_system_prompt_tokens": 346, - "supports_native_streaming": true + "supports_native_streaming": true, + "supports_native_structured_output": true }, "anthropic.claude-3-5-sonnet-20240620-v1:0": { "input_cost_per_token": 3e-06, @@ -967,7 +969,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, @@ -997,7 +1000,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "global.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, @@ -1027,7 +1031,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "us.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, @@ -1057,7 +1062,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "eu.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, @@ -1087,7 +1093,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "au.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, @@ -1117,7 +1124,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, @@ -1147,7 +1155,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "global.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, @@ -1177,7 +1186,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "us.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, @@ -1207,7 +1217,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "eu.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, @@ -1237,7 +1248,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "au.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, @@ -1267,7 +1279,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "anthropic.claude-sonnet-4-20250514-v1:0": { "cache_creation_input_token_cost": 3.75e-06, @@ -1327,7 +1340,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "anthropic.claude-v1": { "input_cost_per_token": 8e-06, @@ -1577,7 +1591,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "apac.anthropic.claude-3-sonnet-20240229-v1:0": { "input_cost_per_token": 3e-06, @@ -1665,7 +1680,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "azure/ada": { "input_cost_per_token": 1e-07, @@ -12187,7 +12203,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "eu.anthropic.claude-3-5-sonnet-20240620-v1:0": { "input_cost_per_token": 3e-06, @@ -12401,7 +12418,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "eu.meta.llama3-2-1b-instruct-v1:0": { "input_cost_per_token": 1.3e-07, @@ -14604,17 +14622,14 @@ "uses_embed_content": true }, "vertex_ai/gemini-embedding-2-preview": { - "input_cost_per_audio_per_second": 0.00016, - "input_cost_per_image": 0.00012, - "input_cost_per_token": 2e-07, - "input_cost_per_video_per_second": 0.00079, + "input_cost_per_token": 1.5e-07, "litellm_provider": "vertex_ai", "max_input_tokens": 8192, "max_tokens": 8192, "mode": "embedding", "output_cost_per_token": 0, "output_vector_size": 3072, - "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing", + "source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal", "supports_multimodal": true, "uses_embed_content": true }, @@ -14631,18 +14646,6 @@ "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing", "uses_embed_content": true }, - "vertex_ai/gemini-embedding-2-preview": { - "input_cost_per_token": 1.5e-07, - "litellm_provider": "vertex_ai", - "max_input_tokens": 8192, - "max_tokens": 8192, - "mode": "embedding", - "output_cost_per_token": 0, - "output_vector_size": 3072, - "source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal", - "supports_multimodal": true, - "uses_embed_content": true - }, "gemini/gemini-embedding-001": { "input_cost_per_token": 1.5e-07, "litellm_provider": "gemini", @@ -16713,7 +16716,8 @@ "mode": "chat", "output_cost_per_token": 2.9e-07, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "google.gemma-3-27b-it": { "input_cost_per_token": 2.3e-07, @@ -16724,7 +16728,8 @@ "mode": "chat", "output_cost_per_token": 3.8e-07, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "google.gemma-3-4b-it": { "input_cost_per_token": 4e-08, @@ -16735,7 +16740,8 @@ "mode": "chat", "output_cost_per_token": 8e-08, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "google_pse/search": { "input_cost_per_query": 0.005, @@ -16770,7 +16776,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "global.anthropic.claude-sonnet-4-20250514-v1:0": { "cache_creation_input_token_cost": 3.75e-06, @@ -16822,7 +16829,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "global.amazon.nova-2-lite-v1:0": { "cache_read_input_token_cost": 7.5e-08, @@ -20551,7 +20559,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "jp.anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.375e-06, @@ -20573,7 +20582,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "lambda_ai/deepseek-llama3.3-70b": { "input_cost_per_token": 2e-07, @@ -21268,7 +21278,8 @@ "max_tokens": 8192, "mode": "chat", "output_cost_per_token": 1.2e-06, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "minimax.minimax-m2.1": { "input_cost_per_token": 3e-07, @@ -21281,7 +21292,8 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_native_structured_output": true }, "minimax/speech-02-hd": { "input_cost_per_character": 0.0001, @@ -21424,7 +21436,8 @@ "mode": "chat", "output_cost_per_token": 2e-07, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.ministral-3-3b-instruct": { "input_cost_per_token": 1e-07, @@ -21435,7 +21448,8 @@ "mode": "chat", "output_cost_per_token": 1e-07, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.ministral-3-8b-instruct": { "input_cost_per_token": 1.5e-07, @@ -21446,7 +21460,8 @@ "mode": "chat", "output_cost_per_token": 1.5e-07, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.mistral-7b-instruct-v0:2": { "input_cost_per_token": 1.5e-07, @@ -21488,7 +21503,8 @@ "mode": "chat", "output_cost_per_token": 1.5e-06, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.mistral-small-2402-v1:0": { "input_cost_per_token": 1e-06, @@ -21519,7 +21535,8 @@ "mode": "chat", "output_cost_per_token": 4e-08, "supports_audio_input": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral.voxtral-small-24b-2507": { "input_cost_per_token": 1e-07, @@ -21530,7 +21547,8 @@ "mode": "chat", "output_cost_per_token": 3e-07, "supports_audio_input": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "mistral/codestral-2405": { "input_cost_per_token": 1e-06, @@ -22217,7 +22235,8 @@ "mode": "chat", "output_cost_per_token": 2.5e-06, "supports_reasoning": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "moonshotai.kimi-k2.5": { "input_cost_per_token": 6e-07, @@ -22231,7 +22250,8 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": true, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_native_structured_output": true }, "moonshot/kimi-k2-0711-preview": { "cache_read_input_token_cost": 1.5e-07, @@ -23069,7 +23089,8 @@ "mode": "chat", "output_cost_per_token": 6e-07, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "nvidia.nemotron-nano-9b-v2": { "input_cost_per_token": 6e-08, @@ -23079,7 +23100,8 @@ "max_tokens": 8192, "mode": "chat", "output_cost_per_token": 2.3e-07, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "nvidia.nemotron-nano-3-30b": { "input_cost_per_token": 6e-08, @@ -23092,7 +23114,8 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_native_structured_output": true }, "o1": { "cache_read_input_token_cost": 7.5e-06, @@ -26176,7 +26199,8 @@ "output_cost_per_token": 1.8e-06, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "qwen.qwen3-235b-a22b-2507-v1:0": { "input_cost_per_token": 2.2e-07, @@ -26188,7 +26212,8 @@ "output_cost_per_token": 8.8e-07, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "qwen.qwen3-coder-30b-a3b-v1:0": { "input_cost_per_token": 1.5e-07, @@ -26200,7 +26225,8 @@ "output_cost_per_token": 6e-07, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "qwen.qwen3-32b-v1:0": { "input_cost_per_token": 1.5e-07, @@ -26212,7 +26238,8 @@ "output_cost_per_token": 6e-07, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "qwen.qwen3-next-80b-a3b": { "input_cost_per_token": 1.5e-07, @@ -26223,7 +26250,8 @@ "mode": "chat", "output_cost_per_token": 1.2e-06, "supports_function_calling": true, - "supports_system_messages": true + "supports_system_messages": true, + "supports_native_structured_output": true }, "qwen.qwen3-vl-235b-a22b": { "input_cost_per_token": 5.3e-07, @@ -26235,7 +26263,8 @@ "output_cost_per_token": 2.66e-06, "supports_function_calling": true, "supports_system_messages": true, - "supports_vision": true + "supports_vision": true, + "supports_native_structured_output": true }, "qwen.qwen3-coder-next": { "input_cost_per_token": 5e-07, @@ -26248,7 +26277,8 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://aws.amazon.com/bedrock/pricing/", + "supports_native_structured_output": true }, "recraft/recraftv2": { "litellm_provider": "recraft", @@ -28260,7 +28290,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "us.anthropic.claude-3-5-sonnet-20240620-v1:0": { "input_cost_per_token": 3e-06, @@ -28418,7 +28449,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "au.anthropic.claude-haiku-4-5-20251001-v1:0": { "cache_creation_input_token_cost": 1.375e-06, @@ -28439,7 +28471,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 346 + "tool_use_system_prompt_tokens": 346, + "supports_native_structured_output": true }, "us.anthropic.claude-opus-4-20250514-v1:0": { "cache_creation_input_token_cost": 1.875e-05, @@ -28491,7 +28524,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "global.anthropic.claude-opus-4-5-20251101-v1:0": { "cache_creation_input_token_cost": 6.25e-06, @@ -28517,7 +28551,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "eu.anthropic.claude-opus-4-5-20251101-v1:0": { "cache_creation_input_token_cost": 6.25e-06, @@ -28543,7 +28578,8 @@ "supports_response_schema": true, "supports_tool_choice": true, "supports_vision": true, - "tool_use_system_prompt_tokens": 159 + "tool_use_system_prompt_tokens": 159, + "supports_native_structured_output": true }, "us.anthropic.claude-sonnet-4-20250514-v1:0": { "cache_creation_input_token_cost": 3.75e-06, @@ -31147,7 +31183,9 @@ "mode": "chat", "output_cost_per_token": 3.2e-06, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#glm-models", - "supported_regions": ["global"], + "supported_regions": [ + "global" + ], "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index e9aaa97a421..bf957567a99 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -6,9 +6,7 @@ import sys import pytest from fastapi.testclient import TestClient -sys.path.insert( - 0, os.path.abspath("../../../../..") -) # Adds the parent directory to the system path +sys.path.insert(0, os.path.abspath("../../../../..")) # Adds the parent directory to the system path from unittest.mock import MagicMock, patch import litellm @@ -37,10 +35,7 @@ def test_transform_usage(): ) assert openai_usage.completion_tokens == usage["outputTokens"] assert openai_usage.total_tokens == usage["totalTokens"] - assert ( - openai_usage.prompt_tokens_details.cached_tokens - == usage["cacheReadInputTokens"] - ) + assert openai_usage.prompt_tokens_details.cached_tokens == usage["cacheReadInputTokens"] assert openai_usage._cache_creation_input_tokens == usage["cacheWriteInputTokens"] assert openai_usage._cache_read_input_tokens == usage["cacheReadInputTokens"] # completion_tokens_details should always be populated @@ -194,14 +189,10 @@ def test_apply_tool_call_transformation_if_needed(): role="user", content=json.dumps(tool_response), ) - transformed_message, _ = config.apply_tool_call_transformation_if_needed( - message, tool_calls - ) + transformed_message, _ = config.apply_tool_call_transformation_if_needed(message, tool_calls) assert len(transformed_message.tool_calls) == 1 assert transformed_message.tool_calls[0].function.name == "test_function" - assert transformed_message.tool_calls[0].function.arguments == json.dumps( - tool_response["parameters"] - ) + assert transformed_message.tool_calls[0].function.arguments == json.dumps(tool_response["parameters"]) def test_transform_tool_call_with_cache_control(): @@ -250,12 +241,7 @@ def test_transform_tool_call_with_cache_control(): print(function_out_msg) assert function_out_msg["toolSpec"]["name"] == "get_location" assert function_out_msg["toolSpec"]["description"] == "Get the user's location" - assert ( - function_out_msg["toolSpec"]["inputSchema"]["json"]["properties"]["location"][ - "type" - ] - == "string" - ) + assert function_out_msg["toolSpec"]["inputSchema"]["json"]["properties"]["location"]["type"] == "string" transformed_cache_msg = result["toolConfig"]["tools"][1] assert "cachePoint" in transformed_cache_msg @@ -285,6 +271,7 @@ def test_reasoning_with_forced_tool_choice_switches_to_auto(): assert optional_params["tool_choice"] == {"auto": {}} + def test_get_supported_openai_params(): config = AmazonConverseConfig() supported_params = config.get_supported_openai_params( @@ -307,15 +294,13 @@ def test_get_supported_openai_params_bedrock_converse(): for model in litellm.BEDROCK_CONVERSE_MODELS: print(f"Testing model: {model}") config = AmazonConverseConfig() - supported_params_without_prefix = config.get_supported_openai_params( - model=model - ) + supported_params_without_prefix = config.get_supported_openai_params(model=model) - supported_params_with_prefix = config.get_supported_openai_params( - model=f"bedrock/converse/{model}" - ) + supported_params_with_prefix = config.get_supported_openai_params(model=f"bedrock/converse/{model}") - assert set(supported_params_without_prefix) == set(supported_params_with_prefix), f"Supported params mismatch for model: {model}. Without prefix: {supported_params_without_prefix}, With prefix: {supported_params_with_prefix}" + assert set(supported_params_without_prefix) == set(supported_params_with_prefix), ( + f"Supported params mismatch for model: {model}. Without prefix: {supported_params_without_prefix}, With prefix: {supported_params_with_prefix}" + ) print(f"✅ Passed for model: {model}") @@ -382,7 +367,7 @@ def test_transform_response_with_computer_use_tool(): }, } } - ] + ], } }, "stopReason": "tool_use", @@ -396,10 +381,12 @@ def test_transform_response_with_computer_use_tool(): "cacheWriteInputTokens": 0, }, } + # Mock httpx.Response class MockResponse: def json(self): return response_json + @property def text(self): return json.dumps(response_json) @@ -468,12 +455,10 @@ def test_transform_response_with_bash_tool(): "toolUse": { "toolUseId": "tooluse_456", "name": "bash", - "input": { - "command": "ls -la *.py" - }, + "input": {"command": "ls -la *.py"}, } } - ] + ], } }, "stopReason": "tool_use", @@ -487,10 +472,12 @@ def test_transform_response_with_bash_tool(): "cacheWriteInputTokens": 0, }, } + # Mock httpx.Response class MockResponse: def json(self): return response_json + @property def text(self): return json.dumps(response_json) @@ -549,10 +536,11 @@ def test_transform_response_with_structured_response_being_called(): "name": "json_tool_call", "input": { "Current_Temperature": 62, - "Weather_Explanation": "San Francisco typically has mild, cool weather year-round due to its coastal location and marine influence. The city is known for its fog, moderate temperatures, and relatively stable climate with little seasonal variation."}, + "Weather_Explanation": "San Francisco typically has mild, cool weather year-round due to its coastal location and marine influence. The city is known for its fog, moderate temperatures, and relatively stable climate with little seasonal variation.", + }, } } - ] + ], } }, "stopReason": "tool_use", @@ -566,10 +554,12 @@ def test_transform_response_with_structured_response_being_called(): "cacheWriteInputTokens": 0, }, } + # Mock httpx.Response class MockResponse: def json(self): return response_json + @property def text(self): return json.dumps(response_json) @@ -580,49 +570,43 @@ def test_transform_response_with_structured_response_being_called(): "json_mode": True, "tools": [ { - 'type': 'function', - 'function': { - 'name': 'get_weather', - 'description': 'Get the current weather in a given location', - 'parameters': { - 'type': 'object', - 'properties': { - 'location': { - 'type': 'string', - 'description': 'The city and state, e.g. San Francisco, CA' - }, - 'unit': { - 'type': 'string', - 'enum': ['celsius', 'fahrenheit'] - } + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the current weather in a given location", + "parameters": { + "type": "object", + "properties": { + "location": {"type": "string", "description": "The city and state, e.g. San Francisco, CA"}, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, }, - 'required': ['location'] - } - } + "required": ["location"], + }, + }, }, { - 'type': 'function', - 'function': { - 'name': 'json_tool_call', - 'parameters': { - '$schema': 'http://json-schema.org/draft-07/schema#', - 'type': 'object', - 'required': ['Weather_Explanation', 'Current_Temperature'], - 'properties': { - 'Weather_Explanation': { - 'type': ['string', 'null'], - 'description': '1-2 sentences explaining the weather in the location' + "type": "function", + "function": { + "name": "json_tool_call", + "parameters": { + "$schema": "http://json-schema.org/draft-07/schema#", + "type": "object", + "required": ["Weather_Explanation", "Current_Temperature"], + "properties": { + "Weather_Explanation": { + "type": ["string", "null"], + "description": "1-2 sentences explaining the weather in the location", + }, + "Current_Temperature": { + "type": ["number", "null"], + "description": "Current temperature in the location", }, - 'Current_Temperature': { - 'type': ['number', 'null'], - 'description': 'Current temperature in the location' - } }, - 'additionalProperties': False - } - } - } - ] + "additionalProperties": False, + }, + }, + }, + ], } # Call the transformation logic result = config._transform_response( @@ -641,7 +625,11 @@ def test_transform_response_with_structured_response_being_called(): assert result.choices[0].message.tool_calls is None assert result.choices[0].message.content is not None - assert result.choices[0].message.content == '{"Current_Temperature": 62, "Weather_Explanation": "San Francisco typically has mild, cool weather year-round due to its coastal location and marine influence. The city is known for its fog, moderate temperatures, and relatively stable climate with little seasonal variation."}' + assert ( + result.choices[0].message.content + == '{"Current_Temperature": 62, "Weather_Explanation": "San Francisco typically has mild, cool weather year-round due to its coastal location and marine influence. The city is known for its fog, moderate temperatures, and relatively stable climate with little seasonal variation."}' + ) + def test_transform_response_with_structured_response_calling_tool(): """Test response transformation with structured response.""" @@ -650,28 +638,20 @@ def test_transform_response_with_structured_response_calling_tool(): # Simulate a Bedrock Converse response with a bash tool call response_json = { - "metrics": { - "latencyMs": 1148 - }, + "metrics": {"latencyMs": 1148}, "output": { - "message": - { + "message": { "content": [ - { - "text": "I\'ll check the current weather in San Francisco for you." - }, + {"text": "I'll check the current weather in San Francisco for you."}, { "toolUse": { - "input": { - "location": "San Francisco, CA", - "unit": "celsius" - }, + "input": {"location": "San Francisco, CA", "unit": "celsius"}, "name": "get_weather", - "toolUseId": "tooluse_oKk__QrqSUmufMw3Q7vGaQ" + "toolUseId": "tooluse_oKk__QrqSUmufMw3Q7vGaQ", } - } + }, ], - "role": "assistant" + "role": "assistant", } }, "stopReason": "tool_use", @@ -682,13 +662,15 @@ def test_transform_response_with_structured_response_calling_tool(): "cacheWriteInputTokens": 0, "inputTokens": 534, "outputTokens": 69, - "totalTokens": 603 - } + "totalTokens": 603, + }, } + # Mock httpx.Response class MockResponse: def json(self): return response_json + @property def text(self): return json.dumps(response_json) @@ -699,49 +681,43 @@ def test_transform_response_with_structured_response_calling_tool(): "json_mode": True, "tools": [ { - 'type': 'function', - 'function': { - 'name': 'get_weather', - 'description': 'Get the current weather in a given location', - 'parameters': { - 'type': 'object', - 'properties': { - 'location': { - 'type': 'string', - 'description': 'The city and state, e.g. San Francisco, CA' - }, - 'unit': { - 'type': 'string', - 'enum': ['celsius', 'fahrenheit'] - } + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the current weather in a given location", + "parameters": { + "type": "object", + "properties": { + "location": {"type": "string", "description": "The city and state, e.g. San Francisco, CA"}, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, }, - 'required': ['location'] - } - } + "required": ["location"], + }, + }, }, { - 'type': 'function', - 'function': { - 'name': 'json_tool_call', - 'parameters': { - '$schema': 'http://json-schema.org/draft-07/schema#', - 'type': 'object', - 'required': ['Weather_Explanation', 'Current_Temperature'], - 'properties': { - 'Weather_Explanation': { - 'type': ['string', 'null'], - 'description': '1-2 sentences explaining the weather in the location' + "type": "function", + "function": { + "name": "json_tool_call", + "parameters": { + "$schema": "http://json-schema.org/draft-07/schema#", + "type": "object", + "required": ["Weather_Explanation", "Current_Temperature"], + "properties": { + "Weather_Explanation": { + "type": ["string", "null"], + "description": "1-2 sentences explaining the weather in the location", + }, + "Current_Temperature": { + "type": ["number", "null"], + "description": "Current temperature in the location", }, - 'Current_Temperature': { - 'type': ['number', 'null'], - 'description': 'Current temperature in the location' - } }, - 'additionalProperties': False - } - } - } - ] + "additionalProperties": False, + }, + }, + }, + ], } # Call the transformation logic result = config._transform_response( @@ -760,7 +736,10 @@ def test_transform_response_with_structured_response_calling_tool(): assert result.choices[0].message.tool_calls is not None assert len(result.choices[0].message.tool_calls) == 1 assert result.choices[0].message.tool_calls[0].function.name == "get_weather" - assert result.choices[0].message.tool_calls[0].function.arguments == '{"location": "San Francisco, CA", "unit": "celsius"}' + assert ( + result.choices[0].message.tool_calls[0].function.arguments + == '{"location": "San Francisco, CA", "unit": "celsius"}' + ) @pytest.mark.asyncio @@ -775,12 +754,7 @@ async def test_bedrock_bash_tool_acompletion(): } ] - messages = [ - { - "role": "user", - "content": "run ls command and find all python files" - } - ] + messages = [{"role": "user", "content": "run ls command and find all python files"}] try: response = await litellm.acompletion( @@ -788,7 +762,7 @@ async def test_bedrock_bash_tool_acompletion(): messages=messages, tools=tools, # Using dummy API key - test should fail with auth error, proving request formatting works - api_key="dummy-key-for-testing" + api_key="dummy-key-for-testing", ) # If we get here, something's wrong - we expect an auth error assert False, "Expected authentication error but got successful response" @@ -797,8 +771,16 @@ async def test_bedrock_bash_tool_acompletion(): # Check if it's an expected authentication/credentials error auth_error_indicators = [ - "credentials", "authentication", "unauthorized", "access denied", - "aws", "region", "profile", "token", "invalid", "signature" + "credentials", + "authentication", + "unauthorized", + "access denied", + "aws", + "region", + "profile", + "token", + "invalid", + "signature", ] if any(auth_error in error_str for auth_error in auth_error_indicators): @@ -828,17 +810,14 @@ async def test_bedrock_computer_use_acompletion(): { "role": "user", "content": [ - { - "type": "text", - "text": "Go to the bedrock console" - }, + {"type": "text", "text": "Go to the bedrock console"}, { "type": "image_url", "image_url": { "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==" - } - } - ] + }, + }, + ], } ] @@ -848,7 +827,7 @@ async def test_bedrock_computer_use_acompletion(): messages=messages, tools=tools, # Using dummy API key - test should fail with auth error, proving request formatting works - api_key="dummy-key-for-testing" + api_key="dummy-key-for-testing", ) # If we get here, something's wrong - we expect an auth error assert False, "Expected authentication error but got successful response" @@ -857,8 +836,16 @@ async def test_bedrock_computer_use_acompletion(): # Check if it's an expected authentication/credentials error auth_error_indicators = [ - "credentials", "authentication", "unauthorized", "access denied", - "aws", "region", "profile", "token", "invalid", "signature" + "credentials", + "authentication", + "unauthorized", + "access denied", + "aws", + "region", + "profile", + "token", + "invalid", + "signature", ] if any(auth_error in error_str for auth_error in auth_error_indicators): @@ -886,15 +873,10 @@ async def test_transformation_directly(): { "type": "bash_20241022", "name": "bash", - } + }, ] - messages = [ - { - "role": "user", - "content": "run ls command and find all python files" - } - ] + messages = [{"role": "user", "content": "run ls command and find all python files"}] # Transform request request_data = config.transform_request( @@ -902,7 +884,7 @@ async def test_transformation_directly(): messages=messages, optional_params={"tools": tools}, litellm_params={}, - headers={} + headers={}, ) # Verify the structure @@ -994,16 +976,11 @@ def test_transform_request_with_multiple_tools(): }, "required": ["location"], }, - } - } + }, + }, ] - messages = [ - { - "role": "user", - "content": "run ls command and find all python files" - } - ] + messages = [{"role": "user", "content": "run ls command and find all python files"}] # Transform request request_data = config.transform_request( @@ -1011,7 +988,7 @@ def test_transform_request_with_multiple_tools(): messages=messages, optional_params={"tools": tools}, litellm_params={}, - headers={} + headers={}, ) # Verify the structure @@ -1054,17 +1031,14 @@ def test_transform_request_with_computer_tool_only(): { "role": "user", "content": [ - { - "type": "text", - "text": "Go to the bedrock console" - }, + {"type": "text", "text": "Go to the bedrock console"}, { "type": "image_url", "image_url": { "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==" - } - } - ] + }, + }, + ], } ] @@ -1074,7 +1048,7 @@ def test_transform_request_with_computer_tool_only(): messages=messages, optional_params={"tools": tools}, litellm_params={}, - headers={} + headers={}, ) # Verify the structure @@ -1102,12 +1076,7 @@ def test_transform_request_with_bash_tool_only(): } ] - messages = [ - { - "role": "user", - "content": "run ls command and find all python files" - } - ] + messages = [{"role": "user", "content": "run ls command and find all python files"}] # Transform request request_data = config.transform_request( @@ -1115,7 +1084,7 @@ def test_transform_request_with_bash_tool_only(): messages=messages, optional_params={"tools": tools}, litellm_params={}, - headers={} + headers={}, ) # Verify the structure @@ -1143,12 +1112,7 @@ def test_transform_request_with_text_editor_tool(): } ] - messages = [ - { - "role": "user", - "content": "Edit this text file" - } - ] + messages = [{"role": "user", "content": "Edit this text file"}] # Transform request request_data = config.transform_request( @@ -1156,7 +1120,7 @@ def test_transform_request_with_text_editor_tool(): messages=messages, optional_params={"tools": tools}, litellm_params={}, - headers={} + headers={}, ) # Verify the structure @@ -1194,16 +1158,11 @@ def test_transform_request_with_function_tool(): }, "required": ["location"], }, - } + }, } ] - messages = [ - { - "role": "user", - "content": "What's the weather like in San Francisco?" - } - ] + messages = [{"role": "user", "content": "What's the weather like in San Francisco?"}] # Transform request request_data = config.transform_request( @@ -1211,7 +1170,7 @@ def test_transform_request_with_function_tool(): messages=messages, optional_params={"tools": tools}, litellm_params={}, - headers={} + headers={}, ) # Verify the structure @@ -1247,7 +1206,7 @@ def test_map_openai_params_with_response_format(): }, "required": ["location"], }, - } + }, } ] @@ -1279,7 +1238,7 @@ def test_map_openai_params_with_response_format(): non_default_params={"response_format": json_schema}, optional_params={"tools": tools}, model="eu.anthropic.claude-sonnet-4-20250514-v1:0", - drop_params=False + drop_params=False, ) assert "tools" in optional_params @@ -1299,31 +1258,21 @@ async def test_assistant_message_cache_control(): # Test assistant message with string content and cache_control messages = [ {"role": "user", "content": "Hello"}, - { - "role": "assistant", - "content": "Hi there!", - "cache_control": {"type": "ephemeral"} - } + {"role": "assistant", "content": "Hi there!", "cache_control": {"type": "ephemeral"}}, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) assert result == async_result async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) assert result == async_result @@ -1353,26 +1302,16 @@ async def test_assistant_message_list_content_cache_control(): {"role": "user", "content": "Hello"}, { "role": "assistant", - "content": [ - { - "type": "text", - "text": "This should be cached", - "cache_control": {"type": "ephemeral"} - } - ] - } + "content": [{"type": "text", "text": "This should be cached", "cache_control": {"type": "ephemeral"}}], + }, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) assert result == async_result @@ -1399,36 +1338,22 @@ async def test_tool_message_cache_control(): "role": "assistant", "content": None, "tool_calls": [ - { - "id": "call_123", - "type": "function", - "function": {"name": "get_weather", "arguments": "{}"} - } - ] + {"id": "call_123", "type": "function", "function": {"name": "get_weather", "arguments": "{}"}} + ], }, { "role": "tool", "tool_call_id": "call_123", - "content": [ - { - "type": "text", - "text": "Weather data: sunny, 25°C", - "cache_control": {"type": "ephemeral"} - } - ] - } + "content": [{"type": "text", "text": "Weather data: sunny, 25°C", "cache_control": {"type": "ephemeral"}}], + }, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) assert result == async_result @@ -1463,31 +1388,23 @@ async def test_tool_message_string_content_cache_control(): "role": "assistant", "content": None, "tool_calls": [ - { - "id": "call_123", - "type": "function", - "function": {"name": "get_weather", "arguments": "{}"} - } - ] + {"id": "call_123", "type": "function", "function": {"name": "get_weather", "arguments": "{}"}} + ], }, { "role": "tool", "tool_call_id": "call_123", "content": "Weather: sunny, 25°C", - "cache_control": {"type": "ephemeral"} - } + "cache_control": {"type": "ephemeral"}, + }, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) assert result == async_result @@ -1523,22 +1440,18 @@ async def test_assistant_tool_calls_cache_control(): "id": "call_proxy_123", "type": "function", "function": {"name": "calc", "arguments": "{}"}, - "cache_control": {"type": "ephemeral"} + "cache_control": {"type": "ephemeral"}, } - ] - } + ], + }, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) assert result == async_result @@ -1575,28 +1488,24 @@ async def test_multiple_tool_calls_with_mixed_cache_control(): "id": "call_1", "type": "function", "function": {"name": "calc", "arguments": '{"expr": "2+2"}'}, - "cache_control": {"type": "ephemeral"} + "cache_control": {"type": "ephemeral"}, }, { "id": "call_2", "type": "function", - "function": {"name": "calc", "arguments": '{"expr": "3+3"}'} + "function": {"name": "calc", "arguments": '{"expr": "3+3"}'}, # No cache_control - } - ] - } + }, + ], + }, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) assert result == async_result @@ -1632,20 +1541,16 @@ async def test_no_cache_control_no_cache_point(): { "role": "tool", "tool_call_id": "call_123", - "content": "Tool result" # No cache_control - } + "content": "Tool result", # No cache_control + }, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) assert result == async_result @@ -1665,6 +1570,7 @@ async def test_no_cache_control_no_cache_point(): # Guarded Text Feature Tests # ============================================================================ + def test_guarded_text_wraps_in_guardrail_converse_content(): """Test that guarded_text content type gets wrapped in guardContent blocks.""" from litellm.litellm_core_utils.prompt_templates.factory import ( @@ -1677,15 +1583,13 @@ def test_guarded_text_wraps_in_guardrail_converse_content(): "content": [ {"type": "text", "text": "Regular text content"}, {"type": "guarded_text", "text": "This should be guarded"}, - {"type": "text", "text": "More regular text"} - ] + {"type": "text", "text": "More regular text"}, + ], } ] result = _bedrock_converse_messages_pt( - messages=messages, - model="us.amazon.nova-pro-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="us.amazon.nova-pro-v1:0", llm_provider="bedrock_converse" ) # Should have 1 message @@ -1705,6 +1609,7 @@ def test_guarded_text_wraps_in_guardrail_converse_content(): assert "guardContent" in content[1] assert content[1]["guardContent"]["text"]["text"] == "This should be guarded" + def test_guarded_text_with_system_messages(): """Test guarded_text with system messages using the full transformation.""" config = AmazonConverseConfig() @@ -1715,24 +1620,22 @@ def test_guarded_text_with_system_messages(): "role": "user", "content": [ {"type": "text", "text": "What is the main topic of this legal document?"}, - {"type": "guarded_text", "text": "This is a set of very long instructions that you will follow. Here is a legal document that you will use to answer the user's question."} - ] - } + { + "type": "guarded_text", + "text": "This is a set of very long instructions that you will follow. Here is a legal document that you will use to answer the user's question.", + }, + ], + }, ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "DRAFT" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "DRAFT"}} result = config._transform_request( model="us.amazon.nova-pro-v1:0", messages=messages, optional_params=optional_params, litellm_params={}, - headers={} + headers={}, ) # Should have system content blocks @@ -1755,7 +1658,10 @@ def test_guarded_text_with_system_messages(): assert content[0]["text"] == "What is the main topic of this legal document?" # Second should be guardContent assert "guardContent" in content[1] - assert content[1]["guardContent"]["text"]["text"] == "This is a set of very long instructions that you will follow. Here is a legal document that you will use to answer the user's question." + assert ( + content[1]["guardContent"]["text"]["text"] + == "This is a set of very long instructions that you will follow. Here is a legal document that you will use to answer the user's question." + ) def test_guarded_text_with_mixed_content_types(): @@ -1770,15 +1676,13 @@ def test_guarded_text_with_mixed_content_types(): "content": [ {"type": "text", "text": "Look at this image"}, {"type": "image_url", "image_url": {"url": "data:image/png;base64,test"}}, - {"type": "guarded_text", "text": "This sensitive content should be guarded"} - ] + {"type": "guarded_text", "text": "This sensitive content should be guarded"}, + ], } ] result = _bedrock_converse_messages_pt( - messages=messages, - model="us.amazon.nova-pro-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="us.amazon.nova-pro-v1:0", llm_provider="bedrock_converse" ) # Should have 1 message @@ -1800,6 +1704,7 @@ def test_guarded_text_with_mixed_content_types(): assert "guardContent" in content[2] assert content[2]["guardContent"]["text"]["text"] == "This sensitive content should be guarded" + @pytest.mark.asyncio async def test_async_guarded_text(): """Test async version of guarded_text processing.""" @@ -1810,17 +1715,12 @@ async def test_async_guarded_text(): messages = [ { "role": "user", - "content": [ - {"type": "text", "text": "Hello"}, - {"type": "guarded_text", "text": "This should be guarded"} - ] + "content": [{"type": "text", "text": "Hello"}, {"type": "guarded_text", "text": "This should be guarded"}], } ] result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( - messages=messages, - model="us.amazon.nova-pro-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="us.amazon.nova-pro-v1:0", llm_provider="bedrock_converse" ) # Should have 1 message @@ -1851,31 +1751,21 @@ def test_guarded_text_with_tool_calls(): "role": "user", "content": [ {"type": "text", "text": "What's the weather?"}, - {"type": "guarded_text", "text": "Please be careful with sensitive information"} - ] + {"type": "guarded_text", "text": "Please be careful with sensitive information"}, + ], }, { "role": "assistant", "content": None, "tool_calls": [ - { - "id": "call_123", - "type": "function", - "function": {"name": "get_weather", "arguments": "{}"} - } - ] + {"id": "call_123", "type": "function", "function": {"name": "get_weather", "arguments": "{}"}} + ], }, - { - "role": "tool", - "tool_call_id": "call_123", - "content": "It's sunny and 25°C" - } + {"role": "tool", "tool_call_id": "call_123", "content": "It's sunny and 25°C"}, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="us.amazon.nova-pro-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="us.amazon.nova-pro-v1:0", llm_provider="bedrock_converse" ) # Should have 3 messages @@ -1909,26 +1799,18 @@ def test_guarded_text_guardrail_config_preserved(): messages = [ { "role": "user", - "content": [ - {"type": "text", "text": "Hello"}, - {"type": "guarded_text", "text": "This should be guarded"} - ] + "content": [{"type": "text", "text": "Hello"}, {"type": "guarded_text", "text": "This should be guarded"}], } ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "DRAFT" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "DRAFT"}} result = config._transform_request( model="us.amazon.nova-pro-v1:0", messages=messages, optional_params=optional_params, litellm_params={}, - headers={} + headers={}, ) # GuardrailConfig should be present at top level @@ -1946,23 +1828,10 @@ def test_auto_convert_last_user_message_to_guarded_text(): config = AmazonConverseConfig() messages = [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "What is the main topic of this legal document?" - } - ] - } + {"role": "user", "content": [{"type": "text", "text": "What is the main topic of this legal document?"}]} ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the helper method directly converted_messages = config._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) @@ -1979,19 +1848,9 @@ def test_auto_convert_last_user_message_string_content(): """Test that last user message with string content is automatically converted to guarded_text when guardrailConfig is present.""" config = AmazonConverseConfig() - messages = [ - { - "role": "user", - "content": "What is the main topic of this legal document?" - } - ] + messages = [{"role": "user", "content": "What is the main topic of this legal document?"}] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the helper method directly converted_messages = config._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) @@ -2009,15 +1868,7 @@ def test_no_conversion_when_no_guardrail_config(): config = AmazonConverseConfig() messages = [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "What is the main topic of this legal document?" - } - ] - } + {"role": "user", "content": [{"type": "text", "text": "What is the main topic of this legal document?"}]} ] optional_params = {} @@ -2033,24 +1884,9 @@ def test_no_conversion_when_guarded_text_already_present(): """Test that no conversion happens when guarded_text is already present in the last user message.""" config = AmazonConverseConfig() - messages = [ - { - "role": "user", - "content": [ - { - "type": "guarded_text", - "text": "This is already guarded" - } - ] - } - ] + messages = [{"role": "user", "content": [{"type": "guarded_text", "text": "This is already guarded"}]}] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the helper method directly converted_messages = config._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) @@ -2067,24 +1903,13 @@ def test_auto_convert_with_mixed_content(): { "role": "user", "content": [ - { - "type": "text", - "text": "What is the main topic of this legal document?" - }, - { - "type": "image_url", - "image_url": {"url": "https://example.com/image.jpg"} - } - ] + {"type": "text", "text": "What is the main topic of this legal document?"}, + {"type": "image_url", "image_url": {"url": "https://example.com/image.jpg"}}, + ], } ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the helper method directly converted_messages = config._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) @@ -2108,23 +1933,10 @@ def test_auto_convert_in_full_transformation(): config = AmazonConverseConfig() messages = [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "What is the main topic of this legal document?" - } - ] - } + {"role": "user", "content": [{"type": "text", "text": "What is the main topic of this legal document?"}]} ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the full transformation result = config._transform_request( @@ -2132,7 +1944,7 @@ def test_auto_convert_in_full_transformation(): messages=messages, optional_params=optional_params, litellm_params={}, - headers={} + headers={}, ) # Verify the transformation worked @@ -2152,45 +1964,13 @@ def test_convert_consecutive_user_messages_to_guarded_text(): config = AmazonConverseConfig() messages = [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "First user message" - } - ] - }, - { - "role": "assistant", - "content": "Assistant response" - }, - { - "role": "user", - "content": [ - { - "type": "text", - "text": "Second user message" - } - ] - }, - { - "role": "user", - "content": [ - { - "type": "text", - "text": "Third user message" - } - ] - } + {"role": "user", "content": [{"type": "text", "text": "First user message"}]}, + {"role": "assistant", "content": "Assistant response"}, + {"role": "user", "content": [{"type": "text", "text": "Second user message"}]}, + {"role": "user", "content": [{"type": "text", "text": "Third user message"}]}, ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the helper method directly converted_messages = config._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) @@ -2223,41 +2003,12 @@ def test_convert_all_user_messages_when_all_consecutive(): config = AmazonConverseConfig() messages = [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "First user message" - } - ] - }, - { - "role": "user", - "content": [ - { - "type": "text", - "text": "Second user message" - } - ] - }, - { - "role": "user", - "content": [ - { - "type": "text", - "text": "Third user message" - } - ] - } + {"role": "user", "content": [{"type": "text", "text": "First user message"}]}, + {"role": "user", "content": [{"type": "text", "text": "Second user message"}]}, + {"role": "user", "content": [{"type": "text", "text": "Third user message"}]}, ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the helper method directly converted_messages = config._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) @@ -2279,26 +2030,12 @@ def test_convert_consecutive_user_messages_with_string_content(): config = AmazonConverseConfig() messages = [ - { - "role": "assistant", - "content": "Assistant response" - }, - { - "role": "user", - "content": "First user message" - }, - { - "role": "user", - "content": "Second user message" - } + {"role": "assistant", "content": "Assistant response"}, + {"role": "user", "content": "First user message"}, + {"role": "user", "content": "Second user message"}, ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the helper method directly converted_messages = config._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) @@ -2327,32 +2064,11 @@ def test_skip_consecutive_user_messages_with_existing_guarded_text(): config = AmazonConverseConfig() messages = [ - { - "role": "user", - "content": [ - { - "type": "guarded_text", - "text": "Already guarded" - } - ] - }, - { - "role": "user", - "content": [ - { - "type": "text", - "text": "Should be converted" - } - ] - } + {"role": "user", "content": [{"type": "guarded_text", "text": "Already guarded"}]}, + {"role": "user", "content": [{"type": "text", "text": "Should be converted"}]}, ] - optional_params = { - "guardrailConfig": { - "guardrailIdentifier": "gr-abc123", - "guardrailVersion": "1" - } - } + optional_params = {"guardrailConfig": {"guardrailIdentifier": "gr-abc123", "guardrailVersion": "1"}} # Test the helper method directly converted_messages = config._convert_consecutive_user_messages_to_guarded_text(messages, optional_params) @@ -2384,11 +2100,7 @@ def test_request_metadata_transformation(): """Test that requestMetadata is properly transformed to top-level field.""" config = AmazonConverseConfig() - request_metadata = { - "cost_center": "engineering", - "user_id": "user123", - "session_id": "sess_abc123" - } + request_metadata = {"cost_center": "engineering", "user_id": "user123", "session_id": "sess_abc123"} messages = [ {"role": "user", "content": "Hello!"}, @@ -2400,7 +2112,7 @@ def test_request_metadata_transformation(): messages=messages, optional_params={"requestMetadata": request_metadata}, litellm_params={}, - headers={} + headers={}, ) # Verify that requestMetadata appears as top-level field @@ -2426,7 +2138,7 @@ def test_request_metadata_validation(): messages=messages, optional_params={"requestMetadata": valid_metadata}, litellm_params={}, - headers={} + headers={}, ) # Test too many items (max 16) @@ -2438,7 +2150,7 @@ def test_request_metadata_validation(): messages=messages, optional_params={"requestMetadata": too_many_items}, litellm_params={}, - headers={} + headers={}, ) assert False, "Should have raised validation error for too many items" except Exception as e: @@ -2461,7 +2173,7 @@ def test_request_metadata_key_constraints(): messages=messages, optional_params={"requestMetadata": invalid_metadata}, litellm_params={}, - headers={} + headers={}, ) assert False, "Should have raised validation error for key too long" except Exception as e: @@ -2476,7 +2188,7 @@ def test_request_metadata_key_constraints(): messages=messages, optional_params={"requestMetadata": invalid_metadata}, litellm_params={}, - headers={} + headers={}, ) assert False, "Should have raised validation error for empty key" except Exception as e: @@ -2499,7 +2211,7 @@ def test_request_metadata_value_constraints(): messages=messages, optional_params={"requestMetadata": invalid_metadata}, litellm_params={}, - headers={} + headers={}, ) assert False, "Should have raised validation error for value too long" except Exception as e: @@ -2514,7 +2226,7 @@ def test_request_metadata_value_constraints(): messages=messages, optional_params={"requestMetadata": valid_metadata}, litellm_params={}, - headers={} + headers={}, ) @@ -2537,7 +2249,7 @@ def test_request_metadata_character_pattern(): messages=messages, optional_params={"requestMetadata": valid_metadata}, litellm_params={}, - headers={} + headers={}, ) @@ -2545,10 +2257,7 @@ def test_request_metadata_with_other_params(): """Test that requestMetadata works alongside other parameters.""" config = AmazonConverseConfig() - request_metadata = { - "experiment": "test_A", - "user_type": "premium" - } + request_metadata = {"experiment": "test_A", "user_type": "premium"} messages = [ {"role": "user", "content": "What's the weather?"}, @@ -2562,12 +2271,10 @@ def test_request_metadata_with_other_params(): "description": "Get the current weather", "parameters": { "type": "object", - "properties": { - "location": {"type": "string"} - }, - "required": ["location"] - } - } + "properties": {"location": {"type": "string"}}, + "required": ["location"], + }, + }, } ] @@ -2575,14 +2282,9 @@ def test_request_metadata_with_other_params(): request_data = config.transform_request( model="anthropic.claude-3-5-sonnet-20240620-v1:0", messages=messages, - optional_params={ - "requestMetadata": request_metadata, - "tools": tools, - "max_tokens": 100, - "temperature": 0.7 - }, + optional_params={"requestMetadata": request_metadata, "tools": tools, "max_tokens": 100, "temperature": 0.7}, litellm_params={}, - headers={} + headers={}, ) # Verify requestMetadata is at top level @@ -2607,7 +2309,7 @@ def test_request_metadata_empty(): messages=messages, optional_params={"requestMetadata": {}}, litellm_params={}, - headers={} + headers={}, ) assert "requestMetadata" in request_data @@ -2626,7 +2328,7 @@ def test_request_metadata_not_provided(): messages=messages, optional_params={}, litellm_params={}, - headers={} + headers={}, ) # requestMetadata should not be in the request @@ -2649,16 +2351,14 @@ def test_empty_assistant_message_handling(): messages = [ {"role": "user", "content": "Hello"}, {"role": "assistant", "content": ""}, # Empty content - {"role": "user", "content": "How are you?"} + {"role": "user", "content": "How are you?"}, ] # Use patch to ensure we modify the litellm reference that factory.py actually uses # This avoids issues with module reloading during parallel test execution with patch.object(factory_module.litellm, "modify_params", True): result = _bedrock_converse_messages_pt( - messages=messages, - model="anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) # Should have 3 messages: user, assistant (with placeholder), user @@ -2676,13 +2376,11 @@ def test_empty_assistant_message_handling(): messages = [ {"role": "user", "content": "Hello"}, {"role": "assistant", "content": " "}, # Whitespace-only content - {"role": "user", "content": "How are you?"} + {"role": "user", "content": "How are you?"}, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) # Assistant message should have placeholder text instead of whitespace @@ -2693,13 +2391,11 @@ def test_empty_assistant_message_handling(): messages = [ {"role": "user", "content": "Hello"}, {"role": "assistant", "content": [{"type": "text", "text": ""}]}, # Empty text in list - {"role": "user", "content": "How are you?"} + {"role": "user", "content": "How are you?"}, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) # Assistant message should have placeholder text instead of empty text @@ -2710,13 +2406,11 @@ def test_empty_assistant_message_handling(): messages = [ {"role": "user", "content": "Hello"}, {"role": "assistant", "content": "I'm doing well, thank you!"}, # Normal content - {"role": "user", "content": "How are you?"} + {"role": "user", "content": "How are you?"}, ] result = _bedrock_converse_messages_pt( - messages=messages, - model="anthropic.claude-3-5-sonnet-20240620-v1:0", - llm_provider="bedrock_converse" + messages=messages, model="anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) # Assistant message should keep original content @@ -2828,6 +2522,7 @@ def test_thinking_with_max_completion_tokens(): assert result["thinking"]["type"] == "enabled" assert result["thinking"]["budget_tokens"] == 5000 + def test_drop_thinking_param_when_thinking_blocks_missing(): """ Test that thinking param is dropped when modify_params=True and @@ -2868,24 +2563,21 @@ def test_drop_thinking_param_when_thinking_blocks_missing(): optional_params = {"thinking": {"type": "enabled", "budget_tokens": 1000}} # Verify the condition is detected - assert last_assistant_with_tool_calls_has_no_thinking_blocks( - messages_without_thinking_blocks - ), "Should detect missing thinking_blocks" + assert last_assistant_with_tool_calls_has_no_thinking_blocks(messages_without_thinking_blocks), ( + "Should detect missing thinking_blocks" + ) # Simulate what _transform_request_helper does if ( optional_params.get("thinking") is not None and messages_without_thinking_blocks is not None - and last_assistant_with_tool_calls_has_no_thinking_blocks( - messages_without_thinking_blocks - ) + and last_assistant_with_tool_calls_has_no_thinking_blocks(messages_without_thinking_blocks) ): if litellm.modify_params: optional_params.pop("thinking", None) assert "thinking" not in optional_params, ( - "thinking param should be dropped when modify_params=True " - "and thinking_blocks are missing" + "thinking param should be dropped when modify_params=True and thinking_blocks are missing" ) # Test case 2: thinking should NOT be dropped when thinking_blocks are present @@ -2901,29 +2593,23 @@ def test_drop_thinking_param_when_thinking_blocks_missing(): "function": {"name": "search", "arguments": "{}"}, } ], - "thinking_blocks": [ - {"type": "thinking", "thinking": "Let me search for weather..."} - ], + "thinking_blocks": [{"type": "thinking", "thinking": "Let me search for weather..."}], }, {"role": "tool", "content": "Weather is sunny", "tool_call_id": "call_123"}, ] - optional_params_with_thinking = { - "thinking": {"type": "enabled", "budget_tokens": 1000} - } + optional_params_with_thinking = {"thinking": {"type": "enabled", "budget_tokens": 1000}} # Verify the condition is NOT detected when thinking_blocks are present - assert not last_assistant_with_tool_calls_has_no_thinking_blocks( - messages_with_thinking_blocks - ), "Should NOT detect missing thinking_blocks when they are present" + assert not last_assistant_with_tool_calls_has_no_thinking_blocks(messages_with_thinking_blocks), ( + "Should NOT detect missing thinking_blocks when they are present" + ) # Simulate what _transform_request_helper does if ( optional_params_with_thinking.get("thinking") is not None and messages_with_thinking_blocks is not None - and last_assistant_with_tool_calls_has_no_thinking_blocks( - messages_with_thinking_blocks - ) + and last_assistant_with_tool_calls_has_no_thinking_blocks(messages_with_thinking_blocks) ): if litellm.modify_params: optional_params_with_thinking.pop("thinking", None) @@ -2935,24 +2621,18 @@ def test_drop_thinking_param_when_thinking_blocks_missing(): # Test case 3: thinking should NOT be dropped when modify_params=False litellm.modify_params = False - optional_params_no_modify = { - "thinking": {"type": "enabled", "budget_tokens": 1000} - } + optional_params_no_modify = {"thinking": {"type": "enabled", "budget_tokens": 1000}} # Simulate what _transform_request_helper does if ( optional_params_no_modify.get("thinking") is not None and messages_without_thinking_blocks is not None - and last_assistant_with_tool_calls_has_no_thinking_blocks( - messages_without_thinking_blocks - ) + and last_assistant_with_tool_calls_has_no_thinking_blocks(messages_without_thinking_blocks) ): if litellm.modify_params: optional_params_no_modify.pop("thinking", None) - assert "thinking" in optional_params_no_modify, ( - "thinking param should NOT be dropped when modify_params=False" - ) + assert "thinking" in optional_params_no_modify, "thinking param should NOT be dropped when modify_params=False" finally: # Restore original modify_params setting @@ -2960,46 +2640,43 @@ def test_drop_thinking_param_when_thinking_blocks_missing(): def test_supports_native_structured_outputs(): - """Test model detection for native structured outputs support.""" + """Test model detection for native structured outputs support. + + Support is driven by the ``supports_native_structured_output`` flag in the + cost JSON (litellm.model_cost), not a hardcoded model set. + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + config = AmazonConverseConfig() - # Supported models - assert config._supports_native_structured_outputs( - "anthropic.claude-sonnet-4-5-20250929-v1:0" - ) - assert config._supports_native_structured_outputs( - "anthropic.claude-haiku-4-5-20251001-v1:0" - ) - assert config._supports_native_structured_outputs( - "anthropic.claude-opus-4-6-v1:0" - ) - assert config._supports_native_structured_outputs( - "eu.anthropic.claude-opus-4-5-20260101-v1:0" - ) - assert config._supports_native_structured_outputs("qwen.qwen3-235b-instruct-v1:0") - assert config._supports_native_structured_outputs("mistral.mistral-large-3-v1:0") - assert config._supports_native_structured_outputs("deepseek.deepseek-v3.1-v1:0") + # Supported models (have supports_native_structured_output=true in cost JSON) + assert config._supports_native_structured_outputs("anthropic.claude-sonnet-4-5-20250929-v1:0") + assert config._supports_native_structured_outputs("anthropic.claude-haiku-4-5-20251001-v1:0") + assert config._supports_native_structured_outputs("anthropic.claude-opus-4-6-v1") + # Version suffix (:0) is stripped when looking up models without it in cost JSON + assert config._supports_native_structured_outputs("anthropic.claude-opus-4-6-v1:0") + # Regional prefix is stripped by get_bedrock_base_model + assert config._supports_native_structured_outputs("eu.anthropic.claude-opus-4-5-20251101-v1:0") + # Claude 4.6 Sonnet + assert config._supports_native_structured_outputs("anthropic.claude-sonnet-4-6") + assert config._supports_native_structured_outputs("us.anthropic.claude-sonnet-4-6") + # Non-Anthropic models + assert config._supports_native_structured_outputs("qwen.qwen3-235b-a22b-2507-v1:0") + assert config._supports_native_structured_outputs("mistral.mistral-large-3-675b-instruct") + assert config._supports_native_structured_outputs("google.gemma-3-27b-it") + assert config._supports_native_structured_outputs("minimax.minimax-m2") + assert config._supports_native_structured_outputs("moonshot.kimi-k2-thinking") + assert config._supports_native_structured_outputs("nvidia.nemotron-nano-12b-v2") - # Unsupported models — should fall back to tool-call approach - assert not config._supports_native_structured_outputs( - "anthropic.claude-3-5-sonnet-20241022-v2:0" - ) - assert not config._supports_native_structured_outputs( - "anthropic.claude-sonnet-4-20250514-v1:0" - ) - assert not config._supports_native_structured_outputs( - "meta.llama3-3-70b-instruct-v1:0" - ) - assert not config._supports_native_structured_outputs( - "amazon.nova-pro-v1:0" - ) + # Unsupported models -- should fall back to tool-call approach + assert not config._supports_native_structured_outputs("anthropic.claude-3-5-sonnet-20241022-v2:0") + assert not config._supports_native_structured_outputs("anthropic.claude-sonnet-4-20250514-v1:0") + assert not config._supports_native_structured_outputs("meta.llama3-3-70b-instruct-v1:0") + assert not config._supports_native_structured_outputs("amazon.nova-pro-v1:0") # Excluded despite AWS listing them: broken constrained decoding on Bedrock - assert not config._supports_native_structured_outputs( - "openai.gpt-oss-120b-1:0" - ) - assert not config._supports_native_structured_outputs( - "mistral.magistral-small-2509" - ) + assert not config._supports_native_structured_outputs("openai.gpt-oss-120b-1:0") + assert not config._supports_native_structured_outputs("mistral.magistral-small-2509") def test_create_output_config_for_response_format(): @@ -3078,10 +2755,7 @@ def test_translate_response_format_native_output_config(): parsed_schema = json.loads(schema_str) expected_schema = {**response_format["json_schema"]["schema"], "additionalProperties": False} assert parsed_schema == expected_schema - assert ( - result["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["name"] - == "WeatherResult" - ) + assert result["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["name"] == "WeatherResult" def test_translate_response_format_fallback_tool_call(): @@ -3118,6 +2792,9 @@ def test_translate_response_format_fallback_tool_call(): def test_native_structured_output_no_fake_stream(): """When using native structured outputs with streaming, fake_stream should NOT be set.""" + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + config = AmazonConverseConfig() response_format = { @@ -3227,11 +2904,7 @@ def test_transform_response_native_structured_output(): "output": { "message": { "role": "assistant", - "content": [ - { - "text": '{"temp": 62, "description": "Mild and foggy"}' - } - ], + "content": [{"text": '{"temp": 62, "description": "Mild and foggy"}'}], } }, "stopReason": "end_turn", @@ -3388,6 +3061,9 @@ def test_add_additional_properties_definitions(): def test_json_object_no_schema_falls_back_to_tool_call(): """response_format: {type: json_object} with no schema should use tool-call fallback, even for models that support native structured outputs.""" + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + config = AmazonConverseConfig() optional_params: dict = {} non_default_params = {"response_format": {"type": "json_object"}} @@ -3427,7 +3103,6 @@ def test_output_config_applies_additional_properties(): assert parsed["properties"]["nested"]["additionalProperties"] is False - _TOOL_PARAM = [ { "type": "function", @@ -3505,9 +3180,7 @@ def test_parallel_tool_calls_older_model_drops_disable_flag(): class TestBedrockMinThinkingBudgetTokens: """Test that thinking.budget_tokens is clamped to the Bedrock minimum (1024).""" - def _map_params( - self, thinking_value, model="anthropic.claude-3-7-sonnet-20250219-v1:0" - ): + def _map_params(self, thinking_value, model="anthropic.claude-3-7-sonnet-20250219-v1:0"): """Helper to call map_openai_params with the given thinking value.""" config = AmazonConverseConfig() non_default_params = {"thinking": thinking_value} @@ -3545,6 +3218,7 @@ class TestBedrockMinThinkingBudgetTokens: ) assert "thinking" not in result or result.get("thinking") is None + def test_transform_response_with_both_json_tool_call_and_real_tool(): """ When Bedrock returns BOTH json_tool_call AND a real tool (get_weather), @@ -3733,9 +3407,7 @@ def test_streaming_filters_json_tool_call_with_real_tools(): # Chunk 2: json_tool_call delta — should become text, not tool_use json_delta = ContentBlockDeltaEvent(toolUse={"input": '{"temp": 62}'}) - text_2, tool_use_2, _, _, _ = decoder._handle_converse_delta_event( - json_delta, index=0 - ) + text_2, tool_use_2, _, _, _ = decoder._handle_converse_delta_event(json_delta, index=0) assert text_2 == '{"temp": 62}' assert tool_use_2 is None @@ -3758,12 +3430,8 @@ def test_streaming_filters_json_tool_call_with_real_tools(): assert decoder.tool_calls_index == 0 # Chunk 5: real tool delta - real_delta = ContentBlockDeltaEvent( - toolUse={"input": '{"location": "SF"}'} - ) - text_5, tool_use_5, _, _, _ = decoder._handle_converse_delta_event( - real_delta, index=1 - ) + real_delta = ContentBlockDeltaEvent(toolUse={"input": '{"location": "SF"}'}) + text_5, tool_use_5, _, _, _ = decoder._handle_converse_delta_event(real_delta, index=1) assert text_5 == "" assert tool_use_5 is not None assert tool_use_5["function"]["arguments"] == '{"location": "SF"}' @@ -3796,9 +3464,7 @@ def test_streaming_without_json_mode_passes_all_tools(): # json_tool_call delta — should be a tool_use, not text json_delta = ContentBlockDeltaEvent(toolUse={"input": '{"data": 1}'}) - text, tool_use_delta, _, _, _ = decoder._handle_converse_delta_event( - json_delta, index=0 - ) + text, tool_use_delta, _, _, _ = decoder._handle_converse_delta_event(json_delta, index=0) assert text == "" assert tool_use_delta is not None assert tool_use_delta["function"]["arguments"] == '{"data": 1}' @@ -3899,4 +3565,3 @@ def test_cache_control_injection_tool_config_not_added_without_injection_point() tools = result["toolConfig"]["tools"] # No cachePoint should be appended assert all("cachePoint" not in tool for tool in tools) - From aba027beed9aab1d2923120f7bef8ebc7adc33c3 Mon Sep 17 00:00:00 2001 From: Nicholas Gigliotti Date: Mon, 16 Mar 2026 19:34:17 -0400 Subject: [PATCH 008/117] Remove native structured output flag from models broken on Bedrock Integration testing confirmed gemma-3 (4b/12b/27b) ignores the JSON schema and returns free text, and nemotron-nano (9b/12b) errors with "Tool calling is not supported in streaming mode" even on sync calls. Remove the flag so these models fall back to the tool-call approach. Also fix test assertions to match (nemotron-nano-3-30b is supported, gemma-3 and nemotron-nano-12b are not). --- .../model_prices_and_context_window_backup.json | 15 +++++---------- model_prices_and_context_window.json | 15 +++++---------- .../bedrock/chat/test_converse_transformation.py | 8 +++++--- 3 files changed, 15 insertions(+), 23 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 13df25c1425..77ba0954b0d 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -16716,8 +16716,7 @@ "mode": "chat", "output_cost_per_token": 2.9e-07, "supports_system_messages": true, - "supports_vision": true, - "supports_native_structured_output": true + "supports_vision": true }, "google.gemma-3-27b-it": { "input_cost_per_token": 2.3e-07, @@ -16728,8 +16727,7 @@ "mode": "chat", "output_cost_per_token": 3.8e-07, "supports_system_messages": true, - "supports_vision": true, - "supports_native_structured_output": true + "supports_vision": true }, "google.gemma-3-4b-it": { "input_cost_per_token": 4e-08, @@ -16740,8 +16738,7 @@ "mode": "chat", "output_cost_per_token": 8e-08, "supports_system_messages": true, - "supports_vision": true, - "supports_native_structured_output": true + "supports_vision": true }, "google_pse/search": { "input_cost_per_query": 0.005, @@ -23089,8 +23086,7 @@ "mode": "chat", "output_cost_per_token": 6e-07, "supports_system_messages": true, - "supports_vision": true, - "supports_native_structured_output": true + "supports_vision": true }, "nvidia.nemotron-nano-9b-v2": { "input_cost_per_token": 6e-08, @@ -23100,8 +23096,7 @@ "max_tokens": 8192, "mode": "chat", "output_cost_per_token": 2.3e-07, - "supports_system_messages": true, - "supports_native_structured_output": true + "supports_system_messages": true }, "nvidia.nemotron-nano-3-30b": { "input_cost_per_token": 6e-08, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 13df25c1425..77ba0954b0d 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -16716,8 +16716,7 @@ "mode": "chat", "output_cost_per_token": 2.9e-07, "supports_system_messages": true, - "supports_vision": true, - "supports_native_structured_output": true + "supports_vision": true }, "google.gemma-3-27b-it": { "input_cost_per_token": 2.3e-07, @@ -16728,8 +16727,7 @@ "mode": "chat", "output_cost_per_token": 3.8e-07, "supports_system_messages": true, - "supports_vision": true, - "supports_native_structured_output": true + "supports_vision": true }, "google.gemma-3-4b-it": { "input_cost_per_token": 4e-08, @@ -16740,8 +16738,7 @@ "mode": "chat", "output_cost_per_token": 8e-08, "supports_system_messages": true, - "supports_vision": true, - "supports_native_structured_output": true + "supports_vision": true }, "google_pse/search": { "input_cost_per_query": 0.005, @@ -23089,8 +23086,7 @@ "mode": "chat", "output_cost_per_token": 6e-07, "supports_system_messages": true, - "supports_vision": true, - "supports_native_structured_output": true + "supports_vision": true }, "nvidia.nemotron-nano-9b-v2": { "input_cost_per_token": 6e-08, @@ -23100,8 +23096,7 @@ "max_tokens": 8192, "mode": "chat", "output_cost_per_token": 2.3e-07, - "supports_system_messages": true, - "supports_native_structured_output": true + "supports_system_messages": true }, "nvidia.nemotron-nano-3-30b": { "input_cost_per_token": 6e-08, diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index bf957567a99..c9fc0c39d67 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -2664,19 +2664,21 @@ def test_supports_native_structured_outputs(): # Non-Anthropic models assert config._supports_native_structured_outputs("qwen.qwen3-235b-a22b-2507-v1:0") assert config._supports_native_structured_outputs("mistral.mistral-large-3-675b-instruct") - assert config._supports_native_structured_outputs("google.gemma-3-27b-it") assert config._supports_native_structured_outputs("minimax.minimax-m2") assert config._supports_native_structured_outputs("moonshot.kimi-k2-thinking") - assert config._supports_native_structured_outputs("nvidia.nemotron-nano-12b-v2") + assert config._supports_native_structured_outputs("nvidia.nemotron-nano-3-30b") # Unsupported models -- should fall back to tool-call approach assert not config._supports_native_structured_outputs("anthropic.claude-3-5-sonnet-20241022-v2:0") assert not config._supports_native_structured_outputs("anthropic.claude-sonnet-4-20250514-v1:0") assert not config._supports_native_structured_outputs("meta.llama3-3-70b-instruct-v1:0") assert not config._supports_native_structured_outputs("amazon.nova-pro-v1:0") - # Excluded despite AWS listing them: broken constrained decoding on Bedrock + # Excluded: broken constrained decoding on Bedrock assert not config._supports_native_structured_outputs("openai.gpt-oss-120b-1:0") assert not config._supports_native_structured_outputs("mistral.magistral-small-2509") + # Excluded: ignores schema or broken on Bedrock + assert not config._supports_native_structured_outputs("google.gemma-3-27b-it") + assert not config._supports_native_structured_outputs("nvidia.nemotron-nano-12b-v2") def test_create_output_config_for_response_format(): From a7ebc72c26b5937cc8f0f7b3f88b8326333cf784 Mon Sep 17 00:00:00 2001 From: Nicholas Gigliotti Date: Mon, 16 Mar 2026 20:18:19 -0400 Subject: [PATCH 009/117] Add native structured output flag for deepseek.v3-v1:0 Integration tested 28/28 (10 sync + 10 streaming + extras) on the native outputConfig.textFormat path in us-west-2. deepseek.v3.2 does not support native structured output (Bedrock returns 400). --- litellm/model_prices_and_context_window_backup.json | 3 ++- model_prices_and_context_window.json | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 77ba0954b0d..7ce3671d332 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -11758,7 +11758,8 @@ "output_cost_per_token": 1.68e-06, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "deepseek.v3.2": { "input_cost_per_token": 6.2e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 77ba0954b0d..7ce3671d332 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -11758,7 +11758,8 @@ "output_cost_per_token": 1.68e-06, "supports_function_calling": true, "supports_reasoning": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_native_structured_output": true }, "deepseek.v3.2": { "input_cost_per_token": 6.2e-07, From ed06dded199fa38fc2d2ef6165c25608020b1fa3 Mon Sep 17 00:00:00 2001 From: Nicholas Gigliotti Date: Mon, 16 Mar 2026 20:30:40 -0400 Subject: [PATCH 010/117] Remove native structured output flag from minimax-m2.1, kimi-k2.5, qwen3-coder-next Integration testing confirmed: - minimax.minimax-m2.1: Bedrock rejects outputConfig.textFormat (400) - moonshotai.kimi-k2.5: Bedrock rejects outputConfig.textFormat (400) - qwen.qwen3-coder-next: unavailable in us-east-1 and us-west-2 --- litellm/model_prices_and_context_window_backup.json | 9 +++------ model_prices_and_context_window.json | 9 +++------ 2 files changed, 6 insertions(+), 12 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7ce3671d332..e18f00508c0 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -21290,8 +21290,7 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/", - "supports_native_structured_output": true + "source": "https://aws.amazon.com/bedrock/pricing/" }, "minimax/speech-02-hd": { "input_cost_per_character": 0.0001, @@ -22248,8 +22247,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": true, - "source": "https://aws.amazon.com/bedrock/pricing/", - "supports_native_structured_output": true + "source": "https://aws.amazon.com/bedrock/pricing/" }, "moonshot/kimi-k2-0711-preview": { "cache_read_input_token_cost": 1.5e-07, @@ -26273,8 +26271,7 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/", - "supports_native_structured_output": true + "source": "https://aws.amazon.com/bedrock/pricing/" }, "recraft/recraftv2": { "litellm_provider": "recraft", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7ce3671d332..e18f00508c0 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -21290,8 +21290,7 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/", - "supports_native_structured_output": true + "source": "https://aws.amazon.com/bedrock/pricing/" }, "minimax/speech-02-hd": { "input_cost_per_character": 0.0001, @@ -22248,8 +22247,7 @@ "supports_system_messages": true, "supports_tool_choice": true, "supports_vision": true, - "source": "https://aws.amazon.com/bedrock/pricing/", - "supports_native_structured_output": true + "source": "https://aws.amazon.com/bedrock/pricing/" }, "moonshot/kimi-k2-0711-preview": { "cache_read_input_token_cost": 1.5e-07, @@ -26273,8 +26271,7 @@ "supports_function_calling": true, "supports_system_messages": true, "supports_tool_choice": true, - "source": "https://aws.amazon.com/bedrock/pricing/", - "supports_native_structured_output": true + "source": "https://aws.amazon.com/bedrock/pricing/" }, "recraft/recraftv2": { "litellm_provider": "recraft", From d7e55bf105e69cf011175c2f5849f276c58806f2 Mon Sep 17 00:00:00 2001 From: Nicholas Gigliotti Date: Mon, 16 Mar 2026 20:40:47 -0400 Subject: [PATCH 011/117] Fix test state leakage: restore env and model_cost after each test Wrap cost-map-dependent tests in try/finally to restore os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] and litellm.model_cost, preventing test-ordering sensitivity. --- .../chat/test_converse_transformation.py | 178 ++++++++++-------- 1 file changed, 101 insertions(+), 77 deletions(-) diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index c9fc0c39d67..2b57fd23a3b 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -2645,40 +2645,48 @@ def test_supports_native_structured_outputs(): Support is driven by the ``supports_native_structured_output`` flag in the cost JSON (litellm.model_cost), not a hardcoded model set. """ + old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") + old_cost = litellm.model_cost os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" litellm.model_cost = litellm.get_model_cost_map(url="") + try: + config = AmazonConverseConfig() - config = AmazonConverseConfig() + # Supported models (have supports_native_structured_output=true in cost JSON) + assert config._supports_native_structured_outputs("anthropic.claude-sonnet-4-5-20250929-v1:0") + assert config._supports_native_structured_outputs("anthropic.claude-haiku-4-5-20251001-v1:0") + assert config._supports_native_structured_outputs("anthropic.claude-opus-4-6-v1") + # Version suffix (:0) is stripped when looking up models without it in cost JSON + assert config._supports_native_structured_outputs("anthropic.claude-opus-4-6-v1:0") + # Regional prefix is stripped by get_bedrock_base_model + assert config._supports_native_structured_outputs("eu.anthropic.claude-opus-4-5-20251101-v1:0") + # Claude 4.6 Sonnet + assert config._supports_native_structured_outputs("anthropic.claude-sonnet-4-6") + assert config._supports_native_structured_outputs("us.anthropic.claude-sonnet-4-6") + # Non-Anthropic models + assert config._supports_native_structured_outputs("qwen.qwen3-235b-a22b-2507-v1:0") + assert config._supports_native_structured_outputs("mistral.mistral-large-3-675b-instruct") + assert config._supports_native_structured_outputs("minimax.minimax-m2") + assert config._supports_native_structured_outputs("moonshot.kimi-k2-thinking") + assert config._supports_native_structured_outputs("nvidia.nemotron-nano-3-30b") - # Supported models (have supports_native_structured_output=true in cost JSON) - assert config._supports_native_structured_outputs("anthropic.claude-sonnet-4-5-20250929-v1:0") - assert config._supports_native_structured_outputs("anthropic.claude-haiku-4-5-20251001-v1:0") - assert config._supports_native_structured_outputs("anthropic.claude-opus-4-6-v1") - # Version suffix (:0) is stripped when looking up models without it in cost JSON - assert config._supports_native_structured_outputs("anthropic.claude-opus-4-6-v1:0") - # Regional prefix is stripped by get_bedrock_base_model - assert config._supports_native_structured_outputs("eu.anthropic.claude-opus-4-5-20251101-v1:0") - # Claude 4.6 Sonnet - assert config._supports_native_structured_outputs("anthropic.claude-sonnet-4-6") - assert config._supports_native_structured_outputs("us.anthropic.claude-sonnet-4-6") - # Non-Anthropic models - assert config._supports_native_structured_outputs("qwen.qwen3-235b-a22b-2507-v1:0") - assert config._supports_native_structured_outputs("mistral.mistral-large-3-675b-instruct") - assert config._supports_native_structured_outputs("minimax.minimax-m2") - assert config._supports_native_structured_outputs("moonshot.kimi-k2-thinking") - assert config._supports_native_structured_outputs("nvidia.nemotron-nano-3-30b") - - # Unsupported models -- should fall back to tool-call approach - assert not config._supports_native_structured_outputs("anthropic.claude-3-5-sonnet-20241022-v2:0") - assert not config._supports_native_structured_outputs("anthropic.claude-sonnet-4-20250514-v1:0") - assert not config._supports_native_structured_outputs("meta.llama3-3-70b-instruct-v1:0") - assert not config._supports_native_structured_outputs("amazon.nova-pro-v1:0") - # Excluded: broken constrained decoding on Bedrock - assert not config._supports_native_structured_outputs("openai.gpt-oss-120b-1:0") - assert not config._supports_native_structured_outputs("mistral.magistral-small-2509") - # Excluded: ignores schema or broken on Bedrock - assert not config._supports_native_structured_outputs("google.gemma-3-27b-it") - assert not config._supports_native_structured_outputs("nvidia.nemotron-nano-12b-v2") + # Unsupported models -- should fall back to tool-call approach + assert not config._supports_native_structured_outputs("anthropic.claude-3-5-sonnet-20241022-v2:0") + assert not config._supports_native_structured_outputs("anthropic.claude-sonnet-4-20250514-v1:0") + assert not config._supports_native_structured_outputs("meta.llama3-3-70b-instruct-v1:0") + assert not config._supports_native_structured_outputs("amazon.nova-pro-v1:0") + # Excluded: broken constrained decoding on Bedrock + assert not config._supports_native_structured_outputs("openai.gpt-oss-120b-1:0") + assert not config._supports_native_structured_outputs("mistral.magistral-small-2509") + # Excluded: ignores schema or broken on Bedrock + assert not config._supports_native_structured_outputs("google.gemma-3-27b-it") + assert not config._supports_native_structured_outputs("nvidia.nemotron-nano-12b-v2") + finally: + litellm.model_cost = old_cost + if old_env is None: + os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None) + else: + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = old_env def test_create_output_config_for_response_format(): @@ -2794,45 +2802,53 @@ def test_translate_response_format_fallback_tool_call(): def test_native_structured_output_no_fake_stream(): """When using native structured outputs with streaming, fake_stream should NOT be set.""" + old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") + old_cost = litellm.model_cost os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" litellm.model_cost = litellm.get_model_cost_map(url="") + try: + config = AmazonConverseConfig() - config = AmazonConverseConfig() - - response_format = { - "type": "json_schema", - "json_schema": { - "name": "Result", - "schema": { - "type": "object", - "properties": { - "answer": {"type": "string"}, + response_format = { + "type": "json_schema", + "json_schema": { + "name": "Result", + "schema": { + "type": "object", + "properties": { + "answer": {"type": "string"}, + }, }, }, - }, - } + } - optional_params: dict = {} - result = config._translate_response_format_param( - value=response_format, - model="anthropic.claude-sonnet-4-5-20250929-v1:0", - optional_params=optional_params, - non_default_params={"response_format": response_format, "stream": True}, - is_thinking_enabled=False, - ) + optional_params: dict = {} + result = config._translate_response_format_param( + value=response_format, + model="anthropic.claude-sonnet-4-5-20250929-v1:0", + optional_params=optional_params, + non_default_params={"response_format": response_format, "stream": True}, + is_thinking_enabled=False, + ) - assert "outputConfig" in result - assert result["json_mode"] is True - # No fake_stream for native approach - assert "fake_stream" not in result + assert "outputConfig" in result + assert result["json_mode"] is True + # No fake_stream for native approach + assert "fake_stream" not in result - # Verify the schema content - schema_str = result["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["schema"] - assert json.loads(schema_str) == { - "type": "object", - "properties": {"answer": {"type": "string"}}, - "additionalProperties": False, - } + # Verify the schema content + schema_str = result["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["schema"] + assert json.loads(schema_str) == { + "type": "object", + "properties": {"answer": {"type": "string"}}, + "additionalProperties": False, + } + finally: + litellm.model_cost = old_cost + if old_env is None: + os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None) + else: + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = old_env def test_transform_request_with_output_config(): @@ -3063,26 +3079,34 @@ def test_add_additional_properties_definitions(): def test_json_object_no_schema_falls_back_to_tool_call(): """response_format: {type: json_object} with no schema should use tool-call fallback, even for models that support native structured outputs.""" + old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") + old_cost = litellm.model_cost os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" litellm.model_cost = litellm.get_model_cost_map(url="") + try: + config = AmazonConverseConfig() + optional_params: dict = {} + non_default_params = {"response_format": {"type": "json_object"}} - config = AmazonConverseConfig() - optional_params: dict = {} - non_default_params = {"response_format": {"type": "json_object"}} + result = config._translate_response_format_param( + value=non_default_params["response_format"], + model="anthropic.claude-sonnet-4-5-20250929-v1:0", + optional_params=optional_params, + non_default_params=non_default_params, + is_thinking_enabled=False, + ) - result = config._translate_response_format_param( - value=non_default_params["response_format"], - model="anthropic.claude-sonnet-4-5-20250929-v1:0", - optional_params=optional_params, - non_default_params=non_default_params, - is_thinking_enabled=False, - ) - - # Should NOT use native outputConfig (no schema provided) - assert "outputConfig" not in result - # Should use tool-call fallback - assert "tools" in result - assert result["json_mode"] is True + # Should NOT use native outputConfig (no schema provided) + assert "outputConfig" not in result + # Should use tool-call fallback + assert "tools" in result + assert result["json_mode"] is True + finally: + litellm.model_cost = old_cost + if old_env is None: + os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None) + else: + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = old_env def test_output_config_applies_additional_properties(): From 0ef8eb612106f4d20d268a8c585decbd76676b15 Mon Sep 17 00:00:00 2001 From: Nicholas Gigliotti Date: Mon, 16 Mar 2026 20:58:40 -0400 Subject: [PATCH 012/117] Add test assertion for deepseek.v3-v1:0 native structured output --- .../llms/bedrock/chat/test_converse_transformation.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index 2b57fd23a3b..a26b2a8bd64 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -2669,6 +2669,8 @@ def test_supports_native_structured_outputs(): assert config._supports_native_structured_outputs("minimax.minimax-m2") assert config._supports_native_structured_outputs("moonshot.kimi-k2-thinking") assert config._supports_native_structured_outputs("nvidia.nemotron-nano-3-30b") + # DeepSeek: old substring "deepseek-v3.1" didn't match real ID + assert config._supports_native_structured_outputs("deepseek.v3-v1:0") # Unsupported models -- should fall back to tool-call approach assert not config._supports_native_structured_outputs("anthropic.claude-3-5-sonnet-20241022-v2:0") From b45ef7f6b9848f669c8bd177b8c047f1b67a9d8a Mon Sep 17 00:00:00 2001 From: Nicholas Gigliotti Date: Mon, 16 Mar 2026 21:03:08 -0400 Subject: [PATCH 013/117] Keep multimodal gemini-embedding-2-preview entry to align with #23599 Main has two duplicate keys for vertex_ai/gemini-embedding-2-preview. Our JSON round-trip collapsed them to the second (text-only) entry, but PR #23599 intentionally keeps the first (multimodal pricing) entry. Restore the multimodal entry to avoid conflicts. --- litellm/model_prices_and_context_window_backup.json | 7 +++++-- model_prices_and_context_window.json | 7 +++++-- 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e18f00508c0..f55b66cc7d5 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -14623,14 +14623,17 @@ "uses_embed_content": true }, "vertex_ai/gemini-embedding-2-preview": { - "input_cost_per_token": 1.5e-07, + "input_cost_per_audio_per_second": 0.00016, + "input_cost_per_image": 0.00012, + "input_cost_per_token": 2e-07, + "input_cost_per_video_per_second": 0.00079, "litellm_provider": "vertex_ai", "max_input_tokens": 8192, "max_tokens": 8192, "mode": "embedding", "output_cost_per_token": 0, "output_vector_size": 3072, - "source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal", + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing", "supports_multimodal": true, "uses_embed_content": true }, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e18f00508c0..f55b66cc7d5 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -14623,14 +14623,17 @@ "uses_embed_content": true }, "vertex_ai/gemini-embedding-2-preview": { - "input_cost_per_token": 1.5e-07, + "input_cost_per_audio_per_second": 0.00016, + "input_cost_per_image": 0.00012, + "input_cost_per_token": 2e-07, + "input_cost_per_video_per_second": 0.00079, "litellm_provider": "vertex_ai", "max_input_tokens": 8192, "max_tokens": 8192, "mode": "embedding", "output_cost_per_token": 0, "output_vector_size": 3072, - "source": "https://ai.google.dev/gemini-api/docs/embeddings#multimodal", + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing", "supports_multimodal": true, "uses_embed_content": true }, From 92654bad37d159dc08daf58b75b0bd0586edab44 Mon Sep 17 00:00:00 2001 From: Nicholas Gigliotti Date: Thu, 26 Mar 2026 21:46:52 -0400 Subject: [PATCH 014/117] Refactor _supports_native_structured_outputs to use standard supports_* utility pattern Addresses Greptile review feedback: replace direct litellm.model_cost lookup with the standard _supports_factory infrastructure used by supports_reasoning, supports_native_streaming, etc. - Add supports_native_structured_output() utility in litellm/utils.py - Add supports_native_structured_output field to ModelInfoBase type - Wire field into _get_model_info_helper return dict - Delegate from Bedrock _supports_native_structured_outputs to utility - Add field to JSON schema validator in test_utils.py --- .../bedrock/chat/converse_transformation.py | 29 +++---- ...odel_prices_and_context_window_backup.json | 4 +- litellm/types/utils.py | 1 + litellm/utils.py | 16 ++++ model_prices_and_context_window.json | 4 +- .../chat/test_converse_transformation.py | 81 ++++++++++--------- tests/test_litellm/test_utils.py | 1 + 7 files changed, 76 insertions(+), 60 deletions(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 78549b160fa..90b501a9cbb 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -703,28 +703,21 @@ class AmazonConverseConfig(BaseConfig): return _tool @staticmethod - def _supports_native_structured_outputs(model: str) -> bool: + def _supports_native_structured_outputs( + model: str, custom_llm_provider: Optional[str] = None + ) -> bool: """Check if the Bedrock model supports native structured outputs (outputConfig.textFormat). - Looks up the ``supports_native_structured_output`` flag in - ``litellm.model_cost`` (set in the cost JSON). + Delegates to the standard ``supports_native_structured_output`` utility + which looks up the flag in ``litellm.model_cost`` via + ``_get_model_info_helper``. Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/structured-output.html """ - from litellm.llms.bedrock.common_utils import get_bedrock_base_model + from litellm.utils import supports_native_structured_output - base_model = get_bedrock_base_model(model) - - # Try direct lookup - info = litellm.model_cost.get(base_model) - - # Try without version suffix (e.g. "model-v1:0" -> "model-v1") - if info is None and ":" in base_model: - info = litellm.model_cost.get(base_model.rsplit(":", 1)[0]) - - if info is not None: - return info.get("supports_native_structured_output", False) is True - - return False + return supports_native_structured_output( + model=model, custom_llm_provider=custom_llm_provider + ) @staticmethod def _add_additional_properties_to_schema(schema: dict) -> dict: @@ -951,7 +944,7 @@ class AmazonConverseConfig(BaseConfig): if "type" in value and value["type"] == "text": return optional_params - if self._supports_native_structured_outputs(model) and json_schema is not None: + if self._supports_native_structured_outputs(model, self.custom_llm_provider) and json_schema is not None: # Use Bedrock's native structured outputs API (outputConfig.textFormat) # No synthetic tool injection, no fake_stream needed. # Requires an explicit schema — json_object with no schema falls through diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index f55b66cc7d5..b4950cdba72 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -31179,9 +31179,7 @@ "mode": "chat", "output_cost_per_token": 3.2e-06, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#glm-models", - "supported_regions": [ - "global" - ], + "supported_regions": ["global"], "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index bd673da8bed..bdc6a0bba9b 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -132,6 +132,7 @@ class ProviderSpecificModelInfo(TypedDict, total=False): supports_audio_output: Optional[bool] supports_pdf_input: Optional[bool] supports_native_streaming: Optional[bool] + supports_native_structured_output: Optional[bool] supports_parallel_function_calling: Optional[bool] supports_web_search: Optional[bool] supports_reasoning: Optional[bool] diff --git a/litellm/utils.py b/litellm/utils.py index 088ee07d630..42029a06f20 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2735,6 +2735,19 @@ def supports_reasoning(model: str, custom_llm_provider: Optional[str] = None) -> ) +def supports_native_structured_output( + model: str, custom_llm_provider: Optional[str] = None +) -> bool: + """ + Check if the given model supports native structured outputs and return a boolean value. + """ + return _supports_factory( + model=model, + custom_llm_provider=custom_llm_provider, + key="supports_native_structured_output", + ) + + def get_supported_regions( model: str, custom_llm_provider: Optional[str] = None ) -> Optional[List[str]]: @@ -5831,6 +5844,9 @@ def _get_model_info_helper( # noqa: PLR0915 supports_native_streaming=_model_info.get( "supports_native_streaming", None ), + supports_native_structured_output=_model_info.get( + "supports_native_structured_output", None + ), supports_web_search=_model_info.get("supports_web_search", None), supports_url_context=_model_info.get("supports_url_context", None), supports_reasoning=_model_info.get("supports_reasoning", None), diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index f55b66cc7d5..b4950cdba72 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -31179,9 +31179,7 @@ "mode": "chat", "output_cost_per_token": 3.2e-06, "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing#glm-models", - "supported_regions": [ - "global" - ], + "supported_regions": ["global"], "supports_function_calling": true, "supports_prompt_caching": true, "supports_reasoning": true, diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index a26b2a8bd64..867a3e61bbd 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -2656,8 +2656,6 @@ def test_supports_native_structured_outputs(): assert config._supports_native_structured_outputs("anthropic.claude-sonnet-4-5-20250929-v1:0") assert config._supports_native_structured_outputs("anthropic.claude-haiku-4-5-20251001-v1:0") assert config._supports_native_structured_outputs("anthropic.claude-opus-4-6-v1") - # Version suffix (:0) is stripped when looking up models without it in cost JSON - assert config._supports_native_structured_outputs("anthropic.claude-opus-4-6-v1:0") # Regional prefix is stripped by get_bedrock_base_model assert config._supports_native_structured_outputs("eu.anthropic.claude-opus-4-5-20251101-v1:0") # Claude 4.6 Sonnet @@ -2728,46 +2726,57 @@ def test_create_output_config_for_response_format(): def test_translate_response_format_native_output_config(): """For supported models, _translate_response_format_param should produce outputConfig.""" - config = AmazonConverseConfig() + old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP") + old_cost = litellm.model_cost + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + try: + config = AmazonConverseConfig() - response_format = { - "type": "json_schema", - "json_schema": { - "name": "WeatherResult", - "description": "Weather info", - "schema": { - "type": "object", - "properties": { - "temp": {"type": "number"}, + response_format = { + "type": "json_schema", + "json_schema": { + "name": "WeatherResult", + "description": "Weather info", + "schema": { + "type": "object", + "properties": { + "temp": {"type": "number"}, + }, + "required": ["temp"], }, - "required": ["temp"], }, - }, - } + } - optional_params: dict = {} - result = config._translate_response_format_param( - value=response_format, - model="anthropic.claude-sonnet-4-5-20250929-v1:0", - optional_params=optional_params, - non_default_params={"response_format": response_format}, - is_thinking_enabled=False, - ) + optional_params: dict = {} + result = config._translate_response_format_param( + value=response_format, + model="anthropic.claude-sonnet-4-5-20250929-v1:0", + optional_params=optional_params, + non_default_params={"response_format": response_format}, + is_thinking_enabled=False, + ) - # Should have outputConfig, NOT tools - assert "outputConfig" in result - assert "tools" not in result - assert "tool_choice" not in result - assert result["json_mode"] is True - # No fake_stream for native approach - assert "fake_stream" not in result + # Should have outputConfig, NOT tools + assert "outputConfig" in result + assert "tools" not in result + assert "tool_choice" not in result + assert result["json_mode"] is True + # No fake_stream for native approach + assert "fake_stream" not in result - # Verify the schema content (additionalProperties: false is added by normalization) - schema_str = result["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["schema"] - parsed_schema = json.loads(schema_str) - expected_schema = {**response_format["json_schema"]["schema"], "additionalProperties": False} - assert parsed_schema == expected_schema - assert result["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["name"] == "WeatherResult" + # Verify the schema content (additionalProperties: false is added by normalization) + schema_str = result["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["schema"] + parsed_schema = json.loads(schema_str) + expected_schema = {**response_format["json_schema"]["schema"], "additionalProperties": False} + assert parsed_schema == expected_schema + assert result["outputConfig"]["textFormat"]["structure"]["jsonSchema"]["name"] == "WeatherResult" + finally: + litellm.model_cost = old_cost + if old_env is None: + os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None) + else: + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = old_env def test_translate_response_format_fallback_tool_call(): diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 38b7b576d4f..f52e993a289 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -831,6 +831,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): }, }, "supports_native_streaming": {"type": "boolean"}, + "supports_native_structured_output": {"type": "boolean"}, "tiered_pricing": { "type": "array", "items": { From dfb543369b22880541f322d67f464c9df82c39fa Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Thu, 26 Mar 2026 21:09:01 -0700 Subject: [PATCH 015/117] fix: address zizmor comments --- .../helm-oci-chart-releaser/action.yml | 36 +++++++++++++++---- .github/dependabot.yaml | 3 ++ .../auto_update_price_and_context_window.yml | 6 ++++ .github/workflows/check_duplicate_issues.yml | 1 + .github/workflows/codeql.yml | 2 ++ .github/workflows/codspeed.yml | 2 ++ .../workflows/create_daily_staging_branch.yml | 6 ++++ .github/workflows/helm_unit_test.yml | 5 +++ .github/workflows/issue-keyword-labeler.yml | 2 ++ .github/workflows/llm-translation-testing.yml | 15 +++++--- .github/workflows/read_pyproject_version.yml | 5 +++ .github/workflows/run_observatory_tests.yml | 11 +++--- .github/workflows/scan_duplicate_issues.yml | 1 + .github/workflows/stale.yml | 4 +++ .github/workflows/test-linting.yml | 5 +++ .github/workflows/test-litellm-matrix.yml | 5 +++ .github/workflows/test-litellm-ui-build.yml | 2 ++ .github/workflows/test-litellm.yml | 5 +++ .github/workflows/test-mcp.yml | 5 +++ .github/workflows/test-model-map.yaml | 5 +++ .../test-proxy-e2e-azure-batches.yml | 5 +++ .github/workflows/test_server_root_path.yml | 2 ++ 22 files changed, 117 insertions(+), 16 deletions(-) diff --git a/.github/actions/helm-oci-chart-releaser/action.yml b/.github/actions/helm-oci-chart-releaser/action.yml index 1823e262832..454c591d436 100644 --- a/.github/actions/helm-oci-chart-releaser/action.yml +++ b/.github/actions/helm-oci-chart-releaser/action.yml @@ -41,32 +41,54 @@ runs: using: composite steps: - name: Helm | Setup - uses: azure/setup-helm@v4 + uses: azure/setup-helm@1a275c3b69536ee54be43f2070a358922e12c8d4 # v4.3.1 with: version: v3.20.0 - name: Helm | Login shell: bash - run: echo ${{ inputs.registry_password }} | helm registry login -u ${{ inputs.registry_username }} --password-stdin ${{ inputs.registry }} + env: + REGISTRY_PASSWORD: ${{ inputs.registry_password }} + REGISTRY_USERNAME: ${{ inputs.registry_username }} + REGISTRY: ${{ inputs.registry }} + run: echo "$REGISTRY_PASSWORD" | helm registry login -u "$REGISTRY_USERNAME" --password-stdin "$REGISTRY" - name: Helm | Dependency if: inputs.update_dependencies == 'true' shell: bash - run: helm dependency update ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} + env: + CHART_PATH: ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} + run: helm dependency update "$CHART_PATH" - name: Helm | Package shell: bash - run: helm package ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} --version ${{ inputs.tag }} --app-version ${{ inputs.app_version }} + env: + CHART_PATH: ${{ inputs.path == null && format('{0}/{1}', 'charts', inputs.name) || inputs.path }} + TAG: ${{ inputs.tag }} + APP_VERSION: ${{ inputs.app_version }} + run: helm package "$CHART_PATH" --version "$TAG" --app-version "$APP_VERSION" - name: Helm | Push shell: bash - run: helm push ${{ inputs.name }}-${{ inputs.tag }}.tgz oci://${{ inputs.registry }}/${{ inputs.repository }} + env: + NAME: ${{ inputs.name }} + TAG: ${{ inputs.tag }} + REGISTRY: ${{ inputs.registry }} + REPOSITORY: ${{ inputs.repository }} + run: helm push "${NAME}-${TAG}.tgz" "oci://${REGISTRY}/${REPOSITORY}" - name: Helm | Logout shell: bash - run: helm registry logout ${{ inputs.registry }} + env: + REGISTRY: ${{ inputs.registry }} + run: helm registry logout "$REGISTRY" - name: Helm | Output id: output shell: bash - run: echo "image=${{ inputs.registry }}/${{ inputs.repository }}/${{ inputs.name }}:${{ inputs.tag }}" >> $GITHUB_OUTPUT + env: + REGISTRY: ${{ inputs.registry }} + REPOSITORY: ${{ inputs.repository }} + NAME: ${{ inputs.name }} + TAG: ${{ inputs.tag }} + run: echo "image=${REGISTRY}/${REPOSITORY}/${NAME}:${TAG}" >> $GITHUB_OUTPUT diff --git a/.github/dependabot.yaml b/.github/dependabot.yaml index 58e7cfe10da..c49882a8d62 100644 --- a/.github/dependabot.yaml +++ b/.github/dependabot.yaml @@ -4,6 +4,9 @@ updates: directory: "/" schedule: interval: "daily" + cooldown: + default-days: 7 + semver-major-days: 14 groups: github-actions: patterns: diff --git a/.github/workflows/auto_update_price_and_context_window.yml b/.github/workflows/auto_update_price_and_context_window.yml index 8265e0d09c5..60e89936219 100644 --- a/.github/workflows/auto_update_price_and_context_window.yml +++ b/.github/workflows/auto_update_price_and_context_window.yml @@ -5,12 +5,18 @@ on: - cron: "0 0 * * 0" # Run every Sundays at midnight #- cron: "0 0 * * *" # Run daily at midnight +permissions: + contents: write + pull-requests: write + jobs: auto_update_price_and_context_window: if: github.repository == 'BerriAI/litellm' runs-on: ubuntu-latest steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Install Dependencies run: | pip install 'aiohttp==3.13.3' diff --git a/.github/workflows/check_duplicate_issues.yml b/.github/workflows/check_duplicate_issues.yml index 539290bfeaf..289d78880ad 100644 --- a/.github/workflows/check_duplicate_issues.yml +++ b/.github/workflows/check_duplicate_issues.yml @@ -33,6 +33,7 @@ jobs: uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: sparse-checkout: .github/scripts + persist-credentials: false - name: Set up Python if: github.event.action == 'opened' diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 0e49b3af138..4fafc0e2737 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -39,6 +39,8 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Initialize CodeQL uses: github/codeql-action/init@ebcb5b36ded6beda4ceefea6a8bc4cc885255bb3 # v3 diff --git a/.github/workflows/codspeed.yml b/.github/workflows/codspeed.yml index 749242b1cd6..52d64addea9 100644 --- a/.github/workflows/codspeed.yml +++ b/.github/workflows/codspeed.yml @@ -26,6 +26,8 @@ jobs: steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Set up Python uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 diff --git a/.github/workflows/create_daily_staging_branch.yml b/.github/workflows/create_daily_staging_branch.yml index 0df5f4f92ea..424d8de0a41 100644 --- a/.github/workflows/create_daily_staging_branch.yml +++ b/.github/workflows/create_daily_staging_branch.yml @@ -9,12 +9,15 @@ jobs: create-staging-branch: if: github.repository == 'BerriAI/litellm' runs-on: ubuntu-latest + permissions: + contents: write steps: - name: Checkout repository uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: fetch-depth: 0 + persist-credentials: false - name: Create daily staging branch env: @@ -46,12 +49,15 @@ jobs: create-internal-dev-branch: if: github.repository == 'BerriAI/litellm' runs-on: ubuntu-latest + permissions: + contents: write steps: - name: Checkout repository uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: fetch-depth: 0 + persist-credentials: false - name: Create internal dev branch env: diff --git a/.github/workflows/helm_unit_test.yml b/.github/workflows/helm_unit_test.yml index 416523f241b..06836b1d1cd 100644 --- a/.github/workflows/helm_unit_test.yml +++ b/.github/workflows/helm_unit_test.yml @@ -6,12 +6,17 @@ on: branches: - main +permissions: + contents: read + jobs: unit-test: runs-on: ubuntu-latest steps: - name: Checkout uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Set up Helm 3.11.1 uses: azure/setup-helm@1a275c3b69536ee54be43f2070a358922e12c8d4 # v4.3.1 diff --git a/.github/workflows/issue-keyword-labeler.yml b/.github/workflows/issue-keyword-labeler.yml index 59b8fd9cf9f..7e2693209b6 100644 --- a/.github/workflows/issue-keyword-labeler.yml +++ b/.github/workflows/issue-keyword-labeler.yml @@ -14,6 +14,8 @@ jobs: steps: - name: Checkout code uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Scan for provider keywords id: scan diff --git a/.github/workflows/llm-translation-testing.yml b/.github/workflows/llm-translation-testing.yml index a6b643dd92e..922013c4b54 100644 --- a/.github/workflows/llm-translation-testing.yml +++ b/.github/workflows/llm-translation-testing.yml @@ -11,6 +11,9 @@ on: tags: - "v*-rc*" # Triggers on release candidate tags like v1.0.0-rc1 +permissions: + contents: read + jobs: run-llm-translation-tests: runs-on: ubuntu-latest @@ -20,6 +23,7 @@ jobs: - name: Checkout code uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: + persist-credentials: false ref: ${{ github.event.inputs.release_candidate_tag || github.ref }} - name: Set up Python @@ -33,8 +37,8 @@ jobs: poetry config virtualenvs.create true poetry config virtualenvs.in-project true - - name: Cache Poetry dependencies - uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.0.0 + - name: Restore Poetry dependencies cache + uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.0.0 with: path: | ~/.cache/pypoetry @@ -60,11 +64,12 @@ jobs: AZURE_API_KEY: ${{ secrets.AZURE_API_KEY }} AZURE_API_BASE: ${{ secrets.AZURE_API_BASE }} AZURE_API_VERSION: ${{ secrets.AZURE_API_VERSION }} - # Add other API keys as needed + RC_TAG: ${{ github.event.inputs.release_candidate_tag || github.ref_name }} + COMMIT_SHA: ${{ github.sha }} run: | python .github/workflows/run_llm_translation_tests.py \ - --tag "${{ github.event.inputs.release_candidate_tag || github.ref_name }}" \ - --commit "${{ github.sha }}" \ + --tag "$RC_TAG" \ + --commit "$COMMIT_SHA" \ || true # Continue even if tests fail - name: Display test summary diff --git a/.github/workflows/read_pyproject_version.yml b/.github/workflows/read_pyproject_version.yml index a9d16ca4413..04b4a38ce19 100644 --- a/.github/workflows/read_pyproject_version.yml +++ b/.github/workflows/read_pyproject_version.yml @@ -5,6 +5,9 @@ on: branches: - main # Change this to the default branch of your repository +permissions: + contents: read + jobs: read-version: runs-on: ubuntu-latest @@ -12,6 +15,8 @@ jobs: steps: - name: Checkout code uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Read version from pyproject.toml id: read-version diff --git a/.github/workflows/run_observatory_tests.yml b/.github/workflows/run_observatory_tests.yml index b0706b7b716..a25b96766d7 100644 --- a/.github/workflows/run_observatory_tests.yml +++ b/.github/workflows/run_observatory_tests.yml @@ -34,6 +34,8 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Validate tag input env: @@ -49,11 +51,12 @@ jobs: TAG: ${{ inputs.tag }} AZURE_API_KEY: ${{ secrets.AZURE_API_KEY }} AZURE_API_BASE: ${{ secrets.AZURE_API_BASE }} + WORKSPACE: ${{ github.workspace }} run: | docker run -d \ --name litellm-rc \ -p 4000:4000 \ - -v "${{ github.workspace }}/.github/observatory/litellm_config.yaml:/app/config.yaml" \ + -v "${WORKSPACE}/.github/observatory/litellm_config.yaml:/app/config.yaml" \ -e LITELLM_MASTER_KEY="${LITELLM_MASTER_KEY}" \ -e AZURE_API_KEY="${AZURE_API_KEY}" \ -e AZURE_API_BASE="${AZURE_API_BASE}" \ @@ -104,11 +107,11 @@ jobs: - name: Verify tunnel connectivity run: | - echo "Testing tunnel at ${{ env.TUNNEL_URL }}..." + echo "Testing tunnel at ${TUNNEL_URL}..." # Quick tunnels need time for DNS propagation; retry to avoid # transient NXDOMAIN (curl exit code 6) on first attempt. for i in $(seq 1 10); do - if curl -sf "${{ env.TUNNEL_URL }}/health/liveliness" > /dev/null 2>&1; then + if curl -sf "${TUNNEL_URL}/health/liveliness" > /dev/null 2>&1; then echo "Tunnel is working (attempt $i)" exit 0 fi @@ -222,5 +225,5 @@ jobs: - name: Cleanup if: always() run: | - kill "${{ env.CLOUDFLARED_PID }}" 2>/dev/null || true + kill "$CLOUDFLARED_PID" 2>/dev/null || true docker rm -f litellm-rc 2>/dev/null || true diff --git a/.github/workflows/scan_duplicate_issues.yml b/.github/workflows/scan_duplicate_issues.yml index 6c88a54554e..222ff11f304 100644 --- a/.github/workflows/scan_duplicate_issues.yml +++ b/.github/workflows/scan_duplicate_issues.yml @@ -24,6 +24,7 @@ jobs: uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: sparse-checkout: .github/scripts + persist-credentials: false - name: Set up Python uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 diff --git a/.github/workflows/stale.yml b/.github/workflows/stale.yml index e7ce48b9fbd..c905bb12312 100644 --- a/.github/workflows/stale.yml +++ b/.github/workflows/stale.yml @@ -5,6 +5,10 @@ on: - cron: "0 0 * * *" # Runs daily at midnight UTC workflow_dispatch: +permissions: + issues: write + pull-requests: write + jobs: stale: if: github.repository == 'BerriAI/litellm' diff --git a/.github/workflows/test-linting.yml b/.github/workflows/test-linting.yml index 1424088eaa2..5bb85716a17 100644 --- a/.github/workflows/test-linting.yml +++ b/.github/workflows/test-linting.yml @@ -4,6 +4,9 @@ on: pull_request: branches: [main] +permissions: + contents: read + jobs: lint: runs-on: ubuntu-latest @@ -14,6 +17,7 @@ jobs: with: fetch-depth: 0 clean: true + persist-credentials: false - name: Set up Python uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 @@ -87,6 +91,7 @@ jobs: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: fetch-depth: 0 + persist-credentials: false - name: Set up Python uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 diff --git a/.github/workflows/test-litellm-matrix.yml b/.github/workflows/test-litellm-matrix.yml index 37c55fc4bf0..c354df565a2 100644 --- a/.github/workflows/test-litellm-matrix.yml +++ b/.github/workflows/test-litellm-matrix.yml @@ -4,6 +4,9 @@ on: pull_request: branches: [main] +permissions: + contents: read + # Cancel in-progress runs for the same PR concurrency: group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} @@ -118,6 +121,8 @@ jobs: steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Set up Python uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 diff --git a/.github/workflows/test-litellm-ui-build.yml b/.github/workflows/test-litellm-ui-build.yml index 4ac268e4e7a..6b0b3a413a6 100644 --- a/.github/workflows/test-litellm-ui-build.yml +++ b/.github/workflows/test-litellm-ui-build.yml @@ -17,6 +17,8 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Setup Node.js uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0 diff --git a/.github/workflows/test-litellm.yml b/.github/workflows/test-litellm.yml index 57051e9fd6c..0c040b3ebe7 100644 --- a/.github/workflows/test-litellm.yml +++ b/.github/workflows/test-litellm.yml @@ -8,6 +8,9 @@ on: # pull_request: # branches: [ main ] +permissions: + contents: read + jobs: test: runs-on: ubuntu-latest @@ -15,6 +18,8 @@ jobs: steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Thank You Message run: | diff --git a/.github/workflows/test-mcp.yml b/.github/workflows/test-mcp.yml index 3b8f2920aba..1b228ab76bb 100644 --- a/.github/workflows/test-mcp.yml +++ b/.github/workflows/test-mcp.yml @@ -4,6 +4,9 @@ on: pull_request: branches: [main] +permissions: + contents: read + jobs: test: runs-on: ubuntu-latest @@ -11,6 +14,8 @@ jobs: steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Thank You Message run: | diff --git a/.github/workflows/test-model-map.yaml b/.github/workflows/test-model-map.yaml index 2874a001e5f..429f9e1ce0a 100644 --- a/.github/workflows/test-model-map.yaml +++ b/.github/workflows/test-model-map.yaml @@ -4,11 +4,16 @@ on: pull_request: branches: [main] +permissions: + contents: read + jobs: validate-model-prices-json: runs-on: ubuntu-latest steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Validate model_prices_and_context_window.json run: | diff --git a/.github/workflows/test-proxy-e2e-azure-batches.yml b/.github/workflows/test-proxy-e2e-azure-batches.yml index b579b37de9c..d5130b07f13 100644 --- a/.github/workflows/test-proxy-e2e-azure-batches.yml +++ b/.github/workflows/test-proxy-e2e-azure-batches.yml @@ -9,6 +9,9 @@ concurrency: group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} cancel-in-progress: true +permissions: + contents: read + jobs: proxy_e2e_azure_batches_tests: runs-on: ubuntu-latest @@ -31,6 +34,8 @@ jobs: steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Set up Python uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 diff --git a/.github/workflows/test_server_root_path.yml b/.github/workflows/test_server_root_path.yml index 647d81849c6..47636ce8e92 100644 --- a/.github/workflows/test_server_root_path.yml +++ b/.github/workflows/test_server_root_path.yml @@ -18,6 +18,8 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Set up Docker Buildx uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12 From 06c84765442854286cf8bbc445c46a7243b836bc Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 10:49:13 +0530 Subject: [PATCH 016/117] feat(gemini): add gemini-3.1-flash-live-preview to model cost map Made-with: Cursor --- ...odel_prices_and_context_window_backup.json | 58 +++++++++++++++++++ model_prices_and_context_window.json | 58 +++++++++++++++++++ .../test_gemini_realtime_transformation.py | 15 +++++ 3 files changed, 131 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c53ee943c58..a063e3a4c6c 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -36830,6 +36830,34 @@ "supports_audio_input": true, "supports_audio_output": true }, + "gemini-3.1-flash-live-preview": { + "input_cost_per_audio_token": 1e-06, + "input_cost_per_token": 3e-07, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 2.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/models", + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_vision": true + }, "gemini/gemini-2.5-flash-native-audio-latest": { "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -36908,6 +36936,36 @@ "tpm": 250000, "rpm": 10 }, + "gemini/gemini-3.1-flash-live-preview": { + "input_cost_per_audio_token": 1e-06, + "input_cost_per_token": 3e-07, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 2.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/models", + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_vision": true, + "tpm": 250000, + "rpm": 10 + }, "gemini-2.5-flash-preview-tts": { "input_cost_per_token": 3e-07, "litellm_provider": "gemini", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c53ee943c58..a063e3a4c6c 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -36830,6 +36830,34 @@ "supports_audio_input": true, "supports_audio_output": true }, + "gemini-3.1-flash-live-preview": { + "input_cost_per_audio_token": 1e-06, + "input_cost_per_token": 3e-07, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 2.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/models", + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_vision": true + }, "gemini/gemini-2.5-flash-native-audio-latest": { "input_cost_per_audio_token": 1e-06, "input_cost_per_token": 3e-07, @@ -36908,6 +36936,36 @@ "tpm": 250000, "rpm": 10 }, + "gemini/gemini-3.1-flash-live-preview": { + "input_cost_per_audio_token": 1e-06, + "input_cost_per_token": 3e-07, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 2.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/models", + "supported_endpoints": [ + "/v1/realtime" + ], + "supported_modalities": [ + "text", + "image", + "audio", + "video" + ], + "supported_output_modalities": [ + "text", + "audio" + ], + "supports_audio_input": true, + "supports_audio_output": true, + "supports_function_calling": true, + "supports_vision": true, + "tpm": 250000, + "rpm": 10 + }, "gemini-2.5-flash-preview-tts": { "input_cost_per_token": 3e-07, "litellm_provider": "gemini", diff --git a/tests/test_litellm/llms/gemini/realtime/test_gemini_realtime_transformation.py b/tests/test_litellm/llms/gemini/realtime/test_gemini_realtime_transformation.py index 69741cdec6f..cc0a32d2ce6 100644 --- a/tests/test_litellm/llms/gemini/realtime/test_gemini_realtime_transformation.py +++ b/tests/test_litellm/llms/gemini/realtime/test_gemini_realtime_transformation.py @@ -10,6 +10,7 @@ sys.path.insert( 0, os.path.abspath("../../../../..") ) # Adds the parent directory to the system path +import litellm from litellm.llms.gemini.realtime.transformation import GeminiRealtimeConfig from litellm.types.llms.openai import OpenAIRealtimeStreamSessionEvents @@ -227,3 +228,17 @@ def test_gemini_realtime_transformation_generation_complete(): contains_audio_delta = True break assert contains_audio_delta, "Expected audio delta event" + + +def test_gemini_3_1_flash_live_preview_model_cost_map_entry(): + for key in ( + "gemini-3.1-flash-live-preview", + "gemini/gemini-3.1-flash-live-preview", + ): + assert key in litellm.model_cost + info = litellm.model_cost[key] + assert "/v1/realtime" in info.get("supported_endpoints", []) + assert info.get("max_input_tokens") == 131072 + assert info.get("max_output_tokens") == 65536 + assert "video" in info.get("supported_modalities", []) + assert info.get("supports_function_calling") is True From 226e18534e06b78a5e0555ff8d7e01043191ff8b Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 11:03:37 +0530 Subject: [PATCH 017/117] Add correct pricing --- ...odel_prices_and_context_window_backup.json | 26 ++++++++++++------- model_prices_and_context_window.json | 26 ++++++++++++------- 2 files changed, 34 insertions(+), 18 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index a063e3a4c6c..ff75c8d3e45 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -36831,15 +36831,18 @@ "supports_audio_output": true }, "gemini-3.1-flash-live-preview": { - "input_cost_per_audio_token": 1e-06, - "input_cost_per_token": 3e-07, + "input_cost_per_audio_token": 3e-06, + "input_cost_per_image_token": 1e-06, + "input_cost_per_token": 7.5e-07, + "input_cost_per_video_per_second": 3.3333333333333335e-05, "litellm_provider": "gemini", "max_input_tokens": 131072, "max_output_tokens": 65536, "max_tokens": 65536, "mode": "chat", - "output_cost_per_token": 2.5e-06, - "source": "https://ai.google.dev/gemini-api/docs/models", + "output_cost_per_audio_token": 1.2e-05, + "output_cost_per_token": 4.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_endpoints": [ "/v1/realtime" ], @@ -36856,7 +36859,8 @@ "supports_audio_input": true, "supports_audio_output": true, "supports_function_calling": true, - "supports_vision": true + "supports_vision": true, + "supports_web_search": true }, "gemini/gemini-2.5-flash-native-audio-latest": { "input_cost_per_audio_token": 1e-06, @@ -36937,15 +36941,18 @@ "rpm": 10 }, "gemini/gemini-3.1-flash-live-preview": { - "input_cost_per_audio_token": 1e-06, - "input_cost_per_token": 3e-07, + "input_cost_per_audio_token": 3e-06, + "input_cost_per_image_token": 1e-06, + "input_cost_per_token": 7.5e-07, + "input_cost_per_video_per_second": 3.3333333333333335e-05, "litellm_provider": "gemini", "max_input_tokens": 131072, "max_output_tokens": 65536, "max_tokens": 65536, "mode": "chat", - "output_cost_per_token": 2.5e-06, - "source": "https://ai.google.dev/gemini-api/docs/models", + "output_cost_per_audio_token": 1.2e-05, + "output_cost_per_token": 4.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_endpoints": [ "/v1/realtime" ], @@ -36963,6 +36970,7 @@ "supports_audio_output": true, "supports_function_calling": true, "supports_vision": true, + "supports_web_search": true, "tpm": 250000, "rpm": 10 }, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a063e3a4c6c..ff75c8d3e45 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -36831,15 +36831,18 @@ "supports_audio_output": true }, "gemini-3.1-flash-live-preview": { - "input_cost_per_audio_token": 1e-06, - "input_cost_per_token": 3e-07, + "input_cost_per_audio_token": 3e-06, + "input_cost_per_image_token": 1e-06, + "input_cost_per_token": 7.5e-07, + "input_cost_per_video_per_second": 3.3333333333333335e-05, "litellm_provider": "gemini", "max_input_tokens": 131072, "max_output_tokens": 65536, "max_tokens": 65536, "mode": "chat", - "output_cost_per_token": 2.5e-06, - "source": "https://ai.google.dev/gemini-api/docs/models", + "output_cost_per_audio_token": 1.2e-05, + "output_cost_per_token": 4.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_endpoints": [ "/v1/realtime" ], @@ -36856,7 +36859,8 @@ "supports_audio_input": true, "supports_audio_output": true, "supports_function_calling": true, - "supports_vision": true + "supports_vision": true, + "supports_web_search": true }, "gemini/gemini-2.5-flash-native-audio-latest": { "input_cost_per_audio_token": 1e-06, @@ -36937,15 +36941,18 @@ "rpm": 10 }, "gemini/gemini-3.1-flash-live-preview": { - "input_cost_per_audio_token": 1e-06, - "input_cost_per_token": 3e-07, + "input_cost_per_audio_token": 3e-06, + "input_cost_per_image_token": 1e-06, + "input_cost_per_token": 7.5e-07, + "input_cost_per_video_per_second": 3.3333333333333335e-05, "litellm_provider": "gemini", "max_input_tokens": 131072, "max_output_tokens": 65536, "max_tokens": 65536, "mode": "chat", - "output_cost_per_token": 2.5e-06, - "source": "https://ai.google.dev/gemini-api/docs/models", + "output_cost_per_audio_token": 1.2e-05, + "output_cost_per_token": 4.5e-06, + "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_endpoints": [ "/v1/realtime" ], @@ -36963,6 +36970,7 @@ "supports_audio_output": true, "supports_function_calling": true, "supports_vision": true, + "supports_web_search": true, "tpm": 250000, "rpm": 10 }, From 265f2eb09030a6bde25c86e94860530604c18167 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 12:11:04 +0530 Subject: [PATCH 018/117] feat(fine-tuning): fix Azure OpenAI fine-tuning job creation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Default trainingType=1 for Azure when omitted to avoid misleading "base model does not support fine-tuning" error - Normalize Azure FineTuningJob responses (pending→queued, null fields→defaults) to match OpenAI schema - Add pending status support to OpenAIFileObject for Azure file uploads - Add test coverage for trainingType default and response normalization Made-with: Cursor --- litellm/fine_tuning/main.py | 9 ++++ litellm/llms/openai/fine_tuning/handler.py | 43 +++++++++++++++--- tests/batches_tests/test_fine_tuning_api.py | 28 ++++++++++++ .../types/llms/test_types_llms_openai.py | 45 +++++++++++++------ 4 files changed, 105 insertions(+), 20 deletions(-) diff --git a/litellm/fine_tuning/main.py b/litellm/fine_tuning/main.py index 08373cda782..a611488b179 100644 --- a/litellm/fine_tuning/main.py +++ b/litellm/fine_tuning/main.py @@ -245,6 +245,15 @@ def create_fine_tuning_job( ) # Azure OpenAI elif custom_llm_provider == "azure": + # Azure requires trainingType (e.g. 1 = supervised). Omitting it yields a misleading + # "The specified base model does not support fine-tuning" error from the service. + if kwargs.get("trainingType") is None: + _eb = ( + kwargs.get("extra_body") or optional_params.get("extra_body") or {} + ) + if not (isinstance(_eb, dict) and _eb.get("trainingType") is not None): + kwargs["trainingType"] = 1 + api_base = optional_params.api_base or litellm.api_base or get_secret_str("AZURE_API_BASE") # type: ignore api_version = ( diff --git a/litellm/llms/openai/fine_tuning/handler.py b/litellm/llms/openai/fine_tuning/handler.py index 9804ff3539e..581ca1eea8c 100644 --- a/litellm/llms/openai/fine_tuning/handler.py +++ b/litellm/llms/openai/fine_tuning/handler.py @@ -1,4 +1,4 @@ -from typing import Any, Coroutine, Optional, Union, cast +from typing import Any, Coroutine, Dict, Optional, Union, cast import httpx from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI @@ -7,6 +7,35 @@ from litellm._logging import verbose_logger from litellm.types.utils import LiteLLMFineTuningJob +def _normalize_fine_tuning_job_dict(data: Dict[str, Any]) -> Dict[str, Any]: + """ + Normalize Azure OpenAI FineTuningJob response to match OpenAI schema. + + Azure differences: + - organization_id: null → "" + - result_files: null → [] + - status: "pending" → "queued" + """ + normalized = data.copy() + + if normalized.get("organization_id") is None: + normalized["organization_id"] = "" + + if normalized.get("result_files") is None: + normalized["result_files"] = [] + + if normalized.get("status") == "pending": + normalized["status"] = "queued" + + return normalized + + +def _litellm_fine_tuning_job_from_response(response: Any) -> LiteLLMFineTuningJob: + return LiteLLMFineTuningJob( + **_normalize_fine_tuning_job_dict(response.model_dump()) + ) + + class OpenAIFineTuningAPI: """ OpenAI methods to support for batches @@ -60,7 +89,7 @@ class OpenAIFineTuningAPI: **create_fine_tuning_job_data ) - return LiteLLMFineTuningJob(**response.model_dump()) + return _litellm_fine_tuning_job_from_response(response) def create_fine_tuning_job( self, @@ -108,7 +137,7 @@ class OpenAIFineTuningAPI: response = cast(OpenAI, openai_client).fine_tuning.jobs.create( **create_fine_tuning_job_data ) - return LiteLLMFineTuningJob(**response.model_dump()) + return _litellm_fine_tuning_job_from_response(response) async def acancel_fine_tuning_job( self, @@ -118,7 +147,7 @@ class OpenAIFineTuningAPI: response = await openai_client.fine_tuning.jobs.cancel( fine_tuning_job_id=fine_tuning_job_id ) - return LiteLLMFineTuningJob(**response.model_dump()) + return _litellm_fine_tuning_job_from_response(response) def cancel_fine_tuning_job( self, @@ -164,7 +193,7 @@ class OpenAIFineTuningAPI: response = cast(OpenAI, openai_client).fine_tuning.jobs.cancel( fine_tuning_job_id=fine_tuning_job_id ) - return LiteLLMFineTuningJob(**response.model_dump()) + return _litellm_fine_tuning_job_from_response(response) async def alist_fine_tuning_jobs( self, @@ -229,7 +258,7 @@ class OpenAIFineTuningAPI: response = await openai_client.fine_tuning.jobs.retrieve( fine_tuning_job_id=fine_tuning_job_id ) - return LiteLLMFineTuningJob(**response.model_dump()) + return _litellm_fine_tuning_job_from_response(response) def retrieve_fine_tuning_job( self, @@ -275,4 +304,4 @@ class OpenAIFineTuningAPI: response = cast(OpenAI, openai_client).fine_tuning.jobs.retrieve( fine_tuning_job_id=fine_tuning_job_id ) - return LiteLLMFineTuningJob(**response.model_dump()) + return _litellm_fine_tuning_job_from_response(response) diff --git a/tests/batches_tests/test_fine_tuning_api.py b/tests/batches_tests/test_fine_tuning_api.py index 7e238173480..b7ff8c7fae7 100644 --- a/tests/batches_tests/test_fine_tuning_api.py +++ b/tests/batches_tests/test_fine_tuning_api.py @@ -180,6 +180,34 @@ async def test_azure_create_fine_tune_jobs_async(): pass +def test_azure_trainingtype_defaults_to_one(): + """ + Azure requires trainingType in extra_body. When omitted, LiteLLM defaults it to 1. + """ + from unittest.mock import MagicMock, patch + from litellm.fine_tuning.main import create_fine_tuning_job + + with patch( + "litellm.fine_tuning.main.azure_fine_tuning_apis_instance.create_fine_tuning_job" + ) as mock_create: + mock_create.return_value = MagicMock( + id="ftjob-test", status="queued", model="gpt-4o-mini" + ) + + create_fine_tuning_job( + model="gpt-4o-mini", + training_file="file-test", + custom_llm_provider="azure", + api_base="https://test.openai.azure.com", + api_key="test-key", + ) + + call_kwargs = mock_create.call_args[1] + create_data = call_kwargs["create_fine_tuning_job_data"] + assert "extra_body" in create_data + assert create_data["extra_body"].get("trainingType") == 1 + + @pytest.mark.asyncio() async def test_create_vertex_fine_tune_jobs_mocked(): load_vertex_ai_credentials() diff --git a/tests/test_litellm/types/llms/test_types_llms_openai.py b/tests/test_litellm/types/llms/test_types_llms_openai.py index 94221bd0efc..384d0dd73d7 100644 --- a/tests/test_litellm/types/llms/test_types_llms_openai.py +++ b/tests/test_litellm/types/llms/test_types_llms_openai.py @@ -219,11 +219,13 @@ class TestAssistantMessageImageUrlContent: # convert to list to consume it — this must not raise ValidationError. content_blocks = list(raw_content) if raw_content is not None else [] - assert len(content_blocks) == 2, ( - f"Expected 2 content blocks (text + image_url), got {len(content_blocks)}: {content_blocks}" - ) + assert ( + len(content_blocks) == 2 + ), f"Expected 2 content blocks (text + image_url), got {len(content_blocks)}: {content_blocks}" types = [b.get("type") for b in content_blocks if isinstance(b, dict)] - assert "image_url" in types, f"image_url block was silently dropped; blocks: {content_blocks}" + assert ( + "image_url" in types + ), f"image_url block was silently dropped; blocks: {content_blocks}" def test_assistant_message_image_url_preserved_in_all_message_values(self): """ @@ -255,14 +257,16 @@ class TestAssistantMessageImageUrlContent: assert assistant is not None, "Assistant message missing after serialisation" content = assistant.get("content", []) - assert isinstance(content, list), f"content should be a list, got {type(content)}" - assert len(content) == 2, ( - f"Expected 2 content blocks (text + image_url), got {len(content)}: {content}" - ) + assert isinstance( + content, list + ), f"content should be a list, got {type(content)}" + assert ( + len(content) == 2 + ), f"Expected 2 content blocks (text + image_url), got {len(content)}: {content}" types = [b.get("type") for b in content if isinstance(b, dict)] - assert "image_url" in types, ( - f"image_url block was silently dropped during AllMessageValues serialisation; blocks: {content}" - ) + assert ( + "image_url" in types + ), f"image_url block was silently dropped during AllMessageValues serialisation; blocks: {content}" class TestResponsesAPIReasoningNullFields: @@ -379,10 +383,14 @@ class TestResponsesAPIReasoningNullFields: ) dumped = response.model_dump() reasoning = [ - o for o in dumped["output"] if isinstance(o, dict) and o.get("type") == "reasoning" + o + for o in dumped["output"] + if isinstance(o, dict) and o.get("type") == "reasoning" ][0] message = [ - o for o in dumped["output"] if isinstance(o, dict) and o.get("type") == "message" + o + for o in dumped["output"] + if isinstance(o, dict) and o.get("type") == "message" ][0] assert "status" not in reasoning assert "content" not in reasoning @@ -410,3 +418,14 @@ class TestResponsesAPIReasoningNullFields: assert dumped["error"] is None assert "instructions" in dumped assert dumped["instructions"] is None + + +def test_normalize_fine_tuning_job_dict_maps_azure_pending(): + from litellm.llms.openai.fine_tuning.handler import _normalize_fine_tuning_job_dict + + out = _normalize_fine_tuning_job_dict( + {"organization_id": None, "result_files": None, "status": "pending"} + ) + assert out["organization_id"] == "" + assert out["result_files"] == [] + assert out["status"] == "queued" From 2484d202f827e838b7d2550ae98d92fc5ef9f402 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 12:26:14 +0530 Subject: [PATCH 019/117] address greptile review feedback (greploop iteration 1) - Move trainingType injection to AzureOpenAIFineTuningAPI handler - Guard normalization with is_azure flag to only apply to Azure responses - Override acreate_fine_tuning_job in Azure handler to use is_azure=True - Update test to directly test _ensure_training_type method - Add test for OpenAI unchanged behavior Made-with: Cursor --- litellm/fine_tuning/main.py | 9 -- litellm/llms/azure/fine_tuning/handler.py | 85 ++++++++++++++++++- litellm/llms/openai/fine_tuning/handler.py | 13 ++- tests/batches_tests/test_fine_tuning_api.py | 27 ++---- .../types/llms/test_types_llms_openai.py | 11 ++- 5 files changed, 110 insertions(+), 35 deletions(-) diff --git a/litellm/fine_tuning/main.py b/litellm/fine_tuning/main.py index a611488b179..08373cda782 100644 --- a/litellm/fine_tuning/main.py +++ b/litellm/fine_tuning/main.py @@ -245,15 +245,6 @@ def create_fine_tuning_job( ) # Azure OpenAI elif custom_llm_provider == "azure": - # Azure requires trainingType (e.g. 1 = supervised). Omitting it yields a misleading - # "The specified base model does not support fine-tuning" error from the service. - if kwargs.get("trainingType") is None: - _eb = ( - kwargs.get("extra_body") or optional_params.get("extra_body") or {} - ) - if not (isinstance(_eb, dict) and _eb.get("trainingType") is not None): - kwargs["trainingType"] = 1 - api_base = optional_params.api_base or litellm.api_base or get_secret_str("AZURE_API_BASE") # type: ignore api_version = ( diff --git a/litellm/llms/azure/fine_tuning/handler.py b/litellm/llms/azure/fine_tuning/handler.py index 429b8349896..2415c7d78ed 100644 --- a/litellm/llms/azure/fine_tuning/handler.py +++ b/litellm/llms/azure/fine_tuning/handler.py @@ -1,10 +1,15 @@ -from typing import Optional, Union +from typing import Any, Coroutine, Dict, Optional, Union, cast import httpx from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI +from litellm._logging import verbose_logger from litellm.llms.azure.common_utils import BaseAzureLLM -from litellm.llms.openai.fine_tuning.handler import OpenAIFineTuningAPI +from litellm.llms.openai.fine_tuning.handler import ( + OpenAIFineTuningAPI, + _litellm_fine_tuning_job_from_response, +) +from litellm.types.utils import LiteLLMFineTuningJob class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM): @@ -12,6 +17,82 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM): AzureOpenAI methods to support fine tuning, inherits from OpenAIFineTuningAPI. """ + @staticmethod + def _ensure_training_type(create_fine_tuning_job_data: Dict[str, Any]) -> None: + """ + Azure requires trainingType in extra_body. Default to 1 (supervised) if omitted. + """ + extra_body = create_fine_tuning_job_data.get("extra_body") or {} + if not isinstance(extra_body, dict): + extra_body = {} + if extra_body.get("trainingType") is None: + extra_body["trainingType"] = 1 + create_fine_tuning_job_data["extra_body"] = extra_body + verbose_logger.debug( + "Azure fine-tuning: defaulting trainingType=1 (supervised)" + ) + + async def acreate_fine_tuning_job( + self, + create_fine_tuning_job_data: dict, + openai_client: Union[AsyncOpenAI, AsyncAzureOpenAI], + ) -> LiteLLMFineTuningJob: + response = await openai_client.fine_tuning.jobs.create( + **create_fine_tuning_job_data + ) + return _litellm_fine_tuning_job_from_response(response, is_azure=True) + + def create_fine_tuning_job( + self, + _is_async: bool, + create_fine_tuning_job_data: dict, + api_key: Optional[str], + api_base: Optional[str], + api_version: Optional[str], + timeout: Union[float, httpx.Timeout], + max_retries: Optional[int], + organization: Optional[str], + client: Optional[ + Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI] + ] = None, + ) -> Union[LiteLLMFineTuningJob, Coroutine[Any, Any, LiteLLMFineTuningJob]]: + self._ensure_training_type(create_fine_tuning_job_data) + + openai_client: Optional[ + Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI] + ] = self.get_openai_client( + api_key=api_key, + api_base=api_base, + timeout=timeout, + max_retries=max_retries, + organization=organization, + client=client, + _is_async=_is_async, + api_version=api_version, + ) + if openai_client is None: + raise ValueError( + "Azure OpenAI client is not initialized. Make sure api_key is passed or AZURE_API_KEY is set in the environment." + ) + + if _is_async is True: + if not isinstance(openai_client, (AsyncOpenAI, AsyncAzureOpenAI)): + raise ValueError( + "OpenAI client is not an instance of AsyncOpenAI. Make sure you passed an AsyncOpenAI client." + ) + return self.acreate_fine_tuning_job( + create_fine_tuning_job_data=create_fine_tuning_job_data, + openai_client=openai_client, + ) + + verbose_logger.debug( + "creating fine tuning job, args= %s", create_fine_tuning_job_data + ) + response = cast(OpenAI, openai_client).fine_tuning.jobs.create( + **create_fine_tuning_job_data + ) + return _litellm_fine_tuning_job_from_response(response, is_azure=True) + def get_openai_client( self, api_key: Optional[str], diff --git a/litellm/llms/openai/fine_tuning/handler.py b/litellm/llms/openai/fine_tuning/handler.py index 581ca1eea8c..2bb39aeaea4 100644 --- a/litellm/llms/openai/fine_tuning/handler.py +++ b/litellm/llms/openai/fine_tuning/handler.py @@ -7,7 +7,9 @@ from litellm._logging import verbose_logger from litellm.types.utils import LiteLLMFineTuningJob -def _normalize_fine_tuning_job_dict(data: Dict[str, Any]) -> Dict[str, Any]: +def _normalize_fine_tuning_job_dict( + data: Dict[str, Any], is_azure: bool = False +) -> Dict[str, Any]: """ Normalize Azure OpenAI FineTuningJob response to match OpenAI schema. @@ -16,6 +18,9 @@ def _normalize_fine_tuning_job_dict(data: Dict[str, Any]) -> Dict[str, Any]: - result_files: null → [] - status: "pending" → "queued" """ + if not is_azure: + return data + normalized = data.copy() if normalized.get("organization_id") is None: @@ -30,9 +35,11 @@ def _normalize_fine_tuning_job_dict(data: Dict[str, Any]) -> Dict[str, Any]: return normalized -def _litellm_fine_tuning_job_from_response(response: Any) -> LiteLLMFineTuningJob: +def _litellm_fine_tuning_job_from_response( + response: Any, is_azure: bool = False +) -> LiteLLMFineTuningJob: return LiteLLMFineTuningJob( - **_normalize_fine_tuning_job_dict(response.model_dump()) + **_normalize_fine_tuning_job_dict(response.model_dump(), is_azure=is_azure) ) diff --git a/tests/batches_tests/test_fine_tuning_api.py b/tests/batches_tests/test_fine_tuning_api.py index b7ff8c7fae7..20867234e53 100644 --- a/tests/batches_tests/test_fine_tuning_api.py +++ b/tests/batches_tests/test_fine_tuning_api.py @@ -182,30 +182,17 @@ async def test_azure_create_fine_tune_jobs_async(): def test_azure_trainingtype_defaults_to_one(): """ - Azure requires trainingType in extra_body. When omitted, LiteLLM defaults it to 1. + Azure requires trainingType in extra_body. When omitted, AzureOpenAIFineTuningAPI defaults it to 1. """ - from unittest.mock import MagicMock, patch - from litellm.fine_tuning.main import create_fine_tuning_job + from litellm.llms.azure.fine_tuning.handler import AzureOpenAIFineTuningAPI - with patch( - "litellm.fine_tuning.main.azure_fine_tuning_apis_instance.create_fine_tuning_job" - ) as mock_create: - mock_create.return_value = MagicMock( - id="ftjob-test", status="queued", model="gpt-4o-mini" - ) + handler = AzureOpenAIFineTuningAPI() + create_data = {"model": "gpt-4o-mini", "training_file": "file-test"} - create_fine_tuning_job( - model="gpt-4o-mini", - training_file="file-test", - custom_llm_provider="azure", - api_base="https://test.openai.azure.com", - api_key="test-key", - ) + handler._ensure_training_type(create_data) - call_kwargs = mock_create.call_args[1] - create_data = call_kwargs["create_fine_tuning_job_data"] - assert "extra_body" in create_data - assert create_data["extra_body"].get("trainingType") == 1 + assert "extra_body" in create_data + assert create_data["extra_body"]["trainingType"] == 1 @pytest.mark.asyncio() diff --git a/tests/test_litellm/types/llms/test_types_llms_openai.py b/tests/test_litellm/types/llms/test_types_llms_openai.py index 384d0dd73d7..323eb5a9424 100644 --- a/tests/test_litellm/types/llms/test_types_llms_openai.py +++ b/tests/test_litellm/types/llms/test_types_llms_openai.py @@ -424,8 +424,17 @@ def test_normalize_fine_tuning_job_dict_maps_azure_pending(): from litellm.llms.openai.fine_tuning.handler import _normalize_fine_tuning_job_dict out = _normalize_fine_tuning_job_dict( - {"organization_id": None, "result_files": None, "status": "pending"} + {"organization_id": None, "result_files": None, "status": "pending"}, + is_azure=True, ) assert out["organization_id"] == "" assert out["result_files"] == [] assert out["status"] == "queued" + + +def test_normalize_fine_tuning_job_dict_openai_unchanged(): + from litellm.llms.openai.fine_tuning.handler import _normalize_fine_tuning_job_dict + + data = {"organization_id": None, "result_files": None, "status": "pending"} + out = _normalize_fine_tuning_job_dict(data, is_azure=False) + assert out is data From a9c7b17bfa47346dcf036e79bfb581551f09f96f Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 12:38:33 +0530 Subject: [PATCH 020/117] address greptile review feedback (greploop iteration 2) - Call _ensure_training_type in acreate_fine_tuning_job async override Made-with: Cursor --- litellm/llms/azure/fine_tuning/handler.py | 1 + 1 file changed, 1 insertion(+) diff --git a/litellm/llms/azure/fine_tuning/handler.py b/litellm/llms/azure/fine_tuning/handler.py index 2415c7d78ed..3d82172c585 100644 --- a/litellm/llms/azure/fine_tuning/handler.py +++ b/litellm/llms/azure/fine_tuning/handler.py @@ -37,6 +37,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM): create_fine_tuning_job_data: dict, openai_client: Union[AsyncOpenAI, AsyncAzureOpenAI], ) -> LiteLLMFineTuningJob: + self._ensure_training_type(create_fine_tuning_job_data) response = await openai_client.fine_tuning.jobs.create( **create_fine_tuning_job_data ) From d4d91684cf77df72d9d45bfd20d83ae3054c8796 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 12:44:55 +0530 Subject: [PATCH 021/117] address greptile review feedback (greploop iteration 3) - Remove redundant _ensure_training_type call from acreate_fine_tuning_job - Use explicit _AZURE_STATUS_MAP for status normalization Made-with: Cursor --- litellm/llms/azure/fine_tuning/handler.py | 1 - litellm/llms/openai/fine_tuning/handler.py | 11 ++++++++--- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/litellm/llms/azure/fine_tuning/handler.py b/litellm/llms/azure/fine_tuning/handler.py index 3d82172c585..2415c7d78ed 100644 --- a/litellm/llms/azure/fine_tuning/handler.py +++ b/litellm/llms/azure/fine_tuning/handler.py @@ -37,7 +37,6 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM): create_fine_tuning_job_data: dict, openai_client: Union[AsyncOpenAI, AsyncAzureOpenAI], ) -> LiteLLMFineTuningJob: - self._ensure_training_type(create_fine_tuning_job_data) response = await openai_client.fine_tuning.jobs.create( **create_fine_tuning_job_data ) diff --git a/litellm/llms/openai/fine_tuning/handler.py b/litellm/llms/openai/fine_tuning/handler.py index 2bb39aeaea4..a9fbd88d2a8 100644 --- a/litellm/llms/openai/fine_tuning/handler.py +++ b/litellm/llms/openai/fine_tuning/handler.py @@ -6,6 +6,10 @@ from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI from litellm._logging import verbose_logger from litellm.types.utils import LiteLLMFineTuningJob +_AZURE_STATUS_MAP = { + "pending": "queued", +} + def _normalize_fine_tuning_job_dict( data: Dict[str, Any], is_azure: bool = False @@ -16,7 +20,7 @@ def _normalize_fine_tuning_job_dict( Azure differences: - organization_id: null → "" - result_files: null → [] - - status: "pending" → "queued" + - status: mapped via _AZURE_STATUS_MAP """ if not is_azure: return data @@ -29,8 +33,9 @@ def _normalize_fine_tuning_job_dict( if normalized.get("result_files") is None: normalized["result_files"] = [] - if normalized.get("status") == "pending": - normalized["status"] = "queued" + status = normalized.get("status") + if status in _AZURE_STATUS_MAP: + normalized["status"] = _AZURE_STATUS_MAP[status] return normalized From 528bac5a2734693e2a8877a878f77cb36ad559ad Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 12:54:33 +0530 Subject: [PATCH 022/117] feat(fine-tuning): address greptile review feedback (greploop iteration 4) - Add cancel/retrieve overrides in AzureOpenAIFineTuningAPI to normalize responses - Expand _AZURE_STATUS_MAP to handle all known Azure statuses - Add "pending" to OpenAIFileObject.status allowed values - Fix async test mock to return awaitable LiteLLMFineTuningJob - Add test_openai_file_object_accepts_pending_status Made-with: Cursor --- litellm/llms/azure/fine_tuning/handler.py | 112 ++++++++++++++++++ litellm/llms/openai/fine_tuning/handler.py | 6 + litellm/types/llms/openai.py | 6 +- tests/batches_tests/test_fine_tuning_api.py | 9 +- .../types/llms/test_types_llms_openai.py | 15 +++ 5 files changed, 142 insertions(+), 6 deletions(-) diff --git a/litellm/llms/azure/fine_tuning/handler.py b/litellm/llms/azure/fine_tuning/handler.py index 2415c7d78ed..7e225a84454 100644 --- a/litellm/llms/azure/fine_tuning/handler.py +++ b/litellm/llms/azure/fine_tuning/handler.py @@ -42,6 +42,26 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM): ) return _litellm_fine_tuning_job_from_response(response, is_azure=True) + async def acancel_fine_tuning_job( + self, + fine_tuning_job_id: str, + openai_client: Union[AsyncOpenAI, AsyncAzureOpenAI], + ) -> LiteLLMFineTuningJob: + response = await openai_client.fine_tuning.jobs.cancel( + fine_tuning_job_id=fine_tuning_job_id + ) + return _litellm_fine_tuning_job_from_response(response, is_azure=True) + + async def aretrieve_fine_tuning_job( + self, + fine_tuning_job_id: str, + openai_client: Union[AsyncOpenAI, AsyncAzureOpenAI], + ) -> LiteLLMFineTuningJob: + response = await openai_client.fine_tuning.jobs.retrieve( + fine_tuning_job_id=fine_tuning_job_id + ) + return _litellm_fine_tuning_job_from_response(response, is_azure=True) + def create_fine_tuning_job( self, _is_async: bool, @@ -93,6 +113,98 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM): ) return _litellm_fine_tuning_job_from_response(response, is_azure=True) + def cancel_fine_tuning_job( + self, + _is_async: bool, + fine_tuning_job_id: str, + api_key: Optional[str], + api_base: Optional[str], + api_version: Optional[str], + timeout: Union[float, httpx.Timeout], + max_retries: Optional[int], + organization: Optional[str], + client: Optional[ + Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI] + ] = None, + ) -> Union[LiteLLMFineTuningJob, Coroutine[Any, Any, LiteLLMFineTuningJob]]: + openai_client: Optional[ + Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI] + ] = self.get_openai_client( + api_key=api_key, + api_base=api_base, + timeout=timeout, + max_retries=max_retries, + organization=organization, + client=client, + _is_async=_is_async, + api_version=api_version, + ) + if openai_client is None: + raise ValueError( + "Azure OpenAI client is not initialized. Make sure api_key is passed or AZURE_API_KEY is set in the environment." + ) + + if _is_async is True: + if not isinstance(openai_client, (AsyncOpenAI, AsyncAzureOpenAI)): + raise ValueError( + "OpenAI client is not an instance of AsyncOpenAI. Make sure you passed an AsyncOpenAI client." + ) + return self.acancel_fine_tuning_job( + fine_tuning_job_id=fine_tuning_job_id, + openai_client=openai_client, + ) + + response = cast(OpenAI, openai_client).fine_tuning.jobs.cancel( + fine_tuning_job_id=fine_tuning_job_id + ) + return _litellm_fine_tuning_job_from_response(response, is_azure=True) + + def retrieve_fine_tuning_job( + self, + _is_async: bool, + fine_tuning_job_id: str, + api_key: Optional[str], + api_base: Optional[str], + api_version: Optional[str], + timeout: Union[float, httpx.Timeout], + max_retries: Optional[int], + organization: Optional[str], + client: Optional[ + Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI] + ] = None, + ) -> Union[LiteLLMFineTuningJob, Coroutine[Any, Any, LiteLLMFineTuningJob]]: + openai_client: Optional[ + Union[OpenAI, AsyncOpenAI, AzureOpenAI, AsyncAzureOpenAI] + ] = self.get_openai_client( + api_key=api_key, + api_base=api_base, + timeout=timeout, + max_retries=max_retries, + organization=organization, + client=client, + _is_async=_is_async, + api_version=api_version, + ) + if openai_client is None: + raise ValueError( + "Azure OpenAI client is not initialized. Make sure api_key is passed or AZURE_API_KEY is set in the environment." + ) + + if _is_async is True: + if not isinstance(openai_client, (AsyncOpenAI, AsyncAzureOpenAI)): + raise ValueError( + "OpenAI client is not an instance of AsyncOpenAI. Make sure you passed an AsyncOpenAI client." + ) + return self.aretrieve_fine_tuning_job( + fine_tuning_job_id=fine_tuning_job_id, + openai_client=openai_client, + ) + + response = cast(OpenAI, openai_client).fine_tuning.jobs.retrieve( + fine_tuning_job_id=fine_tuning_job_id + ) + return _litellm_fine_tuning_job_from_response(response, is_azure=True) + def get_openai_client( self, api_key: Optional[str], diff --git a/litellm/llms/openai/fine_tuning/handler.py b/litellm/llms/openai/fine_tuning/handler.py index a9fbd88d2a8..6800fe81d65 100644 --- a/litellm/llms/openai/fine_tuning/handler.py +++ b/litellm/llms/openai/fine_tuning/handler.py @@ -8,6 +8,12 @@ from litellm.types.utils import LiteLLMFineTuningJob _AZURE_STATUS_MAP = { "pending": "queued", + "notRunning": "queued", + "running": "running", + "succeeded": "succeeded", + "failed": "failed", + "canceled": "cancelled", + "canceling": "cancelled", } diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index 5a80b40d61f..b9c8030c877 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -315,11 +315,11 @@ class OpenAIFileObject(BaseModel): `fine-tune`, `fine-tune-results`, `vision`, and `user_data`. """ - status: Optional[Literal["uploaded", "processed", "error"]] = None + status: Optional[Literal["uploaded", "processed", "error", "pending"]] = None """Deprecated. - The current status of the file, which can be either `uploaded`, `processed`, or - `error`. + The current status of the file, which can be either `uploaded`, `processed`, + `error`, or `pending` (Azure may return `pending` immediately after upload). """ expires_at: Optional[int] = None diff --git a/tests/batches_tests/test_fine_tuning_api.py b/tests/batches_tests/test_fine_tuning_api.py index 20867234e53..3ab15306fcb 100644 --- a/tests/batches_tests/test_fine_tuning_api.py +++ b/tests/batches_tests/test_fine_tuning_api.py @@ -616,11 +616,11 @@ async def test_mock_openai_retrieve_fine_tune_job(): @pytest.mark.asyncio async def test_mock_azure_create_fine_tune_job_with_azure_specific_params(): """Test that Azure-specific parameters are passed through extra_body""" - from openai import AsyncAzureOpenAI from openai.types.fine_tuning.fine_tuning_job import FineTuningJob from openai.types.fine_tuning.fine_tuning_job import Hyperparameters as OAIHyperparameters + from litellm.types.utils import LiteLLMFineTuningJob - mock_response = FineTuningJob( + mock_response = LiteLLMFineTuningJob( id="ft-azure-123", model="gpt-4.1-mini-2025-04-14", created_at=1677610602, @@ -634,8 +634,11 @@ async def test_mock_azure_create_fine_tune_job_with_azure_specific_params(): result_files=[], ) + async def mock_async_create(*args, **kwargs): + return mock_response + with patch("litellm.llms.azure.fine_tuning.handler.AzureOpenAIFineTuningAPI.create_fine_tuning_job") as mock_create: - mock_create.return_value = mock_response + mock_create.return_value = mock_async_create() response = await litellm.acreate_fine_tuning_job( model="gpt-4.1-mini-2025-04-14", diff --git a/tests/test_litellm/types/llms/test_types_llms_openai.py b/tests/test_litellm/types/llms/test_types_llms_openai.py index 323eb5a9424..569743269a5 100644 --- a/tests/test_litellm/types/llms/test_types_llms_openai.py +++ b/tests/test_litellm/types/llms/test_types_llms_openai.py @@ -438,3 +438,18 @@ def test_normalize_fine_tuning_job_dict_openai_unchanged(): data = {"organization_id": None, "result_files": None, "status": "pending"} out = _normalize_fine_tuning_job_dict(data, is_azure=False) assert out is data + + +def test_openai_file_object_accepts_pending_status(): + from litellm.types.llms.openai import OpenAIFileObject + + file_obj = OpenAIFileObject( + id="file-123", + bytes=1024, + created_at=1677610602, + filename="train.jsonl", + object="file", + purpose="fine-tune", + status="pending", + ) + assert file_obj.status == "pending" From e635cee712f7154c472f03c7ae4149ad167fc0d1 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 13:03:10 +0530 Subject: [PATCH 023/117] feat(fine-tuning): address greptile review feedback (greploop iteration 5) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Remove unused FineTuningJob import from test - Document "canceling" → "cancelled" mapping in _AZURE_STATUS_MAP Made-with: Cursor --- litellm/llms/openai/fine_tuning/handler.py | 2 ++ tests/batches_tests/test_fine_tuning_api.py | 1 - 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/litellm/llms/openai/fine_tuning/handler.py b/litellm/llms/openai/fine_tuning/handler.py index 6800fe81d65..c065325254e 100644 --- a/litellm/llms/openai/fine_tuning/handler.py +++ b/litellm/llms/openai/fine_tuning/handler.py @@ -15,6 +15,8 @@ _AZURE_STATUS_MAP = { "canceled": "cancelled", "canceling": "cancelled", } +# Note: Azure's "canceling" (in-progress) is mapped to "cancelled" (terminal) +# because LiteLLMFineTuningJob schema has no intermediate cancellation state. def _normalize_fine_tuning_job_dict( diff --git a/tests/batches_tests/test_fine_tuning_api.py b/tests/batches_tests/test_fine_tuning_api.py index 3ab15306fcb..7e220612ac6 100644 --- a/tests/batches_tests/test_fine_tuning_api.py +++ b/tests/batches_tests/test_fine_tuning_api.py @@ -616,7 +616,6 @@ async def test_mock_openai_retrieve_fine_tune_job(): @pytest.mark.asyncio async def test_mock_azure_create_fine_tune_job_with_azure_specific_params(): """Test that Azure-specific parameters are passed through extra_body""" - from openai.types.fine_tuning.fine_tuning_job import FineTuningJob from openai.types.fine_tuning.fine_tuning_job import Hyperparameters as OAIHyperparameters from litellm.types.utils import LiteLLMFineTuningJob From 5534b40ab3968a6d04da78bb64809b059612c7bc Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 15:31:43 +0530 Subject: [PATCH 024/117] fix(team-routing): use deterministic team model group names Use a deterministic internal model_name for team-scoped deployments so sibling deployments with the same public model share a routing group. This makes team alias writes idempotent and preserves multi-deployment failover/load balancing behavior. Made-with: Cursor --- .../model_management_endpoints.py | 26 ++- .../test_model_management_endpoints.py | 199 ++++++++++++++---- 2 files changed, 172 insertions(+), 53 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 44d41097833..39383d4ee20 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -13,13 +13,13 @@ model/{model_id}/update - PATCH endpoint for model update. import asyncio import datetime import json -from litellm._uuid import uuid from typing import Dict, List, Literal, Optional, Tuple, Union, cast from fastapi import APIRouter, Depends, HTTPException, Request, status from pydantic import BaseModel, ConfigDict, Field from litellm._logging import verbose_proxy_logger +from litellm._uuid import uuid from litellm.constants import LITELLM_PROXY_ADMIN_NAME from litellm.proxy._types import ( CommonProxyErrors, @@ -322,9 +322,13 @@ async def _add_team_model_to_db( """ If 'team_id' is provided, - - generate a unique 'model_name' for the model (e.g. 'model_name_{team_id}_{uuid}) - - store the model in the db with the unique 'model_name' - - store a team model alias mapping {"model_name": "model_name_{team_id}_{uuid}"} + - generate a deterministic 'model_name' for the model (e.g. 'model_name_{team_id}_{public_name}') + - store the model in the db with this shared group name + - store a team model alias mapping {"public_name": "model_name_{team_id}_{public_name}"} + + Using a deterministic name (not UUID) ensures sibling deployments for the + same public model share a model_name, so the router treats them as a single + candidate pool for load balancing and failover. """ _team_id = model_params.model_info.team_id if _team_id is None: @@ -333,9 +337,9 @@ async def _add_team_model_to_db( if original_model_name: model_params.model_info.team_public_model_name = original_model_name - unique_model_name = f"model_name_{_team_id}_{uuid.uuid4()}" + group_model_name = f"model_name_{_team_id}_{original_model_name}" - model_params.model_name = unique_model_name + model_params.model_name = group_model_name ## CREATE MODEL IN DB ## model_response = await _add_model_to_db( @@ -348,7 +352,7 @@ async def _add_team_model_to_db( await update_team( data=UpdateTeamRequest( team_id=_team_id, - model_aliases={original_model_name: unique_model_name}, + model_aliases={original_model_name: group_model_name}, ), user_api_key_dict=user_api_key_dict, http_request=Request(scope={"type": "http"}), @@ -453,14 +457,14 @@ async def _setup_new_team_model_assignment( patch_data: updateDeployment, user_api_key_dict: UserAPIKeyAuth, ) -> None: - """Set up a new team model with unique name, alias, and team membership.""" - unique_model_name = f"model_name_{team_id}_{uuid.uuid4()}" - patch_data.model_name = unique_model_name + """Set up a new team model with deterministic name, alias, and team membership.""" + group_model_name = f"model_name_{team_id}_{public_model_name}" + patch_data.model_name = group_model_name await update_team( data=UpdateTeamRequest( team_id=team_id, - model_aliases={public_model_name: unique_model_name}, + model_aliases={public_model_name: group_model_name}, ), user_api_key_dict=user_api_key_dict, http_request=Request(scope={"type": "http"}), diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index f3c89003105..7f64a3de935 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -1,13 +1,14 @@ import json import os import sys -from litellm._uuid import uuid from typing import Dict, Optional from unittest.mock import AsyncMock, MagicMock, patch import pytest from fastapi.testclient import TestClient +from litellm._uuid import uuid + sys.path.insert( 0, os.path.abspath("../../../..") ) # Adds the parent directory to the system path @@ -399,7 +400,9 @@ class TestClearCache: """ Test that clear_cache clears DB models and preserves config models. """ - from litellm.proxy.management_endpoints.model_management_endpoints import clear_cache + from litellm.proxy.management_endpoints.model_management_endpoints import ( + clear_cache, + ) # Create mock router with mixed DB and config models mock_router = MagicMock() @@ -407,18 +410,18 @@ class TestClearCache: { "model_name": "gpt-4", "model_info": {"id": "db-model-1", "db_model": True}, - "litellm_params": {"model": "gpt-4"} + "litellm_params": {"model": "gpt-4"}, }, { - "model_name": "gpt-3.5-turbo", + "model_name": "gpt-3.5-turbo", "model_info": {"id": "config-model-1", "db_model": False}, - "litellm_params": {"model": "gpt-3.5-turbo"} + "litellm_params": {"model": "gpt-3.5-turbo"}, }, { "model_name": "claude-3", "model_info": {"id": "db-model-2", "db_model": True}, - "litellm_params": {"model": "claude-3"} - } + "litellm_params": {"model": "claude-3"}, + }, ] mock_router.delete_deployment = MagicMock(return_value=True) mock_router.auto_routers = MagicMock() @@ -466,8 +469,8 @@ class TestUpdatePublicModelGroups: """ import litellm from litellm.proxy.management_endpoints.model_management_endpoints import ( - update_public_model_groups, UpdatePublicModelGroupsRequest, + update_public_model_groups, ) old_db_models = ["db-model-1", "db-model-2"] @@ -525,7 +528,10 @@ class TestUpdatePublicModelGroups: ) old_links = {"Old Doc": "https://old.example.com"} - new_links = {"New Doc": "https://new.example.com", "API Ref": "https://api.example.com"} + new_links = { + "New Doc": "https://new.example.com", + "API Ref": "https://api.example.com", + } async def mock_get_config(*args, **kwargs): litellm.public_model_groups_links = old_links @@ -558,6 +564,100 @@ class TestUpdatePublicModelGroups: litellm.public_model_groups_links = original_value +class TestTeamModelAliasSiblingOverwrite: + """ + Verify that two sibling team deployments for the same public model name + produce the same deterministic internal model_name, so the alias write + is idempotent and the router groups both deployments together. + """ + + @pytest.mark.asyncio + async def test_sibling_team_models_share_deterministic_name(self): + from litellm.proxy.management_endpoints.model_management_endpoints import ( + _add_team_model_to_db, + ) + from litellm.types.router import ModelInfo + + team_id = "team_alias_overwrite" + public_name = "gpt-4.1-mini" + + captured_alias_calls = [] + + async def mock_update_team(data, user_api_key_dict, http_request): + if data.model_aliases: + captured_alias_calls.append(dict(data.model_aliases)) + + async def mock_add_model_to_db(model_params, user_api_key_dict, prisma_client): + return MagicMock(model_id=str(uuid.uuid4())) + + async def mock_team_model_add(data, http_request, user_api_key_dict): + pass + + user = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN) + prisma_client = MockPrismaClient(team_exists=True) + + deployment_1 = Deployment( + model_name=public_name, + litellm_params=LiteLLM_Params( + model="azure/gpt-4o-mini", + api_key="key-1", + api_base="https://eastus.example.openai.azure.com", + ), + model_info=ModelInfo(team_id=team_id), + ) + deployment_2 = Deployment( + model_name=public_name, + litellm_params=LiteLLM_Params( + model="azure/gpt-4o-mini", + api_key="key-2", + api_base="https://westus.example.openai.azure.com", + ), + model_info=ModelInfo(team_id=team_id), + ) + + with patch( + "litellm.proxy.management_endpoints.model_management_endpoints.update_team", + side_effect=mock_update_team, + ), patch( + "litellm.proxy.management_endpoints.model_management_endpoints._add_model_to_db", + side_effect=mock_add_model_to_db, + ), patch( + "litellm.proxy.management_endpoints.model_management_endpoints.team_model_add", + side_effect=mock_team_model_add, + ): + await _add_team_model_to_db( + model_params=deployment_1, + user_api_key_dict=user, + prisma_client=prisma_client, + ) + await _add_team_model_to_db( + model_params=deployment_2, + user_api_key_dict=user, + prisma_client=prisma_client, + ) + + assert len(captured_alias_calls) == 2 + + internal_name_1 = captured_alias_calls[0][public_name] + internal_name_2 = captured_alias_calls[1][public_name] + + expected_group_name = f"model_name_{team_id}_{public_name}" + + # Both sibling deployments get the same deterministic group name + assert internal_name_1 == expected_group_name + assert internal_name_2 == expected_group_name + assert internal_name_1 == internal_name_2, ( + "Sibling deployments must share the same model_name so the " + "router treats them as a single candidate pool" + ) + + # The second alias write is idempotent — same key, same value + final_aliases = {} + for alias_call in captured_alias_calls: + final_aliases.update(alias_call) + assert final_aliases == {public_name: expected_group_name} + + class TestTeamModelUpdate: """Test team model update handles team_id consistently with model creation""" @@ -657,27 +757,37 @@ class TestModelInfoEndpoint: user_id="test_user", api_key="test_key", models=["gpt-4", "claude-3"], - team_models=["gpt-3.5-turbo"] + team_models=["gpt-3.5-turbo"], ) - with patch("litellm.proxy.proxy_server.llm_router") as mock_router, \ - patch("litellm.proxy.proxy_server.get_key_models") as mock_get_key_models, \ - patch("litellm.proxy.proxy_server.get_team_models") as mock_get_team_models, \ - patch("litellm.proxy.proxy_server.get_complete_model_list") as mock_get_complete_models, \ - patch("litellm.get_llm_provider") as mock_get_provider: - + with patch("litellm.proxy.proxy_server.llm_router") as mock_router, patch( + "litellm.proxy.proxy_server.get_key_models" + ) as mock_get_key_models, patch( + "litellm.proxy.proxy_server.get_team_models" + ) as mock_get_team_models, patch( + "litellm.proxy.proxy_server.get_complete_model_list" + ) as mock_get_complete_models, patch( + "litellm.get_llm_provider" + ) as mock_get_provider: # Setup mocks - mock_router.get_model_names.return_value = ["gpt-4", "claude-3", "gpt-3.5-turbo"] + mock_router.get_model_names.return_value = [ + "gpt-4", + "claude-3", + "gpt-3.5-turbo", + ] mock_router.get_model_access_groups.return_value = {} mock_get_key_models.return_value = ["gpt-4", "claude-3"] mock_get_team_models.return_value = ["gpt-3.5-turbo"] - mock_get_complete_models.return_value = ["gpt-4", "claude-3", "gpt-3.5-turbo"] + mock_get_complete_models.return_value = [ + "gpt-4", + "claude-3", + "gpt-3.5-turbo", + ] mock_get_provider.return_value = (None, "openai", None, None) # Test accessible model result = await model_info( - model_id="gpt-4", - user_api_key_dict=user_api_key_dict + model_id="gpt-4", user_api_key_dict=user_api_key_dict ) assert result["id"] == "gpt-4" @@ -688,22 +798,25 @@ class TestModelInfoEndpoint: @pytest.mark.asyncio async def test_model_info_inaccessible_model_returns_404(self): """Test model_info returns 404 for inaccessible models""" - from litellm.proxy.proxy_server import model_info from fastapi import HTTPException + from litellm.proxy.proxy_server import model_info + # Mock user with limited access user_api_key_dict = UserAPIKeyAuth( user_id="test_user", api_key="test_key", models=["gpt-4"], # Only has access to gpt-4 - team_models=[] + team_models=[], ) - with patch("litellm.proxy.proxy_server.llm_router") as mock_router, \ - patch("litellm.proxy.proxy_server.get_key_models") as mock_get_key_models, \ - patch("litellm.proxy.proxy_server.get_team_models") as mock_get_team_models, \ - patch("litellm.proxy.proxy_server.get_complete_model_list") as mock_get_complete_models: - + with patch("litellm.proxy.proxy_server.llm_router") as mock_router, patch( + "litellm.proxy.proxy_server.get_key_models" + ) as mock_get_key_models, patch( + "litellm.proxy.proxy_server.get_team_models" + ) as mock_get_team_models, patch( + "litellm.proxy.proxy_server.get_complete_model_list" + ) as mock_get_complete_models: # Setup mocks - user only has access to gpt-4 mock_router.get_model_names.return_value = ["gpt-4", "claude-3"] mock_router.get_model_access_groups.return_value = {} @@ -715,32 +828,35 @@ class TestModelInfoEndpoint: with pytest.raises(HTTPException) as exc_info: await model_info( model_id="claude-3", # Not in user's accessible models - user_api_key_dict=user_api_key_dict + user_api_key_dict=user_api_key_dict, ) - + assert exc_info.value.status_code == 404 assert "does not exist or is not accessible" in exc_info.value.detail - @pytest.mark.asyncio + @pytest.mark.asyncio async def test_model_info_team_model_access(self): """Test model_info works with team model access""" from litellm.proxy.proxy_server import model_info - + # Mock user with team access user_api_key_dict = UserAPIKeyAuth( user_id="test_user", - api_key="test_key", + api_key="test_key", team_id="test_team", models=[], # No direct key models - team_models=["team-model-1"] + team_models=["team-model-1"], ) - with patch("litellm.proxy.proxy_server.llm_router") as mock_router, \ - patch("litellm.proxy.proxy_server.get_key_models") as mock_get_key_models, \ - patch("litellm.proxy.proxy_server.get_team_models") as mock_get_team_models, \ - patch("litellm.proxy.proxy_server.get_complete_model_list") as mock_get_complete_models, \ - patch("litellm.get_llm_provider") as mock_get_provider: - + with patch("litellm.proxy.proxy_server.llm_router") as mock_router, patch( + "litellm.proxy.proxy_server.get_key_models" + ) as mock_get_key_models, patch( + "litellm.proxy.proxy_server.get_team_models" + ) as mock_get_team_models, patch( + "litellm.proxy.proxy_server.get_complete_model_list" + ) as mock_get_complete_models, patch( + "litellm.get_llm_provider" + ) as mock_get_provider: # Setup mocks mock_router.get_model_names.return_value = ["team-model-1"] mock_router.get_model_access_groups.return_value = {} @@ -751,10 +867,9 @@ class TestModelInfoEndpoint: # Test team model access result = await model_info( - model_id="team-model-1", - user_api_key_dict=user_api_key_dict + model_id="team-model-1", user_api_key_dict=user_api_key_dict ) assert result["id"] == "team-model-1" - assert result["object"] == "model" + assert result["object"] == "model" assert result["owned_by"] == "custom" From aeb932d707d2b7d7c2bc345b4a4869472fd2d850 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 16:12:27 +0530 Subject: [PATCH 025/117] fix(team-routing): keep team model routing on public names Remove team model_alias rewrites and resolve team deployments by team_public_model_name with team_id so sibling deployments stay in the routing candidate pool, with explicit logs showing candidate selection before load balancing. Made-with: Cursor --- .../model_management_endpoints.py | 50 ++---- litellm/router.py | 71 +++++++- .../test_model_management_endpoints.py | 160 ++++++++++-------- 3 files changed, 171 insertions(+), 110 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 39383d4ee20..694fa2b1d55 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -322,13 +322,9 @@ async def _add_team_model_to_db( """ If 'team_id' is provided, - - generate a deterministic 'model_name' for the model (e.g. 'model_name_{team_id}_{public_name}') - - store the model in the db with this shared group name - - store a team model alias mapping {"public_name": "model_name_{team_id}_{public_name}"} - - Using a deterministic name (not UUID) ensures sibling deployments for the - same public model share a model_name, so the router treats them as a single - candidate pool for load balancing and failover. + - generate a unique 'model_name' for the model (e.g. 'model_name_{team_id}_{uuid}) + - store the model in the db with the unique 'model_name' + - add the public model name to the team's allowed models list """ _team_id = model_params.model_info.team_id if _team_id is None: @@ -337,9 +333,9 @@ async def _add_team_model_to_db( if original_model_name: model_params.model_info.team_public_model_name = original_model_name - group_model_name = f"model_name_{_team_id}_{original_model_name}" + unique_model_name = f"model_name_{_team_id}_{uuid.uuid4()}" - model_params.model_name = group_model_name + model_params.model_name = unique_model_name ## CREATE MODEL IN DB ## model_response = await _add_model_to_db( @@ -348,17 +344,6 @@ async def _add_team_model_to_db( prisma_client=prisma_client, ) - ## CREATE MODEL ALIAS IN DB ## - await update_team( - data=UpdateTeamRequest( - team_id=_team_id, - model_aliases={original_model_name: group_model_name}, - ), - user_api_key_dict=user_api_key_dict, - http_request=Request(scope={"type": "http"}), - ) - - # add model to team object await team_model_add( data=TeamModelAddRequest( team_id=_team_id, @@ -457,18 +442,9 @@ async def _setup_new_team_model_assignment( patch_data: updateDeployment, user_api_key_dict: UserAPIKeyAuth, ) -> None: - """Set up a new team model with deterministic name, alias, and team membership.""" - group_model_name = f"model_name_{team_id}_{public_model_name}" - patch_data.model_name = group_model_name - - await update_team( - data=UpdateTeamRequest( - team_id=team_id, - model_aliases={public_model_name: group_model_name}, - ), - user_api_key_dict=user_api_key_dict, - http_request=Request(scope={"type": "http"}), - ) + """Set up a new team model with unique name and team membership.""" + unique_model_name = f"model_name_{team_id}_{uuid.uuid4()}" + patch_data.model_name = unique_model_name await team_model_add( data=TeamModelAddRequest( @@ -492,18 +468,16 @@ async def _update_existing_team_model_assignment( db_model.model_info.team_public_model_name if db_model.model_info else None ) - # Update alias only if public name changed if old_public_name and public_model_name != old_public_name: - await update_team( - data=UpdateTeamRequest( + await team_model_add( + data=TeamModelAddRequest( team_id=team_id, - model_aliases={public_model_name: db_model.model_name}, + models=[public_model_name], ), - user_api_key_dict=user_api_key_dict, http_request=Request(scope={"type": "http"}), + user_api_key_dict=user_api_key_dict, ) - # Keep existing unique model_name patch_data.model_name = None diff --git a/litellm/router.py b/litellm/router.py index 25e5c9cb5d9..e146d60e359 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8148,20 +8148,23 @@ class Router: def map_team_model(self, team_model_name: str, team_id: str) -> Optional[str]: """ - Map a team model name to a team-specific model name. + Check if team_model_name resolves to team-specific deployments. + + Returns the public model name (unchanged) so the router can find all + sibling deployments via team_id filtering, instead of collapsing to a + single internal model_name. Returns: - - deployment id: str - the deployment id of the team-specific model - - None: if no team-specific model name is found + - str: the team_model_name if team deployments exist for this team + - None: if no team-specific model is found """ models = self.get_model_list(model_name=team_model_name, team_id=team_id) if not models: return None for model in models: if model.get("model_info", {}).get("team_id") == team_id: - return model.get("model_name") + return team_model_name - ## wildcard models return None def should_include_deployment( @@ -8867,6 +8870,38 @@ class Router: model = _model_from_alias if model not in self.model_names: + # Check for team-specific deployments by team_public_model_name + if request_team_id is not None: + team_deployments = self._get_all_deployments( + model_name=model, team_id=request_team_id + ) + if team_deployments: + candidate_details = [] + for deployment in team_deployments: + deployment_info = deployment.get("model_info", {}) or {} + deployment_params = deployment.get("litellm_params", {}) or {} + candidate_details.append( + { + "model_name": deployment.get("model_name"), + "model_id": deployment_info.get("id"), + "team_public_model_name": deployment_info.get( + "team_public_model_name" + ), + "api_base": deployment_params.get("api_base"), + } + ) + verbose_router_logger.info( + "🔥 routing_candidates_before_lb " + f"model={model} count={len(team_deployments)} " + f"candidates={candidate_details}" + ) + if len(team_deployments) > 1: + verbose_router_logger.info( + "🔥 load_balancer_candidate_pool " + f"model={model} candidate_count={len(team_deployments)}" + ) + return model, team_deployments + # check if provider/ specific wildcard routing use pattern matching pattern_deployments = self.pattern_router.get_deployments_by_pattern( model=model, @@ -8905,6 +8940,32 @@ class Router: # check if the user sent in a deployment name instead healthy_deployments = self._get_deployment_by_litellm_model(model=model) + if isinstance(healthy_deployments, list) and len(healthy_deployments) > 0: + candidate_details = [] + for deployment in healthy_deployments: + deployment_info = deployment.get("model_info", {}) or {} + deployment_params = deployment.get("litellm_params", {}) or {} + candidate_details.append( + { + "model_name": deployment.get("model_name"), + "model_id": deployment_info.get("id"), + "team_public_model_name": deployment_info.get( + "team_public_model_name" + ), + "api_base": deployment_params.get("api_base"), + } + ) + verbose_router_logger.info( + "🔥 routing_candidates_before_lb " + f"model={model} count={len(healthy_deployments)} " + f"candidates={candidate_details}" + ) + if len(healthy_deployments) > 1: + verbose_router_logger.info( + "🔥 load_balancer_candidate_pool " + f"model={model} candidate_count={len(healthy_deployments)}" + ) + if verbose_router_logger.isEnabledFor(logging.DEBUG): verbose_router_logger.debug( f"initial list of deployments: {healthy_deployments}" diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 7f64a3de935..5ef7face1f2 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -564,98 +564,124 @@ class TestUpdatePublicModelGroups: litellm.public_model_groups_links = original_value -class TestTeamModelAliasSiblingOverwrite: +class TestTeamModelSiblingRouting: """ - Verify that two sibling team deployments for the same public model name - produce the same deterministic internal model_name, so the alias write - is idempotent and the router groups both deployments together. + Verify that sibling team deployments (same public model name, different + api_base) are all reachable through routing — no alias overwrite, no + collapse to a single deployment. """ @pytest.mark.asyncio - async def test_sibling_team_models_share_deterministic_name(self): + async def test_no_model_aliases_written_for_team_models(self): + """ + _add_team_model_to_db must NOT write model_aliases (which caused + the second sibling to overwrite the first). It should only call + team_model_add to register the public name on the team's models list. + """ from litellm.proxy.management_endpoints.model_management_endpoints import ( _add_team_model_to_db, ) from litellm.types.router import ModelInfo - team_id = "team_alias_overwrite" + team_id = "team_no_alias" public_name = "gpt-4.1-mini" - captured_alias_calls = [] - - async def mock_update_team(data, user_api_key_dict, http_request): - if data.model_aliases: - captured_alias_calls.append(dict(data.model_aliases)) + mock_update_team = AsyncMock() async def mock_add_model_to_db(model_params, user_api_key_dict, prisma_client): return MagicMock(model_id=str(uuid.uuid4())) - async def mock_team_model_add(data, http_request, user_api_key_dict): - pass + mock_team_model_add = AsyncMock() user = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN) prisma_client = MockPrismaClient(team_exists=True) - deployment_1 = Deployment( - model_name=public_name, - litellm_params=LiteLLM_Params( - model="azure/gpt-4o-mini", - api_key="key-1", - api_base="https://eastus.example.openai.azure.com", - ), - model_info=ModelInfo(team_id=team_id), - ) - deployment_2 = Deployment( - model_name=public_name, - litellm_params=LiteLLM_Params( - model="azure/gpt-4o-mini", - api_key="key-2", - api_base="https://westus.example.openai.azure.com", - ), - model_info=ModelInfo(team_id=team_id), - ) - - with patch( - "litellm.proxy.management_endpoints.model_management_endpoints.update_team", - side_effect=mock_update_team, - ), patch( - "litellm.proxy.management_endpoints.model_management_endpoints._add_model_to_db", - side_effect=mock_add_model_to_db, - ), patch( - "litellm.proxy.management_endpoints.model_management_endpoints.team_model_add", - side_effect=mock_team_model_add, - ): - await _add_team_model_to_db( - model_params=deployment_1, - user_api_key_dict=user, - prisma_client=prisma_client, - ) - await _add_team_model_to_db( - model_params=deployment_2, - user_api_key_dict=user, - prisma_client=prisma_client, + for api_base in ["https://eastus.example.com", "https://westus.example.com"]: + dep = Deployment( + model_name=public_name, + litellm_params=LiteLLM_Params( + model="azure/gpt-4o-mini", + api_key="key", + api_base=api_base, + ), + model_info=ModelInfo(team_id=team_id), ) + with patch( + "litellm.proxy.management_endpoints.model_management_endpoints.update_team", + mock_update_team, + ), patch( + "litellm.proxy.management_endpoints.model_management_endpoints._add_model_to_db", + side_effect=mock_add_model_to_db, + ), patch( + "litellm.proxy.management_endpoints.model_management_endpoints.team_model_add", + mock_team_model_add, + ): + await _add_team_model_to_db( + model_params=dep, + user_api_key_dict=user, + prisma_client=prisma_client, + ) - assert len(captured_alias_calls) == 2 + mock_update_team.assert_not_called() + assert mock_team_model_add.call_count == 2 - internal_name_1 = captured_alias_calls[0][public_name] - internal_name_2 = captured_alias_calls[1][public_name] + @pytest.mark.asyncio + async def test_router_finds_all_sibling_team_deployments(self): + """ + When two team deployments share team_public_model_name="gpt-4.1-mini", + the router's _common_checks_available_deployment must return BOTH as + healthy_deployments (not collapse to one). + """ + import litellm - expected_group_name = f"model_name_{team_id}_{public_name}" + team_id = "teamA" + public_name = "gpt-4.1-mini" - # Both sibling deployments get the same deterministic group name - assert internal_name_1 == expected_group_name - assert internal_name_2 == expected_group_name - assert internal_name_1 == internal_name_2, ( - "Sibling deployments must share the same model_name so the " - "router treats them as a single candidate pool" + router = litellm.Router( + model_list=[ + { + "model_name": f"model_name_{team_id}_uuid1", + "litellm_params": { + "model": "azure/gpt-4o-mini", + "api_key": "key-1", + "api_base": "https://eastus.openai.azure.com", + }, + "model_info": { + "team_id": team_id, + "team_public_model_name": public_name, + }, + }, + { + "model_name": f"model_name_{team_id}_uuid2", + "litellm_params": { + "model": "azure/gpt-4o-mini", + "api_key": "key-2", + "api_base": "https://westus.openai.azure.com", + }, + "model_info": { + "team_id": team_id, + "team_public_model_name": public_name, + }, + }, + ], ) - # The second alias write is idempotent — same key, same value - final_aliases = {} - for alias_call in captured_alias_calls: - final_aliases.update(alias_call) - assert final_aliases == {public_name: expected_group_name} + # map_team_model should return the public name (not an internal UUID) + result = router.map_team_model(public_name, team_id) + assert result == public_name + + # _common_checks_available_deployment should return both deployments + model, healthy = router._common_checks_available_deployment( + model=public_name, + request_kwargs={"metadata": {"user_api_key_team_id": team_id}}, + ) + assert isinstance(healthy, list) + assert len(healthy) == 2 + api_bases = {d["litellm_params"]["api_base"] for d in healthy} + assert api_bases == { + "https://eastus.openai.azure.com", + "https://westus.openai.azure.com", + } class TestTeamModelUpdate: @@ -704,7 +730,7 @@ class TestTeamModelUpdate: assert result.get("model_name", "").startswith("model_name_test_team_123_") assert "team_public_model_name" in str(result.get("model_info", "")) - mock_update_team.assert_called_once() + mock_update_team.assert_not_called() mock_team_model_add.assert_called_once() @pytest.mark.asyncio From 1835e9a25234b7e93cf704e88b2de6feba3e4f76 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 16:14:30 +0530 Subject: [PATCH 026/117] chore(team-routing): remove temporary candidate pool logs Remove temporary fire-emoji router logs used for local verification while keeping team sibling deployment routing behavior unchanged. Made-with: Cursor --- litellm/router.py | 50 ----------------------------------------------- 1 file changed, 50 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index e146d60e359..247f209e338 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8876,30 +8876,6 @@ class Router: model_name=model, team_id=request_team_id ) if team_deployments: - candidate_details = [] - for deployment in team_deployments: - deployment_info = deployment.get("model_info", {}) or {} - deployment_params = deployment.get("litellm_params", {}) or {} - candidate_details.append( - { - "model_name": deployment.get("model_name"), - "model_id": deployment_info.get("id"), - "team_public_model_name": deployment_info.get( - "team_public_model_name" - ), - "api_base": deployment_params.get("api_base"), - } - ) - verbose_router_logger.info( - "🔥 routing_candidates_before_lb " - f"model={model} count={len(team_deployments)} " - f"candidates={candidate_details}" - ) - if len(team_deployments) > 1: - verbose_router_logger.info( - "🔥 load_balancer_candidate_pool " - f"model={model} candidate_count={len(team_deployments)}" - ) return model, team_deployments # check if provider/ specific wildcard routing use pattern matching @@ -8940,32 +8916,6 @@ class Router: # check if the user sent in a deployment name instead healthy_deployments = self._get_deployment_by_litellm_model(model=model) - if isinstance(healthy_deployments, list) and len(healthy_deployments) > 0: - candidate_details = [] - for deployment in healthy_deployments: - deployment_info = deployment.get("model_info", {}) or {} - deployment_params = deployment.get("litellm_params", {}) or {} - candidate_details.append( - { - "model_name": deployment.get("model_name"), - "model_id": deployment_info.get("id"), - "team_public_model_name": deployment_info.get( - "team_public_model_name" - ), - "api_base": deployment_params.get("api_base"), - } - ) - verbose_router_logger.info( - "🔥 routing_candidates_before_lb " - f"model={model} count={len(healthy_deployments)} " - f"candidates={candidate_details}" - ) - if len(healthy_deployments) > 1: - verbose_router_logger.info( - "🔥 load_balancer_candidate_pool " - f"model={model} candidate_count={len(healthy_deployments)}" - ) - if verbose_router_logger.isEnabledFor(logging.DEBUG): verbose_router_logger.debug( f"initial list of deployments: {healthy_deployments}" From 7b5e7e05b1fb41e55a4f10adbc3082358aa76a84 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 16:25:26 +0530 Subject: [PATCH 027/117] fix(router): address Greptile review comments - Add None guard for original_model_name in _add_team_model_to_db - Remove stale old public name when renaming team model - Add comment clarifying team deployment early-return priority Made-with: Cursor --- .../model_management_endpoints.py | 27 +++++++++++++------ litellm/router.py | 4 ++- 2 files changed, 22 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 694fa2b1d55..c091f6b5812 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -32,6 +32,7 @@ from litellm.proxy._types import ( ProxyErrorTypes, ProxyException, TeamModelAddRequest, + TeamModelDeleteRequest, UpdateTeamRequest, UserAPIKeyAuth, ) @@ -40,6 +41,7 @@ from litellm.proxy.common_utils.encrypt_decrypt_utils import encrypt_value_helpe from litellm.proxy.management_endpoints.common_utils import _is_user_team_admin from litellm.proxy.management_endpoints.team_endpoints import ( team_model_add, + team_model_delete, update_team, ) from litellm.proxy.management_helpers.audit_logs import create_object_audit_log @@ -344,14 +346,15 @@ async def _add_team_model_to_db( prisma_client=prisma_client, ) - await team_model_add( - data=TeamModelAddRequest( - team_id=_team_id, - models=[original_model_name], - ), - http_request=Request(scope={"type": "http"}), - user_api_key_dict=user_api_key_dict, - ) + if original_model_name: + await team_model_add( + data=TeamModelAddRequest( + team_id=_team_id, + models=[original_model_name], + ), + http_request=Request(scope={"type": "http"}), + user_api_key_dict=user_api_key_dict, + ) return model_response @@ -469,6 +472,14 @@ async def _update_existing_team_model_assignment( ) if old_public_name and public_model_name != old_public_name: + await team_model_delete( + data=TeamModelDeleteRequest( + team_id=team_id, + models=[old_public_name], + ), + http_request=Request(scope={"type": "http"}), + user_api_key_dict=user_api_key_dict, + ) await team_model_add( data=TeamModelAddRequest( team_id=team_id, diff --git a/litellm/router.py b/litellm/router.py index 247f209e338..767938d11fb 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8870,7 +8870,9 @@ class Router: model = _model_from_alias if model not in self.model_names: - # Check for team-specific deployments by team_public_model_name + # Check for team-specific deployments by team_public_model_name. + # This intentionally takes priority over team pattern routers below, + # so that named team deployments shadow wildcard/pattern routes. if request_team_id is not None: team_deployments = self._get_all_deployments( model_name=model, team_id=request_team_id From 248fb8bc90799de32f793173ff34f40562c3dae4 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 16:33:26 +0530 Subject: [PATCH 028/117] fix(router): address remaining Greptile P0/P1 issues - Update map_team_model test to expect public name return - Only remove old public name if no sibling deployments use it Made-with: Cursor --- .../model_management_endpoints.py | 34 ++++++++++++++----- .../test_get_model_list_alias_optimization.py | 2 +- 2 files changed, 27 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index c091f6b5812..eb85112a66c 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -472,14 +472,32 @@ async def _update_existing_team_model_assignment( ) if old_public_name and public_model_name != old_public_name: - await team_model_delete( - data=TeamModelDeleteRequest( - team_id=team_id, - models=[old_public_name], - ), - http_request=Request(scope={"type": "http"}), - user_api_key_dict=user_api_key_dict, - ) + from litellm.proxy.proxy_server import llm_router + + other_deployments_with_old_name = [] + if llm_router: + all_deployments = llm_router.get_model_list( + model_name=old_public_name, team_id=team_id + ) + if all_deployments: + other_deployments_with_old_name = [ + d + for d in all_deployments + if d.get("model_name") != db_model.model_name + and d.get("model_info", {}).get("team_public_model_name") + == old_public_name + ] + + if not other_deployments_with_old_name: + await team_model_delete( + data=TeamModelDeleteRequest( + team_id=team_id, + models=[old_public_name], + ), + http_request=Request(scope={"type": "http"}), + user_api_key_dict=user_api_key_dict, + ) + await team_model_add( data=TeamModelAddRequest( team_id=team_id, diff --git a/tests/router_unit_tests/test_get_model_list_alias_optimization.py b/tests/router_unit_tests/test_get_model_list_alias_optimization.py index 31d992b6646..62baf0a3d22 100644 --- a/tests/router_unit_tests/test_get_model_list_alias_optimization.py +++ b/tests/router_unit_tests/test_get_model_list_alias_optimization.py @@ -46,5 +46,5 @@ def test_map_team_model_should_not_iterate_aliases_for_non_alias_team_model_name assert ( router.map_team_model(team_model_name="team-model", team_id="team-1") - == "gpt-3.5-turbo" + == "team-model" ) From ef9ea1f8f20b53cae0ff8050dee04591b6f053f8 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 16:44:30 +0530 Subject: [PATCH 029/117] fix(router): address Greptile P1/P2 performance issues - Guard against llm_router=None to prevent silent deletion - Add O(1) team_model index to avoid O(n) scan on every team request Made-with: Cursor --- .../model_management_endpoints.py | 26 ++++---- litellm/router.py | 61 +++++++++++++++++++ 2 files changed, 76 insertions(+), 11 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index eb85112a66c..bfd67ea4e59 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -474,11 +474,15 @@ async def _update_existing_team_model_assignment( if old_public_name and public_model_name != old_public_name: from litellm.proxy.proxy_server import llm_router - other_deployments_with_old_name = [] - if llm_router: + if llm_router is None: + verbose_proxy_logger.warning( + "llm_router not initialized; skipping old public name cleanup to preserve sibling deployments" + ) + else: all_deployments = llm_router.get_model_list( model_name=old_public_name, team_id=team_id ) + other_deployments_with_old_name = [] if all_deployments: other_deployments_with_old_name = [ d @@ -488,15 +492,15 @@ async def _update_existing_team_model_assignment( == old_public_name ] - if not other_deployments_with_old_name: - await team_model_delete( - data=TeamModelDeleteRequest( - team_id=team_id, - models=[old_public_name], - ), - http_request=Request(scope={"type": "http"}), - user_api_key_dict=user_api_key_dict, - ) + if not other_deployments_with_old_name: + await team_model_delete( + data=TeamModelDeleteRequest( + team_id=team_id, + models=[old_public_name], + ), + http_request=Request(scope={"type": "http"}), + user_api_key_dict=user_api_key_dict, + ) await team_model_add( data=TeamModelAddRequest( diff --git a/litellm/router.py b/litellm/router.py index 767938d11fb..c9d79beacce 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -467,6 +467,8 @@ class Router: # Initialize model name to deployment indices mapping for O(1) lookups # Maps model_name -> list of indices in model_list self.model_name_to_deployment_indices: Dict[str, List[int]] = {} + # Maps (team_id, team_public_model_name) -> list of indices in model_list + self.team_model_to_deployment_indices: Dict[Tuple[str, str], List[int]] = {} if model_list is not None: # set_model_list will build indices automatically @@ -6835,6 +6837,7 @@ class Router: self.model_list = [] self.model_id_to_deployment_index_map = {} # Reset the index self.model_name_to_deployment_indices = {} # Reset the model_name index + self.team_model_to_deployment_indices = {} # Reset the team_model index self._invalidate_model_group_info_cache() self._invalidate_access_groups_cache() # we add api_base/api_key each model so load balancing between azure/gpt on api_base1 and api_base2 works @@ -7150,6 +7153,26 @@ class Router: else: del self.model_name_to_deployment_indices[model_name] + # Update team_model_to_deployment_indices + for key, indices in list(self.team_model_to_deployment_indices.items()): + # Remove the deleted index + if removal_idx in indices: + indices.remove(removal_idx) + + # Decrement all indices greater than removal_idx + updated_indices = [] + for idx in indices: + if idx > removal_idx: + updated_indices.append(idx - 1) + else: + updated_indices.append(idx) + + # Update or remove the entry + if len(updated_indices) > 0: + self.team_model_to_deployment_indices[key] = updated_indices + else: + del self.team_model_to_deployment_indices[key] + def _add_model_to_list_and_index_map( self, model: dict, model_id: Optional[str] = None ) -> None: @@ -7178,6 +7201,17 @@ class Router: self.model_name_to_deployment_indices[model_name] = [] self.model_name_to_deployment_indices[model_name].append(idx) + # Update team_model index for O(1) team-scoped lookup + team_id = model.get("model_info", {}).get("team_id") + team_public_model_name = model.get("model_info", {}).get( + "team_public_model_name" + ) + if team_id and team_public_model_name: + key = (team_id, team_public_model_name) + if key not in self.team_model_to_deployment_indices: + self.team_model_to_deployment_indices[key] = [] + self.team_model_to_deployment_indices[key].append(idx) + def upsert_deployment(self, deployment: Deployment) -> Optional[Deployment]: """ Add or update deployment @@ -8008,6 +8042,7 @@ class Router: instead of O(n) linear scan through the entire model_list. """ self.model_name_to_deployment_indices.clear() + self.team_model_to_deployment_indices.clear() for idx, model in enumerate(model_list): model_name = model.get("model_name") @@ -8016,6 +8051,16 @@ class Router: self.model_name_to_deployment_indices[model_name] = [] self.model_name_to_deployment_indices[model_name].append(idx) + team_id = model.get("model_info", {}).get("team_id") + team_public_model_name = model.get("model_info", {}).get( + "team_public_model_name" + ) + if team_id and team_public_model_name: + key = (team_id, team_public_model_name) + if key not in self.team_model_to_deployment_indices: + self.team_model_to_deployment_indices[key] = [] + self.team_model_to_deployment_indices[key].append(idx) + def _build_model_id_to_deployment_index_map(self, model_list: list): """ Build model index from model list to enable O(1) lookups immediately. @@ -8200,6 +8245,22 @@ class Router: """ returned_models: List[DeploymentTypedDict] = [] + # O(1) lookup in team_model index when team_id is provided + if team_id is not None: + key = (team_id, model_name) + if key in self.team_model_to_deployment_indices: + indices = self.team_model_to_deployment_indices[key] + # O(k) where k = team deployments for this model_name (typically 1-10) + for idx in indices: + model = self.model_list[idx] + if model_alias is not None: + alias_model = model.copy() + alias_model["model_name"] = model_alias + returned_models.append(alias_model) + else: + returned_models.append(model) + return returned_models + # O(1) lookup in model_name index if model_name in self.model_name_to_deployment_indices: indices = self.model_name_to_deployment_indices[model_name] From 4f302f10d0522a3eccf458990402b23567509fdd Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 17:00:30 +0530 Subject: [PATCH 030/117] fix(router): prevent cross-team deployment leakage in fallback path Guard should_include_deployment fallback to only return deployments matching the requested team_id, preventing public-name collisions from leaking deployments across teams Made-with: Cursor --- litellm/router.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/litellm/router.py b/litellm/router.py index c9d79beacce..7f428b18edf 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8225,7 +8225,8 @@ class Router: ): return True elif model_name is not None and model["model_name"] == model_name: - return True + if team_id is None or model["model_info"].get("team_id") == team_id: + return True return False def _get_all_deployments( From f5b72988540c2e195c1ed2c5abfdda4897e9e792 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 17:10:11 +0530 Subject: [PATCH 031/117] fix(management): query DB directly for sibling deployments on rename - Add clarifying comments to test assertions - Query prisma DB instead of in-memory router to avoid stale state - Prevents incorrect deletion of old public name when siblings exist Made-with: Cursor --- .../model_management_endpoints.py | 32 ++++++++++--------- .../test_model_management_endpoints.py | 9 ++++++ 2 files changed, 26 insertions(+), 15 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index bfd67ea4e59..c38a00e18a1 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -420,6 +420,7 @@ async def _update_team_model_in_db( db_model=db_model, patch_data=patch_data, user_api_key_dict=user_api_key_dict, + prisma_client=prisma_client, ) return update_db_model(db_model=db_model, updated_patch=patch_data) @@ -465,6 +466,7 @@ async def _update_existing_team_model_assignment( db_model: Deployment, patch_data: updateDeployment, user_api_key_dict: UserAPIKeyAuth, + prisma_client: PrismaClient, ) -> None: """Update an existing team model if the public name changed.""" old_public_name = ( @@ -472,25 +474,25 @@ async def _update_existing_team_model_assignment( ) if old_public_name and public_model_name != old_public_name: - from litellm.proxy.proxy_server import llm_router - - if llm_router is None: + if prisma_client is None: verbose_proxy_logger.warning( - "llm_router not initialized; skipping old public name cleanup to preserve sibling deployments" + "prisma_client not initialized; skipping old public name cleanup to preserve sibling deployments" ) else: - all_deployments = llm_router.get_model_list( - model_name=old_public_name, team_id=team_id + response = await prisma_client.db.litellm_proxymodeltable.find_many( + where={ + "model_info": { + "path": ["team_id"], + "equals": team_id, + } + } ) - other_deployments_with_old_name = [] - if all_deployments: - other_deployments_with_old_name = [ - d - for d in all_deployments - if d.get("model_name") != db_model.model_name - and d.get("model_info", {}).get("team_public_model_name") - == old_public_name - ] + other_deployments_with_old_name = [ + d + for d in response + if d.model_name != db_model.model_name + and d.model_info.get("team_public_model_name") == old_public_name + ] if not other_deployments_with_old_name: await team_model_delete( diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 5ef7face1f2..fd4f3d56b10 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -46,10 +46,17 @@ class MockPrismaClient: ) return None + async def find_many(self, where): + return [] + @property def litellm_teamtable(self): return self + @property + def litellm_proxymodeltable(self): + return self + class MockLLMRouter: def __init__(self): @@ -730,7 +737,9 @@ class TestTeamModelUpdate: assert result.get("model_name", "").startswith("model_name_test_team_123_") assert "team_public_model_name" in str(result.get("model_info", "")) + # update_team must not be called (no model_aliases writes for team models) mock_update_team.assert_not_called() + # team_model_add must be called to add public name to team's models list mock_team_model_add.assert_called_once() @pytest.mark.asyncio From 298df75066bb5fda8f5852a5a8aacb127da2815d Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 17:19:11 +0530 Subject: [PATCH 032/117] fix(router): guard None model_info and deduplicate team index logic - Guard against None model_info in sibling deployment check - Extract _update_team_model_index helper to eliminate duplication Made-with: Cursor --- .../model_management_endpoints.py | 3 +- litellm/router.py | 38 ++++++++++--------- 2 files changed, 22 insertions(+), 19 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index c38a00e18a1..4b06c4460e7 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -491,7 +491,8 @@ async def _update_existing_team_model_assignment( d for d in response if d.model_name != db_model.model_name - and d.model_info.get("team_public_model_name") == old_public_name + and (d.model_info or {}).get("team_public_model_name") + == old_public_name ] if not other_deployments_with_old_name: diff --git a/litellm/router.py b/litellm/router.py index 7f428b18edf..130979d25bc 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7173,6 +7173,24 @@ class Router: else: del self.team_model_to_deployment_indices[key] + def _update_team_model_index(self, model: dict, idx: int) -> None: + """ + Helper to update team_model_to_deployment_indices for a single deployment. + + Parameters: + - model: dict - the deployment to index + - idx: int - the index in model_list + """ + team_id = model.get("model_info", {}).get("team_id") + team_public_model_name = model.get("model_info", {}).get( + "team_public_model_name" + ) + if team_id and team_public_model_name: + key = (team_id, team_public_model_name) + if key not in self.team_model_to_deployment_indices: + self.team_model_to_deployment_indices[key] = [] + self.team_model_to_deployment_indices[key].append(idx) + def _add_model_to_list_and_index_map( self, model: dict, model_id: Optional[str] = None ) -> None: @@ -7202,15 +7220,7 @@ class Router: self.model_name_to_deployment_indices[model_name].append(idx) # Update team_model index for O(1) team-scoped lookup - team_id = model.get("model_info", {}).get("team_id") - team_public_model_name = model.get("model_info", {}).get( - "team_public_model_name" - ) - if team_id and team_public_model_name: - key = (team_id, team_public_model_name) - if key not in self.team_model_to_deployment_indices: - self.team_model_to_deployment_indices[key] = [] - self.team_model_to_deployment_indices[key].append(idx) + self._update_team_model_index(model, idx) def upsert_deployment(self, deployment: Deployment) -> Optional[Deployment]: """ @@ -8051,15 +8061,7 @@ class Router: self.model_name_to_deployment_indices[model_name] = [] self.model_name_to_deployment_indices[model_name].append(idx) - team_id = model.get("model_info", {}).get("team_id") - team_public_model_name = model.get("model_info", {}).get( - "team_public_model_name" - ) - if team_id and team_public_model_name: - key = (team_id, team_public_model_name) - if key not in self.team_model_to_deployment_indices: - self.team_model_to_deployment_indices[key] = [] - self.team_model_to_deployment_indices[key].append(idx) + self._update_team_model_index(model, idx) def _build_model_id_to_deployment_index_map(self, model_list: list): """ From 8aa58bdcaaa3a67000ea0752e7f90c96f92d028d Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 17:33:07 +0530 Subject: [PATCH 033/117] fix(routing): prevent stale model_aliases from interfering with team routing - Skip model_aliases rewrite if model resolves to team deployments - Add test coverage for sibling-preservation branch - Update MockPrismaClient to support sibling deployment scenarios Made-with: Cursor --- litellm/proxy/litellm_pre_call_utils.py | 15 ++++ .../test_model_management_endpoints.py | 70 ++++++++++++++++++- 2 files changed, 83 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 4ca0d876a1c..1a7bbe0474b 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -1296,6 +1296,10 @@ def _update_model_if_team_alias_exists( "gpt-4o": "gpt-4o-team-1" } - requested_model = "gpt-4o-team-1" + + Note: model_aliases for team models are deprecated. This function only applies + to legacy non-team-scoped aliases. Team-scoped deployments use team_public_model_name + and are resolved via map_team_model in route_llm_request. """ _model = data.get("model") if ( @@ -1303,6 +1307,17 @@ def _update_model_if_team_alias_exists( and user_api_key_dict.team_model_aliases and _model in user_api_key_dict.team_model_aliases ): + from litellm.proxy.proxy_server import llm_router + + # Skip alias rewrite if this model resolves to team-specific deployments + # (team models use team_public_model_name, not model_aliases) + if ( + llm_router + and user_api_key_dict.team_id + and llm_router.map_team_model(_model, user_api_key_dict.team_id) is not None + ): + return + data["model"] = user_api_key_dict.team_model_aliases[_model] return diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index fd4f3d56b10..dcfd5847bd5 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -28,9 +28,15 @@ from litellm.types.router import Deployment, LiteLLM_Params, updateDeployment class MockPrismaClient: - def __init__(self, team_exists: bool = True, user_admin: bool = True): + def __init__( + self, + team_exists: bool = True, + user_admin: bool = True, + sibling_deployments: list = None, + ): self.team_exists = team_exists self.user_admin = user_admin + self.sibling_deployments = sibling_deployments or [] self.db = self async def find_unique(self, where): @@ -47,7 +53,7 @@ class MockPrismaClient: return None async def find_many(self, where): - return [] + return self.sibling_deployments @property def litellm_teamtable(self): @@ -742,6 +748,66 @@ class TestTeamModelUpdate: # team_model_add must be called to add public name to team's models list mock_team_model_add.assert_called_once() + @pytest.mark.asyncio + async def test_rename_preserves_old_name_when_siblings_exist(self): + """Test that renaming a deployment preserves old public name when sibling deployments still use it""" + from unittest.mock import MagicMock + + from litellm.proxy.management_endpoints.model_management_endpoints import ( + _update_existing_team_model_assignment, + ) + from litellm.types.router import ModelInfo + + # Create a deployment being renamed + db_model = Deployment( + model_name="model_name_team_123_uuid1", + litellm_params=LiteLLM_Params(model="azure/gpt-4o-mini"), + model_info=ModelInfo( + team_id="team_123", team_public_model_name="old-public-name" + ), + ) + + # Create a sibling deployment that still uses the old public name + sibling_deployment = MagicMock() + sibling_deployment.model_name = "model_name_team_123_uuid2" + sibling_deployment.model_info = { + "team_id": "team_123", + "team_public_model_name": "old-public-name", + } + + prisma_client = MockPrismaClient( + team_exists=True, sibling_deployments=[sibling_deployment] + ) + + patch_data = updateDeployment( + model_name="new-public-name", + model_info=ModelInfo(team_id="team_123"), + ) + + user_api_key_dict = UserAPIKeyAuth( + user_id="test_user", + user_role=LitellmUserRoles.PROXY_ADMIN, + ) + + with patch( + "litellm.proxy.management_endpoints.model_management_endpoints.team_model_delete" + ) as mock_delete, patch( + "litellm.proxy.management_endpoints.model_management_endpoints.team_model_add" + ) as mock_add: + await _update_existing_team_model_assignment( + team_id="team_123", + public_model_name="new-public-name", + db_model=db_model, + patch_data=patch_data, + user_api_key_dict=user_api_key_dict, + prisma_client=prisma_client, # type: ignore + ) + + # team_model_delete should NOT be called because sibling exists + mock_delete.assert_not_called() + # team_model_add should be called to add new public name + mock_add.assert_called_once() + @pytest.mark.asyncio async def test_patch_model_with_team_id_validates_permissions(self): """Test PATCH with team_id runs same validation as POST for team permissions""" From e8fb7762b345d643a0431f8e9f5bd7b1a074d8d6 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 17:56:03 +0530 Subject: [PATCH 034/117] perf(routing): optimize team model checks and improve test coverage - Use O(1) team index lookup instead of map_team_model in alias guard - Fix MockPrismaClient to validate where clause filters - Add comment explaining DB query trade-off for team deployments Made-with: Cursor --- litellm/proxy/litellm_pre_call_utils.py | 11 ++++----- .../model_management_endpoints.py | 4 ++++ .../test_model_management_endpoints.py | 23 +++++++++++++++++++ 3 files changed, 32 insertions(+), 6 deletions(-) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 1a7bbe0474b..48e83f1395b 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -1311,12 +1311,11 @@ def _update_model_if_team_alias_exists( # Skip alias rewrite if this model resolves to team-specific deployments # (team models use team_public_model_name, not model_aliases) - if ( - llm_router - and user_api_key_dict.team_id - and llm_router.map_team_model(_model, user_api_key_dict.team_id) is not None - ): - return + # Use O(1) index lookup instead of map_team_model to avoid O(n) scan + if llm_router and user_api_key_dict.team_id: + key = (user_api_key_dict.team_id, _model) + if key in llm_router.team_model_to_deployment_indices: + return data["model"] = user_api_key_dict.team_model_aliases[_model] return diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 4b06c4460e7..7682b65657a 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -479,6 +479,10 @@ async def _update_existing_team_model_assignment( "prisma_client not initialized; skipping old public name cleanup to preserve sibling deployments" ) else: + # Query DB for all deployments in this team, then filter by public name. + # Note: Prisma's JSON filtering doesn't support compound AND conditions + # across multiple JSON paths, so we filter team_public_model_name in Python. + # For most teams (typically <100 deployments), this is acceptable. response = await prisma_client.db.litellm_proxymodeltable.find_many( where={ "model_info": { diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index dcfd5847bd5..09410c19d34 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -53,6 +53,29 @@ class MockPrismaClient: return None async def find_many(self, where): + # Filter sibling deployments by team_id if where clause specifies it + if not self.sibling_deployments: + return [] + + # Extract team_id from where clause if present + team_id_filter = None + if where and "model_info" in where: + model_info_filter = where["model_info"] + if isinstance(model_info_filter, dict) and "path" in model_info_filter: + if ( + model_info_filter["path"] == ["team_id"] + and "equals" in model_info_filter + ): + team_id_filter = model_info_filter["equals"] + + # Filter deployments by team_id if specified + if team_id_filter: + return [ + d + for d in self.sibling_deployments + if d.model_info.get("team_id") == team_id_filter + ] + return self.sibling_deployments @property From 8db867c51c2bb4a8995308439ad2a4f2cc474390 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 18:13:35 +0530 Subject: [PATCH 035/117] fix(routing): address state consistency and type safety issues - Check alias target pattern to detect stale team aliases - Fix PrismaClient type annotation to Optional - Eliminate in-place mutation in index update logic Made-with: Cursor --- litellm/proxy/litellm_pre_call_utils.py | 19 +++++++++----- .../model_management_endpoints.py | 2 +- litellm/router.py | 26 ++++++++++--------- 3 files changed, 28 insertions(+), 19 deletions(-) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 48e83f1395b..96a271dd027 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -1311,13 +1311,20 @@ def _update_model_if_team_alias_exists( # Skip alias rewrite if this model resolves to team-specific deployments # (team models use team_public_model_name, not model_aliases) - # Use O(1) index lookup instead of map_team_model to avoid O(n) scan - if llm_router and user_api_key_dict.team_id: - key = (user_api_key_dict.team_id, _model) - if key in llm_router.team_model_to_deployment_indices: - return + aliased_target = user_api_key_dict.team_model_aliases[_model] - data["model"] = user_api_key_dict.team_model_aliases[_model] + # Check if the alias points to a stale team-scoped UUID name + # (format: "model_name_{team_id}_{uuid}") + if aliased_target.startswith(f"model_name_{user_api_key_dict.team_id}_"): + # This is a stale alias from pre-PR deployments. + # Check if current team deployments exist for the public name. + if llm_router: + key = (user_api_key_dict.team_id, _model) + if key in llm_router.team_model_to_deployment_indices: + # Team deployments exist; skip stale alias + return + + data["model"] = aliased_target return diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 7682b65657a..7d0181a3687 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -466,7 +466,7 @@ async def _update_existing_team_model_assignment( db_model: Deployment, patch_data: updateDeployment, user_api_key_dict: UserAPIKeyAuth, - prisma_client: PrismaClient, + prisma_client: Optional[PrismaClient], ) -> None: """Update an existing team model if the public name changed.""" old_public_name = ( diff --git a/litellm/router.py b/litellm/router.py index 130979d25bc..1a57d2f8c0d 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7135,16 +7135,17 @@ class Router: # Update model_name_to_deployment_indices for model_name, indices in list(self.model_name_to_deployment_indices.items()): - # Remove the deleted index - if removal_idx in indices: - indices.remove(removal_idx) - - # Decrement all indices greater than removal_idx + # Build new list without mutating the original updated_indices = [] for idx in indices: - if idx > removal_idx: + if idx == removal_idx: + # Skip the removed index + continue + elif idx > removal_idx: + # Decrement indices after removal updated_indices.append(idx - 1) else: + # Keep indices before removal unchanged updated_indices.append(idx) # Update or remove the entry @@ -7155,16 +7156,17 @@ class Router: # Update team_model_to_deployment_indices for key, indices in list(self.team_model_to_deployment_indices.items()): - # Remove the deleted index - if removal_idx in indices: - indices.remove(removal_idx) - - # Decrement all indices greater than removal_idx + # Build new list without mutating the original updated_indices = [] for idx in indices: - if idx > removal_idx: + if idx == removal_idx: + # Skip the removed index + continue + elif idx > removal_idx: + # Decrement indices after removal updated_indices.append(idx - 1) else: + # Keep indices before removal unchanged updated_indices.append(idx) # Update or remove the entry From 173695f5e0ed6e2e8933fbec341cfdc6162843dc Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 18:26:43 +0530 Subject: [PATCH 036/117] Fix greptile comments --- litellm/proxy/litellm_pre_call_utils.py | 12 +++- .../model_management_endpoints.py | 20 +++++- tests/proxy_unit_tests/test_proxy_utils.py | 43 ++++++++++++ .../test_model_management_endpoints.py | 70 ++++++++++++++++++- 4 files changed, 140 insertions(+), 5 deletions(-) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 96a271dd027..4a12a0a5774 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -26,6 +26,7 @@ _SPECIAL_HEADERS_CACHE = frozenset( v.value.lower() for v in SpecialHeaders._member_map_.values() ) from litellm.router import Router +from litellm.secret_managers.main import get_secret_bool from litellm.types.llms.anthropic import ANTHROPIC_API_HEADERS from litellm.types.services import ServiceTypes from litellm.types.utils import ( @@ -1313,9 +1314,16 @@ def _update_model_if_team_alias_exists( # (team models use team_public_model_name, not model_aliases) aliased_target = user_api_key_dict.team_model_aliases[_model] - # Check if the alias points to a stale team-scoped UUID name + # Optional bypass for stale aliases from pre-PR deployments: + # only enabled via feature flag to preserve backwards compatibility. + enable_stale_alias_bypass = get_secret_bool( + "LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS", False + ) + # Check if the alias points to a team-scoped UUID name # (format: "model_name_{team_id}_{uuid}") - if aliased_target.startswith(f"model_name_{user_api_key_dict.team_id}_"): + if enable_stale_alias_bypass and aliased_target.startswith( + f"model_name_{user_api_key_dict.team_id}_" + ): # This is a stale alias from pre-PR deployments. # Check if current team deployments exist for the public name. if llm_router: diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 7d0181a3687..40f4d722dc6 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -469,6 +469,23 @@ async def _update_existing_team_model_assignment( prisma_client: Optional[PrismaClient], ) -> None: """Update an existing team model if the public name changed.""" + + def _get_team_public_model_name( + model_info: Optional[Union[dict, str]] + ) -> Optional[str]: + if isinstance(model_info, dict): + value = model_info.get("team_public_model_name") + return value if isinstance(value, str) else None + if isinstance(model_info, str): + try: + parsed = json.loads(model_info) + except (TypeError, ValueError): + return None + if isinstance(parsed, dict): + value = parsed.get("team_public_model_name") + return value if isinstance(value, str) else None + return None + old_public_name = ( db_model.model_info.team_public_model_name if db_model.model_info else None ) @@ -495,8 +512,7 @@ async def _update_existing_team_model_assignment( d for d in response if d.model_name != db_model.model_name - and (d.model_info or {}).get("team_public_model_name") - == old_public_name + and _get_team_public_model_name(d.model_info) == old_public_name ] if not other_deployments_with_old_name: diff --git a/tests/proxy_unit_tests/test_proxy_utils.py b/tests/proxy_unit_tests/test_proxy_utils.py index 00d4cd24e4b..5e75890388c 100644 --- a/tests/proxy_unit_tests/test_proxy_utils.py +++ b/tests/proxy_unit_tests/test_proxy_utils.py @@ -2044,6 +2044,49 @@ def test_update_model_if_team_alias_exists(data, user_api_key_dict, expected_mod assert test_data.get("model") == expected_model +def test_team_alias_stale_bypass_disabled_by_default(): + from litellm.proxy.litellm_pre_call_utils import _update_model_if_team_alias_exists + + class _MockRouter: + team_model_to_deployment_indices = {("team-1", "gpt-4o"): [0]} + + test_data = {"model": "gpt-4o"} + user_api_key_dict = UserAPIKeyAuth( + api_key="test_key", + team_id="team-1", + team_model_aliases={"gpt-4o": "model_name_team-1_legacy-uuid"}, + ) + + with patch("litellm.proxy.proxy_server.llm_router", _MockRouter()): + _update_model_if_team_alias_exists( + data=test_data, user_api_key_dict=user_api_key_dict + ) + + assert test_data.get("model") == "model_name_team-1_legacy-uuid" + + +def test_team_alias_stale_bypass_enabled_by_flag(monkeypatch): + from litellm.proxy.litellm_pre_call_utils import _update_model_if_team_alias_exists + + class _MockRouter: + team_model_to_deployment_indices = {("team-1", "gpt-4o"): [0]} + + test_data = {"model": "gpt-4o"} + user_api_key_dict = UserAPIKeyAuth( + api_key="test_key", + team_id="team-1", + team_model_aliases={"gpt-4o": "model_name_team-1_legacy-uuid"}, + ) + monkeypatch.setenv("LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS", "true") + + with patch("litellm.proxy.proxy_server.llm_router", _MockRouter()): + _update_model_if_team_alias_exists( + data=test_data, user_api_key_dict=user_api_key_dict + ) + + assert test_data.get("model") == "gpt-4o" + + @pytest.fixture def mock_prisma_client(): client = MagicMock() diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 09410c19d34..83e6b0c93a5 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -70,10 +70,23 @@ class MockPrismaClient: # Filter deployments by team_id if specified if team_id_filter: + + def _get_team_id(model_info): + if isinstance(model_info, dict): + return model_info.get("team_id") + if isinstance(model_info, str): + try: + parsed = json.loads(model_info) + except (TypeError, ValueError): + return None + if isinstance(parsed, dict): + return parsed.get("team_id") + return None + return [ d for d in self.sibling_deployments - if d.model_info.get("team_id") == team_id_filter + if _get_team_id(d.model_info) == team_id_filter ] return self.sibling_deployments @@ -831,6 +844,61 @@ class TestTeamModelUpdate: # team_model_add should be called to add new public name mock_add.assert_called_once() + @pytest.mark.asyncio + async def test_rename_handles_legacy_string_model_info(self): + """Test rename path handles legacy string-encoded model_info rows without crashing.""" + from unittest.mock import MagicMock + + from litellm.proxy.management_endpoints.model_management_endpoints import ( + _update_existing_team_model_assignment, + ) + from litellm.types.router import ModelInfo + + db_model = Deployment( + model_name="model_name_team_123_uuid1", + litellm_params=LiteLLM_Params(model="azure/gpt-4o-mini"), + model_info=ModelInfo( + team_id="team_123", team_public_model_name="old-public-name" + ), + ) + + sibling_deployment = MagicMock() + sibling_deployment.model_name = "model_name_team_123_uuid2" + sibling_deployment.model_info = ( + '{"team_id":"team_123","team_public_model_name":"old-public-name"}' + ) + + prisma_client = MockPrismaClient( + team_exists=True, sibling_deployments=[sibling_deployment] + ) + + patch_data = updateDeployment( + model_name="new-public-name", + model_info=ModelInfo(team_id="team_123"), + ) + + user_api_key_dict = UserAPIKeyAuth( + user_id="test_user", + user_role=LitellmUserRoles.PROXY_ADMIN, + ) + + with patch( + "litellm.proxy.management_endpoints.model_management_endpoints.team_model_delete" + ) as mock_delete, patch( + "litellm.proxy.management_endpoints.model_management_endpoints.team_model_add" + ) as mock_add: + await _update_existing_team_model_assignment( + team_id="team_123", + public_model_name="new-public-name", + db_model=db_model, + patch_data=patch_data, + user_api_key_dict=user_api_key_dict, + prisma_client=prisma_client, # type: ignore + ) + + mock_delete.assert_not_called() + mock_add.assert_called_once() + @pytest.mark.asyncio async def test_patch_model_with_team_id_validates_permissions(self): """Test PATCH with team_id runs same validation as POST for team permissions""" From 303072dc44ed6a1b0d4b0d7decde03cf7af33e63 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 19:16:38 +0530 Subject: [PATCH 037/117] Fix greptile comments --- .../model_management_endpoints.py | 27 +++++++++++-------- litellm/router.py | 4 +++ 2 files changed, 20 insertions(+), 11 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 40f4d722dc6..1acedfb346a 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -468,7 +468,13 @@ async def _update_existing_team_model_assignment( user_api_key_dict: UserAPIKeyAuth, prisma_client: Optional[PrismaClient], ) -> None: - """Update an existing team model if the public name changed.""" + """Update an existing team model if the public name changed. + + Note on DB scan: Prisma's JSON filtering does not support compound AND conditions + across multiple JSON paths, so we fetch all deployments for the team and filter + team_public_model_name in Python. For teams with many deployments this scan grows + linearly; if team deployment counts become large this should be revisited. + """ def _get_team_public_model_name( model_info: Optional[Union[dict, str]] @@ -496,10 +502,6 @@ async def _update_existing_team_model_assignment( "prisma_client not initialized; skipping old public name cleanup to preserve sibling deployments" ) else: - # Query DB for all deployments in this team, then filter by public name. - # Note: Prisma's JSON filtering doesn't support compound AND conditions - # across multiple JSON paths, so we filter team_public_model_name in Python. - # For most teams (typically <100 deployments), this is acceptable. response = await prisma_client.db.litellm_proxymodeltable.find_many( where={ "model_info": { @@ -508,12 +510,15 @@ async def _update_existing_team_model_assignment( } } ) - other_deployments_with_old_name = [ - d - for d in response - if d.model_name != db_model.model_name - and _get_team_public_model_name(d.model_info) == old_public_name - ] + if not response: + other_deployments_with_old_name = [] + else: + other_deployments_with_old_name = [ + d + for d in response + if d.model_name != db_model.model_name + and _get_team_public_model_name(d.model_info) == old_public_name + ] if not other_deployments_with_old_name: await team_model_delete( diff --git a/litellm/router.py b/litellm/router.py index 1a57d2f8c0d..ac965a0af5b 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8258,6 +8258,10 @@ class Router: # O(k) where k = team deployments for this model_name (typically 1-10) for idx in indices: model = self.model_list[idx] + if not self.should_include_deployment( + model_name=model_name, model=model, team_id=team_id + ): + continue if model_alias is not None: alias_model = model.copy() alias_model["model_name"] = model_alias From fc6865c3a3c460a3b01ba6ebe84718e5d5f0046f Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 19:37:05 +0530 Subject: [PATCH 038/117] Fix greptile comments --- litellm/router.py | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index ac965a0af5b..19a0f250dc1 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7183,8 +7183,8 @@ class Router: - model: dict - the deployment to index - idx: int - the index in model_list """ - team_id = model.get("model_info", {}).get("team_id") - team_public_model_name = model.get("model_info", {}).get( + team_id = (model.get("model_info") or {}).get("team_id") + team_public_model_name = (model.get("model_info") or {}).get( "team_public_model_name" ) if team_id and team_public_model_name: @@ -7242,7 +7242,10 @@ class Router: ) if _deployment_on_router is not None: # deployment with this model_id exists on the router - if deployment.litellm_params == _deployment_on_router.litellm_params: + if ( + deployment.litellm_params == _deployment_on_router.litellm_params + and deployment.model_info == _deployment_on_router.model_info + ): # No need to update return None @@ -8268,7 +8271,8 @@ class Router: returned_models.append(alias_model) else: returned_models.append(model) - return returned_models + if returned_models: + return returned_models # O(1) lookup in model_name index if model_name in self.model_name_to_deployment_indices: From d02a70ab4e632f4bec3e7e5ab42c2e6ebf2fa74b Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 19:50:33 +0530 Subject: [PATCH 039/117] Fix greptile comments --- .../model_management_endpoints.py | 57 ++++++++++--------- litellm/router.py | 10 +++- 2 files changed, 36 insertions(+), 31 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 1acedfb346a..d8a1075a168 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -499,36 +499,37 @@ async def _update_existing_team_model_assignment( if old_public_name and public_model_name != old_public_name: if prisma_client is None: verbose_proxy_logger.warning( - "prisma_client not initialized; skipping old public name cleanup to preserve sibling deployments" + "prisma_client not initialized; skipping public name update entirely to avoid orphaned entries" ) - else: - response = await prisma_client.db.litellm_proxymodeltable.find_many( - where={ - "model_info": { - "path": ["team_id"], - "equals": team_id, - } - } - ) - if not response: - other_deployments_with_old_name = [] - else: - other_deployments_with_old_name = [ - d - for d in response - if d.model_name != db_model.model_name - and _get_team_public_model_name(d.model_info) == old_public_name - ] + return - if not other_deployments_with_old_name: - await team_model_delete( - data=TeamModelDeleteRequest( - team_id=team_id, - models=[old_public_name], - ), - http_request=Request(scope={"type": "http"}), - user_api_key_dict=user_api_key_dict, - ) + response = await prisma_client.db.litellm_proxymodeltable.find_many( + where={ + "model_info": { + "path": ["team_id"], + "equals": team_id, + } + } + ) + if not response: + other_deployments_with_old_name = [] + else: + other_deployments_with_old_name = [ + d + for d in response + if d.model_name != db_model.model_name + and _get_team_public_model_name(d.model_info) == old_public_name + ] + + if not other_deployments_with_old_name: + await team_model_delete( + data=TeamModelDeleteRequest( + team_id=team_id, + models=[old_public_name], + ), + http_request=Request(scope={"type": "http"}), + user_api_key_dict=user_api_key_dict, + ) await team_model_add( data=TeamModelAddRequest( diff --git a/litellm/router.py b/litellm/router.py index 19a0f250dc1..76c13443e72 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8227,12 +8227,16 @@ class Router: """ if ( team_id is not None - and model["model_info"].get("team_id") == team_id - and model_name == model["model_info"].get("team_public_model_name") + and (model.get("model_info") or {}).get("team_id") == team_id + and model_name + == (model.get("model_info") or {}).get("team_public_model_name") ): return True elif model_name is not None and model["model_name"] == model_name: - if team_id is None or model["model_info"].get("team_id") == team_id: + if ( + team_id is None + or (model.get("model_info") or {}).get("team_id") == team_id + ): return True return False From 316a742945494ae9784591d6ecb6f2d314a0428c Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 20:12:06 +0530 Subject: [PATCH 040/117] Fix greptile comments --- .../model_management_endpoints.py | 19 +++++---- litellm/router.py | 4 +- .../test_model_management_endpoints.py | 41 +++++++++++++++++++ 3 files changed, 54 insertions(+), 10 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index d8a1075a168..721a8c2a1c4 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -521,6 +521,16 @@ async def _update_existing_team_model_assignment( and _get_team_public_model_name(d.model_info) == old_public_name ] + # Add new name first, then delete old name to prevent access loss on partial failure + await team_model_add( + data=TeamModelAddRequest( + team_id=team_id, + models=[public_model_name], + ), + http_request=Request(scope={"type": "http"}), + user_api_key_dict=user_api_key_dict, + ) + if not other_deployments_with_old_name: await team_model_delete( data=TeamModelDeleteRequest( @@ -531,15 +541,6 @@ async def _update_existing_team_model_assignment( user_api_key_dict=user_api_key_dict, ) - await team_model_add( - data=TeamModelAddRequest( - team_id=team_id, - models=[public_model_name], - ), - http_request=Request(scope={"type": "http"}), - user_api_key_dict=user_api_key_dict, - ) - patch_data.model_name = None diff --git a/litellm/router.py b/litellm/router.py index 76c13443e72..d7f5d42eac7 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8233,9 +8233,11 @@ class Router: ): return True elif model_name is not None and model["model_name"] == model_name: + model_team_id = (model.get("model_info") or {}).get("team_id") if ( team_id is None - or (model.get("model_info") or {}).get("team_id") == team_id + or model_team_id is None # global deployment - accessible to all teams + or model_team_id == team_id ): return True return False diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 83e6b0c93a5..2dd29fd5c9c 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -712,6 +712,15 @@ class TestTeamModelSiblingRouting: "team_public_model_name": public_name, }, }, + { + "model_name": "global-gpt-4o", + "litellm_params": { + "model": "azure/gpt-4o", + "api_key": "global-key", + "api_base": "https://global.openai.azure.com", + }, + "model_info": {}, # No team_id - global deployment + }, ], ) @@ -732,6 +741,38 @@ class TestTeamModelSiblingRouting: "https://westus.openai.azure.com", } + def test_global_deployments_accessible_to_teams(self): + """Test that global deployments (no team_id) are accessible to all teams""" + import litellm + + router = litellm.Router( + model_list=[ + { + "model_name": "global-gpt-4o", + "litellm_params": { + "model": "azure/gpt-4o", + "api_key": "global-key", + "api_base": "https://global.openai.azure.com", + }, + "model_info": {}, # No team_id - global deployment + }, + ], + ) + + # Global deployment should be accessible when team_id is provided + deployments = router._get_all_deployments( + model_name="global-gpt-4o", team_id="teamA" + ) + assert len(deployments) == 1 + assert deployments[0]["model_name"] == "global-gpt-4o" + + # should_include_deployment should return True for global deployments + assert router.should_include_deployment( + model_name="global-gpt-4o", + model={"model_name": "global-gpt-4o", "model_info": {}}, + team_id="teamA", + ) + class TestTeamModelUpdate: """Test team model update handles team_id consistently with model creation""" From 9a0a21619514ba11a22153ebd5f725cfd5d1fd30 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 22:02:50 +0530 Subject: [PATCH 041/117] Fix code qa issues --- .../model_management_endpoints.py | 2 -- .../test_router_index_management.py | 22 +++++++++++++++++++ 2 files changed, 22 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 721a8c2a1c4..8f6b8a626e4 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -33,7 +33,6 @@ from litellm.proxy._types import ( ProxyException, TeamModelAddRequest, TeamModelDeleteRequest, - UpdateTeamRequest, UserAPIKeyAuth, ) from litellm.proxy.auth.user_api_key_auth import user_api_key_auth @@ -42,7 +41,6 @@ from litellm.proxy.management_endpoints.common_utils import _is_user_team_admin from litellm.proxy.management_endpoints.team_endpoints import ( team_model_add, team_model_delete, - update_team, ) from litellm.proxy.management_helpers.audit_logs import create_object_audit_log from litellm.proxy.utils import PrismaClient diff --git a/tests/router_unit_tests/test_router_index_management.py b/tests/router_unit_tests/test_router_index_management.py index 90d98b8ab0a..2694c62827c 100644 --- a/tests/router_unit_tests/test_router_index_management.py +++ b/tests/router_unit_tests/test_router_index_management.py @@ -118,6 +118,28 @@ class TestRouterIndexManagement: assert router.model_id_to_deployment_index_map["id-2"] == 1 assert router.model_id_to_deployment_index_map["id-3"] == 2 + def test_update_team_model_index(self, router): + """Test _update_team_model_index updates team_model_to_deployment_indices.""" + model = { + "model_name": "team-alias", + "model_info": { + "id": "dep-1", + "team_id": "team-abc", + "team_public_model_name": "gpt-4o", + }, + } + router._update_team_model_index(model, 0) + assert router.team_model_to_deployment_indices[("team-abc", "gpt-4o")] == [0] + router._update_team_model_index(model, 2) + assert router.team_model_to_deployment_indices[("team-abc", "gpt-4o")] == [0, 2] + + router._update_team_model_index( + {"model_name": "x", "model_info": {"id": "dep-2"}}, 5 + ) + assert router.team_model_to_deployment_indices == { + ("team-abc", "gpt-4o"): [0, 2], + } + def test_has_model_id(self, router): """Test has_model_id function for O(1) membership check""" # Setup: Add models to router From c6cc0341f61836e74ee41a6afba197268e91fe99 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 22:26:08 +0530 Subject: [PATCH 042/117] Fix greptile reviews and mock test --- .../model_management_endpoints.py | 11 ++++ .../test_model_management_endpoints.py | 51 +++++++++++++++---- 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 8f6b8a626e4..5952aede853 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -538,6 +538,17 @@ async def _update_existing_team_model_assignment( http_request=Request(scope={"type": "http"}), user_api_key_dict=user_api_key_dict, ) + elif not old_public_name and public_model_name: + # First-time assignment of public name on an existing team deployment: + # ensure the team's models list is updated so team routing can resolve it. + await team_model_add( + data=TeamModelAddRequest( + team_id=team_id, + models=[public_model_name], + ), + http_request=Request(scope={"type": "http"}), + user_api_key_dict=user_api_key_dict, + ) patch_data.model_name = None diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 2dd29fd5c9c..751d0a02ff0 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -635,8 +635,6 @@ class TestTeamModelSiblingRouting: team_id = "team_no_alias" public_name = "gpt-4.1-mini" - mock_update_team = AsyncMock() - async def mock_add_model_to_db(model_params, user_api_key_dict, prisma_client): return MagicMock(model_id=str(uuid.uuid4())) @@ -656,9 +654,6 @@ class TestTeamModelSiblingRouting: model_info=ModelInfo(team_id=team_id), ) with patch( - "litellm.proxy.management_endpoints.model_management_endpoints.update_team", - mock_update_team, - ), patch( "litellm.proxy.management_endpoints.model_management_endpoints._add_model_to_db", side_effect=mock_add_model_to_db, ), patch( @@ -671,7 +666,6 @@ class TestTeamModelSiblingRouting: prisma_client=prisma_client, ) - mock_update_team.assert_not_called() assert mock_team_model_add.call_count == 2 @pytest.mark.asyncio @@ -807,8 +801,6 @@ class TestTeamModelUpdate: "litellm.proxy.proxy_server.premium_user", True, ), patch( - "litellm.proxy.management_endpoints.model_management_endpoints.update_team" - ) as mock_update_team, patch( "litellm.proxy.management_endpoints.model_management_endpoints.team_model_add" ) as mock_team_model_add: result = await _update_team_model_in_db( @@ -820,8 +812,6 @@ class TestTeamModelUpdate: assert result.get("model_name", "").startswith("model_name_test_team_123_") assert "team_public_model_name" in str(result.get("model_info", "")) - # update_team must not be called (no model_aliases writes for team models) - mock_update_team.assert_not_called() # team_model_add must be called to add public name to team's models list mock_team_model_add.assert_called_once() @@ -885,6 +875,47 @@ class TestTeamModelUpdate: # team_model_add should be called to add new public name mock_add.assert_called_once() + @pytest.mark.asyncio + async def test_first_time_public_name_assignment_adds_team_model(self): + """If existing team deployment had no public name, first assignment must call team_model_add.""" + from litellm.proxy.management_endpoints.model_management_endpoints import ( + _update_existing_team_model_assignment, + ) + from litellm.types.router import ModelInfo + + db_model = Deployment( + model_name="model_name_team_123_uuid1", + litellm_params=LiteLLM_Params(model="azure/gpt-4o-mini"), + model_info=ModelInfo(team_id="team_123"), + ) + + patch_data = updateDeployment( + model_name="new-public-name", + model_info=ModelInfo(team_id="team_123"), + ) + + user_api_key_dict = UserAPIKeyAuth( + user_id="test_user", + user_role=LitellmUserRoles.PROXY_ADMIN, + ) + + with patch( + "litellm.proxy.management_endpoints.model_management_endpoints.team_model_delete" + ) as mock_delete, patch( + "litellm.proxy.management_endpoints.model_management_endpoints.team_model_add" + ) as mock_add: + await _update_existing_team_model_assignment( + team_id="team_123", + public_model_name="new-public-name", + db_model=db_model, + patch_data=patch_data, + user_api_key_dict=user_api_key_dict, + prisma_client=None, + ) + + mock_add.assert_called_once() + mock_delete.assert_not_called() + @pytest.mark.asyncio async def test_rename_handles_legacy_string_model_info(self): """Test rename path handles legacy string-encoded model_info rows without crashing.""" From fb8d9c2e9a33a2a653335ed095526c29ef7f1417 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 22:38:54 +0530 Subject: [PATCH 043/117] Fix greptile reviews and mock test --- docs/my-website/docs/proxy/config_settings.md | 1 + docs/my-website/docs/proxy/load_balancing.md | 15 ++++++++++++ litellm/proxy/litellm_pre_call_utils.py | 23 +++++++++++++++---- litellm/router.py | 6 +++++ 4 files changed, 40 insertions(+), 5 deletions(-) diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md index 02f5c2be9c7..dce979ab89f 100644 --- a/docs/my-website/docs/proxy/config_settings.md +++ b/docs/my-website/docs/proxy/config_settings.md @@ -804,6 +804,7 @@ router_settings: | LITELLM_OTEL_INTEGRATION_ENABLE_EVENTS | Optionally enable semantic logs for OTEL | LITELLM_OTEL_INTEGRATION_ENABLE_METRICS | Optionally enable emantic metrics for OTEL | LITELLM_ENABLE_PYROSCOPE | If true, enables Pyroscope CPU profiling. Profiles are sent to PYROSCOPE_SERVER_ADDRESS. Off by default. See [Pyroscope profiling](/proxy/pyroscope_profiling). +| LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS | When `true`, if a team's legacy `model_aliases` entry maps a public model name to an internal `model_name__` deployment, pre-call handling can skip that rewrite when team-scoped sibling deployments exist for the public name—so load balancing / `order` apply across siblings. Default is `false` for backwards compatibility. See [Team-scoped models and legacy aliases](./load_balancing#team-scoped-models-and-legacy-model_aliases). When stale aliases are detected and this flag is off, the proxy may log a one-time warning. | PYROSCOPE_APP_NAME | Application name reported to Pyroscope. Required when LITELLM_ENABLE_PYROSCOPE is true. No default. | PYROSCOPE_SERVER_ADDRESS | Pyroscope server URL to send profiles to. Required when LITELLM_ENABLE_PYROSCOPE is true. No default. | PYROSCOPE_SAMPLE_RATE | Optional. Sample rate for Pyroscope profiling (integer). No default; when unset, the pyroscope-io library default is used. diff --git a/docs/my-website/docs/proxy/load_balancing.md b/docs/my-website/docs/proxy/load_balancing.md index 74b3e8a5117..897c04b2b00 100644 --- a/docs/my-website/docs/proxy/load_balancing.md +++ b/docs/my-website/docs/proxy/load_balancing.md @@ -336,6 +336,21 @@ The `order` parameter requires `enable_pre_call_checks: true` in `router_setting If `order=1` deployment is unavailable (e.g., rate-limited), the router falls back to `order=2` deployments. +### Team-scoped models and legacy `model_aliases` {#team-scoped-models-and-legacy-model_aliases} + +Team-scoped deployments are identified by `model_info.team_id` and `model_info.team_public_model_name`. Requests should use the **public** model name; the router resolves all sibling deployments (same public name, different `api_base` / `order`, etc.) for routing, failover, and deployment `order`. + +For router internals: when a `team_id` is in scope, optimized lookups key off `(team_id, team_public_model_name)`. If code passes an internal deployment id (e.g. `model_name__`) instead of the public name, routing still works via the usual deployment-name paths, but the team-specific fast path applies only to the public name. + +**Legacy teams:** Older proxy versions could persist `model_aliases` on the team row mapping a public name to a single internal deployment id (`model_name__`). On each request, pre-call logic may still rewrite `model` to that internal name **before** routing, which collapses to one deployment and can make newer sibling deployments unreachable. + +**Migration options:** + +1. **Recommended for upgrades:** Set environment variable `LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS=true` so that when sibling team deployments exist for the public name, the stale alias rewrite is skipped and team-scoped routing (including `order` and failover) applies. See the [Environment variables](./config_settings) table in the proxy settings doc. +2. **Data cleanup:** Remove obsolete `model_aliases` entries for team public names from the team record in the database so only `team_public_model_name` + team model list drive access. + +If a stale alias is detected and the bypass is **not** enabled, the proxy may emit a **one-time** warning in logs explaining that sibling deployments may be unreachable until the flag is set or aliases are cleaned up. + ### When You'll See Load Balancing in Action **Immediate Effects:** diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 4a12a0a5774..64ef2405002 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -37,6 +37,7 @@ from litellm.types.utils import ( ) service_logger_obj = ServiceLogging() # used for tracking latency on OTEL +_STALE_TEAM_ALIAS_WARNING_KEYS: set[str] = set() if TYPE_CHECKING: @@ -1321,16 +1322,28 @@ def _update_model_if_team_alias_exists( ) # Check if the alias points to a team-scoped UUID name # (format: "model_name_{team_id}_{uuid}") - if enable_stale_alias_bypass and aliased_target.startswith( + is_stale_team_alias = aliased_target.startswith( f"model_name_{user_api_key_dict.team_id}_" - ): + ) + if is_stale_team_alias and llm_router: # This is a stale alias from pre-PR deployments. # Check if current team deployments exist for the public name. - if llm_router: - key = (user_api_key_dict.team_id, _model) - if key in llm_router.team_model_to_deployment_indices: + key = (user_api_key_dict.team_id, _model) + if key in llm_router.team_model_to_deployment_indices: + if enable_stale_alias_bypass: # Team deployments exist; skip stale alias return + warning_key = f"{user_api_key_dict.team_id}:{_model}:{aliased_target}" + if warning_key not in _STALE_TEAM_ALIAS_WARNING_KEYS: + _STALE_TEAM_ALIAS_WARNING_KEYS.add(warning_key) + verbose_proxy_logger.warning( + "Stale team model alias detected for model='%s', team_id='%s'. " + "New sibling deployments may be unreachable. " + "Set LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS=true to enable " + "team-scoped sibling routing.", + _model, + user_api_key_dict.team_id, + ) data["model"] = aliased_target return diff --git a/litellm/router.py b/litellm/router.py index d7f5d42eac7..64d29d8bceb 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8256,6 +8256,12 @@ class Router: if team_id specified, only return team-specific models Optimized with O(1) index lookup instead of O(n) linear scan. + + Note: when team_id is provided, O(1) lookup in + `team_model_to_deployment_indices` only applies when `model_name` is the + team public model name. If a caller passes an internal deployment model + name (for example, `model_name__`), this method falls back + to the standard model-name index / scan path. """ returned_models: List[DeploymentTypedDict] = [] From 1a0b30aaac029808cd25029837306d7e10f228eb Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 23 Mar 2026 22:49:57 +0530 Subject: [PATCH 044/117] Fix greptile reviews and mock test --- litellm/proxy/litellm_pre_call_utils.py | 12 +++++-- .../model_management_endpoints.py | 3 ++ .../test_model_management_endpoints.py | 35 +++++++++++++++++++ 3 files changed, 48 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 64ef2405002..f8f299b1481 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -1,6 +1,7 @@ import asyncio import copy import time +from collections import OrderedDict from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union from fastapi import Request @@ -37,7 +38,9 @@ from litellm.types.utils import ( ) service_logger_obj = ServiceLogging() # used for tracking latency on OTEL -_STALE_TEAM_ALIAS_WARNING_KEYS: set[str] = set() +# Bounded dedup for stale-alias warnings (FIFO eviction when over cap). +_MAX_STALE_ALIAS_WARNING_KEYS = 10_000 +_STALE_TEAM_ALIAS_WARNING_KEYS: OrderedDict[str, None] = OrderedDict() if TYPE_CHECKING: @@ -1335,7 +1338,12 @@ def _update_model_if_team_alias_exists( return warning_key = f"{user_api_key_dict.team_id}:{_model}:{aliased_target}" if warning_key not in _STALE_TEAM_ALIAS_WARNING_KEYS: - _STALE_TEAM_ALIAS_WARNING_KEYS.add(warning_key) + _STALE_TEAM_ALIAS_WARNING_KEYS[warning_key] = None + while ( + len(_STALE_TEAM_ALIAS_WARNING_KEYS) + > _MAX_STALE_ALIAS_WARNING_KEYS + ): + _STALE_TEAM_ALIAS_WARNING_KEYS.popitem(last=False) verbose_proxy_logger.warning( "Stale team model alias detected for model='%s', team_id='%s'. " "New sibling deployments may be unreachable. " diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 5952aede853..95c44a431b5 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -495,6 +495,9 @@ async def _update_existing_team_model_assignment( ) if old_public_name and public_model_name != old_public_name: + # Clear user-supplied public name from patch before any early return so the + # caller does not overwrite the internal UUID-based model_name in the DB. + patch_data.model_name = None if prisma_client is None: verbose_proxy_logger.warning( "prisma_client not initialized; skipping public name update entirely to avoid orphaned entries" diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 751d0a02ff0..2a85e24d780 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -916,6 +916,41 @@ class TestTeamModelUpdate: mock_add.assert_called_once() mock_delete.assert_not_called() + @pytest.mark.asyncio + async def test_rename_with_prisma_none_clears_patch_model_name(self): + """Rename path must clear patch_data.model_name even when prisma is unavailable (P1).""" + from litellm.proxy.management_endpoints.model_management_endpoints import ( + _update_existing_team_model_assignment, + ) + from litellm.types.router import ModelInfo + + db_model = Deployment( + model_name="model_name_team_123_uuid1", + litellm_params=LiteLLM_Params(model="azure/gpt-4o-mini"), + model_info=ModelInfo( + team_id="team_123", team_public_model_name="old-public-name" + ), + ) + patch_data = updateDeployment( + model_name="new-public-name", + model_info=ModelInfo(team_id="team_123"), + ) + user_api_key_dict = UserAPIKeyAuth( + user_id="test_user", + user_role=LitellmUserRoles.PROXY_ADMIN, + ) + + await _update_existing_team_model_assignment( + team_id="team_123", + public_model_name="new-public-name", + db_model=db_model, + patch_data=patch_data, + user_api_key_dict=user_api_key_dict, + prisma_client=None, + ) + + assert patch_data.model_name is None + @pytest.mark.asyncio async def test_rename_handles_legacy_string_model_info(self): """Test rename path handles legacy string-encoded model_info rows without crashing.""" From 592ac98ddc1154a00115528cada41654a0091cdc Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 24 Mar 2026 20:10:57 +0530 Subject: [PATCH 045/117] fix(router): address Greptile P1/P2 review comments - Add deduplication guard in _update_team_model_index to prevent duplicate indices - Add wildcard comment in map_team_model for clarity - Add monkeypatch to test_team_alias_stale_bypass_disabled_by_default for determinism - Extract _get_team_deployments helper to centralize DB access pattern - Add clarifying comments for team_public_model_name assignment ordering Made-with: Cursor --- .../model_management_endpoints.py | 53 ++++++++++++------- litellm/router.py | 5 +- tests/proxy_unit_tests/test_proxy_utils.py | 3 +- 3 files changed, 40 insertions(+), 21 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 95c44a431b5..4ab52ac5c0b 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -329,12 +329,17 @@ async def _add_team_model_to_db( _team_id = model_params.model_info.team_id if _team_id is None: return None + # Capture the original public name before mutating model_params.model_name original_model_name = model_params.model_name + + # Generate unique internal model_name for team-scoped deployment + unique_model_name = f"model_name_{_team_id}_{uuid.uuid4()}" + + # Store public name in model_info BEFORE overwriting model_name + # so _add_model_to_db serializes the correct team_public_model_name if original_model_name: model_params.model_info.team_public_model_name = original_model_name - unique_model_name = f"model_name_{_team_id}_{uuid.uuid4()}" - model_params.model_name = unique_model_name ## CREATE MODEL IN DB ## @@ -458,6 +463,25 @@ async def _setup_new_team_model_assignment( ) +async def _get_team_deployments( + team_id: str, prisma_client: PrismaClient +) -> List[LiteLLM_ProxyModelTable]: + """ + Fetch all deployments for a given team_id from the database. + + Centralizes team deployment queries to ensure consistent filtering and error handling. + """ + response = await prisma_client.db.litellm_proxymodeltable.find_many( + where={ + "model_info": { + "path": ["team_id"], + "equals": team_id, + } + } + ) + return response if response else [] + + async def _update_existing_team_model_assignment( team_id: str, public_model_name: str, @@ -504,23 +528,14 @@ async def _update_existing_team_model_assignment( ) return - response = await prisma_client.db.litellm_proxymodeltable.find_many( - where={ - "model_info": { - "path": ["team_id"], - "equals": team_id, - } - } - ) - if not response: - other_deployments_with_old_name = [] - else: - other_deployments_with_old_name = [ - d - for d in response - if d.model_name != db_model.model_name - and _get_team_public_model_name(d.model_info) == old_public_name - ] + # Query DB for all team deployments to check for sibling deployments + team_deployments = await _get_team_deployments(team_id, prisma_client) + other_deployments_with_old_name = [ + d + for d in team_deployments + if d.model_name != db_model.model_name + and _get_team_public_model_name(d.model_info) == old_public_name + ] # Add new name first, then delete old name to prevent access loss on partial failure await team_model_add( diff --git a/litellm/router.py b/litellm/router.py index 64d29d8bceb..8f2785a3838 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7191,7 +7191,8 @@ class Router: key = (team_id, team_public_model_name) if key not in self.team_model_to_deployment_indices: self.team_model_to_deployment_indices[key] = [] - self.team_model_to_deployment_indices[key].append(idx) + if idx not in self.team_model_to_deployment_indices[key]: + self.team_model_to_deployment_indices[key].append(idx) def _add_model_to_list_and_index_map( self, model: dict, model_id: Optional[str] = None @@ -8217,6 +8218,8 @@ class Router: if model.get("model_info", {}).get("team_id") == team_id: return team_model_name + # No team-scoped deployment found; wildcard/pattern routes are + # handled downstream by the pattern_router in _common_checks_available_deployment. return None def should_include_deployment( diff --git a/tests/proxy_unit_tests/test_proxy_utils.py b/tests/proxy_unit_tests/test_proxy_utils.py index 5e75890388c..9bfb466c0eb 100644 --- a/tests/proxy_unit_tests/test_proxy_utils.py +++ b/tests/proxy_unit_tests/test_proxy_utils.py @@ -2044,7 +2044,8 @@ def test_update_model_if_team_alias_exists(data, user_api_key_dict, expected_mod assert test_data.get("model") == expected_model -def test_team_alias_stale_bypass_disabled_by_default(): +def test_team_alias_stale_bypass_disabled_by_default(monkeypatch): + monkeypatch.delenv("LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS", raising=False) from litellm.proxy.litellm_pre_call_utils import _update_model_if_team_alias_exists class _MockRouter: From 2321d7759916c321da686ad9e267bd4f645c8d49 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 24 Mar 2026 20:32:01 +0530 Subject: [PATCH 046/117] fix(router): address remaining Greptile review comments - Cache LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS at module level to avoid hot-path secret lookups - Add clarifying comments for should_include_deployment team isolation logic - Add negative assertion for update_team.assert_not_called() in test - Add docstring clarification for _get_team_deployments helper pattern - Add explicit assertion message in test_get_model_list_alias_optimization Made-with: Cursor --- litellm/proxy/litellm_pre_call_utils.py | 12 +++++++++--- .../model_management_endpoints.py | 4 ++++ litellm/router.py | 7 +++++-- .../test_get_model_list_alias_optimization.py | 8 ++++---- .../test_model_management_endpoints.py | 6 +++++- 5 files changed, 27 insertions(+), 10 deletions(-) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index f8f299b1481..a605f3ee23b 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -41,6 +41,8 @@ service_logger_obj = ServiceLogging() # used for tracking latency on OTEL # Bounded dedup for stale-alias warnings (FIFO eviction when over cap). _MAX_STALE_ALIAS_WARNING_KEYS = 10_000 _STALE_TEAM_ALIAS_WARNING_KEYS: OrderedDict[str, None] = OrderedDict() +# Cache the stale alias bypass flag at module load to avoid hot-path secret lookups +_ENABLE_TEAM_STALE_ALIAS_BYPASS: Optional[bool] = None if TYPE_CHECKING: @@ -1320,9 +1322,13 @@ def _update_model_if_team_alias_exists( # Optional bypass for stale aliases from pre-PR deployments: # only enabled via feature flag to preserve backwards compatibility. - enable_stale_alias_bypass = get_secret_bool( - "LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS", False - ) + # Cached at module level to avoid hot-path secret lookups on every request. + global _ENABLE_TEAM_STALE_ALIAS_BYPASS + if _ENABLE_TEAM_STALE_ALIAS_BYPASS is None: + _ENABLE_TEAM_STALE_ALIAS_BYPASS = get_secret_bool( + "LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS", False + ) + enable_stale_alias_bypass = _ENABLE_TEAM_STALE_ALIAS_BYPASS # Check if the alias points to a team-scoped UUID name # (format: "model_name_{team_id}_{uuid}") is_stale_team_alias = aliased_target.startswith( diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 4ab52ac5c0b..ab29dde6e38 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -470,6 +470,10 @@ async def _get_team_deployments( Fetch all deployments for a given team_id from the database. Centralizes team deployment queries to ensure consistent filtering and error handling. + This is the established helper pattern for team deployment DB access in this module. + + Note: Direct Prisma call is intentional here as this IS the helper function that + encapsulates the DB access pattern for team deployments. """ response = await prisma_client.db.litellm_proxymodeltable.find_many( where={ diff --git a/litellm/router.py b/litellm/router.py index 8f2785a3838..64ad6fc2215 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -8236,13 +8236,16 @@ class Router: ): return True elif model_name is not None and model["model_name"] == model_name: + # Fallback: check by internal model_name for non-team deployments + # or deployments that haven't been migrated to team_public_model_name yet model_team_id = (model.get("model_info") or {}).get("team_id") if ( - team_id is None + team_id is None # requester has no team constraint or model_team_id is None # global deployment - accessible to all teams - or model_team_id == team_id + or model_team_id == team_id # deployment belongs to requester's team ): return True + # No match: deployment is for a different team or doesn't match the requested model return False def _get_all_deployments( diff --git a/tests/router_unit_tests/test_get_model_list_alias_optimization.py b/tests/router_unit_tests/test_get_model_list_alias_optimization.py index 62baf0a3d22..2c2df3be945 100644 --- a/tests/router_unit_tests/test_get_model_list_alias_optimization.py +++ b/tests/router_unit_tests/test_get_model_list_alias_optimization.py @@ -44,7 +44,7 @@ def test_map_team_model_should_not_iterate_aliases_for_non_alias_team_model_name {f"alias-{idx}": "gpt-4" for idx in range(200)} ) - assert ( - router.map_team_model(team_model_name="team-model", team_id="team-1") - == "team-model" - ) + # map_team_model should return the public name unchanged (not the internal UUID name) + # so the router can find all sibling deployments via team_id filtering + result = router.map_team_model(team_model_name="team-model", team_id="team-1") + assert result == "team-model", f"Expected public name 'team-model', got {result}" diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py index 2a85e24d780..2e566ab6222 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_model_management_endpoints.py @@ -802,7 +802,9 @@ class TestTeamModelUpdate: True, ), patch( "litellm.proxy.management_endpoints.model_management_endpoints.team_model_add" - ) as mock_team_model_add: + ) as mock_team_model_add, patch( + "litellm.proxy.management_endpoints.model_management_endpoints.update_team" + ) as mock_update_team: result = await _update_team_model_in_db( db_model=db_model, patch_data=patch_data, @@ -814,6 +816,8 @@ class TestTeamModelUpdate: assert "team_public_model_name" in str(result.get("model_info", "")) # team_model_add must be called to add public name to team's models list mock_team_model_add.assert_called_once() + # update_team (model_aliases write) must NOT be called in the new implementation + mock_update_team.assert_not_called() @pytest.mark.asyncio async def test_rename_preserves_old_name_when_siblings_exist(self): From 7436f889caff61876880217362ddcac5b3773b23 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 24 Mar 2026 20:34:12 +0530 Subject: [PATCH 047/117] fix(router): address final Greptile P1/P2 comments - Reorder team_public_model_name assignment to happen before model_name mutation for clarity - Add comment explaining no-rename fast-exit case in _update_existing_team_model_assignment - Add comment explaining final patch_data.model_name = None applies to all code paths Made-with: Cursor --- .../model_management_endpoints.py | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index ab29dde6e38..355e2011237 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -329,17 +329,19 @@ async def _add_team_model_to_db( _team_id = model_params.model_info.team_id if _team_id is None: return None - # Capture the original public name before mutating model_params.model_name + + # Capture the original public name FIRST, before any mutations original_model_name = model_params.model_name - # Generate unique internal model_name for team-scoped deployment - unique_model_name = f"model_name_{_team_id}_{uuid.uuid4()}" - - # Store public name in model_info BEFORE overwriting model_name - # so _add_model_to_db serializes the correct team_public_model_name + # Set team_public_model_name in model_info using the captured original_model_name + # This must happen BEFORE mutating model_params.model_name so _add_model_to_db + # serializes the correct team_public_model_name (not the internal UUID name) if original_model_name: model_params.model_info.team_public_model_name = original_model_name + # Generate and assign unique internal model_name LAST + # (after team_public_model_name is safely stored) + unique_model_name = f"model_name_{_team_id}_{uuid.uuid4()}" model_params.model_name = unique_model_name ## CREATE MODEL IN DB ## @@ -571,7 +573,11 @@ async def _update_existing_team_model_assignment( http_request=Request(scope={"type": "http"}), user_api_key_dict=user_api_key_dict, ) + # else: old_public_name == public_model_name (no rename needed) + # No team_model_add/delete calls required; public name is already registered + # Always clear patch_data.model_name to prevent caller from overwriting + # the internal UUID-based model_name in the DB with the user-supplied public name patch_data.model_name = None From 1fac58abb370596e88d2048e24116879979450b2 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 24 Mar 2026 20:41:46 +0530 Subject: [PATCH 048/117] fix(tests): reset module-level cache in stale alias bypass tests Reset _ENABLE_TEAM_STALE_ALIAS_BYPASS to None in both test functions to ensure test isolation and prevent ordering-dependent failures Made-with: Cursor --- tests/proxy_unit_tests/test_proxy_utils.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/tests/proxy_unit_tests/test_proxy_utils.py b/tests/proxy_unit_tests/test_proxy_utils.py index 9bfb466c0eb..6c6ec7bcd60 100644 --- a/tests/proxy_unit_tests/test_proxy_utils.py +++ b/tests/proxy_unit_tests/test_proxy_utils.py @@ -2046,7 +2046,11 @@ def test_update_model_if_team_alias_exists(data, user_api_key_dict, expected_mod def test_team_alias_stale_bypass_disabled_by_default(monkeypatch): monkeypatch.delenv("LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS", raising=False) + import litellm.proxy.litellm_pre_call_utils as pre_call_utils from litellm.proxy.litellm_pre_call_utils import _update_model_if_team_alias_exists + + # Reset module-level cache to ensure test isolation + pre_call_utils._ENABLE_TEAM_STALE_ALIAS_BYPASS = None class _MockRouter: team_model_to_deployment_indices = {("team-1", "gpt-4o"): [0]} @@ -2067,7 +2071,11 @@ def test_team_alias_stale_bypass_disabled_by_default(monkeypatch): def test_team_alias_stale_bypass_enabled_by_flag(monkeypatch): + import litellm.proxy.litellm_pre_call_utils as pre_call_utils from litellm.proxy.litellm_pre_call_utils import _update_model_if_team_alias_exists + + # Reset module-level cache to ensure test isolation + pre_call_utils._ENABLE_TEAM_STALE_ALIAS_BYPASS = None class _MockRouter: team_model_to_deployment_indices = {("team-1", "gpt-4o"): [0]} From 15f5dc38c47aaf8c5dc61d3bfeb665495df6769c Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 19:41:43 +0530 Subject: [PATCH 049/117] Fix tests --- .../model_management_endpoints.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/litellm/proxy/management_endpoints/model_management_endpoints.py b/litellm/proxy/management_endpoints/model_management_endpoints.py index 355e2011237..2ca8e3daba3 100644 --- a/litellm/proxy/management_endpoints/model_management_endpoints.py +++ b/litellm/proxy/management_endpoints/model_management_endpoints.py @@ -42,6 +42,9 @@ from litellm.proxy.management_endpoints.team_endpoints import ( team_model_add, team_model_delete, ) +from litellm.proxy.management_endpoints.team_endpoints import ( + update_team as _legacy_update_team, +) from litellm.proxy.management_helpers.audit_logs import create_object_audit_log from litellm.proxy.utils import PrismaClient from litellm.types.proxy.management_endpoints.model_management_endpoints import ( @@ -58,6 +61,14 @@ from litellm.utils import get_utc_datetime router = APIRouter() +async def update_team(*args, **kwargs): + """ + Backward-compatible shim for tests/legacy call sites that patch this symbol. + Team model management now uses team_model_add/team_model_delete directly. + """ + return await _legacy_update_team(*args, **kwargs) + + class UpdatePublicModelGroupsRequest(BaseModel): """Request model for updating public model groups""" From d3568efad07a65da97f319a572e576c4d9d4e58e Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 26 Mar 2026 22:06:20 -0700 Subject: [PATCH 050/117] Merge pull request #24611 from Sameerlite/Sameerlite/order-fallback2 feat(router): order-based fallback across deployment priority levels --- docs/my-website/docs/proxy/load_balancing.md | 42 ++- docs/my-website/docs/routing.md | 15 +- litellm/router.py | 76 +++- litellm/utils.py | 16 +- .../test_router_order_fallback.py | 331 ++++++++++++++++++ 5 files changed, 455 insertions(+), 25 deletions(-) create mode 100644 tests/test_litellm/test_router_order_fallback.py diff --git a/docs/my-website/docs/proxy/load_balancing.md b/docs/my-website/docs/proxy/load_balancing.md index 897c04b2b00..93f3d944340 100644 --- a/docs/my-website/docs/proxy/load_balancing.md +++ b/docs/my-website/docs/proxy/load_balancing.md @@ -324,17 +324,43 @@ model_list: litellm_params: model: azure/gpt-4-fallback api_key: os.environ/AZURE_API_KEY_2 - order: 2 # 👈 Used when order=1 is unavailable - -router_settings: - enable_pre_call_checks: true # 👈 Required for 'order' to work + order: 2 # 👈 Used when order=1 fails ``` -:::important -The `order` parameter requires `enable_pre_call_checks: true` in `router_settings`. -::: +### How order-based fallback works -If `order=1` deployment is unavailable (e.g., rate-limited), the router falls back to `order=2` deployments. +When a request to an `order=1` deployment fails (connection error, 404, 429, etc.), the router automatically tries `order=2` deployments, then `order=3`, and so on. Each order level gets its own set of retries before escalating to the next. + +If all order levels are exhausted, the router falls through to any configured [model-level fallbacks](#fallbacks). + +```yaml +model_list: + - model_name: gpt-4 + litellm_params: + model: azure/gpt-4-primary + api_key: os.environ/AZURE_API_KEY + order: 1 + + - model_name: gpt-4 + litellm_params: + model: azure/gpt-4-secondary + api_key: os.environ/AZURE_API_KEY_2 + order: 2 + + - model_name: gpt-4-fallback + litellm_params: + model: openai/gpt-4 + api_key: os.environ/OPENAI_API_KEY + +router_settings: + fallbacks: + - gpt-4: + - gpt-4-fallback # tried after all order levels fail +``` + +The fallback chain for the above config: `order=1` → `order=2` → `gpt-4-fallback`. + +For 429 (rate limit) errors specifically, the failed deployment is immediately placed on cooldown. If all `order=1` deployments are on cooldown, the router picks `order=2` deployments directly during retries without waiting for the fallback path. ### Team-scoped models and legacy `model_aliases` {#team-scoped-models-and-legacy-model_aliases} diff --git a/docs/my-website/docs/routing.md b/docs/my-website/docs/routing.md index 67e7f681147..5aa655ae212 100644 --- a/docs/my-website/docs/routing.md +++ b/docs/my-website/docs/routing.md @@ -842,6 +842,8 @@ Traffic mirroring allows you to "mimic" production traffic to a secondary (silen Set `order` in `litellm_params` to prioritize deployments. Lower values = higher priority. When multiple deployments share the same `order`, the routing strategy picks among them. +When a request to an `order=1` deployment fails (connection error, 404, 429, etc.), the router automatically tries `order=2` deployments, then `order=3`, and so on. Each order level gets its own set of retries before escalating to the next. If all order levels are exhausted, the router falls through to any configured [fallbacks](#fallbacks). + @@ -862,18 +864,14 @@ model_list = [ "litellm_params": { "model": "azure/gpt-4-fallback", "api_key": os.getenv("AZURE_API_KEY_2"), - "order": 2, # 👈 Used when order=1 is unavailable + "order": 2, # 👈 Tried when order=1 fails }, }, ] -router = Router(model_list=model_list, enable_pre_call_checks=True) # 👈 Required for 'order' to work +router = Router(model_list=model_list) ``` -:::important -The `order` parameter requires `enable_pre_call_checks=True` to be set on the Router. -::: - @@ -889,10 +887,7 @@ model_list: litellm_params: model: azure/gpt-4-fallback api_key: os.environ/AZURE_API_KEY_2 - order: 2 # 👈 Used when order=1 is unavailable - -router_settings: - enable_pre_call_checks: true # 👈 Required for 'order' to work + order: 2 # 👈 Tried when order=1 fails ``` diff --git a/litellm/router.py b/litellm/router.py index 64ad6fc2215..5cd4f837782 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -5290,6 +5290,64 @@ class Router: if "fallback_depth" not in input_kwargs: input_kwargs["fallback_depth"] = 0 + # ORDER-BASED FALLBACKS: prepend higher order levels to the fallback list + # Skip for error types that have their own dedicated fallback handlers + _skip_order_fallback = isinstance( + e, + (litellm.ContextWindowExceededError, litellm.ContentPolicyViolationError), + ) + all_deployments = self._get_all_deployments(model_name=original_model_group) + _order_set: set = { + d.get("litellm_params", {}).get("order") + for d in all_deployments + if d.get("litellm_params", {}).get("order") is not None + } + order_values: list = sorted(_order_set) + if len(order_values) > 1 and not _skip_order_fallback: + # Determine which order levels have already been tried + current_target = kwargs.get("_target_order") + skip_up_to = ( + current_target if current_target is not None else order_values[0] + ) + # Build order-based fallback entries (skip already-tried levels) + order_fallback_entries: List = [ + {"model": original_model_group, "_target_order": o} + for o in order_values + if o > skip_up_to + ] + # Get external fallbacks — handle both standard and non-standard formats + external_fallback_group: Optional[List] = None + if fallbacks is not None and model_group is not None: + if _check_non_standard_fallback_format(fallbacks=fallbacks): + # Non-standard formats (e.g. ["claude-3-haiku"] or + # [{"model": "...", "messages": [...]}]) are passed through directly + external_fallback_group = fallbacks + else: + external_fallback_group, generic_idx = get_fallback_model_group( + fallbacks=fallbacks, + model_group=cast(str, model_group), + ) + if external_fallback_group is None and generic_idx is not None: + external_fallback_group = fallbacks[generic_idx]["*"] + + # Combined list: order fallbacks first, then external + combined_fallbacks = order_fallback_entries + ( + external_fallback_group or [] + ) + + if combined_fallbacks: + input_kwargs.update( + { + "fallback_model_group": combined_fallbacks, + "original_model_group": original_model_group, + } + ) + response = await run_async_fallback( + *args, + **input_kwargs, + ) + return response + try: verbose_router_logger.info("Trying to fallback b/w models") @@ -8886,12 +8944,6 @@ class Router: if i not in invalid_model_indices ] - ## ORDER FILTERING ## -> if user set 'order' in deployments, return deployments with lowest order (e.g. order=1 > order=2) - if len(_returned_deployments) > 0: - _returned_deployments = litellm.utils._get_order_filtered_deployments( - _returned_deployments - ) - return _returned_deployments def _get_model_from_alias(self, model: str) -> Optional[str]: @@ -9140,6 +9192,12 @@ class Router: ), ) + ## ORDER FILTERING ## -> if user set 'order' in deployments, return deployments with lowest order (e.g. order=1 > order=2) + _target_order = (request_kwargs or {}).pop("_target_order", None) + healthy_deployments = litellm.utils._get_order_filtered_deployments( + cast(List[Dict], healthy_deployments), target_order=_target_order + ) + if len(healthy_deployments) == 0: exception = await async_raise_no_deployment_exception( litellm_router_instance=self, @@ -9544,6 +9602,12 @@ class Router: request_kwargs=request_kwargs, ) + ## ORDER FILTERING ## -> if user set 'order' in deployments, return deployments with lowest order (e.g. order=1 > order=2) + _target_order = (request_kwargs or {}).pop("_target_order", None) + healthy_deployments = litellm.utils._get_order_filtered_deployments( + healthy_deployments, target_order=_target_order + ) + if len(healthy_deployments) == 0: model_ids = self.get_model_ids(model_name=model) _cooldown_time = self.cooldown_cache.get_min_cooldown( diff --git a/litellm/utils.py b/litellm/utils.py index 088ee07d630..0e3792773aa 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4866,7 +4866,21 @@ def calculate_max_parallel_requests( return None -def _get_order_filtered_deployments(healthy_deployments: List[Dict]) -> List: +def _get_order_filtered_deployments( + healthy_deployments: List[Dict], target_order: Optional[int] = None +) -> List: + if target_order is not None: + filtered = [ + d + for d in healthy_deployments + if d["litellm_params"].get("order") == target_order + ] + if filtered: + return filtered + # target_order doesn't match any deployment (e.g., external fallback model) — return all + return healthy_deployments + + # Default: pick min order group min_order = min( ( deployment["litellm_params"]["order"] diff --git a/tests/test_litellm/test_router_order_fallback.py b/tests/test_litellm/test_router_order_fallback.py new file mode 100644 index 00000000000..760766a7461 --- /dev/null +++ b/tests/test_litellm/test_router_order_fallback.py @@ -0,0 +1,331 @@ +""" +Tests for order-based fallback routing. + +When deployments have `order` set in litellm_params, lower order deployments +should be tried first, and higher order deployments should be used as fallbacks +when lower order deployments fail. +""" + +from typing import Optional + +import pytest + +from litellm import Router +from litellm.utils import _get_order_filtered_deployments + +# --------------------------------------------------------------------------- +# Unit tests for _get_order_filtered_deployments +# --------------------------------------------------------------------------- + + +class TestGetOrderFilteredDeployments: + def _make_deployment(self, order: Optional[int], dep_id: str) -> dict: + params: dict = {"model": "gpt-4o", "api_key": "key"} + if order is not None: + params["order"] = order + return { + "model_name": "test-model", + "litellm_params": params, + "model_info": {"id": dep_id}, + } + + def test_returns_min_order_group(self): + deps = [ + self._make_deployment(1, "a"), + self._make_deployment(2, "b"), + self._make_deployment(1, "c"), + ] + result = _get_order_filtered_deployments(deps) + assert len(result) == 2 + assert all(d["model_info"]["id"] in ("a", "c") for d in result) + + def test_target_order_filters_to_exact_level(self): + deps = [ + self._make_deployment(1, "a"), + self._make_deployment(2, "b"), + self._make_deployment(3, "c"), + ] + result = _get_order_filtered_deployments(deps, target_order=2) + assert len(result) == 1 + assert result[0]["model_info"]["id"] == "b" + + def test_target_order_no_match_returns_all(self): + deps = [ + self._make_deployment(1, "a"), + self._make_deployment(2, "b"), + ] + result = _get_order_filtered_deployments(deps, target_order=99) + assert len(result) == 2 + + def test_no_order_set_returns_all(self): + deps = [ + self._make_deployment(None, "a"), + self._make_deployment(None, "b"), + ] + result = _get_order_filtered_deployments(deps) + assert len(result) == 2 + + def test_empty_list(self): + result = _get_order_filtered_deployments([]) + assert result == [] + + def test_single_order_returns_all_with_that_order(self): + deps = [ + self._make_deployment(1, "a"), + self._make_deployment(1, "b"), + ] + result = _get_order_filtered_deployments(deps) + assert len(result) == 2 + + +# --------------------------------------------------------------------------- +# Integration tests for order-based fallback in Router +# --------------------------------------------------------------------------- + + +def test_router_order_without_pre_call_checks(): + """Order filtering should work even when enable_pre_call_checks=False (default).""" + router = Router( + model_list=[ + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "key", + "mock_response": "from order 1", + "order": 1, + }, + "model_info": {"id": "1"}, + }, + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "key", + "mock_response": "from order 2", + "order": 2, + }, + "model_info": {"id": "2"}, + }, + ], + num_retries=0, + enable_pre_call_checks=False, + ) + + for _ in range(20): + response = router.completion( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + ) + assert response._hidden_params["model_id"] == "1" + + +def test_router_order_no_fallback_when_healthy(): + """When order=1 is healthy, order=2 should never be used.""" + router = Router( + model_list=[ + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "key", + "mock_response": "from order 1", + "order": 1, + }, + "model_info": {"id": "1"}, + }, + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "key", + "mock_response": "from order 2", + "order": 2, + }, + "model_info": {"id": "2"}, + }, + ], + num_retries=0, + ) + + for _ in range(50): + response = router.completion( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + ) + assert response._hidden_params["model_id"] == "1" + + +@pytest.mark.asyncio +async def test_router_order_fallback_on_failure(): + """When order=1 fails, order=2 should be tried as fallback.""" + router = Router( + model_list=[ + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "bad-key", + "mock_response": Exception("connection error"), + "order": 1, + }, + "model_info": {"id": "1"}, + }, + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "good-key", + "mock_response": "success from order 2", + "order": 2, + }, + "model_info": {"id": "2"}, + }, + ], + num_retries=0, + ) + + response = await router.acompletion( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + ) + assert response._hidden_params["model_id"] == "2" + + +@pytest.mark.asyncio +async def test_router_order_fallback_three_levels(): + """When order=1 and order=2 both fail, order=3 should be tried.""" + router = Router( + model_list=[ + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "bad", + "mock_response": Exception("fail 1"), + "order": 1, + }, + "model_info": {"id": "1"}, + }, + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "bad", + "mock_response": Exception("fail 2"), + "order": 2, + }, + "model_info": {"id": "2"}, + }, + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "good", + "mock_response": "success from order 3", + "order": 3, + }, + "model_info": {"id": "3"}, + }, + ], + num_retries=0, + ) + + response = await router.acompletion( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + ) + assert response._hidden_params["model_id"] == "3" + + +@pytest.mark.asyncio +async def test_router_order_fallback_then_external_fallback(): + """When all order levels fail, external fallbacks should be tried.""" + router = Router( + model_list=[ + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "bad", + "mock_response": Exception("fail order 1"), + "order": 1, + }, + "model_info": {"id": "1"}, + }, + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "bad", + "mock_response": Exception("fail order 2"), + "order": 2, + }, + "model_info": {"id": "2"}, + }, + { + "model_name": "fallback-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "good", + "mock_response": "success from external fallback", + }, + "model_info": {"id": "fallback"}, + }, + ], + fallbacks=[{"test-model": ["fallback-model"]}], + num_retries=0, + ) + + response = await router.acompletion( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + ) + assert response._hidden_params["model_id"] == "fallback" + + +@pytest.mark.asyncio +async def test_router_order_fallback_with_non_standard_fallbacks(): + """Non-standard fallback formats (e.g. fallbacks=["model-name"]) passed + per-request should still be tried after all order levels are exhausted.""" + router = Router( + model_list=[ + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "bad", + "mock_response": Exception("fail order 1"), + "order": 1, + }, + "model_info": {"id": "1"}, + }, + { + "model_name": "test-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "bad", + "mock_response": Exception("fail order 2"), + "order": 2, + }, + "model_info": {"id": "2"}, + }, + { + "model_name": "fallback-model", + "litellm_params": { + "model": "gpt-4o", + "api_key": "good", + "mock_response": "success from non-standard fallback", + }, + "model_info": {"id": "fallback"}, + }, + ], + num_retries=0, + ) + + response = await router.acompletion( + model="test-model", + messages=[{"role": "user", "content": "hi"}], + fallbacks=["fallback-model"], # non-standard format, passed per-request + ) + assert response._hidden_params["model_id"] == "fallback" From 00a810e92d61dc0e0072a8588f94bbda3024acd2 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Wed, 25 Mar 2026 14:30:15 +0530 Subject: [PATCH 051/117] feat(openai): round-trip Responses API reasoning_items in chat completions Made-with: Cursor --- docs/my-website/docs/providers/openai.md | 84 ++++++ .../transformation.py | 106 ++++++- .../litellm_core_utils/streaming_handler.py | 4 + litellm/types/llms/openai.py | 14 + litellm/types/utils.py | 20 ++ ...responses_transformation_transformation.py | 259 +++++++++++++++++- 6 files changed, 474 insertions(+), 13 deletions(-) diff --git a/docs/my-website/docs/providers/openai.md b/docs/my-website/docs/providers/openai.md index 2907cdf9f47..1f4a1687e8b 100644 --- a/docs/my-website/docs/providers/openai.md +++ b/docs/my-website/docs/providers/openai.md @@ -581,6 +581,90 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ See [OpenAI Reasoning documentation](https://platform.openai.com/docs/guides/reasoning) for more details on organization verification requirements. +### Multi-turn Conversations with `reasoning_items` + +For multi-turn conversations you need `reasoning_items`: structured blocks that include the `encrypted_content` token OpenAI uses to restore reasoning state on the next request. Pass `include=["reasoning.encrypted_content"]` on every call where you want that token returned. + + + + +```python showLineNumbers title="Non-streaming: round-trip reasoning_items" +import litellm + +messages = [{"role": "user", "content": "Solve this step by step: 2 + 2"}] + +# Turn 1 — get reasoning_items (encrypted_content); +response = litellm.completion( + model="openai/responses/gpt-5-mini", + messages=messages, + reasoning_effort="low", + include=["reasoning.encrypted_content"], +) + +assistant_msg = response.choices[0].message + +# Turn 2 — pass reasoning_items back; LiteLLM converts to the correct Responses API format +messages.append({ + "role": "assistant", + "content": assistant_msg.content, + "reasoning_items": assistant_msg.reasoning_items, +}) +messages.append({"role": "user", "content": "Now summarize your reasoning."}) + +response2 = litellm.completion( + model="openai/responses/gpt-5-mini", + messages=messages, + reasoning_effort="low", + include=["reasoning.encrypted_content"], +) +``` + + + + +`reasoning_items` (with `encrypted_content`) arrive on the final chunk when the full response completes: + +```python showLineNumbers title="Streaming: collect and round-trip reasoning_items" +import litellm + +messages = [{"role": "user", "content": "Solve this step by step: 2 + 2"}] + +collected_content = [] +collected_reasoning_items = [] + +stream = litellm.completion( + model="openai/responses/gpt-5-mini", + messages=messages, + stream=True, + reasoning_effort="low", + include=["reasoning.encrypted_content"], +) + +for chunk in stream: + delta = chunk.choices[0].delta + if delta.content: + collected_content.append(delta.content) + if getattr(delta, "reasoning_items", None): + collected_reasoning_items.extend(delta.reasoning_items) + +messages.append({ + "role": "assistant", + "content": "".join(collected_content), + "reasoning_items": collected_reasoning_items or None, +}) +messages.append({"role": "user", "content": "Continue the conversation."}) + +response2 = litellm.completion( + model="openai/responses/gpt-5-mini", + messages=messages, + reasoning_effort="low", + include=["reasoning.encrypted_content"], +) +``` + + + + ### Verbosity Control for GPT-5 Models The `verbosity` parameter controls the length and detail of responses from GPT-5 family models. It accepts three values: `"low"`, `"medium"`, or `"high"`. diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index ee4cdbcdf36..6e6070f7f2c 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -32,6 +32,7 @@ from litellm.llms.base_llm.bridges.completion_transformation import ( ) from litellm.types.llms.openai import ( ChatCompletionAnnotation, + ChatCompletionReasoningItem, ChatCompletionToolParamFunctionChunk, Reasoning, ResponsesAPIOptionalRequestParams, @@ -55,6 +56,49 @@ if TYPE_CHECKING: ) +def _build_reasoning_item( + item_id: str, + encrypted_content: Optional[str], + summary_raw: Any, +) -> Dict[str, Any]: + """Build a ChatCompletionReasoningItem-shaped dict from raw response data. + + Handles both pydantic objects (attribute access) and plain dicts. + """ + summary: List[Dict[str, Any]] = [] + for s in summary_raw or []: + if isinstance(s, dict): + summary.append( + {"type": s.get("type", "summary_text"), "text": s.get("text", "")} + ) + else: + summary.append( + { + "type": getattr(s, "type", "summary_text"), + "text": getattr(s, "text", ""), + } + ) + return { + "id": item_id, + "type": "reasoning", + "encrypted_content": encrypted_content, + "summary": summary, + } + + +def _reasoning_item_to_response_input(r_item: Dict[str, Any]) -> Dict[str, Any]: + """Convert a stored ChatCompletionReasoningItem back to a Responses API input item.""" + r_input: Dict[str, Any] = { + "type": "reasoning", + "id": r_item.get("id") or f"rs_{id(r_item)}", + # summary is always required by the Responses API, even when empty + "summary": r_item.get("summary") or [], + } + if r_item.get("encrypted_content"): + r_input["encrypted_content"] = r_item["encrypted_content"] + return r_input + + class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): """ Handler for transforming /chat/completions api requests to litellm.responses requests @@ -202,10 +246,12 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): } ) elif role == "assistant" and tool_calls and isinstance(tool_calls, list): + for r_item in msg.get("reasoning_items") or []: + input_items.append(_reasoning_item_to_response_input(r_item)) for tool_call in tool_calls: function = tool_call.get("function") if function: - input_tool_call = { + input_tool_call: Dict[str, Any] = { "type": "function_call", "call_id": tool_call["id"], } @@ -217,7 +263,9 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): else: raise ValueError(f"tool call not supported: {tool_call}") elif content is not None: - # Regular user/assistant message + if role == "assistant": + for r_item in msg.get("reasoning_items") or []: + input_items.append(_reasoning_item_to_response_input(r_item)) input_items.append( { "type": "message", @@ -411,6 +459,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): choices: List[Choices] = [] index = 0 reasoning_content: Optional[str] = None + pending_reasoning_item: Optional[Dict[str, Any]] = None # Collect all tool calls to put them in a single choice # (Chat Completions API expects all tool calls in one message) @@ -419,9 +468,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): for item in output_items: if isinstance(item, ResponseReasoningItem): - for summary_item in item.summary: - response_text = getattr(summary_item, "text", "") - reasoning_content = response_text if response_text else "" + pending_reasoning_item = _build_reasoning_item( + item_id=item.id, + encrypted_content=getattr(item, "encrypted_content", None), + summary_raw=item.summary, + ) + reasoning_content = " ".join( + s["text"] + for s in pending_reasoning_item["summary"] + if s.get("text") + ) elif isinstance(item, ResponseOutputMessage): for content in item.content: @@ -436,6 +492,12 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): content=response_text if response_text else "", reasoning_content=reasoning_content, annotations=annotations, + reasoning_items=cast( + Optional[List[ChatCompletionReasoningItem]], + [pending_reasoning_item] + if pending_reasoning_item is not None + else None, + ), ) choices.append( @@ -446,7 +508,8 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): ) ) - reasoning_content = None # flush reasoning content + reasoning_content = None # flush + pending_reasoning_item = None # flush index += 1 elif isinstance(item, ResponseFunctionToolCall): @@ -489,11 +552,18 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): content=None, tool_calls=accumulated_tool_calls, reasoning_content=reasoning_content, + reasoning_items=cast( + Optional[List[ChatCompletionReasoningItem]], + [pending_reasoning_item] + if pending_reasoning_item is not None + else None, + ), ) choices.append( Choices(message=msg, finish_reason="tool_calls", index=index) ) reasoning_content = None + pending_reasoning_item = None return choices @@ -1232,6 +1302,25 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): finish_reason = "tool_calls" if has_function_calls else "stop" + # Extract reasoning items with encrypted_content for round-tripping + completed_reasoning_items: Optional[List[Dict[str, Any]]] = None + for item in output_items: + if not isinstance(item, dict) or item.get("type") != "reasoning": + continue + if completed_reasoning_items is None: + completed_reasoning_items = [] + completed_reasoning_items.append( + _build_reasoning_item( + item_id=item.get("id", ""), + encrypted_content=item.get("encrypted_content"), + summary_raw=item.get("summary"), + ) + ) + completed_reasoning_items_typed = cast( + Optional[List[ChatCompletionReasoningItem]], + completed_reasoning_items, + ) + usage = None if response_data.get("usage"): from litellm.responses.utils import ResponseAPILoggingUtils @@ -1245,7 +1334,10 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): choices=[ StreamingChoices( index=0, - delta=Delta(content=""), + delta=Delta( + content="", + reasoning_items=completed_reasoning_items_typed, + ), finish_reason=finish_reason, ) ], diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index 67e4fadf638..1bb2b99c015 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -831,6 +831,10 @@ class CustomStreamWrapper: "annotations" in model_response.choices[0].delta and model_response.choices[0].delta.annotations is not None ) + or ( + getattr(model_response.choices[0].delta, "reasoning_items", None) + is not None + ) ): return True else: diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index 5a80b40d61f..aa5719bc7ac 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -536,6 +536,20 @@ class ChatCompletionRedactedThinkingBlock(TypedDict, total=False): cache_control: Optional[Union[dict, ChatCompletionCachedContent]] +class ChatCompletionReasoningSummaryTextBlock(TypedDict, total=False): + type: Required[Literal["summary_text"]] + text: str + + +class ChatCompletionReasoningItem(TypedDict, total=False): + """Represents an OpenAI Responses API reasoning item for round-tripping in conversation history.""" + + type: Required[Literal["reasoning"]] + id: str + encrypted_content: Optional[str] + summary: List["ChatCompletionReasoningSummaryTextBlock"] + + class WebSearchOptionsUserLocationApproximate(TypedDict, total=False): city: str """Free text input for the city of the user, e.g. `San Francisco`.""" diff --git a/litellm/types/utils.py b/litellm/types/utils.py index bd673da8bed..8b94fbdad6d 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -58,6 +58,7 @@ from .llms.openai import ( AllMessageValues, Batch, ChatCompletionAnnotation, + ChatCompletionReasoningItem, ChatCompletionRedactedThinkingBlock, ChatCompletionThinkingBlock, ChatCompletionToolCallChunk, @@ -1132,6 +1133,7 @@ class Message(SafeAttributeModel, OpenAIObject): thinking_blocks: Optional[ List[Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]] ] = None + reasoning_items: Optional[List[ChatCompletionReasoningItem]] = None provider_specific_fields: Optional[Dict[str, Any]] = Field(default=None) annotations: Optional[List[ChatCompletionAnnotation]] = None @@ -1150,6 +1152,7 @@ class Message(SafeAttributeModel, OpenAIObject): Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock] ] ] = None, + reasoning_items: Optional[List[ChatCompletionReasoningItem]] = None, annotations: Optional[List[ChatCompletionAnnotation]] = None, **params, ): @@ -1182,6 +1185,9 @@ class Message(SafeAttributeModel, OpenAIObject): if thinking_blocks is not None: init_values["thinking_blocks"] = thinking_blocks + if reasoning_items is not None: + init_values["reasoning_items"] = reasoning_items + if annotations is not None: init_values["annotations"] = annotations @@ -1219,6 +1225,11 @@ class Message(SafeAttributeModel, OpenAIObject): if hasattr(self, "thinking_blocks"): del self.thinking_blocks + if reasoning_items is None: + # ensure default response matches OpenAI spec + if hasattr(self, "reasoning_items"): + del self.reasoning_items + add_provider_specific_fields(self, provider_specific_fields) def get(self, key, default=None): @@ -1246,6 +1257,7 @@ class Delta(SafeAttributeModel, OpenAIObject): thinking_blocks: Optional[ List[Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock]] ] = None + reasoning_items: Optional[List[ChatCompletionReasoningItem]] = None provider_specific_fields: Optional[Dict[str, Any]] = Field(default=None) def __init__( @@ -1262,6 +1274,7 @@ class Delta(SafeAttributeModel, OpenAIObject): Union[ChatCompletionThinkingBlock, ChatCompletionRedactedThinkingBlock] ] ] = None, + reasoning_items: Optional[List[ChatCompletionReasoningItem]] = None, annotations: Optional[List[ChatCompletionAnnotation]] = None, **params, ): @@ -1295,6 +1308,13 @@ class Delta(SafeAttributeModel, OpenAIObject): # ensure default response matches OpenAI spec del self.thinking_blocks + if reasoning_items is not None: + self.reasoning_items = reasoning_items + else: + # ensure default response matches OpenAI spec + if hasattr(self, "reasoning_items"): + del self.reasoning_items + # Add annotations to the delta, ensure they are only on Delta if they exist (Match OpenAI spec) if annotations is not None: self.annotations = annotations diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 8bc6ffc0505..e40543e01a0 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -2127,9 +2127,10 @@ def test_convert_chat_completion_file_type_to_input_file(): } ] - input_items, instructions = ( - handler.convert_chat_completion_messages_to_responses_api(messages) - ) + ( + input_items, + instructions, + ) = handler.convert_chat_completion_messages_to_responses_api(messages) assert len(input_items) == 1 msg = input_items[0] @@ -2176,11 +2177,257 @@ def test_convert_chat_completion_file_type_with_file_id(): } ] - input_items, instructions = ( - handler.convert_chat_completion_messages_to_responses_api(messages) - ) + ( + input_items, + instructions, + ) = handler.convert_chat_completion_messages_to_responses_api(messages) content = input_items[0]["content"] assert content[1]["type"] == "input_file" assert content[1]["file_id"] == "file-abc123" assert "file_data" not in content[1] + + +# ============================================================================= +# Tests for reasoning_items round-trip (encrypted_content preservation) +# ============================================================================= + + +def test_reasoning_items_non_streaming_round_trip(): + """ + Non-streaming: verify that reasoning_items (with encrypted_content) are: + 1. Extracted from ResponseReasoningItem and attached to the Message. + 2. Emitted as a 'reasoning' input item when the assistant message is + passed back to convert_chat_completion_messages_to_responses_api. + """ + from unittest.mock import Mock + + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + from openai.types.responses.response_reasoning_item import ( + ResponseReasoningItem, + Summary, + ) + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + from litellm.types.llms.openai import ( + InputTokensDetails, + OutputTokensDetails, + ResponseAPIUsage, + ResponsesAPIResponse, + ) + from litellm.types.utils import ModelResponse, Usage + + handler = LiteLLMResponsesTransformationHandler() + + encrypted = "gAAAAABpw5abc123FAKE==" + summary_text = "**Thinking about it**\n\nSome reasoning here." + + reasoning_item = ResponseReasoningItem( + id="rs_test001", + summary=[Summary(text=summary_text, type="summary_text")], + type="reasoning", + content=None, + encrypted_content=encrypted, + status=None, + ) + output_message = ResponseOutputMessage( + id="msg_test001", + content=[ + ResponseOutputText( + annotations=[], + text="The answer is 42.", + type="output_text", + logprobs=[], + ) + ], + role="assistant", + status="completed", + type="message", + ) + usage = ResponseAPIUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + audio_tokens=None, cached_tokens=0, text_tokens=None + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None), + total_tokens=30, + cost=None, + ) + raw_response = ResponsesAPIResponse( + id="resp_test001", + created_at=1234567890, + error=None, + incomplete_details=None, + instructions=None, + metadata={}, + model="gpt-5-mini", + object="response", + output=[reasoning_item, output_message], + parallel_tool_calls=True, + temperature=1.0, + tool_choice="auto", + tools=[], + top_p=1.0, + max_output_tokens=None, + previous_response_id=None, + reasoning={"effort": "low", "summary": "detailed"}, + status="completed", + text={"format": {"type": "text"}, "verbosity": "medium"}, + truncation="disabled", + usage=usage, + user=None, + store=True, + background=False, + billing={"payer": "developer"}, + max_tool_calls=None, + prompt_cache_key=None, + safety_identifier=None, + service_tier="default", + top_logprobs=0, + ) + model_response = ModelResponse( + id="chatcmpl-test001", + created=1234567890, + model=None, + object="chat.completion", + system_fingerprint=None, + choices=[], + usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0), + ) + + result = handler.transform_response( + model="gpt-5-mini", + raw_response=raw_response, + model_response=model_response, + logging_obj=Mock(), + request_data={"model": "gpt-5-mini"}, + messages=[{"role": "user", "content": "What is the answer?"}], + optional_params={}, + litellm_params={}, + encoding=Mock(), + ) + + # ── Part 1: reasoning_items on the response message ────────────────────── + assert len(result.choices) == 1 + msg = result.choices[0].message + + assert ( + msg.reasoning_content == summary_text + ), "reasoning_content should equal summary text" + + assert msg.reasoning_items is not None, "reasoning_items should be set" + assert len(msg.reasoning_items) == 1 + ri = msg.reasoning_items[0] + assert ri["type"] == "reasoning" + assert ri["id"] == "rs_test001" + assert ri["encrypted_content"] == encrypted, "encrypted_content must be preserved" + assert len(ri["summary"]) == 1 + assert ri["summary"][0]["text"] == summary_text + + # ── Part 2: reasoning item round-trips through message history ──────────── + history = [ + {"role": "user", "content": "What is the answer?"}, + { + "role": "assistant", + "content": msg.content, + "reasoning_items": msg.reasoning_items, + }, + {"role": "user", "content": "Can you elaborate?"}, + ] + input_items, _ = handler.convert_chat_completion_messages_to_responses_api(history) + + # The reasoning input item must appear before the assistant message item + types = [item.get("type") for item in input_items] + assert ( + "reasoning" in types + ), "reasoning input item must be emitted for the assistant turn" + + reasoning_input = next( + item for item in input_items if item.get("type") == "reasoning" + ) + assert reasoning_input["id"] == "rs_test001" + assert reasoning_input["encrypted_content"] == encrypted + assert reasoning_input["summary"][0]["text"] == summary_text + + # reasoning item must come before the assistant message item + reasoning_idx = types.index("reasoning") + assistant_msg_idx = next( + i + for i, item in enumerate(input_items) + if item.get("type") == "message" and item.get("role") == "assistant" + ) + assert ( + reasoning_idx < assistant_msg_idx + ), "reasoning input item must precede the assistant message item" + + +def test_reasoning_items_streaming_emitted_on_response_completed(): + """ + Streaming: verify that reasoning_items (with encrypted_content) are emitted + on the delta of the response.completed chunk, enabling the caller to + round-trip them in subsequent requests. + """ + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + iterator = OpenAiResponsesToChatCompletionStreamIterator( + streaming_response=None, sync_stream=True + ) + + encrypted = "gAAAAABpw5xyz987FAKE==" + summary_text = "**Reasoning summary**\n\nModel thought about this carefully." + + chunk = { + "type": "response.completed", + "response": { + "id": "resp_stream001", + "status": "completed", + "output": [ + { + "type": "reasoning", + "id": "rs_stream001", + "encrypted_content": encrypted, + "summary": [{"type": "summary_text", "text": summary_text}], + }, + { + "type": "message", + "id": "msg_stream001", + "role": "assistant", + "content": [{"type": "output_text", "text": "The answer."}], + "status": "completed", + }, + ], + "usage": { + "input_tokens": 10, + "output_tokens": 5, + "total_tokens": 15, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens_details": {"reasoning_tokens": 0}, + }, + }, + } + + result = iterator.chunk_parser(chunk) + + assert len(result.choices) == 1 + delta = result.choices[0].delta + + # finish_reason must be set (response is complete) + assert result.choices[0].finish_reason == "stop" + + # reasoning_items must be on the delta + assert ( + getattr(delta, "reasoning_items", None) is not None + ), "reasoning_items must be present on the response.completed delta" + assert len(delta.reasoning_items) == 1 + ri = delta.reasoning_items[0] + assert ri["type"] == "reasoning" + assert ri["id"] == "rs_stream001" + assert ( + ri["encrypted_content"] == encrypted + ), "encrypted_content must be preserved in streaming" + assert ri["summary"][0]["text"] == summary_text From bbd8ca3b3dbaf4d5ad536501477984811929ce5d Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Wed, 25 Mar 2026 16:17:57 +0530 Subject: [PATCH 052/117] feat(prometheus): add metrics for managed batch lifecycle - Add Prometheus metrics for managed batch and file operations - Track batch creation, file size, duration, and deletion events - Add CheckBatchCost polling metrics (jobs polled/processed, errors) - Record metrics in managed_files hook and check_batch_cost utility - Metrics include labels for model, provider, user, and status Made-with: Cursor --- .../proxy/common_utils/check_batch_cost.py | 51 ++++- .../proxy/hooks/managed_files.py | 68 ++++++ litellm/integrations/prometheus.py | 202 ++++++++++++++++++ litellm/types/integrations/prometheus.py | 64 ++++++ 4 files changed, 384 insertions(+), 1 deletion(-) diff --git a/enterprise/litellm_enterprise/proxy/common_utils/check_batch_cost.py b/enterprise/litellm_enterprise/proxy/common_utils/check_batch_cost.py index cbe8d449b42..356f6ecd4b5 100644 --- a/enterprise/litellm_enterprise/proxy/common_utils/check_batch_cost.py +++ b/enterprise/litellm_enterprise/proxy/common_utils/check_batch_cost.py @@ -3,7 +3,7 @@ Polls LiteLLM_ManagedObjectTable to check if the batch job is complete, and if t """ from datetime import datetime, timedelta, timezone -from typing import TYPE_CHECKING, Optional +from typing import TYPE_CHECKING, List, Optional, Tuple from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid @@ -118,6 +118,15 @@ class CheckBatchCost: get_model_id_from_unified_batch_id, ) + try: + from litellm.integrations.prometheus import PrometheusLogger + prom_logger = PrometheusLogger.get_instance() + except Exception as e: + verbose_proxy_logger.error(f"CheckBatchCost: could not get Prometheus logger: {e}") + prom_logger = None + + processed_models: List[Tuple[Optional[str], Optional[str]]] = [] + try: await self._cleanup_stale_managed_objects() except Exception as cleanup_err: @@ -172,6 +181,8 @@ class CheckBatchCost: verbose_proxy_logger.info( f"Skipping job {unified_object_id} because it is not a valid unified object id" ) + if prom_logger: + prom_logger.record_check_batch_cost_error("invalid_unified_id") continue else: unified_object_id = decoded_unified_object_id @@ -183,6 +194,8 @@ class CheckBatchCost: verbose_proxy_logger.info( f"Skipping job {unified_object_id} because it is not a valid model id" ) + if prom_logger: + prom_logger.record_check_batch_cost_error("invalid_model_id") continue verbose_proxy_logger.info( @@ -202,6 +215,8 @@ class CheckBatchCost: verbose_proxy_logger.info( f"Skipping job {unified_object_id} because of error querying model ID: {model_id} for cost and usage of batch ID: {batch_id}: {e}" ) + if prom_logger: + prom_logger.record_check_batch_cost_error("provider_retrieval_error") continue ## RETRIEVE THE BATCH JOB OUTPUT FILE @@ -257,11 +272,25 @@ class CheckBatchCost: content_bytes # type: ignore[arg-type] ) + # Record output file size + if prom_logger and content_bytes: + try: + prom_logger.record_managed_file_size( + size_bytes=len(content_bytes), # type: ignore + purpose="batch", + file_type="output", + model=model_id, + ) + except Exception: + pass + deployment_info = self.llm_router.get_deployment(model_id=model_id) if deployment_info is None: verbose_proxy_logger.info( f"Skipping job {unified_object_id} because it is not a valid deployment info" ) + if prom_logger: + prom_logger.record_check_batch_cost_error("deployment_not_found") continue custom_llm_provider = deployment_info.litellm_params.custom_llm_provider litellm_model_name = deployment_info.litellm_params.model @@ -318,6 +347,19 @@ class CheckBatchCost: batch_models=batch_models, ) + # Record batch duration (completed_at - created_at) + if prom_logger and response.completed_at and response.created_at: + duration_seconds = float(response.completed_at - response.created_at) + if duration_seconds >= 0: + prom_logger.record_managed_batch_duration( + duration_seconds=duration_seconds, + model=model_name, + api_provider=str(llm_provider) if llm_provider else None, + ) + + # Track this job for the final metrics summary + processed_models.append((model_name, str(llm_provider) if llm_provider else None)) + # mark the job as complete try: update_data: dict = { @@ -334,3 +376,10 @@ class CheckBatchCost: verbose_proxy_logger.error( f"CheckBatchCost: failed to mark job {job.id} complete in DB: {db_err}" ) + + # Record polling run metrics (always, even if nothing was processed) + if prom_logger: + prom_logger.record_check_batch_cost_run( + jobs_polled=len(jobs), + processed_models=processed_models if processed_models else None, + ) diff --git a/enterprise/litellm_enterprise/proxy/hooks/managed_files.py b/enterprise/litellm_enterprise/proxy/hooks/managed_files.py index dc14937d46b..60c564072a0 100644 --- a/enterprise/litellm_enterprise/proxy/hooks/managed_files.py +++ b/enterprise/litellm_enterprise/proxy/hooks/managed_files.py @@ -74,6 +74,13 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): self.internal_usage_cache = internal_usage_cache self.prisma_client = prisma_client + @staticmethod + def _get_prometheus_logger(): + """Find PrometheusLogger from litellm.callbacks, if registered.""" + from litellm.integrations.prometheus import PrometheusLogger + + return PrometheusLogger.get_instance() + async def store_unified_file_id( self, file_id: str, @@ -905,6 +912,31 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): model_mappings=model_mappings, user_api_key_dict=user_api_key_dict, ) + + # Emit Prometheus metrics for managed file creation + prom_logger = self._get_prometheus_logger() + if prom_logger: + first_model = target_model_names_list[0] if target_model_names_list else None + first_provider = "" + if responses: + first_provider = getattr(responses[0], "_hidden_params", {}).get("custom_llm_provider") or "" + prom_logger.record_managed_file_created( + model=first_model or "", + api_provider=first_provider, + user=user_api_key_dict.user_id or "", + user_email=getattr(user_api_key_dict, "user_email", None) or "", + api_key_alias=user_api_key_dict.key_alias or "", + ) + if response.bytes and response.bytes > 0: + prom_logger.record_managed_file_size( + size_bytes=response.bytes, + purpose=response.purpose or "batch", + file_type="input", + model=first_model, + api_provider=first_provider, + user=user_api_key_dict.user_id, + ) + return response @staticmethod @@ -1083,6 +1115,31 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): file_purpose="batch", user_api_key_dict=user_api_key_dict, ) + + # Only record batch creation metric on actual create (not retrieve/cancel). + # unified_file_id in _hidden_params is only set by the create_batch endpoint. + original_unified_file_id = response._hidden_params.get("unified_file_id") + if original_unified_file_id: + prom_logger = self._get_prometheus_logger() + if prom_logger: + batch_provider = "" + if model_name: + try: + from litellm.litellm_core_utils.get_llm_provider_logic import ( + get_llm_provider, + ) + _, batch_provider, _, _ = get_llm_provider(model=model_name) + except Exception: + if "/" in model_name: + batch_provider = model_name.split("/")[0] + prom_logger.record_managed_batch_created( + model=model_name or "", + api_provider=batch_provider, + user=user_api_key_dict.user_id or "", + user_email=getattr(user_api_key_dict, "user_email", None) or "", + api_key_alias=user_api_key_dict.key_alias or "", + ) + elif isinstance(response, LiteLLMFineTuningJob): ## Check if unified_file_id is in the response unified_file_id = response._hidden_params.get( @@ -1332,6 +1389,11 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): f"Alternatively, wait for all batches to complete and for cost to be computed (batch_processed=true)." ) + # Record blocked deletion metric + prom_logger = self._get_prometheus_logger() + if prom_logger: + prom_logger.record_managed_file_deleted(result="blocked") + raise HTTPException( status_code=400, detail=error_message, @@ -1365,6 +1427,12 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): file_id, litellm_parent_otel_span ) + # Record successful deletion metric only on actual success + if stored_file_object or delete_response: + prom_logger = self._get_prometheus_logger() + if prom_logger: + prom_logger.record_managed_file_deleted(result="success") + if stored_file_object: return stored_file_object elif delete_response: diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 90306c11a42..8de01aac60b 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -65,6 +65,17 @@ def _get_cached_end_user_id_for_cost_tracking(): class PrometheusLogger(CustomLogger): # Class variables or attributes + + @staticmethod + def get_instance() -> Optional["PrometheusLogger"]: + """Find the PrometheusLogger instance from litellm.callbacks, if registered.""" + import litellm + + for cb in litellm.callbacks: + if isinstance(cb, PrometheusLogger): + return cb + return None + def __init__( # noqa: PLR0915 self, **kwargs, @@ -440,6 +451,76 @@ class PrometheusLogger(CustomLogger): labelnames=[], ) + ######################################## + # Managed Batch Metrics + ######################################## + self.litellm_managed_batch_created_total = self._counter_factory( + name="litellm_managed_batch_created_total", + documentation="Total number of managed batches created", + labelnames=[ + "model", + "api_provider", + "user", + "user_email", + "api_key_alias", + ], + ) + + self.litellm_managed_file_size_bytes = self._gauge_factory( + "litellm_managed_file_size_bytes", + "Size of the most recent managed batch file in bytes (last-seen value per label combination)", + labelnames=["purpose", "file_type", "model", "api_provider", "user"], + ) + + self.litellm_managed_batch_duration_seconds = self._histogram_factory( + "litellm_managed_batch_duration_seconds", + "Duration of completed managed batches in seconds (completed_at - created_at)", + labelnames=["model", "api_provider"], + buckets=BATCH_DURATION_BUCKETS, + ) + + self.litellm_managed_file_created_total = self._counter_factory( + name="litellm_managed_file_created_total", + documentation="Total number of managed files created", + labelnames=[ + "model", + "api_provider", + "user", + "user_email", + "api_key_alias", + ], + ) + + self.litellm_managed_file_deleted_total = self._counter_factory( + name="litellm_managed_file_deleted_total", + documentation="Total number of managed file deletions (success or blocked)", + labelnames=["result"], + ) + + self.litellm_check_batch_cost_jobs_polled = self._gauge_factory( + "litellm_check_batch_cost_jobs_polled", + "Number of unprocessed batches found by the last CheckBatchCost poll", + labelnames=[], + ) + + self.litellm_check_batch_cost_jobs_processed_total = self._counter_factory( + name="litellm_check_batch_cost_jobs_processed_total", + documentation="Total number of batches successfully cost-tracked by CheckBatchCost", + labelnames=["model", "api_provider"], + ) + + self.litellm_check_batch_cost_errors_total = self._counter_factory( + name="litellm_check_batch_cost_errors_total", + documentation="Total number of errors in CheckBatchCost by error type", + labelnames=["error_type"], + ) + + self.litellm_check_batch_cost_last_run_timestamp = self._gauge_factory( + "litellm_check_batch_cost_last_run_timestamp", + "Unix timestamp of the last CheckBatchCost job run", + labelnames=[], + ) + except Exception as e: print_verbose(f"Got exception on init prometheus client {str(e)}") raise e @@ -2158,6 +2239,127 @@ class PrometheusLogger(CustomLogger): except Exception as e: verbose_logger.debug(f"Error recording guardrail metrics: {str(e)}") + ######################################## + # Managed Batch Metric Recording Methods + ######################################## + + def record_managed_batch_created( + self, + model: Optional[str], + api_provider: Optional[str], + user: Optional[str], + user_email: Optional[str], + api_key_alias: Optional[str], + ): + try: + self.litellm_managed_batch_created_total.labels( + model=model, + api_provider=api_provider, + user=user, + user_email=user_email, + api_key_alias=api_key_alias, + ).inc() + except Exception as e: + verbose_logger.warning(f"Error recording batch created metric: {e}") + + def record_managed_file_size( + self, + size_bytes: int, + purpose: str, + file_type: str, + model: Optional[str] = None, + api_provider: Optional[str] = None, + user: Optional[str] = None, + ): + """Record the size of a managed file. Uses a gauge (last-seen value per label combination).""" + try: + self.litellm_managed_file_size_bytes.labels( + purpose=purpose, + file_type=file_type, + model=model or "", + api_provider=api_provider or "", + user=user or "", + ).set(size_bytes) + except Exception as e: + verbose_logger.warning(f"Error recording file size metric: {e}") + + def record_managed_batch_duration( + self, + duration_seconds: float, + model: Optional[str] = None, + api_provider: Optional[str] = None, + ): + try: + self.litellm_managed_batch_duration_seconds.labels( + model=model or "", + api_provider=api_provider or "", + ).observe(duration_seconds) + except Exception as e: + verbose_logger.warning(f"Error recording batch duration metric: {e}") + + def record_managed_file_created( + self, + model: Optional[str], + api_provider: Optional[str], + user: Optional[str], + user_email: Optional[str], + api_key_alias: Optional[str], + ): + try: + self.litellm_managed_file_created_total.labels( + model=model, + api_provider=api_provider, + user=user, + user_email=user_email, + api_key_alias=api_key_alias, + ).inc() + except Exception as e: + verbose_logger.warning(f"Error recording file created metric: {e}") + + def record_managed_file_deleted(self, result: str): + """Record a managed file deletion attempt. result is 'success' or 'blocked'.""" + try: + self.litellm_managed_file_deleted_total.labels(result=result).inc() + except Exception as e: + verbose_logger.warning(f"Error recording file deleted metric: {e}") + + def record_check_batch_cost_run( + self, + jobs_polled: int, + processed_models: Optional[List[Tuple[Optional[str], Optional[str]]]] = None, + ): + """ + Record CheckBatchCost polling metrics. + + Args: + jobs_polled: Number of unprocessed batches found + processed_models: List of (model, api_provider) tuples for processed jobs + """ + import time + + try: + self.litellm_check_batch_cost_last_run_timestamp.set(time.time()) + self.litellm_check_batch_cost_jobs_polled.set(jobs_polled) + + if processed_models: + for model, api_provider in processed_models: + self.litellm_check_batch_cost_jobs_processed_total.labels( + model=model or "", + api_provider=api_provider or "", + ).inc() + except Exception as e: + verbose_logger.warning(f"Error recording check batch cost metrics: {e}") + + def record_check_batch_cost_error(self, error_type: str): + try: + self.litellm_check_batch_cost_errors_total.labels( + error_type=error_type, + ).inc() + except Exception as e: + verbose_logger.warning( + f"Error recording check batch cost error metric: {e}" + ) + @staticmethod def _get_exception_class_name(exception: Exception) -> str: exception_class_name = "" diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index 0856d8a6f9b..05913e609f6 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -159,6 +159,23 @@ LATENCY_BUCKETS = ( float("inf"), ) +# Batch jobs can run for minutes to hours; buckets span 1 min → 24 h. +BATCH_DURATION_BUCKETS = ( + 60.0, + 120.0, + 300.0, + 600.0, + 900.0, + 1800.0, + 3600.0, + 7200.0, + 14400.0, + 28800.0, + 43200.0, + 86400.0, + float("inf"), +) + class UserAPIKeyLabelNames(Enum): END_USER = "end_user" @@ -238,6 +255,16 @@ DEFINED_PROMETHEUS_METRICS = Literal[ "litellm_llm_api_failed_requests_metric", "litellm_callback_logging_failures_metric", "litellm_in_flight_requests", + # Managed batch metrics + "litellm_managed_batch_created_total", + "litellm_managed_file_size_bytes", + "litellm_managed_batch_duration_seconds", + "litellm_managed_file_created_total", + "litellm_managed_file_deleted_total", + "litellm_check_batch_cost_jobs_polled", + "litellm_check_batch_cost_jobs_processed_total", + "litellm_check_batch_cost_errors_total", + "litellm_check_batch_cost_last_run_timestamp", ] @@ -618,6 +645,43 @@ class PrometheusMetricLabels: litellm_cache_misses_metric = _cache_metric_labels litellm_cached_tokens_metric = _cache_metric_labels + # Managed batch metrics + _batch_user_labels = [ + UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, + UserAPIKeyLabelNames.API_PROVIDER.value, + UserAPIKeyLabelNames.USER.value, + UserAPIKeyLabelNames.USER_EMAIL.value, + UserAPIKeyLabelNames.API_KEY_ALIAS.value, + ] + + litellm_managed_batch_created_total = _batch_user_labels + + litellm_managed_file_size_bytes: List[ + str + ] = [] # labels: purpose, file_type, model, api_provider, user (custom) + + litellm_managed_batch_duration_seconds = [ + UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, + UserAPIKeyLabelNames.API_PROVIDER.value, + ] + + litellm_managed_file_created_total = _batch_user_labels + + litellm_managed_file_deleted_total: List[ + str + ] = [] # only "result" label, added at metric creation + + litellm_check_batch_cost_jobs_polled: List[str] = [] + + litellm_check_batch_cost_jobs_processed_total = [ + UserAPIKeyLabelNames.v1_LITELLM_MODEL_NAME.value, + UserAPIKeyLabelNames.API_PROVIDER.value, + ] + + litellm_check_batch_cost_errors_total: List[str] = [] # label: error_type (custom) + + litellm_check_batch_cost_last_run_timestamp: List[str] = [] + @staticmethod def get_labels(label_name: DEFINED_PROMETHEUS_METRICS) -> List[str]: default_labels = getattr(PrometheusMetricLabels, label_name) From 9d7fc307b80fcb8efeb46fdde408a184c68851e1 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 09:28:53 +0530 Subject: [PATCH 053/117] fix(openrouter): strip LiteLLM prefix when proxy sets custom_llm_provider Wildcard openrouter/* deployments pass custom_llm_provider=openrouter with the full openrouter/provider/model id; OpenRouter expects provider/model. Strip the outer openrouter/ only when the remainder contains a slash so native ids like openrouter/auto stay intact. Adds regression test for proxy wildcard path. Made-with: Cursor --- .../litellm_core_utils/get_llm_provider_logic.py | 13 ++++++++----- .../openrouter/test_openrouter_provider_routing.py | 14 +++++++++++++- 2 files changed, 21 insertions(+), 6 deletions(-) diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 36218417377..c26759eaf0a 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -158,12 +158,15 @@ def get_llm_provider( # noqa: PLR0915 ): # handle scenario where model="azure/*" and custom_llm_provider="azure" model = custom_llm_provider + "/" + model - # Native OpenRouter models have IDs like "openrouter/free" where the - # "openrouter/" prefix is part of the actual model name on the API. - # When called from a bridge (e.g. anthropic_messages adapter), - # custom_llm_provider is already resolved, so return early to prevent - # the provider-list stripping below from removing the prefix. + # OpenRouter: when the router/proxy already set custom_llm_provider, + # the model may still carry LiteLLM's "openrouter/" routing prefix. + # Native IDs like "openrouter/auto" must stay intact for the API; IDs + # like "openrouter/anthropic/claude-3.5-sonnet" must become + # "anthropic/claude-3.5-sonnet" (OpenRouter expects provider/model). if custom_llm_provider == "openrouter" and model.startswith("openrouter/"): + remainder = model[len("openrouter/") :] + if "/" in remainder: + return remainder, custom_llm_provider, dynamic_api_key, api_base return model, custom_llm_provider, dynamic_api_key, api_base if api_key and api_key.startswith("os.environ/"): diff --git a/tests/test_litellm/llms/openrouter/test_openrouter_provider_routing.py b/tests/test_litellm/llms/openrouter/test_openrouter_provider_routing.py index 72cf2eec371..0815b15c873 100644 --- a/tests/test_litellm/llms/openrouter/test_openrouter_provider_routing.py +++ b/tests/test_litellm/llms/openrouter/test_openrouter_provider_routing.py @@ -80,7 +80,10 @@ class TestOpenRouterNativeModelRouting: "input_model,expected_model", [ ("openrouter/anthropic/claude-3-haiku", "anthropic/claude-3-haiku"), - ("openrouter/meta-llama/llama-3-70b-instruct", "meta-llama/llama-3-70b-instruct"), + ( + "openrouter/meta-llama/llama-3-70b-instruct", + "meta-llama/llama-3-70b-instruct", + ), ], ) def test_regular_models_still_strip_normally(self, input_model, expected_model): @@ -88,3 +91,12 @@ class TestOpenRouterNativeModelRouting: result_model, provider, _, _ = litellm.get_llm_provider(model=input_model) assert provider == "openrouter" assert result_model == expected_model + + def test_wildcard_deployment_strips_routing_prefix(self): + """openrouter/* proxy deployments pass custom_llm_provider; strip LiteLLM prefix.""" + result_model, provider, _, _ = litellm.get_llm_provider( + model="openrouter/anthropic/claude-3.5-sonnet", + custom_llm_provider="openrouter", + ) + assert provider == "openrouter" + assert result_model == "anthropic/claude-3.5-sonnet" From cc73ae776a703ba9dd0f7d78ae45e8d82580c656 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 11:41:10 +0530 Subject: [PATCH 054/117] feat(gemini): add Lyria 3 preview models to cost map and docs Made-with: Cursor --- docs/my-website/docs/providers/gemini.md | 1 + .../my-website/docs/providers/gemini/music.md | 28 ++++++++++ docs/my-website/sidebars.js | 1 + ...odel_prices_and_context_window_backup.json | 54 +++++++++++++++++++ model_prices_and_context_window.json | 54 +++++++++++++++++++ tests/test_litellm/test_utils.py | 18 +++++++ 6 files changed, 156 insertions(+) create mode 100644 docs/my-website/docs/providers/gemini/music.md diff --git a/docs/my-website/docs/providers/gemini.md b/docs/my-website/docs/providers/gemini.md index c8c9114ea87..87ab5ad40f4 100644 --- a/docs/my-website/docs/providers/gemini.md +++ b/docs/my-website/docs/providers/gemini.md @@ -11,6 +11,7 @@ import TabItem from '@theme/TabItem'; | Provider Doc | [Google AI Studio ↗](https://aistudio.google.com/) | | API Endpoint for Provider | https://generativelanguage.googleapis.com | | Supported OpenAI Endpoints | `/chat/completions`, [`/embeddings`](../embedding/supported_embedding#gemini-ai-embedding-models), `/completions`, [`/videos`](./gemini/videos.md), [`/images/edits`](../image_edits.md) | +| Lyria (music) | [Cost map & notes](./gemini/music.md) | | Pass-through Endpoint | [Supported](../pass_through/google_ai_studio.md) |
diff --git a/docs/my-website/docs/providers/gemini/music.md b/docs/my-website/docs/providers/gemini/music.md new file mode 100644 index 00000000000..f3968f2db39 --- /dev/null +++ b/docs/my-website/docs/providers/gemini/music.md @@ -0,0 +1,28 @@ +# Gemini — Lyria (music generation) + +Google Lyria 3 preview models are listed in LiteLLM’s [model cost map](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json) under the `gemini/` provider for metadata and spend tracking. + +| Property | Details | +|----------|---------| +| Provider route | `gemini/` | +| Models | `gemini/lyria-3-clip-preview`, `gemini/lyria-3-pro-preview` | +| Provider docs | [Gemini API pricing / models ↗](https://ai.google.dev/gemini-api/docs/pricing) | + +## Models + +| Model | Notes | +|-------|--------| +| `gemini/lyria-3-clip-preview` | ~30s clip; paid tier listed as per generated song in Google’s pricing | +| `gemini/lyria-3-pro-preview` | Full song; paid tier listed as per generated song in Google’s pricing | + +Input context limit in the cost map: **131,072** tokens. For modalities, limits, and features, see [Google’s Gemini API docs ↗](https://ai.google.dev/gemini-api/docs/models). + +## LiteLLM behavior + +- **Cost map**: Per-song paid pricing is stored as `output_cost_per_image` on those entries (flat per generation unit). Token-based completion cost may not reflect music billing until a dedicated path exists. +- **API calls**: Use the Gemini API as documented by Google. LiteLLM does not ship a separate `music_generation` helper like Veo’s `video_generation`. + +## Auth + +Same as other Gemini API models: `GEMINI_API_KEY` or `GOOGLE_API_KEY`. + diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 6446e227d99..cc2d7800a21 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -850,6 +850,7 @@ const sidebars = { items: [ "providers/gemini", "providers/gemini/videos", + "providers/gemini/music", "providers/google_ai_studio/files", "providers/google_ai_studio/image_gen", "providers/google_ai_studio/realtime", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c53ee943c58..38faed43e70 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -15931,6 +15931,60 @@ "supports_tool_choice": true, "supports_vision": true }, + "gemini/lyria-3-clip-preview": { + "input_cost_per_token": 0, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_image": 0.04, + "output_cost_per_token": 0, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "audio", + "text" + ], + "supports_audio_input": false, + "supports_audio_output": true, + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": false, + "supports_vision": true, + "supports_web_search": false + }, + "gemini/lyria-3-pro-preview": { + "input_cost_per_token": 0, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_image": 0.08, + "output_cost_per_token": 0, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "audio", + "text" + ], + "supports_audio_input": false, + "supports_audio_output": true, + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": false, + "supports_vision": true, + "supports_web_search": false + }, "gemini/veo-2.0-generate-001": { "litellm_provider": "gemini", "max_input_tokens": 1024, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c53ee943c58..38faed43e70 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -15931,6 +15931,60 @@ "supports_tool_choice": true, "supports_vision": true }, + "gemini/lyria-3-clip-preview": { + "input_cost_per_token": 0, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_image": 0.04, + "output_cost_per_token": 0, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "audio", + "text" + ], + "supports_audio_input": false, + "supports_audio_output": true, + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": false, + "supports_vision": true, + "supports_web_search": false + }, + "gemini/lyria-3-pro-preview": { + "input_cost_per_token": 0, + "litellm_provider": "gemini", + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_image": 0.08, + "output_cost_per_token": 0, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "audio", + "text" + ], + "supports_audio_input": false, + "supports_audio_output": true, + "supports_function_calling": false, + "supports_prompt_caching": false, + "supports_response_schema": false, + "supports_system_messages": false, + "supports_vision": true, + "supports_web_search": false + }, "gemini/veo-2.0-generate-001": { "litellm_provider": "gemini", "max_input_tokens": 1024, diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 38b7b576d4f..21cd1a37496 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -961,6 +961,7 @@ def test_get_model_info_gemini(): and not "learnlm" in model and not "imagen" in model and not "veo" in model + and not "lyria" in model and not "robotics" in model ): assert info.get("tpm") is not None, f"{model} does not have tpm" @@ -2788,6 +2789,23 @@ def test_model_info_for_openrouter_kimi_k2_5(): print("openrouter kimi-k2.5 model info", model_info) +def test_gemini_lyria_3_preview_models_in_cost_map(): + import json + from pathlib import Path + + json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json" + with open(json_path) as f: + model_cost = json.load(f) + + clip = model_cost.get("gemini/lyria-3-clip-preview") + pro = model_cost.get("gemini/lyria-3-pro-preview") + assert clip is not None and pro is not None + assert clip["litellm_provider"] == "gemini" and pro["litellm_provider"] == "gemini" + assert clip["max_input_tokens"] == 131072 == pro["max_input_tokens"] + assert clip["output_cost_per_image"] == 0.04 + assert pro["output_cost_per_image"] == 0.08 + + def test_model_info_for_fireworks_short_form_models(): """ Test that fireworks_ai short-form model entries (fireworks_ai/) From 3bbc6944613878597e24a0bd6fc90b1b3a8ccb6a Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 16:17:33 +0530 Subject: [PATCH 055/117] Fix model map --- .../model_prices_and_context_window_backup.json | 16 ++++++---------- model_prices_and_context_window.json | 16 ++++++---------- 2 files changed, 12 insertions(+), 20 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 38faed43e70..ad54a29f013 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -15942,12 +15942,10 @@ "output_cost_per_token": 0, "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_modalities": [ - "text", - "image" + "text" ], "supported_output_modalities": [ - "audio", - "text" + "audio" ], "supports_audio_input": false, "supports_audio_output": true, @@ -15955,7 +15953,7 @@ "supports_prompt_caching": false, "supports_response_schema": false, "supports_system_messages": false, - "supports_vision": true, + "supports_vision": false, "supports_web_search": false }, "gemini/lyria-3-pro-preview": { @@ -15969,12 +15967,10 @@ "output_cost_per_token": 0, "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_modalities": [ - "text", - "image" + "text" ], "supported_output_modalities": [ - "audio", - "text" + "audio" ], "supports_audio_input": false, "supports_audio_output": true, @@ -15982,7 +15978,7 @@ "supports_prompt_caching": false, "supports_response_schema": false, "supports_system_messages": false, - "supports_vision": true, + "supports_vision": false, "supports_web_search": false }, "gemini/veo-2.0-generate-001": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 38faed43e70..ad54a29f013 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -15942,12 +15942,10 @@ "output_cost_per_token": 0, "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_modalities": [ - "text", - "image" + "text" ], "supported_output_modalities": [ - "audio", - "text" + "audio" ], "supports_audio_input": false, "supports_audio_output": true, @@ -15955,7 +15953,7 @@ "supports_prompt_caching": false, "supports_response_schema": false, "supports_system_messages": false, - "supports_vision": true, + "supports_vision": false, "supports_web_search": false }, "gemini/lyria-3-pro-preview": { @@ -15969,12 +15967,10 @@ "output_cost_per_token": 0, "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_modalities": [ - "text", - "image" + "text" ], "supported_output_modalities": [ - "audio", - "text" + "audio" ], "supports_audio_input": false, "supports_audio_output": true, @@ -15982,7 +15978,7 @@ "supports_prompt_caching": false, "supports_response_schema": false, "supports_system_messages": false, - "supports_vision": true, + "supports_vision": false, "supports_web_search": false }, "gemini/veo-2.0-generate-001": { From 1aa90f9bd16e624a8a9fa9c9ea26e67a2fc64689 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 16:18:43 +0530 Subject: [PATCH 056/117] Fix model map --- litellm/model_prices_and_context_window_backup.json | 1 - model_prices_and_context_window.json | 1 - 2 files changed, 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ad54a29f013..953c144d0c7 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -15963,7 +15963,6 @@ "max_output_tokens": 8192, "max_tokens": 8192, "mode": "chat", - "output_cost_per_image": 0.08, "output_cost_per_token": 0, "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_modalities": [ diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ad54a29f013..953c144d0c7 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -15963,7 +15963,6 @@ "max_output_tokens": 8192, "max_tokens": 8192, "mode": "chat", - "output_cost_per_image": 0.08, "output_cost_per_token": 0, "source": "https://ai.google.dev/gemini-api/docs/pricing", "supported_modalities": [ From cdc1dd5c373bd971fa02b9f6316c38313bf5cf68 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 19:51:35 +0530 Subject: [PATCH 057/117] Fix the tests --- tests/test_litellm/test_utils.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 21cd1a37496..eb2f510e239 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -2803,7 +2803,6 @@ def test_gemini_lyria_3_preview_models_in_cost_map(): assert clip["litellm_provider"] == "gemini" and pro["litellm_provider"] == "gemini" assert clip["max_input_tokens"] == 131072 == pro["max_input_tokens"] assert clip["output_cost_per_image"] == 0.04 - assert pro["output_cost_per_image"] == 0.08 def test_model_info_for_fireworks_short_form_models(): From 8112fbf27428dfd2cd2e7c075692966a3b36a563 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 15:33:25 +0530 Subject: [PATCH 058/117] fix(proxy): sanitize user_id input and block dangerous env var keys Add input validation to get_user_id_from_request (length limit, control char rejection) and a blocklist of dangerous environment variable keys in _load_environment_variables to prevent PATH/LD_PRELOAD/PYTHONPATH override via config. Co-Authored-By: Claude Opus 4.6 --- .../internal_user_endpoints.py | 17 +- litellm/proxy/proxy_server.py | 28 +- .../test_internal_user_endpoints.py | 177 +++++-- tests/test_litellm/proxy/test_proxy_server.py | 466 +++++++++++------- 4 files changed, 459 insertions(+), 229 deletions(-) diff --git a/litellm/proxy/management_endpoints/internal_user_endpoints.py b/litellm/proxy/management_endpoints/internal_user_endpoints.py index ca8c345f46c..b2df9a1a2cb 100644 --- a/litellm/proxy/management_endpoints/internal_user_endpoints.py +++ b/litellm/proxy/management_endpoints/internal_user_endpoints.py @@ -24,13 +24,13 @@ import litellm from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid from litellm.proxy._types import * +from litellm.proxy.auth.auth_checks import get_team_object, get_user_object from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.proxy.hooks.user_management_event_hooks import UserManagementEventHooks from litellm.proxy.management_endpoints.common_daily_activity import ( get_daily_activity, get_daily_activity_aggregated, ) -from litellm.proxy.auth.auth_checks import get_team_object, get_user_object from litellm.proxy.management_endpoints.common_utils import ( _is_user_team_admin, _user_has_admin_view, @@ -557,6 +557,18 @@ def get_team_from_list( return None +def _is_valid_user_id(user_id: str) -> bool: + """Validate that a decoded user_id is safe to use downstream.""" + MAX_USER_ID_LENGTH = 512 + if len(user_id) > MAX_USER_ID_LENGTH: + return False + # Reject null bytes and control characters (< 0x20) except space + for ch in user_id: + if ch == "\x00" or (ord(ch) < 0x20 and ch != " "): + return False + return True + + def get_user_id_from_request(request: Request) -> Optional[str]: """ Get the user id from the request @@ -573,7 +585,8 @@ def get_user_id_from_request(request: Request) -> Optional[str]: if match: # Use unquote instead of unquote_plus to preserve + characters raw_user_id = unquote(match.group(1)) - user_id = raw_user_id + if _is_valid_user_id(raw_user_id): + user_id = raw_user_id return user_id diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 7d3d2ceb533..203344d452c 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -480,11 +480,11 @@ from litellm.proxy.search_endpoints.search_tool_management import ( router as search_tool_management_router, ) from litellm.proxy.spend_tracking.cloudzero_endpoints import router as cloudzero_router -from litellm.proxy.spend_tracking.vantage_endpoints import router as vantage_router from litellm.proxy.spend_tracking.spend_management_endpoints import ( router as spend_management_router, ) from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload +from litellm.proxy.spend_tracking.vantage_endpoints import router as vantage_router from litellm.proxy.types_utils.utils import get_instance_fn from litellm.proxy.ui_crud_endpoints.proxy_setting_endpoints import ( router as ui_crud_endpoints_router, @@ -2695,12 +2695,38 @@ class ProxyConfig: return search_tools_parsed if search_tools_parsed else None + # Environment variable keys that must not be overridden via config because + # they can alter process execution, library loading, or network routing. + _BLOCKED_ENV_KEYS: Set[str] = { + "PATH", + "LD_PRELOAD", + "LD_LIBRARY_PATH", + "DYLD_LIBRARY_PATH", + "DYLD_INSERT_LIBRARIES", + "PYTHONPATH", + "PYTHONSTARTUP", + "PYTHONHOME", + "HOME", + "USER", + "SHELL", + "LOGNAME", + "http_proxy", + "https_proxy", + "HTTP_PROXY", + "HTTPS_PROXY", + } + def _load_environment_variables(self, config: dict): ## ENVIRONMENT VARIABLES global premium_user environment_variables = config.get("environment_variables", None) if environment_variables: for key, value in environment_variables.items(): + if key in self._BLOCKED_ENV_KEYS: + verbose_proxy_logger.warning( + "Skipping blocked environment variable key: %s", key + ) + continue ######################################################### # handles this scenario: # ```yaml diff --git a/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py index e358cbe3be4..0f90d236aed 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py @@ -85,6 +85,7 @@ async def test_ui_view_users_proxy_admin_no_org_filter(mocker): Proxy admin: find_many is called without organization_memberships in where. """ mock_prisma_client = mocker.MagicMock() + async def mock_find_many(*args, **kwargs): assert "organization_memberships" not in (kwargs.get("where") or {}) return [] @@ -327,6 +328,7 @@ async def test_ui_view_users_flag_on_team_admin_non_org_team_403(mocker): Flag ON, team admin for non-org team: returns 403. """ from fastapi import HTTPException + from litellm.proxy._types import LiteLLM_TeamTableCachedObj mock_prisma_client = mocker.MagicMock() @@ -372,9 +374,7 @@ async def test_ui_view_users_flag_on_team_admin_non_org_team_403(mocker): with pytest.raises(HTTPException) as exc_info: await ui_view_users( - user_api_key_dict=UserAPIKeyAuth( - user_id="team-admin-user", user_role=None - ), + user_api_key_dict=UserAPIKeyAuth(user_id="team-admin-user", user_role=None), user_id=None, user_email="u", team_id=tid, @@ -633,7 +633,9 @@ async def test_get_users_includes_timestamps(mocker): # Call get_users function directly with proxy admin auth admin_key = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN) - response = await get_users(page=1, page_size=1, user_api_key_dict=admin_key, organization_ids=None) + response = await get_users( + page=1, page_size=1, user_api_key_dict=admin_key, organization_ids=None + ) print("user /list response: ", response) @@ -855,7 +857,9 @@ async def test_new_user_non_admin_cannot_create_admin(mocker): # Verify the exception details assert exc_info.value.code == 403 or exc_info.value.code == "403" - assert "Only proxy admins can create administrative users" in str(exc_info.value.message) + assert "Only proxy admins can create administrative users" in str( + exc_info.value.message + ) assert "proxy_admin" in str(exc_info.value.message) assert "proxy_admin_viewer" in str(exc_info.value.message) assert str(LitellmUserRoles.PROXY_ADMIN) in str(exc_info.value.message) @@ -896,14 +900,14 @@ async def test_user_info_url_encoding_plus_character(mocker): # Mock the prisma client mock_prisma_client = mocker.MagicMock() - + # Create a real LiteLLM_UserTable instance (BaseModel) so isinstance check passes mock_user = LiteLLM_UserTable( user_id="machine-user+alp-air-admin-b58-b@tempus.com", user_email="machine-user+alp-air-admin-b58-b@tempus.com", teams=[], ) - + # Mock get_data to return user when called with user_id, empty list for keys async def mock_get_data(*args, **kwargs): if kwargs.get("table_name") == "key": @@ -913,7 +917,7 @@ async def test_user_info_url_encoding_plus_character(mocker): elif kwargs.get("user_id") is not None: return mock_user return None - + mock_prisma_client.get_data = mocker.AsyncMock(side_effect=mock_get_data) # Mock list_team to return None (patch it from where it's imported) @@ -941,7 +945,7 @@ async def test_user_info_url_encoding_plus_character(mocker): "machine-user alp-air-admin-b58-b@tempus.com" # What FastAPI gives us ) expected_user_id = "machine-user+alp-air-admin-b58-b@tempus.com" - + response = await user_info( user_id=decoded_user_id, user_api_key_dict=mock_user_api_key_dict, @@ -955,7 +959,7 @@ async def test_user_info_url_encoding_plus_character(mocker): if call.kwargs.get("user_id") and not call.kwargs.get("table_name"): user_call = call break - + assert user_call is not None, "get_data should be called with user_id" assert user_call.kwargs["user_id"] == expected_user_id @@ -972,7 +976,7 @@ async def test_user_info_nonexistent_user(mocker): # Mock the prisma client mock_prisma_client = mocker.MagicMock() - + # Mock get_data to return None (user doesn't exist) async def mock_get_data(*args, **kwargs): if kwargs.get("table_name") == "key": @@ -980,7 +984,7 @@ async def test_user_info_nonexistent_user(mocker): elif kwargs.get("user_id") is not None: return None # User not found return None - + mock_prisma_client.get_data = mocker.AsyncMock(side_effect=mock_get_data) # Patch the prisma client import in the endpoint @@ -996,7 +1000,7 @@ async def test_user_info_nonexistent_user(mocker): # Call user_info function with a non-existent user_id nonexistent_user_id = "nonexistent-user@example.com" - + # Should raise ProxyException with 404 status code (HTTPException is converted by decorator) with pytest.raises(ProxyException) as exc_info: await user_info( @@ -1370,9 +1374,7 @@ async def test_check_duplicate_user_id(mocker): await _check_duplicate_user_id("existing-user-id", mock_prisma_client) assert exc_info.value.status_code == 409 - assert "User with id existing-user-id already exists" in str( - exc_info.value.detail - ) + assert "User with id existing-user-id already exists" in str(exc_info.value.detail) # No duplicate should pass async def mock_find_first_no_duplicate(*args, **kwargs): @@ -1393,7 +1395,7 @@ async def test_check_duplicate_user_id(mocker): def test_process_keys_for_user_info_filters_dashboard_keys(monkeypatch): """ Test that _process_keys_for_user_info filters out keys with team_id='litellm-dashboard' - + UI session tokens (team_id='litellm-dashboard') should be excluded from user info responses to prevent confusion, as these are automatically created during dashboard login. """ @@ -1412,7 +1414,7 @@ def test_process_keys_for_user_info_filters_dashboard_keys(monkeypatch): "user_id": "test-user", "key_alias": "dashboard-session-key", } - + mock_key_regular = MagicMock() mock_key_regular.model_dump.return_value = { "token": "sk-regular-token", @@ -1420,7 +1422,7 @@ def test_process_keys_for_user_info_filters_dashboard_keys(monkeypatch): "user_id": "test-user", "key_alias": "regular-key", } - + mock_key_no_team = MagicMock() mock_key_no_team.model_dump.return_value = { "token": "sk-no-team-token", @@ -1446,20 +1448,24 @@ def test_process_keys_for_user_info_filters_dashboard_keys(monkeypatch): # Verify that dashboard key is filtered out assert len(result) == 2, "Should return 2 keys (dashboard key filtered out)" - + # Verify dashboard key is not in results result_team_ids = [key.get("team_id") for key in result] - assert UI_SESSION_TOKEN_TEAM_ID not in result_team_ids, "Dashboard key should be filtered out" - + assert ( + UI_SESSION_TOKEN_TEAM_ID not in result_team_ids + ), "Dashboard key should be filtered out" + # Verify regular keys are included assert "regular-team" in result_team_ids, "Regular team key should be included" assert None in result_team_ids, "No-team key should be included" - + # Verify the correct keys are returned result_tokens = [key.get("token") for key in result] assert "sk-regular-token" in result_tokens, "Regular key should be included" assert "sk-no-team-token" in result_tokens, "No-team key should be included" - assert "sk-dashboard-token" not in result_tokens, "Dashboard key should not be included" + assert ( + "sk-dashboard-token" not in result_tokens + ), "Dashboard key should not be included" def test_process_keys_for_user_info_handles_none_keys(monkeypatch): @@ -1558,7 +1564,13 @@ async def test_get_users_user_id_partial_match(mocker): admin_key = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN) captured_where_conditions.clear() - await get_users(user_ids="test-user", page=1, page_size=1, user_api_key_dict=admin_key, organization_ids=None) + await get_users( + user_ids="test-user", + page=1, + page_size=1, + user_api_key_dict=admin_key, + organization_ids=None, + ) assert "user_id" in captured_where_conditions assert "contains" in captured_where_conditions["user_id"] @@ -1566,7 +1578,13 @@ async def test_get_users_user_id_partial_match(mocker): assert captured_where_conditions["user_id"]["mode"] == "insensitive" captured_where_conditions.clear() - await get_users(user_ids="user1,user2,user3", page=1, page_size=1, user_api_key_dict=admin_key, organization_ids=None) + await get_users( + user_ids="user1,user2,user3", + page=1, + page_size=1, + user_api_key_dict=admin_key, + organization_ids=None, + ) assert "user_id" in captured_where_conditions assert "in" in captured_where_conditions["user_id"] @@ -1578,7 +1596,7 @@ def test_update_internal_user_params_reset_max_budget_with_none(): Test that _update_internal_user_params allows setting max_budget to None. This verifies the fix for unsetting/resetting the budget to unlimited. """ - + # Case 1: max_budget is explicitly None in the input dictionary data_json = {"max_budget": None, "user_id": "test_user"} data = UpdateUserRequest(max_budget=None, user_id="test_user") @@ -1610,7 +1628,7 @@ def test_update_internal_user_params_ignores_other_nones(): def test_update_internal_user_params_keeps_original_max_budget_when_not_provided(): """ - Test that _update_internal_user_params does not include max_budget + Test that _update_internal_user_params does not include max_budget when it's not provided in the request (should keep original value). """ # Create test data without max_budget @@ -1631,7 +1649,7 @@ def test_generate_request_base_validator(): Test that GenerateRequestBase validator converts empty string to None for max_budget """ from litellm.proxy._types import GenerateRequestBase - + # Test with empty string req = GenerateRequestBase(max_budget="") assert req.max_budget is None @@ -1662,9 +1680,7 @@ async def test_get_user_daily_activity_non_admin_cannot_view_other_users(monkeyp # Mock the prisma client so the DB-not-connected check passes mock_prisma_client = MagicMock() - monkeypatch.setattr( - "litellm.proxy.proxy_server.prisma_client", mock_prisma_client - ) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client) # Non-admin caller non_admin_key_dict = UserAPIKeyAuth( @@ -1731,9 +1747,7 @@ async def test_get_user_daily_activity_aggregated_admin_global_view(monkeypatch) # Mock the prisma client mock_prisma_client = MagicMock() - monkeypatch.setattr( - "litellm.proxy.proxy_server.prisma_client", mock_prisma_client - ) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", mock_prisma_client) # Mock the downstream helper so we don't need a real DB mock_response = MagicMock() @@ -1846,7 +1860,9 @@ async def test_delete_user_cleans_up_created_by_invitation_links(mocker): call_kwargs = mock_prisma_client.db.litellm_invitationlink.delete_many.call_args where_clause = call_kwargs.kwargs.get("where") or call_kwargs[1].get("where") - assert "OR" in where_clause, "Should use OR to match user_id, created_by, and updated_by" + assert ( + "OR" in where_clause + ), "Should use OR to match user_id, created_by, and updated_by" or_conditions = where_clause["OR"] assert len(or_conditions) == 3, "Should have 3 OR conditions" @@ -2188,9 +2204,20 @@ async def test_user_info_v2_response_shape(mocker): # Verify all expected fields are present response_dict = response.model_dump() expected_fields = { - "user_id", "user_email", "user_alias", "user_role", "spend", - "max_budget", "models", "budget_duration", "budget_reset_at", - "metadata", "created_at", "updated_at", "sso_user_id", "teams", + "user_id", + "user_email", + "user_alias", + "user_role", + "spend", + "max_budget", + "models", + "budget_duration", + "budget_reset_at", + "metadata", + "created_at", + "updated_at", + "sso_user_id", + "teams", } assert set(response_dict.keys()) == expected_fields @@ -2418,4 +2445,74 @@ async def test_user_info_v2_url_encoding_plus_character(mocker): ) assert isinstance(response, UserInfoV2Response) - assert response.user_id == expected_user_id \ No newline at end of file + assert response.user_id == expected_user_id + + +class TestGetUserIdFromRequestValidation: + """Tests for user_id input validation in get_user_id_from_request.""" + + def _make_request(self, query_string: str): + from unittest.mock import MagicMock + + from starlette.requests import Request + + request = MagicMock(spec=Request) + request.url.query = query_string + return request + + def test_valid_uuid(self): + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + get_user_id_from_request, + ) + + request = self._make_request("user_id=550e8400-e29b-41d4-a716-446655440000") + result = get_user_id_from_request(request) + assert result == "550e8400-e29b-41d4-a716-446655440000" + + def test_valid_email(self): + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + get_user_id_from_request, + ) + + request = self._make_request("user_id=user%40example.com") + result = get_user_id_from_request(request) + assert result == "user@example.com" + + def test_rejects_overlong_user_id(self): + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + get_user_id_from_request, + ) + + long_id = "a" * 513 + request = self._make_request(f"user_id={long_id}") + result = get_user_id_from_request(request) + assert result is None + + def test_rejects_null_byte(self): + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + get_user_id_from_request, + ) + + request = self._make_request("user_id=admin%00evil") + result = get_user_id_from_request(request) + assert result is None + + def test_rejects_control_characters(self): + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + get_user_id_from_request, + ) + + # Tab character (0x09) + request = self._make_request("user_id=admin%09evil") + result = get_user_id_from_request(request) + assert result is None + + def test_allows_512_char_user_id(self): + from litellm.proxy.management_endpoints.internal_user_endpoints import ( + get_user_id_from_request, + ) + + exact_id = "a" * 512 + request = self._make_request(f"user_id={exact_id}") + result = get_user_id_from_request(request) + assert result == exact_id diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py index bd6162f225a..334cd7395d6 100644 --- a/tests/test_litellm/proxy/test_proxy_server.py +++ b/tests/test_litellm/proxy/test_proxy_server.py @@ -104,10 +104,7 @@ def test_login_v2_returns_redirect_url_and_sets_cookie(monkeypatch): ) assert response.status_code == 200 - assert ( - response.json() - == {"redirect_url": "http://testserver/ui/?login=success"} - ) + assert response.json() == {"redirect_url": "http://testserver/ui/?login=success"} assert response.cookies.get("token") == "signed-token" mock_authenticate_user.assert_awaited_once_with( @@ -516,15 +513,11 @@ def test_restructure_ui_html_files_handles_nested_routes(tmp_path): assert (ui_root / "home" / "index.html").read_text() == "home" assert not (ui_root / "mcp" / "oauth" / "callback.html").exists() assert ( - (ui_root / "mcp" / "oauth" / "callback" / "index.html").read_text() - == "callback" - ) + ui_root / "mcp" / "oauth" / "callback" / "index.html" + ).read_text() == "callback" assert (ui_root / "existing" / "index.html").read_text() == "keep" assert (ui_root / "_next" / "ignore.html").read_text() == "asset" - assert ( - (ui_root / "litellm-asset-prefix" / "ignore.html").read_text() - == "asset" - ) + assert (ui_root / "litellm-asset-prefix" / "ignore.html").read_text() == "asset" def test_ui_extensionless_route_requires_restructure(tmp_path): @@ -541,9 +534,7 @@ def test_ui_extensionless_route_requires_restructure(tmp_path): (ui_root / "login.html").write_text("login") fastapi_app = FastAPI() - fastapi_app.mount( - "/ui", StaticFiles(directory=str(ui_root), html=True), name="ui" - ) + fastapi_app.mount("/ui", StaticFiles(directory=str(ui_root), html=True), name="ui") client = TestClient(fastapi_app) assert client.get("/ui/login.html").status_code == 200 @@ -564,37 +555,37 @@ def test_restructure_always_happens(monkeypatch): """ # Test Case 1: is_non_root is True - restructuring happens in /var/lib/litellm/ui monkeypatch.setenv("LITELLM_NON_ROOT", "true") - + runtime_ui_path = "/var/lib/litellm/ui" packaged_ui_path = "/some/packaged/ui/path" - + # Simulate the logic from proxy_server.py is_non_root = os.getenv("LITELLM_NON_ROOT", "").lower() == "true" if is_non_root: ui_path = runtime_ui_path else: ui_path = packaged_ui_path - + # Restructuring always happens now, regardless of ui_path vs packaged_ui_path should_restructure = True - + assert is_non_root is True assert should_restructure is True assert ui_path == runtime_ui_path - + # Test Case 2: is_non_root is False - restructuring happens directly in packaged_ui_path monkeypatch.delenv("LITELLM_NON_ROOT", raising=False) - + # Simulate the logic from proxy_server.py is_non_root = os.getenv("LITELLM_NON_ROOT", "").lower() == "true" if is_non_root: ui_path = runtime_ui_path else: ui_path = packaged_ui_path - + # Restructuring always happens now, even when ui_path == packaged_ui_path should_restructure = True - + assert is_non_root is False assert should_restructure is True assert ui_path == packaged_ui_path @@ -691,9 +682,7 @@ def test_update_config_fields_deep_merge_db_wins(): "hidden": True, }, # Demonstrate that None values from DB are skipped (preserve existing) - "legacy-sonnet": { - "hidden": None # should not clobber current True - }, + "legacy-sonnet": {"hidden": None}, # should not clobber current True } } @@ -743,9 +732,7 @@ def test_get_config_custom_callback_api_env_vars(monkeypatch): mock_router = MagicMock() mock_router.get_settings.return_value = {} monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", mock_router) - monkeypatch.setattr( - proxy_config, "get_config", AsyncMock(return_value=config_data) - ) + monkeypatch.setattr(proxy_config, "get_config", AsyncMock(return_value=config_data)) # Bypass auth dependency original_overrides = app.dependency_overrides.copy() @@ -923,7 +910,9 @@ def test_embedding_input_array_of_tokens(client_no_auth): assert response.status_code == 200 result = response.json() print(len(result["data"][0]["embedding"])) - assert len(result["data"][0]["embedding"]) > 10 # this usually has len==1536 so + assert ( + len(result["data"][0]["embedding"]) > 10 + ) # this usually has len==1536 so except Exception as e: pytest.fail(f"LiteLLM Proxy test failed. Exception - {str(e)}") @@ -1180,7 +1169,6 @@ async def test_delete_deployment_type_mismatch(): with patch("litellm.proxy.proxy_server.llm_router", mock_llm_router), patch( "litellm.proxy.proxy_server.user_config_file_path", "test_config.yaml" ): - # Call the function under test deleted_count = await pc._delete_deployment(db_models=[]) @@ -1322,6 +1310,7 @@ def test_normalize_datetime_for_sorting(): # Test Case 6: Timezone-aware datetime object (non-UTC) from datetime import timedelta + aware_dt = datetime(2024, 1, 15, 10, 30, 0, tzinfo=timezone(timedelta(hours=5))) result = _normalize_datetime_for_sorting(aware_dt) assert result is not None @@ -1574,6 +1563,70 @@ async def test_load_environment_variables_litellm_license_and_edge_cases(): assert "FAILED_SECRET" not in os.environ +@pytest.mark.asyncio +async def test_load_environment_variables_blocks_dangerous_keys(): + """ + Test that _load_environment_variables rejects dangerous env var keys + like PATH, LD_PRELOAD, PYTHONPATH, etc. + """ + import logging + + from litellm.proxy.proxy_server import ProxyConfig + + proxy_config = ProxyConfig() + + original_path = os.environ.get("PATH", "") + + test_config = { + "environment_variables": { + "PATH": "/tmp/evil", + "LD_PRELOAD": "/tmp/evil.so", + "PYTHONPATH": "/tmp/evil", + "SAFE_CUSTOM_VAR": "safe_value", + } + } + + with patch.dict(os.environ, {}, clear=False): + proxy_config._load_environment_variables(test_config) + + # Blocked keys should not be set to the attacker value + assert os.environ.get("PATH") != "/tmp/evil" + assert ( + "LD_PRELOAD" not in os.environ or os.environ["LD_PRELOAD"] != "/tmp/evil.so" + ) + assert os.environ.get("PYTHONPATH") != "/tmp/evil" + + # Safe keys should still be set + assert os.environ["SAFE_CUSTOM_VAR"] == "safe_value" + + +@pytest.mark.asyncio +async def test_load_environment_variables_blocks_proxy_keys(): + """ + Test that _load_environment_variables rejects proxy-related env var keys. + """ + from litellm.proxy.proxy_server import ProxyConfig + + proxy_config = ProxyConfig() + + test_config = { + "environment_variables": { + "HTTP_PROXY": "http://evil-proxy:8080", + "HTTPS_PROXY": "http://evil-proxy:8080", + "http_proxy": "http://evil-proxy:8080", + "https_proxy": "http://evil-proxy:8080", + } + } + + with patch.dict(os.environ, {}, clear=False): + proxy_config._load_environment_variables(test_config) + + assert os.environ.get("HTTP_PROXY") != "http://evil-proxy:8080" + assert os.environ.get("HTTPS_PROXY") != "http://evil-proxy:8080" + assert os.environ.get("http_proxy") != "http://evil-proxy:8080" + assert os.environ.get("https_proxy") != "http://evil-proxy:8080" + + @pytest.mark.asyncio async def test_write_config_to_file(monkeypatch): """ @@ -1882,7 +1935,6 @@ async def test_chat_completion_result_no_nested_none_values(): "litellm.proxy.proxy_server.ProxyBaseLLMRequestProcessing", return_value=mock_base_processor, ): - # Call the chat_completion function result = await chat_completion( request=mock_request, @@ -2027,9 +2079,7 @@ class TestPriceDataReloadAPI: # Mock the database connection with patch("litellm.proxy.proxy_server.prisma_client") as mock_prisma: - mock_prisma.db.litellm_config.find_unique = AsyncMock( - return_value=None - ) + mock_prisma.db.litellm_config.find_unique = AsyncMock(return_value=None) mock_prisma.db.litellm_config.upsert = AsyncMock(return_value=None) response = client_with_auth.post("/reload/model_cost_map") @@ -2372,8 +2422,13 @@ class TestPriceDataReloadIntegration: with patch("litellm.proxy.proxy_server.prisma_client") as mock_prisma: # Simulate existing config with a schedule mock_existing = MagicMock() - mock_existing.param_value = {"interval_hours": 12, "force_reload": False} - mock_prisma.db.litellm_config.find_unique = AsyncMock(return_value=mock_existing) + mock_existing.param_value = { + "interval_hours": 12, + "force_reload": False, + } + mock_prisma.db.litellm_config.find_unique = AsyncMock( + return_value=mock_existing + ) mock_prisma.db.litellm_config.upsert = AsyncMock(return_value=None) response = client.post("/reload/model_cost_map") @@ -2415,7 +2470,9 @@ class TestPriceDataReloadIntegration: ) as mock_reload: mock_reload.return_value = {"anthropic": {"beta_header": "test-value"}} - asyncio.run(proxy_config._check_and_reload_anthropic_beta_headers(mock_prisma)) + asyncio.run( + proxy_config._check_and_reload_anthropic_beta_headers(mock_prisma) + ) # Verify the upsert update branch preserves interval_hours mock_prisma.db.litellm_config.upsert.assert_called() @@ -2456,7 +2513,9 @@ class TestPriceDataReloadIntegration: # Simulate existing config with a schedule mock_existing = MagicMock() mock_existing.param_value = {"interval_hours": 8, "force_reload": False} - mock_prisma.db.litellm_config.find_unique = AsyncMock(return_value=mock_existing) + mock_prisma.db.litellm_config.find_unique = AsyncMock( + return_value=mock_existing + ) mock_prisma.db.litellm_config.upsert = AsyncMock(return_value=None) response = client.post("/reload/anthropic_beta_headers") @@ -2821,7 +2880,7 @@ async def test_model_info_v1_oci_secrets_not_leaked(): mock_user_api_key_dict.api_key = "test-key" mock_user_api_key_dict.team_models = [] mock_user_api_key_dict.models = ["oci-grok-test"] - + # Mock model data with OCI sensitive information mock_model_data = { "model_name": "oci-grok-test", @@ -2834,59 +2893,73 @@ async def test_model_info_v1_oci_secrets_not_leaked(): "oci_tenancy": "ocid1.tenancy.oc1..aaaaaaaa7kbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbk", "oci_key_file": "/path/to/oci_api_key.pem", "oci_compartment_id": "ocid1.compartment.oc1..aaaaaaaa7kbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbk", - "drop_params": True + "drop_params": True, }, - "model_info": { - "mode": "completion", - "id": "test-model-id" - } + "model_info": {"mode": "completion", "id": "test-model-id"}, } - + # Mock the llm_router to return our test data mock_router = MagicMock() mock_router.get_model_names.return_value = ["oci-grok-test"] mock_router.get_model_access_groups.return_value = {} mock_router.get_model_list.return_value = [mock_model_data] - + # Mock global variables - with patch("litellm.proxy.proxy_server.llm_router", mock_router), \ - patch("litellm.proxy.proxy_server.llm_model_list", [mock_model_data]), \ - patch("litellm.proxy.proxy_server.general_settings", {"infer_model_from_keys": False}), \ - patch("litellm.proxy.proxy_server.user_model", None): - + with patch("litellm.proxy.proxy_server.llm_router", mock_router), patch( + "litellm.proxy.proxy_server.llm_model_list", [mock_model_data] + ), patch( + "litellm.proxy.proxy_server.general_settings", {"infer_model_from_keys": False} + ), patch( + "litellm.proxy.proxy_server.user_model", None + ): # Call the model_info_v1 endpoint result = await model_info_v1( - user_api_key_dict=mock_user_api_key_dict, - litellm_model_id=None + user_api_key_dict=mock_user_api_key_dict, litellm_model_id=None ) - + # Verify the result structure assert "data" in result assert len(result["data"]) == 1 - + model_info = result["data"][0] litellm_params = model_info["litellm_params"] - + # Verify that sensitive OCI fields are masked assert "****" in litellm_params["oci_key"], "oci_key should be masked" - assert "****" in litellm_params["oci_fingerprint"], "oci_fingerprint should be masked" + assert ( + "****" in litellm_params["oci_fingerprint"] + ), "oci_fingerprint should be masked" assert "****" in litellm_params["oci_tenancy"], "oci_tenancy should be masked" assert "****" in litellm_params["oci_key_file"], "oci_key_file should be masked" - + # Verify that non-sensitive fields are NOT masked - assert litellm_params["model"] == "oci/xai.grok-4", "model field should not be masked" - assert litellm_params["oci_region"] == "us-phoenix-1", "oci_region should not be masked" + assert ( + litellm_params["model"] == "oci/xai.grok-4" + ), "model field should not be masked" + assert ( + litellm_params["oci_region"] == "us-phoenix-1" + ), "oci_region should not be masked" assert litellm_params["drop_params"] is True, "drop_params should not be masked" - + # Verify the model field specifically is not masked (this was the original issue) - assert "****" not in litellm_params["model"], "model field should never be masked" - assert litellm_params["model"].startswith("oci/"), "model should retain its full value" - + assert ( + "****" not in litellm_params["model"] + ), "model field should never be masked" + assert litellm_params["model"].startswith( + "oci/" + ), "model should retain its full value" + # Verify that actual secret values are not present in the response result_str = str(result) - assert "ocid1.api_key.oc1..aaaaaaaa7kbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbk" not in result_str + assert ( + "ocid1.api_key.oc1..aaaaaaaa7kbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbk" + not in result_str + ) assert "aa:bb:cc:dd:ee:ff:11:22:33:44:55:66:77:88:99:00" not in result_str - assert "ocid1.tenancy.oc1..aaaaaaaa7kbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbk" not in result_str + assert ( + "ocid1.tenancy.oc1..aaaaaaaa7kbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbkbk" + not in result_str + ) assert "/path/to/oci_api_key.pem" not in result_str @@ -2898,17 +2971,17 @@ def test_add_callback_from_db_to_in_memory_litellm_callbacks(): from unittest.mock import MagicMock, patch from litellm.proxy.proxy_server import ProxyConfig - + proxy_config = ProxyConfig() - + # Mock the callback manager mock_callback_manager = MagicMock() - + with patch("litellm.proxy.proxy_server.litellm") as mock_litellm: # Set up mock litellm attributes mock_litellm._known_custom_logger_compatible_callbacks = [] mock_litellm.logging_callback_manager = mock_callback_manager - + # Test Case 1: Add success callback mock_success_callbacks = [] proxy_config._add_callback_from_db_to_in_memory_litellm_callbacks( @@ -2916,9 +2989,11 @@ def test_add_callback_from_db_to_in_memory_litellm_callbacks(): event_types=["success"], existing_callbacks=mock_success_callbacks, ) - mock_callback_manager.add_litellm_success_callback.assert_called_once_with("prometheus") + mock_callback_manager.add_litellm_success_callback.assert_called_once_with( + "prometheus" + ) mock_callback_manager.reset_mock() - + # Test Case 2: Add failure callback mock_failure_callbacks = [] proxy_config._add_callback_from_db_to_in_memory_litellm_callbacks( @@ -2926,9 +3001,11 @@ def test_add_callback_from_db_to_in_memory_litellm_callbacks(): event_types=["failure"], existing_callbacks=mock_failure_callbacks, ) - mock_callback_manager.add_litellm_failure_callback.assert_called_once_with("langfuse") + mock_callback_manager.add_litellm_failure_callback.assert_called_once_with( + "langfuse" + ) mock_callback_manager.reset_mock() - + # Test Case 3: Add callback for both success and failure mock_callbacks = [] proxy_config._add_callback_from_db_to_in_memory_litellm_callbacks( @@ -2938,7 +3015,7 @@ def test_add_callback_from_db_to_in_memory_litellm_callbacks(): ) mock_callback_manager.add_litellm_callback.assert_called_once_with("s3") mock_callback_manager.reset_mock() - + # Test Case 4: Don't add callback if it already exists existing_callbacks_with_item = ["prometheus"] proxy_config._add_callback_from_db_to_in_memory_litellm_callbacks( @@ -2952,7 +3029,7 @@ def test_add_callback_from_db_to_in_memory_litellm_callbacks(): def test_should_load_db_object_with_supported_db_objects(): """ Test _should_load_db_object method with supported_db_objects configuration. - + Verifies that when supported_db_objects is set, only specified object types are loaded from the database. """ @@ -3056,8 +3133,12 @@ async def test_tag_cache_update_called(): "spend": 10.0, } - with patch.object(cache, "async_get_cache", new=AsyncMock(return_value=mock_tag_obj)) as mock_get_cache: - with patch.object(cache, "async_set_cache_pipeline", new=AsyncMock()) as mock_set_cache: + with patch.object( + cache, "async_get_cache", new=AsyncMock(return_value=mock_tag_obj) + ) as mock_get_cache: + with patch.object( + cache, "async_set_cache_pipeline", new=AsyncMock() + ) as mock_set_cache: await litellm.proxy.proxy_server.update_cache( token=None, user_id=None, @@ -3108,8 +3189,12 @@ async def test_tag_cache_update_multiple_tags(): return mock_tag2_obj return None - with patch.object(cache, "async_get_cache", new=AsyncMock(side_effect=mock_get_cache_side_effect)) as mock_get_cache: - with patch.object(cache, "async_set_cache_pipeline", new=AsyncMock()) as mock_set_cache: + with patch.object( + cache, "async_get_cache", new=AsyncMock(side_effect=mock_get_cache_side_effect) + ) as mock_get_cache: + with patch.object( + cache, "async_set_cache_pipeline", new=AsyncMock() + ) as mock_set_cache: await litellm.proxy.proxy_server.update_cache( token=None, user_id=None, @@ -3130,7 +3215,9 @@ async def test_tag_cache_update_multiple_tags(): assert len(cache_list) == 2 - tag_updates = {cache_key: cache_value for cache_key, cache_value in cache_list} + tag_updates = { + cache_key: cache_value for cache_key, cache_value in cache_list + } assert "tag:tag1" in tag_updates assert "tag:tag2" in tag_updates assert tag_updates["tag:tag1"]["spend"] == 15.0 @@ -3250,7 +3337,9 @@ async def test_init_sso_settings_in_db_error_handling(): assert True except Exception as e: # The exception should be caught and logged, not propagated - pytest.fail(f"Exception should have been caught and logged, but was raised: {e}") + pytest.fail( + f"Exception should have been caught and logged, but was raised: {e}" + ) @pytest.mark.asyncio @@ -3353,11 +3442,15 @@ def test_get_prompt_spec_for_db_prompt_with_versions(): } # Test version 1 - prompt_spec_v1 = proxy_config._get_prompt_spec_for_db_prompt(db_prompt=mock_prompt_v1) + prompt_spec_v1 = proxy_config._get_prompt_spec_for_db_prompt( + db_prompt=mock_prompt_v1 + ) assert prompt_spec_v1.prompt_id == "chat_prompt.v1" # Test version 2 - prompt_spec_v2 = proxy_config._get_prompt_spec_for_db_prompt(db_prompt=mock_prompt_v2) + prompt_spec_v2 = proxy_config._get_prompt_spec_for_db_prompt( + db_prompt=mock_prompt_v2 + ) assert prompt_spec_v2.prompt_id == "chat_prompt.v2" @@ -3372,15 +3465,15 @@ def test_root_redirect_when_docs_url_not_root_and_redirect_url_set(monkeypatch): config_fp = f"{filepath}/test_configs/test_config_no_auth.yaml" # Ensure docs are mounted on a non-root path to trigger redirect logic monkeypatch.setenv("DOCS_URL", "/docs") - + test_redirect_url = "/ui" monkeypatch.setenv("ROOT_REDIRECT_URL", test_redirect_url) - + asyncio.run(initialize(config=config_fp, debug=True)) - + docs_url = _get_docs_url() root_redirect_url = os.getenv("ROOT_REDIRECT_URL") - + # Remove any existing "/" route that might interfere routes_to_remove = [] for route in app.routes: @@ -3389,16 +3482,17 @@ def test_root_redirect_when_docs_url_not_root_and_redirect_url_set(monkeypatch): routes_to_remove.append(route) elif not hasattr(route, "methods"): # Catch-all routes routes_to_remove.append(route) - + for route in routes_to_remove: app.routes.remove(route) - + # Add the redirect route if conditions are met (matching the actual implementation) if docs_url != "/" and root_redirect_url: + @app.get("/", include_in_schema=False) async def root_redirect(): return RedirectResponse(url=root_redirect_url) - + client = TestClient(app) response = client.get("/", follow_redirects=False) assert response.status_code == 307 @@ -3422,12 +3516,13 @@ async def test_get_image_non_root_uses_var_lib_assets_dir(monkeypatch): def exists_side_effect(path): return False if path == "/var/lib/litellm/assets" else True - with patch("litellm.proxy.proxy_server.os.makedirs") as mock_makedirs, \ - patch("litellm.proxy.proxy_server.os.path.exists", side_effect=exists_side_effect), \ - patch("litellm.proxy.proxy_server.os.access", return_value=True), \ - patch("litellm.proxy.proxy_server.os.getenv") as mock_getenv, \ - patch("litellm.proxy.proxy_server.FileResponse") as mock_file_response: - + with patch("litellm.proxy.proxy_server.os.makedirs") as mock_makedirs, patch( + "litellm.proxy.proxy_server.os.path.exists", side_effect=exists_side_effect + ), patch("litellm.proxy.proxy_server.os.access", return_value=True), patch( + "litellm.proxy.proxy_server.os.getenv" + ) as mock_getenv, patch( + "litellm.proxy.proxy_server.FileResponse" + ) as mock_file_response: # Setup mock_getenv to return empty string for UI_LOGO_PATH def getenv_side_effect(key, default=""): if key == "UI_LOGO_PATH": @@ -3471,12 +3566,13 @@ async def test_get_image_non_root_fallback_to_default_logo(monkeypatch): return True # Mock os.path operations - with patch("litellm.proxy.proxy_server.os.makedirs") as mock_makedirs, \ - patch("litellm.proxy.proxy_server.os.path.exists", side_effect=exists_side_effect), \ - patch("litellm.proxy.proxy_server.os.access", return_value=True), \ - patch("litellm.proxy.proxy_server.os.getenv") as mock_getenv, \ - patch("litellm.proxy.proxy_server.FileResponse") as mock_file_response: - + with patch("litellm.proxy.proxy_server.os.makedirs") as mock_makedirs, patch( + "litellm.proxy.proxy_server.os.path.exists", side_effect=exists_side_effect + ), patch("litellm.proxy.proxy_server.os.access", return_value=True), patch( + "litellm.proxy.proxy_server.os.getenv" + ) as mock_getenv, patch( + "litellm.proxy.proxy_server.FileResponse" + ) as mock_file_response: # Setup mock_getenv def getenv_side_effect(key, default=""): if key == "UI_LOGO_PATH": @@ -3495,8 +3591,9 @@ async def test_get_image_non_root_fallback_to_default_logo(monkeypatch): # Verify that exists was called to check /var/lib/litellm/assets/logo.jpg assets_logo_path = "/var/lib/litellm/assets/logo.jpg" - assert any(assets_logo_path in str(call) for call in exists_calls), \ - f"Should check if {assets_logo_path} exists" + assert any( + assets_logo_path in str(call) for call in exists_calls + ), f"Should check if {assets_logo_path} exists" # Verify FileResponse was called (with fallback logo) assert mock_file_response.called, "FileResponse should be called" @@ -3516,11 +3613,11 @@ async def test_get_image_root_case_uses_current_dir(monkeypatch): monkeypatch.delenv("UI_LOGO_PATH", raising=False) # Mock os.path operations - with patch("litellm.proxy.proxy_server.os.makedirs") as mock_makedirs, \ - patch("litellm.proxy.proxy_server.os.path.exists", return_value=True), \ - patch("litellm.proxy.proxy_server.os.getenv") as mock_getenv, \ - patch("litellm.proxy.proxy_server.FileResponse") as mock_file_response: - + with patch("litellm.proxy.proxy_server.os.makedirs") as mock_makedirs, patch( + "litellm.proxy.proxy_server.os.path.exists", return_value=True + ), patch("litellm.proxy.proxy_server.os.getenv") as mock_getenv, patch( + "litellm.proxy.proxy_server.FileResponse" + ) as mock_file_response: # Setup mock_getenv def getenv_side_effect(key, default=""): if key == "UI_LOGO_PATH": @@ -3536,10 +3633,13 @@ async def test_get_image_root_case_uses_current_dir(monkeypatch): # Verify makedirs was NOT called with /var/lib/litellm/assets (should not create it for root case) var_lib_assets_calls = [ - call for call in mock_makedirs.call_args_list + call + for call in mock_makedirs.call_args_list if "/var/lib/litellm/assets" in str(call) ] - assert len(var_lib_assets_calls) == 0, "Should not create /var/lib/litellm/assets for root case" + assert ( + len(var_lib_assets_calls) == 0 + ), "Should not create /var/lib/litellm/assets for root case" # Verify FileResponse was called assert mock_file_response.called, "FileResponse should be called" @@ -3569,13 +3669,14 @@ async def test_get_image_custom_local_logo_bypasses_cache(monkeypatch): calls_to_file_response.append(path) return MagicMock() - with patch("litellm.proxy.proxy_server.os.path.exists", return_value=True), \ - patch("litellm.proxy.proxy_server.os.access", return_value=True), \ - patch("litellm.proxy.proxy_server.FileResponse", side_effect=fake_file_response): - + with patch("litellm.proxy.proxy_server.os.path.exists", return_value=True), patch( + "litellm.proxy.proxy_server.os.access", return_value=True + ), patch("litellm.proxy.proxy_server.FileResponse", side_effect=fake_file_response): await get_image() - assert len(calls_to_file_response) == 1, "FileResponse should be called exactly once" + assert ( + len(calls_to_file_response) == 1 + ), "FileResponse should be called exactly once" assert calls_to_file_response[0] == "/app/custom_logo.jpg", ( f"Expected custom logo path, got {calls_to_file_response[0]}. " "A stale cached_logo.jpg may have been returned instead." @@ -3602,17 +3703,18 @@ async def test_get_image_default_logo_still_uses_cache(monkeypatch): calls_to_file_response.append(path) return MagicMock() - with patch("litellm.proxy.proxy_server.os.path.exists", return_value=True), \ - patch("litellm.proxy.proxy_server.os.access", return_value=True), \ - patch("litellm.proxy.proxy_server.FileResponse", side_effect=fake_file_response): - + with patch("litellm.proxy.proxy_server.os.path.exists", return_value=True), patch( + "litellm.proxy.proxy_server.os.access", return_value=True + ), patch("litellm.proxy.proxy_server.FileResponse", side_effect=fake_file_response): await get_image() - assert len(calls_to_file_response) == 1, "FileResponse should be called exactly once" + assert ( + len(calls_to_file_response) == 1 + ), "FileResponse should be called exactly once" served_path = calls_to_file_response[0] - assert served_path.endswith("cached_logo.jpg"), ( - f"Expected cached_logo.jpg for default logo, got {served_path}" - ) + assert served_path.endswith( + "cached_logo.jpg" + ), f"Expected cached_logo.jpg for default logo, got {served_path}" @pytest.mark.asyncio @@ -3641,20 +3743,23 @@ async def test_get_image_custom_logo_missing_falls_through_to_default(monkeypatc return False return True - with patch("litellm.proxy.proxy_server.os.path.exists", side_effect=exists_side_effect), \ - patch("litellm.proxy.proxy_server.os.access", return_value=True), \ - patch("litellm.proxy.proxy_server.FileResponse", side_effect=fake_file_response): - + with patch( + "litellm.proxy.proxy_server.os.path.exists", side_effect=exists_side_effect + ), patch("litellm.proxy.proxy_server.os.access", return_value=True), patch( + "litellm.proxy.proxy_server.FileResponse", side_effect=fake_file_response + ): await get_image() - assert len(calls_to_file_response) == 1, "FileResponse should be called exactly once" + assert ( + len(calls_to_file_response) == 1 + ), "FileResponse should be called exactly once" served_path = calls_to_file_response[0] - assert served_path != "/app/nonexistent_logo.jpg", ( - "Should not attempt to serve a non-existent custom logo" - ) - assert served_path.endswith("cached_logo.jpg"), ( - f"Expected fallback to cached_logo.jpg, got {served_path}" - ) + assert ( + served_path != "/app/nonexistent_logo.jpg" + ), "Should not attempt to serve a non-existent custom logo" + assert served_path.endswith( + "cached_logo.jpg" + ), f"Expected fallback to cached_logo.jpg, got {served_path}" @pytest.mark.asyncio @@ -3686,20 +3791,23 @@ async def test_get_image_custom_logo_missing_no_cache_serves_default(monkeypatch return False return True - with patch("litellm.proxy.proxy_server.os.path.exists", side_effect=exists_side_effect), \ - patch("litellm.proxy.proxy_server.os.access", return_value=True), \ - patch("litellm.proxy.proxy_server.FileResponse", side_effect=fake_file_response): - + with patch( + "litellm.proxy.proxy_server.os.path.exists", side_effect=exists_side_effect + ), patch("litellm.proxy.proxy_server.os.access", return_value=True), patch( + "litellm.proxy.proxy_server.FileResponse", side_effect=fake_file_response + ): await get_image() - assert len(calls_to_file_response) == 1, "FileResponse should be called exactly once" + assert ( + len(calls_to_file_response) == 1 + ), "FileResponse should be called exactly once" served_path = calls_to_file_response[0] - assert served_path != "/app/nonexistent_logo.jpg", ( - "Should not attempt to serve a non-existent custom logo" - ) - assert served_path.endswith("logo.jpg"), ( - f"Expected fallback to default logo.jpg, got {served_path}" - ) + assert ( + served_path != "/app/nonexistent_logo.jpg" + ), "Should not attempt to serve a non-existent custom logo" + assert served_path.endswith( + "logo.jpg" + ), f"Expected fallback to default logo.jpg, got {served_path}" def test_get_config_normalizes_string_callbacks(monkeypatch): @@ -3721,9 +3829,7 @@ def test_get_config_normalizes_string_callbacks(monkeypatch): mock_router = MagicMock() mock_router.get_settings.return_value = {} monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", mock_router) - monkeypatch.setattr( - proxy_config, "get_config", AsyncMock(return_value=config_data) - ) + monkeypatch.setattr(proxy_config, "get_config", AsyncMock(return_value=config_data)) original_overrides = app.dependency_overrides.copy() app.dependency_overrides[user_api_key_auth] = lambda: MagicMock() @@ -4127,9 +4233,9 @@ async def test_update_general_settings_store_model_in_db_false(): proxy_config = ProxyConfig() - with patch( - "litellm.proxy.proxy_server.store_model_in_db", True - ), patch("litellm.proxy.proxy_server.general_settings", {}): + with patch("litellm.proxy.proxy_server.store_model_in_db", True), patch( + "litellm.proxy.proxy_server.general_settings", {} + ): await proxy_config._update_general_settings( db_general_settings={"store_model_in_db": False} ) @@ -4150,9 +4256,9 @@ async def test_update_general_settings_store_model_in_db_string_normalization(): proxy_config = ProxyConfig() # Test "true" string - with patch( - "litellm.proxy.proxy_server.store_model_in_db", False - ), patch("litellm.proxy.proxy_server.general_settings", {}): + with patch("litellm.proxy.proxy_server.store_model_in_db", False), patch( + "litellm.proxy.proxy_server.general_settings", {} + ): await proxy_config._update_general_settings( db_general_settings={"store_model_in_db": "true"} ) @@ -4161,9 +4267,9 @@ async def test_update_general_settings_store_model_in_db_string_normalization(): assert ps.store_model_in_db is True # Test "True" string - with patch( - "litellm.proxy.proxy_server.store_model_in_db", False - ), patch("litellm.proxy.proxy_server.general_settings", {}): + with patch("litellm.proxy.proxy_server.store_model_in_db", False), patch( + "litellm.proxy.proxy_server.general_settings", {} + ): await proxy_config._update_general_settings( db_general_settings={"store_model_in_db": "True"} ) @@ -4172,9 +4278,9 @@ async def test_update_general_settings_store_model_in_db_string_normalization(): assert ps.store_model_in_db is True # Test "false" string - with patch( - "litellm.proxy.proxy_server.store_model_in_db", True - ), patch("litellm.proxy.proxy_server.general_settings", {}): + with patch("litellm.proxy.proxy_server.store_model_in_db", True), patch( + "litellm.proxy.proxy_server.general_settings", {} + ): await proxy_config._update_general_settings( db_general_settings={"store_model_in_db": "false"} ) @@ -4194,9 +4300,9 @@ async def test_update_general_settings_store_model_in_db_none_keeps_current(): proxy_config = ProxyConfig() # When current is True and DB sends None, should stay True - with patch( - "litellm.proxy.proxy_server.store_model_in_db", True - ), patch("litellm.proxy.proxy_server.general_settings", {}): + with patch("litellm.proxy.proxy_server.store_model_in_db", True), patch( + "litellm.proxy.proxy_server.general_settings", {} + ): await proxy_config._update_general_settings( db_general_settings={"store_model_in_db": None} ) @@ -4205,9 +4311,9 @@ async def test_update_general_settings_store_model_in_db_none_keeps_current(): assert ps.store_model_in_db is True # When current is False and DB sends None, should stay False - with patch( - "litellm.proxy.proxy_server.store_model_in_db", False - ), patch("litellm.proxy.proxy_server.general_settings", {}): + with patch("litellm.proxy.proxy_server.store_model_in_db", False), patch( + "litellm.proxy.proxy_server.general_settings", {} + ): await proxy_config._update_general_settings( db_general_settings={"store_model_in_db": None} ) @@ -4238,13 +4344,9 @@ async def test_store_model_in_db_db_override_when_config_false(): mock_proxy_logging.slack_alerting_instance = MagicMock() mock_proxy_config = AsyncMock() - with patch( - "litellm.proxy.proxy_server.proxy_config", mock_proxy_config - ), patch( + with patch("litellm.proxy.proxy_server.proxy_config", mock_proxy_config), patch( "litellm.proxy.proxy_server.store_model_in_db", False - ), patch( - "litellm.proxy.proxy_server.get_secret_bool", return_value=False - ): + ), patch("litellm.proxy.proxy_server.get_secret_bool", return_value=False): await ProxyStartupEvent.initialize_scheduled_background_jobs( general_settings={}, prisma_client=mock_prisma_client, @@ -4282,13 +4384,9 @@ async def test_store_model_in_db_db_check_skipped_when_already_true(monkeypatch) mock_proxy_logging.slack_alerting_instance = MagicMock() mock_proxy_config = AsyncMock() - with patch( - "litellm.proxy.proxy_server.proxy_config", mock_proxy_config - ), patch( + with patch("litellm.proxy.proxy_server.proxy_config", mock_proxy_config), patch( "litellm.proxy.proxy_server.store_model_in_db", True - ), patch( - "litellm.proxy.proxy_server.get_secret_bool", return_value=True - ): + ), patch("litellm.proxy.proxy_server.get_secret_bool", return_value=True): await ProxyStartupEvent.initialize_scheduled_background_jobs( general_settings={}, prisma_client=mock_prisma_client, @@ -4328,13 +4426,9 @@ async def test_store_model_in_db_db_failure_graceful(monkeypatch): mock_proxy_logging.slack_alerting_instance = MagicMock() mock_proxy_config = AsyncMock() - with patch( - "litellm.proxy.proxy_server.proxy_config", mock_proxy_config - ), patch( + with patch("litellm.proxy.proxy_server.proxy_config", mock_proxy_config), patch( "litellm.proxy.proxy_server.store_model_in_db", False - ), patch( - "litellm.proxy.proxy_server.get_secret_bool", return_value=False - ): + ), patch("litellm.proxy.proxy_server.get_secret_bool", return_value=False): # Should not raise an exception await ProxyStartupEvent.initialize_scheduled_background_jobs( general_settings={}, From 92a07e2d6e2850bef28f8c7859d6c7632cfb3c45 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 16:00:26 +0530 Subject: [PATCH 059/117] fix(proxy): address Greptile review feedback - Remove HTTP_PROXY/HTTPS_PROXY from blocklist (legitimately used in corporate envs) - Add NO_PROXY/no_proxy to blocklist (prevents bypassing proxy monitoring) - Remove dead code in _is_valid_user_id (space exception was unreachable) - Update tests accordingly Co-Authored-By: Claude Opus 4.6 --- .../internal_user_endpoints.py | 4 +- litellm/proxy/proxy_server.py | 6 +-- tests/test_litellm/proxy/test_proxy_server.py | 41 ++++++++++++++----- 3 files changed, 35 insertions(+), 16 deletions(-) diff --git a/litellm/proxy/management_endpoints/internal_user_endpoints.py b/litellm/proxy/management_endpoints/internal_user_endpoints.py index b2df9a1a2cb..ff7b89fe192 100644 --- a/litellm/proxy/management_endpoints/internal_user_endpoints.py +++ b/litellm/proxy/management_endpoints/internal_user_endpoints.py @@ -562,9 +562,9 @@ def _is_valid_user_id(user_id: str) -> bool: MAX_USER_ID_LENGTH = 512 if len(user_id) > MAX_USER_ID_LENGTH: return False - # Reject null bytes and control characters (< 0x20) except space + # Reject ASCII control characters (U+0000–U+001F) for ch in user_id: - if ch == "\x00" or (ord(ch) < 0x20 and ch != " "): + if ord(ch) < 0x20: return False return True diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 203344d452c..6765feebb14 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -2710,10 +2710,8 @@ class ProxyConfig: "USER", "SHELL", "LOGNAME", - "http_proxy", - "https_proxy", - "HTTP_PROXY", - "HTTPS_PROXY", + "NO_PROXY", + "no_proxy", } def _load_environment_variables(self, config: dict): diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py index 334cd7395d6..d7f1774fc1b 100644 --- a/tests/test_litellm/proxy/test_proxy_server.py +++ b/tests/test_litellm/proxy/test_proxy_server.py @@ -1601,9 +1601,10 @@ async def test_load_environment_variables_blocks_dangerous_keys(): @pytest.mark.asyncio -async def test_load_environment_variables_blocks_proxy_keys(): +async def test_load_environment_variables_allows_proxy_keys(): """ - Test that _load_environment_variables rejects proxy-related env var keys. + Test that HTTP_PROXY/HTTPS_PROXY are allowed since they are commonly used + in corporate environments to route outbound API calls. """ from litellm.proxy.proxy_server import ProxyConfig @@ -1611,20 +1612,40 @@ async def test_load_environment_variables_blocks_proxy_keys(): test_config = { "environment_variables": { - "HTTP_PROXY": "http://evil-proxy:8080", - "HTTPS_PROXY": "http://evil-proxy:8080", - "http_proxy": "http://evil-proxy:8080", - "https_proxy": "http://evil-proxy:8080", + "HTTP_PROXY": "http://corp-proxy:8080", + "HTTPS_PROXY": "http://corp-proxy:8080", } } with patch.dict(os.environ, {}, clear=False): proxy_config._load_environment_variables(test_config) - assert os.environ.get("HTTP_PROXY") != "http://evil-proxy:8080" - assert os.environ.get("HTTPS_PROXY") != "http://evil-proxy:8080" - assert os.environ.get("http_proxy") != "http://evil-proxy:8080" - assert os.environ.get("https_proxy") != "http://evil-proxy:8080" + assert os.environ["HTTP_PROXY"] == "http://corp-proxy:8080" + assert os.environ["HTTPS_PROXY"] == "http://corp-proxy:8080" + + +@pytest.mark.asyncio +async def test_load_environment_variables_blocks_no_proxy(): + """ + Test that NO_PROXY/no_proxy are blocked to prevent bypassing proxy-based + network monitoring. + """ + from litellm.proxy.proxy_server import ProxyConfig + + proxy_config = ProxyConfig() + + test_config = { + "environment_variables": { + "NO_PROXY": "internal-service", + "no_proxy": "internal-service", + } + } + + with patch.dict(os.environ, {}, clear=False): + proxy_config._load_environment_variables(test_config) + + assert os.environ.get("NO_PROXY") != "internal-service" + assert os.environ.get("no_proxy") != "internal-service" @pytest.mark.asyncio From cdb3037c4e54b60687b2b8ec8fff72be145d5290 Mon Sep 17 00:00:00 2001 From: Conductor Date: Fri, 27 Mar 2026 20:37:02 +0530 Subject: [PATCH 060/117] Saving uncommitted changes before archiving --- litellm/proxy/_experimental/out/{404.html => 404/index.html} | 0 .../_experimental/out/{_not-found.html => _not-found/index.html} | 0 .../out/{api-reference.html => api-reference/index.html} | 0 litellm/proxy/_experimental/out/{chat.html => chat/index.html} | 0 .../{api-playground.html => api-playground/index.html} | 0 .../out/experimental/{budgets.html => budgets/index.html} | 0 .../out/experimental/{caching.html => caching/index.html} | 0 .../{claude-code-plugins.html => claude-code-plugins/index.html} | 0 .../out/experimental/{old-usage.html => old-usage/index.html} | 0 .../out/experimental/{prompts.html => prompts/index.html} | 0 .../{tag-management.html => tag-management/index.html} | 0 .../_experimental/out/{guardrails.html => guardrails/index.html} | 0 litellm/proxy/_experimental/out/{login.html => login/index.html} | 0 litellm/proxy/_experimental/out/{logs.html => logs/index.html} | 0 .../out/mcp/oauth/{callback.html => callback/index.html} | 0 .../_experimental/out/{model-hub.html => model-hub/index.html} | 0 .../_experimental/out/{model_hub.html => model_hub/index.html} | 0 .../out/{model_hub_table.html => model_hub_table/index.html} | 0 .../index.html} | 0 .../_experimental/out/{onboarding.html => onboarding/index.html} | 0 .../out/{organizations.html => organizations/index.html} | 0 .../_experimental/out/{playground.html => playground/index.html} | 0 .../_experimental/out/{policies.html => policies/index.html} | 0 .../settings/{admin-settings.html => admin-settings/index.html} | 0 .../{logging-and-alerts.html => logging-and-alerts/index.html} | 0 .../settings/{router-settings.html => router-settings/index.html} | 0 .../out/settings/{ui-theme.html => ui-theme/index.html} | 0 litellm/proxy/_experimental/out/{teams.html => teams/index.html} | 0 .../_experimental/out/{test-key.html => test-key/index.html} | 0 .../out/tools/{mcp-servers.html => mcp-servers/index.html} | 0 .../out/tools/{vector-stores.html => vector-stores/index.html} | 0 litellm/proxy/_experimental/out/{usage.html => usage/index.html} | 0 litellm/proxy/_experimental/out/{users.html => users/index.html} | 0 .../out/{virtual-keys.html => virtual-keys/index.html} | 0 34 files changed, 0 insertions(+), 0 deletions(-) rename litellm/proxy/_experimental/out/{404.html => 404/index.html} (100%) rename litellm/proxy/_experimental/out/{_not-found.html => _not-found/index.html} (100%) rename litellm/proxy/_experimental/out/{api-reference.html => api-reference/index.html} (100%) rename litellm/proxy/_experimental/out/{chat.html => chat/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{api-playground.html => api-playground/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{budgets.html => budgets/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{caching.html => caching/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{claude-code-plugins.html => claude-code-plugins/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{old-usage.html => old-usage/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{prompts.html => prompts/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{tag-management.html => tag-management/index.html} (100%) rename litellm/proxy/_experimental/out/{guardrails.html => guardrails/index.html} (100%) rename litellm/proxy/_experimental/out/{login.html => login/index.html} (100%) rename litellm/proxy/_experimental/out/{logs.html => logs/index.html} (100%) rename litellm/proxy/_experimental/out/mcp/oauth/{callback.html => callback/index.html} (100%) rename litellm/proxy/_experimental/out/{model-hub.html => model-hub/index.html} (100%) rename litellm/proxy/_experimental/out/{model_hub.html => model_hub/index.html} (100%) rename litellm/proxy/_experimental/out/{model_hub_table.html => model_hub_table/index.html} (100%) rename litellm/proxy/_experimental/out/{models-and-endpoints.html => models-and-endpoints/index.html} (100%) rename litellm/proxy/_experimental/out/{onboarding.html => onboarding/index.html} (100%) rename litellm/proxy/_experimental/out/{organizations.html => organizations/index.html} (100%) rename litellm/proxy/_experimental/out/{playground.html => playground/index.html} (100%) rename litellm/proxy/_experimental/out/{policies.html => policies/index.html} (100%) rename litellm/proxy/_experimental/out/settings/{admin-settings.html => admin-settings/index.html} (100%) rename litellm/proxy/_experimental/out/settings/{logging-and-alerts.html => logging-and-alerts/index.html} (100%) rename litellm/proxy/_experimental/out/settings/{router-settings.html => router-settings/index.html} (100%) rename litellm/proxy/_experimental/out/settings/{ui-theme.html => ui-theme/index.html} (100%) rename litellm/proxy/_experimental/out/{teams.html => teams/index.html} (100%) rename litellm/proxy/_experimental/out/{test-key.html => test-key/index.html} (100%) rename litellm/proxy/_experimental/out/tools/{mcp-servers.html => mcp-servers/index.html} (100%) rename litellm/proxy/_experimental/out/tools/{vector-stores.html => vector-stores/index.html} (100%) rename litellm/proxy/_experimental/out/{usage.html => usage/index.html} (100%) rename litellm/proxy/_experimental/out/{users.html => users/index.html} (100%) rename litellm/proxy/_experimental/out/{virtual-keys.html => virtual-keys/index.html} (100%) diff --git a/litellm/proxy/_experimental/out/404.html b/litellm/proxy/_experimental/out/404/index.html similarity index 100% rename from litellm/proxy/_experimental/out/404.html rename to litellm/proxy/_experimental/out/404/index.html diff --git a/litellm/proxy/_experimental/out/_not-found.html b/litellm/proxy/_experimental/out/_not-found/index.html similarity index 100% rename from litellm/proxy/_experimental/out/_not-found.html rename to litellm/proxy/_experimental/out/_not-found/index.html diff --git a/litellm/proxy/_experimental/out/api-reference.html b/litellm/proxy/_experimental/out/api-reference/index.html similarity index 100% rename from litellm/proxy/_experimental/out/api-reference.html rename to litellm/proxy/_experimental/out/api-reference/index.html diff --git a/litellm/proxy/_experimental/out/chat.html b/litellm/proxy/_experimental/out/chat/index.html similarity index 100% rename from litellm/proxy/_experimental/out/chat.html rename to litellm/proxy/_experimental/out/chat/index.html diff --git a/litellm/proxy/_experimental/out/experimental/api-playground.html b/litellm/proxy/_experimental/out/experimental/api-playground/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/api-playground.html rename to litellm/proxy/_experimental/out/experimental/api-playground/index.html diff --git a/litellm/proxy/_experimental/out/experimental/budgets.html b/litellm/proxy/_experimental/out/experimental/budgets/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/budgets.html rename to litellm/proxy/_experimental/out/experimental/budgets/index.html diff --git a/litellm/proxy/_experimental/out/experimental/caching.html b/litellm/proxy/_experimental/out/experimental/caching/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/caching.html rename to litellm/proxy/_experimental/out/experimental/caching/index.html diff --git a/litellm/proxy/_experimental/out/experimental/claude-code-plugins.html b/litellm/proxy/_experimental/out/experimental/claude-code-plugins/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/claude-code-plugins.html rename to litellm/proxy/_experimental/out/experimental/claude-code-plugins/index.html diff --git a/litellm/proxy/_experimental/out/experimental/old-usage.html b/litellm/proxy/_experimental/out/experimental/old-usage/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/old-usage.html rename to litellm/proxy/_experimental/out/experimental/old-usage/index.html diff --git a/litellm/proxy/_experimental/out/experimental/prompts.html b/litellm/proxy/_experimental/out/experimental/prompts/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/prompts.html rename to litellm/proxy/_experimental/out/experimental/prompts/index.html diff --git a/litellm/proxy/_experimental/out/experimental/tag-management.html b/litellm/proxy/_experimental/out/experimental/tag-management/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/tag-management.html rename to litellm/proxy/_experimental/out/experimental/tag-management/index.html diff --git a/litellm/proxy/_experimental/out/guardrails.html b/litellm/proxy/_experimental/out/guardrails/index.html similarity index 100% rename from litellm/proxy/_experimental/out/guardrails.html rename to litellm/proxy/_experimental/out/guardrails/index.html diff --git a/litellm/proxy/_experimental/out/login.html b/litellm/proxy/_experimental/out/login/index.html similarity index 100% rename from litellm/proxy/_experimental/out/login.html rename to litellm/proxy/_experimental/out/login/index.html diff --git a/litellm/proxy/_experimental/out/logs.html b/litellm/proxy/_experimental/out/logs/index.html similarity index 100% rename from litellm/proxy/_experimental/out/logs.html rename to litellm/proxy/_experimental/out/logs/index.html diff --git a/litellm/proxy/_experimental/out/mcp/oauth/callback.html b/litellm/proxy/_experimental/out/mcp/oauth/callback/index.html similarity index 100% rename from litellm/proxy/_experimental/out/mcp/oauth/callback.html rename to litellm/proxy/_experimental/out/mcp/oauth/callback/index.html diff --git a/litellm/proxy/_experimental/out/model-hub.html b/litellm/proxy/_experimental/out/model-hub/index.html similarity index 100% rename from litellm/proxy/_experimental/out/model-hub.html rename to litellm/proxy/_experimental/out/model-hub/index.html diff --git a/litellm/proxy/_experimental/out/model_hub.html b/litellm/proxy/_experimental/out/model_hub/index.html similarity index 100% rename from litellm/proxy/_experimental/out/model_hub.html rename to litellm/proxy/_experimental/out/model_hub/index.html diff --git a/litellm/proxy/_experimental/out/model_hub_table.html b/litellm/proxy/_experimental/out/model_hub_table/index.html similarity index 100% rename from litellm/proxy/_experimental/out/model_hub_table.html rename to litellm/proxy/_experimental/out/model_hub_table/index.html diff --git a/litellm/proxy/_experimental/out/models-and-endpoints.html b/litellm/proxy/_experimental/out/models-and-endpoints/index.html similarity index 100% rename from litellm/proxy/_experimental/out/models-and-endpoints.html rename to litellm/proxy/_experimental/out/models-and-endpoints/index.html diff --git a/litellm/proxy/_experimental/out/onboarding.html b/litellm/proxy/_experimental/out/onboarding/index.html similarity index 100% rename from litellm/proxy/_experimental/out/onboarding.html rename to litellm/proxy/_experimental/out/onboarding/index.html diff --git a/litellm/proxy/_experimental/out/organizations.html b/litellm/proxy/_experimental/out/organizations/index.html similarity index 100% rename from litellm/proxy/_experimental/out/organizations.html rename to litellm/proxy/_experimental/out/organizations/index.html diff --git a/litellm/proxy/_experimental/out/playground.html b/litellm/proxy/_experimental/out/playground/index.html similarity index 100% rename from litellm/proxy/_experimental/out/playground.html rename to litellm/proxy/_experimental/out/playground/index.html diff --git a/litellm/proxy/_experimental/out/policies.html b/litellm/proxy/_experimental/out/policies/index.html similarity index 100% rename from litellm/proxy/_experimental/out/policies.html rename to litellm/proxy/_experimental/out/policies/index.html diff --git a/litellm/proxy/_experimental/out/settings/admin-settings.html b/litellm/proxy/_experimental/out/settings/admin-settings/index.html similarity index 100% rename from litellm/proxy/_experimental/out/settings/admin-settings.html rename to litellm/proxy/_experimental/out/settings/admin-settings/index.html diff --git a/litellm/proxy/_experimental/out/settings/logging-and-alerts.html b/litellm/proxy/_experimental/out/settings/logging-and-alerts/index.html similarity index 100% rename from litellm/proxy/_experimental/out/settings/logging-and-alerts.html rename to litellm/proxy/_experimental/out/settings/logging-and-alerts/index.html diff --git a/litellm/proxy/_experimental/out/settings/router-settings.html b/litellm/proxy/_experimental/out/settings/router-settings/index.html similarity index 100% rename from litellm/proxy/_experimental/out/settings/router-settings.html rename to litellm/proxy/_experimental/out/settings/router-settings/index.html diff --git a/litellm/proxy/_experimental/out/settings/ui-theme.html b/litellm/proxy/_experimental/out/settings/ui-theme/index.html similarity index 100% rename from litellm/proxy/_experimental/out/settings/ui-theme.html rename to litellm/proxy/_experimental/out/settings/ui-theme/index.html diff --git a/litellm/proxy/_experimental/out/teams.html b/litellm/proxy/_experimental/out/teams/index.html similarity index 100% rename from litellm/proxy/_experimental/out/teams.html rename to litellm/proxy/_experimental/out/teams/index.html diff --git a/litellm/proxy/_experimental/out/test-key.html b/litellm/proxy/_experimental/out/test-key/index.html similarity index 100% rename from litellm/proxy/_experimental/out/test-key.html rename to litellm/proxy/_experimental/out/test-key/index.html diff --git a/litellm/proxy/_experimental/out/tools/mcp-servers.html b/litellm/proxy/_experimental/out/tools/mcp-servers/index.html similarity index 100% rename from litellm/proxy/_experimental/out/tools/mcp-servers.html rename to litellm/proxy/_experimental/out/tools/mcp-servers/index.html diff --git a/litellm/proxy/_experimental/out/tools/vector-stores.html b/litellm/proxy/_experimental/out/tools/vector-stores/index.html similarity index 100% rename from litellm/proxy/_experimental/out/tools/vector-stores.html rename to litellm/proxy/_experimental/out/tools/vector-stores/index.html diff --git a/litellm/proxy/_experimental/out/usage.html b/litellm/proxy/_experimental/out/usage/index.html similarity index 100% rename from litellm/proxy/_experimental/out/usage.html rename to litellm/proxy/_experimental/out/usage/index.html diff --git a/litellm/proxy/_experimental/out/users.html b/litellm/proxy/_experimental/out/users/index.html similarity index 100% rename from litellm/proxy/_experimental/out/users.html rename to litellm/proxy/_experimental/out/users/index.html diff --git a/litellm/proxy/_experimental/out/virtual-keys.html b/litellm/proxy/_experimental/out/virtual-keys/index.html similarity index 100% rename from litellm/proxy/_experimental/out/virtual-keys.html rename to litellm/proxy/_experimental/out/virtual-keys/index.html From 38e80032976b500f85a8de39a6f0bbaacd1fa653 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 08:33:11 +0530 Subject: [PATCH 061/117] fix(anthropic): strip undocumented keys from metadata before sending to API --- litellm/llms/anthropic/chat/transformation.py | 10 +++ .../test_anthropic_completion.py | 78 +++++++++++++++++++ 2 files changed, 88 insertions(+) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 73d1b02c76d..9a99f9efc82 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1421,6 +1421,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): ): optional_params["metadata"] = {"user_id": _litellm_metadata["user_id"]} + ## Ensure metadata only contains user_id (only documented field in Anthropic Messages API) + if "metadata" in optional_params and isinstance( + optional_params["metadata"], dict + ): + _user_id = optional_params["metadata"].get("user_id") + if _user_id is not None: + optional_params["metadata"] = {"user_id": _user_id} + else: + optional_params.pop("metadata") + # Remove internal LiteLLM parameters that should not be sent to Anthropic API optional_params.pop("is_vertex_request", None) diff --git a/tests/llm_translation/test_anthropic_completion.py b/tests/llm_translation/test_anthropic_completion.py index 8630ba65610..3c1162cd016 100644 --- a/tests/llm_translation/test_anthropic_completion.py +++ b/tests/llm_translation/test_anthropic_completion.py @@ -1800,3 +1800,81 @@ def test_anthropic_structured_output_chat_completion_api(): ) assert response is not None print(f"response: {response}") + + +def _make_transform_request(optional_params: dict, litellm_params: dict) -> dict: + from litellm.llms.anthropic.chat.transformation import AnthropicConfig + + return AnthropicConfig().transform_request( + model="claude-3-5-sonnet-20241022", + messages=[{"role": "user", "content": "hi"}], + optional_params=optional_params, + litellm_params=litellm_params, + headers={}, + ) + + +def test_metadata_only_user_id_passes_through(): + """metadata with only user_id is forwarded as-is.""" + data = _make_transform_request( + optional_params={"metadata": {"user_id": "abc123"}}, + litellm_params={}, + ) + assert data.get("metadata") == {"user_id": "abc123"} + + +def test_metadata_extra_keys_are_stripped(): + """Extra keys in metadata are removed; only user_id is sent.""" + data = _make_transform_request( + optional_params={"metadata": {"user_id": "abc123", "extra_key": "val"}}, + litellm_params={}, + ) + assert data.get("metadata") == {"user_id": "abc123"} + + +def test_metadata_without_user_id_is_dropped(): + """metadata with no user_id is removed entirely.""" + data = _make_transform_request( + optional_params={"metadata": {"only_other_key": "val"}}, + litellm_params={}, + ) + assert "metadata" not in data + + +def test_metadata_user_id_from_litellm_params_strips_extras(): + """user_id from litellm_params metadata is extracted; extra keys are not forwarded.""" + data = _make_transform_request( + optional_params={}, + litellm_params={"metadata": {"user_id": "abc123", "trace_id": "xyz"}}, + ) + assert data.get("metadata") == {"user_id": "abc123"} + + +def test_metadata_filter_applies_to_vertex_anthropic(): + """VertexAIAnthropicConfig inherits the metadata filter.""" + from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import ( + VertexAIAnthropicConfig, + ) + + data = VertexAIAnthropicConfig().transform_request( + model="claude-3-5-sonnet-20241022", + messages=[{"role": "user", "content": "hi"}], + optional_params={"metadata": {"user_id": "u1", "extra": "drop_me"}}, + litellm_params={}, + headers={}, + ) + assert data.get("metadata") == {"user_id": "u1"} + + +def test_metadata_filter_applies_to_azure_anthropic(): + """AzureAnthropicConfig inherits the metadata filter.""" + from litellm.llms.azure_ai.anthropic.transformation import AzureAnthropicConfig + + data = AzureAnthropicConfig().transform_request( + model="claude-3-5-sonnet-20241022", + messages=[{"role": "user", "content": "hi"}], + optional_params={"metadata": {"user_id": "u2", "extra": "drop_me"}}, + litellm_params={}, + headers={}, + ) + assert data.get("metadata") == {"user_id": "u2"} From b212b340ab82fe4acaac6c9b54604709115941b4 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 08:49:33 +0530 Subject: [PATCH 062/117] feat(gemini): normalize AI Studio file retrieve URL and harden tests Made-with: Cursor --- litellm/llms/gemini/files/transformation.py | 59 +++++++++++---- .../files/test_gemini_files_transformation.py | 71 ++++++++++++++----- 2 files changed, 98 insertions(+), 32 deletions(-) diff --git a/litellm/llms/gemini/files/transformation.py b/litellm/llms/gemini/files/transformation.py index bdfb0ee1e52..a29ed66e63d 100644 --- a/litellm/llms/gemini/files/transformation.py +++ b/litellm/llms/gemini/files/transformation.py @@ -5,6 +5,7 @@ For vertex ai, check out the vertex_ai/files/handler.py file. """ import time from typing import Any, List, Literal, Optional +from urllib.parse import urlparse import httpx from openai.types.file_deleted import FileDeleted @@ -209,27 +210,58 @@ class GoogleAIStudioFilesHandler(GeminiModelInfo, BaseFilesConfig): """ Get the URL to retrieve a file from Google AI Studio. - We expect file_id to be the URI (e.g. https://generativelanguage.googleapis.com/v1beta/files/...) - as returned by the upload response. + Endpoint: + GET https://generativelanguage.googleapis.com/v1beta/{name=files/*} + + The URL should look like: + https://generativelanguage.googleapis.com/v1beta/files/{file_id}?key=API_KEY + + We expect file_id to be just the file identifier (e.g., files/abc123 or abc123) + as returned by the upload response. (If it's a full URL, extract the file name.) """ api_key = litellm_params.get("api_key") or self.get_api_key() if not api_key: raise ValueError("api_key is required") - if file_id.startswith("http"): - url = "{}?key={}".format(file_id, api_key) - else: - # Fallback for just file name (files/...) - api_base = ( - self.get_api_base(litellm_params.get("api_base")) - or "https://generativelanguage.googleapis.com" - ) - api_base = api_base.rstrip("/") - url = "{}/v1beta/{}?key={}".format(api_base, file_id, api_key) + file_part = self._normalize_gemini_file_id(file_id) + + api_base = ( + self.get_api_base(litellm_params.get("api_base")) + or "https://generativelanguage.googleapis.com" + ) + api_base = api_base.rstrip("/") + + url = f"{api_base}/v1beta/{file_part}?key={api_key}" # Return empty params dict - API key is already in URL, no query params needed return url, {} + def _normalize_gemini_file_id(self, file_id: str) -> str: + """ + Normalize file identifier into `files/{id}` form. + + Supports: + - `abc123` + - `files/abc123` + - `https://generativelanguage.googleapis.com/v1beta/files/abc123` + """ + if file_id.startswith(("http://", "https://")): + parsed = urlparse(file_id) + path = parsed.path.lstrip("/") + files_index = path.find("files/") + if files_index != -1: + normalized_file_id = path[files_index:] + else: + normalized_file_id = path + else: + normalized_file_id = file_id + + normalized_file_id = normalized_file_id.strip("/") + if not normalized_file_id.startswith("files/"): + normalized_file_id = f"files/{normalized_file_id}" + + return normalized_file_id + def transform_retrieve_file_response( self, raw_response: httpx.Response, @@ -240,8 +272,9 @@ class GoogleAIStudioFilesHandler(GeminiModelInfo, BaseFilesConfig): Transform Gemini's file retrieval response into OpenAI-style FileObject """ try: + verbose_logger.debug(f"Retrieve file response: {raw_response.text}") response_json = raw_response.json() - + verbose_logger.debug(f"Response JSON: {response_json}") # Map Gemini state to OpenAI status gemini_state = response_json.get("state", "STATE_UNSPECIFIED") # Explicitly type status as the Literal union diff --git a/tests/test_litellm/llms/gemini/files/test_gemini_files_transformation.py b/tests/test_litellm/llms/gemini/files/test_gemini_files_transformation.py index a5f72fc08c3..6cc97cd95e6 100644 --- a/tests/test_litellm/llms/gemini/files/test_gemini_files_transformation.py +++ b/tests/test_litellm/llms/gemini/files/test_gemini_files_transformation.py @@ -3,10 +3,10 @@ Test Google AI Studio (Gemini) files transformation functionality """ import os -import pytest from unittest.mock import Mock, patch import httpx +import pytest from litellm.llms.gemini.files.transformation import GoogleAIStudioFilesHandler from litellm.types.llms.openai import OpenAIFileObject @@ -23,7 +23,7 @@ class TestGoogleAIStudioFilesTransformation: """ Test that transform_retrieve_file_request returns empty params dict to avoid 'Content-Type' query parameter error - + Regression test for: https://github.com/BerriAI/litellm/issues/XXX When retrieving a file, the API was incorrectly trying to pass Content-Type as a query parameter, which Gemini API rejected. @@ -37,14 +37,19 @@ class TestGoogleAIStudioFilesTransformation: litellm_params=litellm_params, ) - # Verify URL is constructed correctly with API key - assert "key=test-api-key" in url - assert file_id in url + # Verify URL is constructed exactly as required: + # https://generativelanguage.googleapis.com/v1beta/files/{file_id}?key=API_KEY + assert ( + url + == "https://generativelanguage.googleapis.com/v1beta/files/test123?key=test-api-key" + ) # CRITICAL: params should be empty dict, not contain Content-Type or any other params # These would be incorrectly interpreted as query parameters assert params == {}, f"Expected empty params dict, got: {params}" - assert "Content-Type" not in params, "Content-Type should not be in query params" + assert ( + "Content-Type" not in params + ), "Content-Type should not be in query params" def test_transform_retrieve_file_request_with_file_name_only(self): """ @@ -59,17 +64,44 @@ class TestGoogleAIStudioFilesTransformation: litellm_params=litellm_params, ) - # Verify URL is constructed correctly - assert "generativelanguage.googleapis.com" in url - assert file_id in url - assert "key=test-api-key" in url + # Verify URL is constructed exactly as required: + # https://generativelanguage.googleapis.com/v1beta/files/{file_id}?key=API_KEY + assert ( + url + == "https://generativelanguage.googleapis.com/v1beta/files/test123?key=test-api-key" + ) # CRITICAL: params should be empty dict assert params == {}, f"Expected empty params dict, got: {params}" - assert "Content-Type" not in params, "Content-Type should not be in query params" + assert ( + "Content-Type" not in params + ), "Content-Type should not be in query params" - @patch.dict('os.environ', {}, clear=True) - @patch('litellm.llms.gemini.common_utils.get_secret_str', return_value=None) + def test_transform_retrieve_file_request_with_raw_id_only(self): + """ + Regression guard for the exact retrieval URL format. + + If someone changes the method and stops producing: + https://generativelanguage.googleapis.com/v1beta/files/{file_id}?key=API_KEY + this test should fail. + """ + file_id = "cctqueckiggb" + litellm_params = {"api_key": "test-api-key"} + + url, params = self.handler.transform_retrieve_file_request( + file_id=file_id, + optional_params={}, + litellm_params=litellm_params, + ) + + assert ( + url + == "https://generativelanguage.googleapis.com/v1beta/files/cctqueckiggb?key=test-api-key" + ) + assert params == {} + + @patch.dict("os.environ", {}, clear=True) + @patch("litellm.llms.gemini.common_utils.get_secret_str", return_value=None) def test_transform_retrieve_file_request_missing_api_key(self, mock_get_secret): """Test that transform_retrieve_file_request raises error when API key is missing""" file_id = "files/test123" @@ -178,7 +210,7 @@ class TestGoogleAIStudioFilesTransformation: def test_transform_retrieve_file_response_missing_createTime(self): """ Test that transform_retrieve_file_response raises proper error when createTime is missing - + This tests the error scenario that occurs when API returns an error response without the expected file metadata fields. """ @@ -221,14 +253,15 @@ class TestGoogleAIStudioFilesTransformation: assert "x-goog-api-key" in result_headers assert result_headers["x-goog-api-key"] == api_key - @patch.dict('os.environ', {}, clear=True) - @patch('litellm.llms.gemini.common_utils.get_secret_str', return_value=None) + @patch.dict("os.environ", {}, clear=True) + @patch("litellm.llms.gemini.common_utils.get_secret_str", return_value=None) def test_validate_environment_missing_api_key(self, mock_get_secret): """Test that validate_environment raises error when API key is missing""" headers = {} with pytest.raises( - ValueError, match="GEMINI_API_KEY is required for Google AI Studio file operations" + ValueError, + match="GEMINI_API_KEY is required for Google AI Studio file operations", ): self.handler.validate_environment( headers=headers, @@ -243,7 +276,7 @@ class TestGoogleAIStudioFilesTransformation: """Test that get_complete_url constructs proper upload URL""" api_base = "https://generativelanguage.googleapis.com" api_key = "test-api-key" - + url = self.handler.get_complete_url( api_base=api_base, api_key=api_key, @@ -274,7 +307,7 @@ class TestGoogleAIStudioFilesTransformation: # Verify URL extraction assert "files/test123" in url assert "generativelanguage.googleapis.com" in url - + # Params should be empty (API key goes in header via validate_environment) assert params == {} From 76754886400285c36e6b130cb98524e9a15393d7 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 14:43:16 +0530 Subject: [PATCH 063/117] feat(router): add health-check-driven routing behind opt-in flag MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Background health checks now feed deployment health state into the router candidate-filtering pipeline. Unhealthy deployments are excluded proactively instead of waiting for request failures to trigger cooldown. Gated by `enable_health_check_routing: true` in general_settings. Off by default — zero behavior change for existing users. Co-Authored-By: Claude Opus 4.6 --- docs/my-website/docs/proxy/health.md | 83 ++++++++ litellm/constants.py | 3 + litellm/proxy/health_check.py | 58 +++++- litellm/proxy/proxy_server.py | 55 ++++- litellm/router.py | 100 ++++++++- litellm/router_utils/health_state_cache.py | 100 +++++++++ .../router_utils/test_health_check_routing.py | 197 ++++++++++++++++++ .../router_utils/test_health_state_cache.py | 113 ++++++++++ 8 files changed, 696 insertions(+), 13 deletions(-) create mode 100644 litellm/router_utils/health_state_cache.py create mode 100644 tests/test_litellm/router_utils/test_health_check_routing.py create mode 100644 tests/test_litellm/router_utils/test_health_state_cache.py diff --git a/docs/my-website/docs/proxy/health.md b/docs/my-website/docs/proxy/health.md index 2764a6f0d4f..530bea3d06b 100644 --- a/docs/my-website/docs/proxy/health.md +++ b/docs/my-website/docs/proxy/health.md @@ -314,6 +314,89 @@ general_settings: health_check_details: False ``` +## Health Check Driven Routing + +By default, background health checks are observability-only — they populate the `/health` endpoint but don't affect routing. Unhealthy deployments still receive traffic until request failures trigger cooldown. + +With `enable_health_check_routing: true`, the router **excludes deployments that failed their last background health check** before selecting a candidate. This gives you proactive failover instead of reactive cooldown. + +### How it works + +1. Background health checks run on their configured interval +2. After each cycle, every deployment is marked healthy or unhealthy +3. On each incoming request, the router filters out unhealthy deployments **before** cooldown filtering and load balancing +4. If all deployments are unhealthy, the filter is bypassed (safety net — never causes a total outage) +5. If health state is stale (older than `health_check_staleness_threshold`), it is ignored + +### Quick start + +```yaml +model_list: + - model_name: gpt-4 + litellm_params: + model: openai/gpt-4 + api_key: os.environ/OPENAI_API_KEY + - model_name: gpt-4 + litellm_params: + model: openai/gpt-4 + api_key: os.environ/OPENAI_API_KEY_SECONDARY + +general_settings: + background_health_checks: true + health_check_interval: 60 + enable_health_check_routing: true +``` + +### Configuration + +| Setting | Where | Default | Description | +|---------|-------|---------|-------------| +| `enable_health_check_routing` | `general_settings` | `false` | Enable/disable health-check-driven routing | +| `health_check_staleness_threshold` | `general_settings` | `health_check_interval * 2` | Seconds before health state is considered stale and ignored | +| `background_health_checks` | `general_settings` | `false` | Must be `true` for health check routing to work | +| `health_check_interval` | `general_settings` | `300` | Seconds between health check cycles | + +### Interaction with cooldown + +Health check filtering and cooldown are **additive**. A deployment can be excluded by either mechanism: + +- **Health check filter** — proactive, runs on the configured interval, excludes deployments that failed the last check +- **Cooldown** — reactive, triggered by request failures, excludes deployments for a short TTL + +This means request failures still provide fast detection between health check intervals. + +### Staleness + +If a health check result is older than `health_check_staleness_threshold`, it is ignored and the deployment is treated as eligible. This prevents stale data from permanently excluding a deployment if the health check loop stops or slows down. + +The default staleness threshold is `health_check_interval * 2`. For a 60s interval, health state expires after 120s. + +### Example: custom staleness + +```yaml +general_settings: + background_health_checks: true + health_check_interval: 30 + enable_health_check_routing: true + health_check_staleness_threshold: 90 # ignore health state older than 90s +``` + +### Debugging + +Run the proxy with `--detailed_debug` and look for: + +``` +health_check_routing_state_updated healthy=3 unhealthy=1 +``` + +This is logged after each health check cycle when routing state is written. + +If the safety net triggers (all deployments unhealthy), you'll see: + +``` +All deployments marked unhealthy by health checks, bypassing health filter +``` + ## Health Check Timeout The health check timeout is set in `litellm/constants.py` and defaults to 60 seconds. diff --git a/litellm/constants.py b/litellm/constants.py index 423f01afac1..252068bd7b0 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1402,6 +1402,9 @@ DEFAULT_SHARED_HEALTH_CHECK_TTL = int( DEFAULT_SHARED_HEALTH_CHECK_LOCK_TTL = int( os.getenv("DEFAULT_SHARED_HEALTH_CHECK_LOCK_TTL", 60) ) # 1 minute - TTL for health check lock +DEFAULT_HEALTH_CHECK_STALENESS_MULTIPLIER = ( + 2 # health state is stale after interval * this +) PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS = int( os.getenv("PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS", 9) ) diff --git a/litellm/proxy/health_check.py b/litellm/proxy/health_check.py index a8d0e3e9af2..058f2f4ed9d 100644 --- a/litellm/proxy/health_check.py +++ b/litellm/proxy/health_check.py @@ -207,21 +207,65 @@ async def _perform_health_check( for is_healthy, model in zip(results, model_list): litellm_params = model["litellm_params"] + _model_id = (model.get("model_info") or {}).get("id") if isinstance(is_healthy, dict) and "error" not in is_healthy: - healthy_endpoints.append( - _clean_endpoint_data({**litellm_params, **is_healthy}, details) - ) + endpoint_data = {**litellm_params, **is_healthy} + if _model_id: + endpoint_data["model_id"] = _model_id + healthy_endpoints.append(_clean_endpoint_data(endpoint_data, details)) elif isinstance(is_healthy, dict): - unhealthy_endpoints.append( - _clean_endpoint_data({**litellm_params, **is_healthy}, details) - ) + endpoint_data = {**litellm_params, **is_healthy} + if _model_id: + endpoint_data["model_id"] = _model_id + unhealthy_endpoints.append(_clean_endpoint_data(endpoint_data, details)) else: - unhealthy_endpoints.append(_clean_endpoint_data(litellm_params, details)) + endpoint_data = {**litellm_params} + if _model_id: + endpoint_data["model_id"] = _model_id + unhealthy_endpoints.append(_clean_endpoint_data(endpoint_data, details)) return healthy_endpoints, unhealthy_endpoints +def build_deployment_health_states( + healthy_endpoints: list, + unhealthy_endpoints: list, +) -> dict: + """ + Build a dict mapping deployment_id -> DeploymentHealthStateValue from + health check endpoint results. + + Each endpoint dict includes a 'model_id' field (added by _perform_health_check) + that maps back to the deployment's model_info.id. + + Used by the background health check loop to feed health state into + the router's DeploymentHealthCache for health-check-driven routing. + """ + now = time.time() + states: dict = {} + + for ep in healthy_endpoints: + model_id = ep.get("model_id") + if model_id: + states[model_id] = { + "is_healthy": True, + "timestamp": now, + "reason": "", + } + + for ep in unhealthy_endpoints: + model_id = ep.get("model_id") + if model_id: + states[model_id] = { + "is_healthy": False, + "timestamp": now, + "reason": "background_health_check_failed", + } + + return states + + def _update_litellm_params_for_health_check( model_info: dict, litellm_params: dict ) -> dict: diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 7d3d2ceb533..42740c24f45 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -37,7 +37,7 @@ import websockets import websockets.exceptions from pydantic import BaseModel, Json -from litellm._uuid import uuid +from litellm._litellm_uuid import uuid from litellm.constants import ( AIOHTTP_CONNECTOR_LIMIT, AIOHTTP_CONNECTOR_LIMIT_PER_HOST, @@ -480,11 +480,11 @@ from litellm.proxy.search_endpoints.search_tool_management import ( router as search_tool_management_router, ) from litellm.proxy.spend_tracking.cloudzero_endpoints import router as cloudzero_router -from litellm.proxy.spend_tracking.vantage_endpoints import router as vantage_router from litellm.proxy.spend_tracking.spend_management_endpoints import ( router as spend_management_router, ) from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload +from litellm.proxy.spend_tracking.vantage_endpoints import router as vantage_router from litellm.proxy.types_utils.utils import get_instance_fn from litellm.proxy.ui_crud_endpoints.proxy_setting_endpoints import ( router as ui_crud_endpoints_router, @@ -2112,6 +2112,37 @@ def _schedule_background_health_check_db_save( ) +def _write_health_state_to_router_cache( + healthy_endpoints: list, + unhealthy_endpoints: list, +) -> None: + """ + Write deployment health states to the router's health state cache + for health-check-driven routing. No-op if the feature is disabled. + """ + from litellm.proxy.health_check import build_deployment_health_states + + try: + if llm_router is None or not llm_router.enable_health_check_routing: + return + + states = build_deployment_health_states( + healthy_endpoints=healthy_endpoints, + unhealthy_endpoints=unhealthy_endpoints, + ) + if states: + llm_router.health_state_cache.set_deployment_health_states(states) + verbose_proxy_logger.debug( + "health_check_routing_state_updated healthy=%d unhealthy=%d", + sum(1 for s in states.values() if s.get("is_healthy")), + sum(1 for s in states.values() if not s.get("is_healthy")), + ) + except Exception as e: + verbose_proxy_logger.debug( + "Failed to write health state to router cache: %s", str(e) + ) + + async def _run_background_health_check(): """ Periodically run health checks in the background on the endpoints. @@ -2281,6 +2312,9 @@ async def _run_background_health_check(): unhealthy_endpoints, ) + # Write health state to router cache for health-check-driven routing + _write_health_state_to_router_cache(healthy_endpoints, unhealthy_endpoints) + await asyncio.sleep(health_check_interval) @@ -3048,6 +3082,8 @@ class ProxyConfig: general_settings = config.get("general_settings", {}) if general_settings is None: general_settings = {} + _enable_hc_routing = False + _hc_staleness = None if general_settings: ### LOAD KEY MANAGEMENT SETTINGS FIRST (needed for custom secret manager) ### key_management_settings = general_settings.get( @@ -3227,13 +3263,21 @@ class ProxyConfig: "health_check_concurrency", None ) health_check_details = general_settings.get("health_check_details", True) + # Health-check-driven routing (opt-in, passes through to Router later) + _enable_hc_routing = general_settings.get( + "enable_health_check_routing", False + ) + _hc_staleness = general_settings.get( + "health_check_staleness_threshold", None + ) verbose_proxy_logger.info( - "background_health_check_config enabled=%s shared=%s interval_seconds=%s max_concurrency=%s details=%s", + "background_health_check_config enabled=%s shared=%s interval_seconds=%s max_concurrency=%s details=%s health_check_routing=%s", use_background_health_checks, use_shared_health_check, health_check_interval, health_check_concurrency, health_check_details, + _enable_hc_routing, ) ### RBAC ### @@ -3263,6 +3307,11 @@ class ProxyConfig: "cache_responses": litellm.cache is not None, # cache if user passed in cache values } + # Health-check-driven routing params (from general_settings) + if _enable_hc_routing: + router_params["enable_health_check_routing"] = True + if _hc_staleness is not None: + router_params["health_check_staleness_threshold"] = _hc_staleness ## MODEL LIST model_list = config.get("model_list", None) if model_list: diff --git a/litellm/router.py b/litellm/router.py index 5cd4f837782..8d0e3334cb2 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -46,15 +46,19 @@ import litellm import litellm.litellm_core_utils import litellm.litellm_core_utils.exception_mapping_utils from litellm import get_secret_str +from litellm._litellm_uuid import uuid from litellm._logging import verbose_router_logger -from litellm._uuid import uuid from litellm.caching.caching import ( DualCache, InMemoryCache, RedisCache, RedisClusterCache, ) -from litellm.constants import DEFAULT_MAX_LRU_CACHE_SIZE +from litellm.constants import ( + DEFAULT_HEALTH_CHECK_INTERVAL, + DEFAULT_HEALTH_CHECK_STALENESS_MULTIPLIER, + DEFAULT_MAX_LRU_CACHE_SIZE, +) from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.asyncify import run_async_function from litellm.litellm_core_utils.core_helpers import ( @@ -113,6 +117,7 @@ from litellm.router_utils.handle_error import ( async_raise_no_deployment_exception, send_llm_exception_alert, ) +from litellm.router_utils.health_state_cache import DeploymentHealthCache from litellm.router_utils.pre_call_checks.deployment_affinity_check import ( DeploymentAffinityCheck, ) @@ -303,6 +308,8 @@ class Router: deployment_affinity_ttl_seconds: int = 3600, model_group_affinity_config: Optional[Dict[str, List[str]]] = None, ignore_invalid_deployments: bool = False, + enable_health_check_routing: bool = False, + health_check_staleness_threshold: Optional[int] = None, ) -> None: """ Initialize the Router class with the given parameters for caching, reliability, and routing strategy. @@ -493,6 +500,13 @@ class Router: cache=self.cache, default_cooldown_time=self.cooldown_time ) self.disable_cooldowns = disable_cooldowns + self.enable_health_check_routing = enable_health_check_routing + _staleness = health_check_staleness_threshold or ( + DEFAULT_HEALTH_CHECK_INTERVAL * DEFAULT_HEALTH_CHECK_STALENESS_MULTIPLIER + ) + self.health_state_cache = DeploymentHealthCache( + cache=self.cache, staleness_threshold=float(_staleness) + ) self.failed_calls = ( InMemoryCache() ) # cache to track failed call per deployment, if num failed calls within 1 minute > allowed fails, then add it to cooldown @@ -9154,6 +9168,14 @@ class Router: if isinstance(healthy_deployments, dict): return healthy_deployments + # Health-check-based filtering (before cooldown) + healthy_deployments = ( + await self._async_filter_health_check_unhealthy_deployments( + healthy_deployments=healthy_deployments, + parent_otel_span=parent_otel_span, + ) + ) + cooldown_deployments = await _async_get_cooldown_deployments( litellm_router_instance=self, parent_otel_span=parent_otel_span ) @@ -9585,6 +9607,13 @@ class Router: parent_otel_span: Optional[Span] = _get_parent_otel_span_from_kwargs( request_kwargs ) + + # Health-check-based filtering (before cooldown) + healthy_deployments = self._filter_health_check_unhealthy_deployments( + healthy_deployments=healthy_deployments, + parent_otel_span=parent_otel_span, + ) + cooldown_deployments = _get_cooldown_deployments( litellm_router_instance=self, parent_otel_span=parent_otel_span ) @@ -9750,10 +9779,14 @@ class Router: llm_provider="", ) - # 4. Apply cooldown filtering + # 4. Apply health-check and cooldown filtering parent_otel_span: Optional[Span] = _get_parent_otel_span_from_kwargs( request_kwargs ) + pass_through_deployments = self._filter_health_check_unhealthy_deployments( + healthy_deployments=pass_through_deployments, + parent_otel_span=parent_otel_span, + ) cooldown_deployments = _get_cooldown_deployments( litellm_router_instance=self, parent_otel_span=parent_otel_span ) @@ -9875,6 +9908,67 @@ class Router: if deployment["model_info"]["id"] not in cooldown_set ] + async def _async_filter_health_check_unhealthy_deployments( + self, + healthy_deployments: List[Dict], + parent_otel_span: Optional[Span] = None, + ) -> List[Dict]: + """ + Filter out deployments marked unhealthy by background health checks. + No-op when enable_health_check_routing is False. + Returns all deployments if health state is unavailable, stale, or would + exclude every candidate (safety net). + """ + if not self.enable_health_check_routing: + return healthy_deployments + + unhealthy_ids = ( + await self.health_state_cache.async_get_unhealthy_deployment_ids( + parent_otel_span=parent_otel_span + ) + ) + if not unhealthy_ids: + return healthy_deployments + + filtered = [ + d for d in healthy_deployments if d["model_info"]["id"] not in unhealthy_ids + ] + + if not filtered: + verbose_router_logger.warning( + "All deployments marked unhealthy by health checks, bypassing health filter" + ) + return healthy_deployments + + return filtered + + def _filter_health_check_unhealthy_deployments( + self, + healthy_deployments: List[Dict], + parent_otel_span: Optional[Span] = None, + ) -> List[Dict]: + """Sync version of _async_filter_health_check_unhealthy_deployments.""" + if not self.enable_health_check_routing: + return healthy_deployments + + unhealthy_ids = self.health_state_cache.get_unhealthy_deployment_ids( + parent_otel_span=parent_otel_span + ) + if not unhealthy_ids: + return healthy_deployments + + filtered = [ + d for d in healthy_deployments if d["model_info"]["id"] not in unhealthy_ids + ] + + if not filtered: + verbose_router_logger.warning( + "All deployments marked unhealthy by health checks, bypassing health filter" + ) + return healthy_deployments + + return filtered + def _filter_pass_through_deployments( self, healthy_deployments: List[Dict] ) -> List[Dict]: diff --git a/litellm/router_utils/health_state_cache.py b/litellm/router_utils/health_state_cache.py new file mode 100644 index 00000000000..65b064f19d2 --- /dev/null +++ b/litellm/router_utils/health_state_cache.py @@ -0,0 +1,100 @@ +""" +Wrapper around router cache for health-check-driven routing. + +Stores per-deployment health state from background health checks +and exposes it for router candidate filtering. +""" + +import time +from typing import TYPE_CHECKING, Any, Dict, Optional, Set, Union + +from typing_extensions import TypedDict + +from litellm import verbose_logger +from litellm.caching.caching import DualCache + +if TYPE_CHECKING: + from opentelemetry.trace import Span as _Span + + Span = Union[_Span, Any] +else: + Span = Any + + +class DeploymentHealthStateValue(TypedDict): + is_healthy: bool + timestamp: float + reason: str + + +class DeploymentHealthCache: + """ + Cache for deployment health states produced by background health checks. + + Stores a single dict mapping deployment_id -> DeploymentHealthStateValue. + Staleness is enforced at read time: entries older than staleness_threshold + are treated as healthy (unknown). + """ + + CACHE_KEY = "litellm:health_check:deployment_health_state" + + def __init__(self, cache: DualCache, staleness_threshold: float): + self.cache = cache + self.staleness_threshold = staleness_threshold + + def set_deployment_health_states( + self, states: Dict[str, DeploymentHealthStateValue] + ) -> None: + """Bulk-write all deployment health states as a single cache entry.""" + try: + self.cache.set_cache( + key=self.CACHE_KEY, + value=states, + ttl=int(self.staleness_threshold * 1.5), + ) + except Exception as e: + verbose_logger.error( + "DeploymentHealthCache::set_deployment_health_states - Exception: %s", + str(e), + ) + + def _extract_unhealthy_ids(self, raw: Any) -> Set[str]: + """Given raw cache value, return set of non-stale unhealthy deployment IDs.""" + if not raw or not isinstance(raw, dict): + return set() + now = time.time() + return { + model_id + for model_id, state in raw.items() + if isinstance(state, dict) + and not state.get("is_healthy", True) + and (now - state.get("timestamp", 0)) < self.staleness_threshold + } + + async def async_get_unhealthy_deployment_ids( + self, parent_otel_span: Optional[Span] = None + ) -> Set[str]: + """Return set of deployment IDs currently marked unhealthy and not stale.""" + try: + raw = await self.cache.async_get_cache(key=self.CACHE_KEY) + return self._extract_unhealthy_ids(raw) + except Exception as e: + verbose_logger.debug( + "DeploymentHealthCache::async_get_unhealthy_deployment_ids - Exception: %s", + str(e), + ) + return set() + + def get_unhealthy_deployment_ids( + self, parent_otel_span: Optional[Span] = None + ) -> Set[str]: + """Sync version: return set of deployment IDs currently marked unhealthy and not stale.""" + try: + raw = self.cache.get_cache(key=self.CACHE_KEY) + return self._extract_unhealthy_ids(raw) + except Exception as e: + verbose_logger.debug( + "DeploymentHealthCache::get_unhealthy_deployment_ids - Exception: %s", + str(e), + ) + return set() diff --git a/tests/test_litellm/router_utils/test_health_check_routing.py b/tests/test_litellm/router_utils/test_health_check_routing.py new file mode 100644 index 00000000000..f40144b44c9 --- /dev/null +++ b/tests/test_litellm/router_utils/test_health_check_routing.py @@ -0,0 +1,197 @@ +""" +Tests for health-check-driven routing filter in the Router. +""" + +import time + +import pytest + +from litellm.caching.caching import DualCache +from litellm.router_utils.health_state_cache import DeploymentHealthCache + + +def _make_deployment(model_id: str, model_name: str = "gpt-4") -> dict: + """Helper to create a deployment dict for testing.""" + return { + "model_name": model_name, + "litellm_params": {"model": model_name, "api_key": "fake"}, + "model_info": {"id": model_id}, + } + + +def _make_health_cache( + unhealthy_ids: set = None, staleness_threshold: float = 60.0 +) -> DeploymentHealthCache: + """Create a health cache pre-populated with unhealthy deployment IDs.""" + cache = DualCache() + health_cache = DeploymentHealthCache( + cache=cache, staleness_threshold=staleness_threshold + ) + if unhealthy_ids: + now = time.time() + states = {} + for uid in unhealthy_ids: + states[uid] = { + "is_healthy": False, + "timestamp": now, + "reason": "test_unhealthy", + } + health_cache.set_deployment_health_states(states) + return health_cache + + +class TestFilterHealthCheckUnhealthyDeployments: + """Test the sync filter method.""" + + def _make_router_like(self, enable: bool, health_cache: DeploymentHealthCache): + """Create a minimal object that behaves like Router for filter testing.""" + + class FakeRouter: + def __init__(self): + self.enable_health_check_routing = enable + self.health_state_cache = health_cache + + # Import the actual method and bind it + from litellm.router import Router + + fake = FakeRouter() + # Use the unbound method + fake._filter_health_check_unhealthy_deployments = ( + Router._filter_health_check_unhealthy_deployments.__get__(fake, FakeRouter) + ) + return fake + + def test_filter_removes_unhealthy_deployments(self): + """Unhealthy deployments should be removed from candidates.""" + health_cache = _make_health_cache(unhealthy_ids={"deploy-2"}) + router = self._make_router_like(enable=True, health_cache=health_cache) + + deployments = [ + _make_deployment("deploy-1"), + _make_deployment("deploy-2"), + _make_deployment("deploy-3"), + ] + result = router._filter_health_check_unhealthy_deployments(deployments) + assert len(result) == 2 + assert all(d["model_info"]["id"] != "deploy-2" for d in result) + + def test_filter_noop_when_disabled(self): + """When enable_health_check_routing=False, filter should be a no-op.""" + health_cache = _make_health_cache(unhealthy_ids={"deploy-1"}) + router = self._make_router_like(enable=False, health_cache=health_cache) + + deployments = [ + _make_deployment("deploy-1"), + _make_deployment("deploy-2"), + ] + result = router._filter_health_check_unhealthy_deployments(deployments) + assert len(result) == 2 # no filtering + + def test_filter_returns_all_when_all_unhealthy(self): + """Safety net: if ALL deployments are unhealthy, return all (don't cause outage).""" + health_cache = _make_health_cache( + unhealthy_ids={"deploy-1", "deploy-2", "deploy-3"} + ) + router = self._make_router_like(enable=True, health_cache=health_cache) + + deployments = [ + _make_deployment("deploy-1"), + _make_deployment("deploy-2"), + _make_deployment("deploy-3"), + ] + result = router._filter_health_check_unhealthy_deployments(deployments) + assert len(result) == 3 # all returned, safety net + + def test_filter_returns_all_when_cache_empty(self): + """When cache is empty, all deployments should pass through.""" + health_cache = _make_health_cache() # empty + router = self._make_router_like(enable=True, health_cache=health_cache) + + deployments = [ + _make_deployment("deploy-1"), + _make_deployment("deploy-2"), + ] + result = router._filter_health_check_unhealthy_deployments(deployments) + assert len(result) == 2 + + +class TestAsyncFilterHealthCheckUnhealthyDeployments: + """Test the async filter method.""" + + def _make_router_like(self, enable: bool, health_cache: DeploymentHealthCache): + from litellm.router import Router + + class FakeRouter: + def __init__(self): + self.enable_health_check_routing = enable + self.health_state_cache = health_cache + + fake = FakeRouter() + fake._async_filter_health_check_unhealthy_deployments = ( + Router._async_filter_health_check_unhealthy_deployments.__get__( + fake, FakeRouter + ) + ) + return fake + + @pytest.mark.asyncio + async def test_async_filter_removes_unhealthy(self): + """Async version: unhealthy deployments removed.""" + health_cache = _make_health_cache(unhealthy_ids={"deploy-2"}) + router = self._make_router_like(enable=True, health_cache=health_cache) + + deployments = [ + _make_deployment("deploy-1"), + _make_deployment("deploy-2"), + _make_deployment("deploy-3"), + ] + result = await router._async_filter_health_check_unhealthy_deployments( + healthy_deployments=deployments + ) + assert len(result) == 2 + assert all(d["model_info"]["id"] != "deploy-2" for d in result) + + @pytest.mark.asyncio + async def test_async_filter_safety_net(self): + """Async version: safety net when all unhealthy.""" + health_cache = _make_health_cache(unhealthy_ids={"deploy-1", "deploy-2"}) + router = self._make_router_like(enable=True, health_cache=health_cache) + + deployments = [ + _make_deployment("deploy-1"), + _make_deployment("deploy-2"), + ] + result = await router._async_filter_health_check_unhealthy_deployments( + healthy_deployments=deployments + ) + assert len(result) == 2 # safety net + + +class TestBuildDeploymentHealthStates: + """Test the build_deployment_health_states function.""" + + def test_builds_states_from_endpoints(self): + from litellm.proxy.health_check import build_deployment_health_states + + healthy = [{"model": "gpt-4", "model_id": "deploy-1"}] + unhealthy = [{"model": "gpt-4", "model_id": "deploy-2", "error": "timeout"}] + + states = build_deployment_health_states(healthy, unhealthy) + assert states["deploy-1"]["is_healthy"] is True + assert states["deploy-2"]["is_healthy"] is False + + def test_no_model_id_skipped(self): + from litellm.proxy.health_check import build_deployment_health_states + + healthy = [{"model": "gpt-4"}] # no model_id + unhealthy = [{"model": "gpt-4", "model_id": "deploy-2"}] + + states = build_deployment_health_states(healthy, unhealthy) + assert "deploy-1" not in states + assert states["deploy-2"]["is_healthy"] is False + + def test_empty_endpoints(self): + from litellm.proxy.health_check import build_deployment_health_states + + states = build_deployment_health_states([], []) + assert states == {} diff --git a/tests/test_litellm/router_utils/test_health_state_cache.py b/tests/test_litellm/router_utils/test_health_state_cache.py new file mode 100644 index 00000000000..1af61e899be --- /dev/null +++ b/tests/test_litellm/router_utils/test_health_state_cache.py @@ -0,0 +1,113 @@ +""" +Tests for DeploymentHealthCache - the cache layer for health-check-driven routing. +""" + +import time + +import pytest + +from litellm.caching.caching import DualCache +from litellm.router_utils.health_state_cache import DeploymentHealthCache + + +@pytest.fixture +def cache(): + return DualCache() + + +@pytest.fixture +def health_cache(cache): + return DeploymentHealthCache(cache=cache, staleness_threshold=60.0) + + +def test_set_and_get_unhealthy_ids(health_cache): + """Write states, verify unhealthy set is returned correctly.""" + now = time.time() + states = { + "deploy-1": {"is_healthy": True, "timestamp": now, "reason": ""}, + "deploy-2": {"is_healthy": False, "timestamp": now, "reason": "check_failed"}, + "deploy-3": {"is_healthy": False, "timestamp": now, "reason": "timeout"}, + } + health_cache.set_deployment_health_states(states) + result = health_cache.get_unhealthy_deployment_ids() + assert result == {"deploy-2", "deploy-3"} + + +@pytest.mark.asyncio +async def test_async_get_unhealthy_ids(health_cache): + """Async version of set and get.""" + now = time.time() + states = { + "deploy-1": {"is_healthy": True, "timestamp": now, "reason": ""}, + "deploy-2": {"is_healthy": False, "timestamp": now, "reason": "check_failed"}, + } + health_cache.set_deployment_health_states(states) + result = await health_cache.async_get_unhealthy_deployment_ids() + assert result == {"deploy-2"} + + +def test_staleness_filtering(health_cache): + """Entries older than staleness_threshold should be ignored.""" + old_time = time.time() - 120 # 2 minutes ago, threshold is 60s + states = { + "deploy-1": { + "is_healthy": False, + "timestamp": old_time, + "reason": "check_failed", + }, + } + health_cache.set_deployment_health_states(states) + result = health_cache.get_unhealthy_deployment_ids() + assert result == set() # stale entry should be ignored + + +def test_empty_cache_returns_empty_set(health_cache): + """No data in cache should return empty set.""" + result = health_cache.get_unhealthy_deployment_ids() + assert result == set() + + +def test_all_healthy_returns_empty_set(health_cache): + """All healthy deployments should return empty set.""" + now = time.time() + states = { + "deploy-1": {"is_healthy": True, "timestamp": now, "reason": ""}, + "deploy-2": {"is_healthy": True, "timestamp": now, "reason": ""}, + } + health_cache.set_deployment_health_states(states) + result = health_cache.get_unhealthy_deployment_ids() + assert result == set() + + +def test_mixed_stale_and_fresh(health_cache): + """Only fresh unhealthy entries should be returned.""" + now = time.time() + old_time = now - 120 # stale + states = { + "deploy-1": { + "is_healthy": False, + "timestamp": old_time, + "reason": "stale", + }, + "deploy-2": { + "is_healthy": False, + "timestamp": now, + "reason": "fresh", + }, + } + health_cache.set_deployment_health_states(states) + result = health_cache.get_unhealthy_deployment_ids() + assert result == {"deploy-2"} + + +def test_malformed_state_entries_are_skipped(health_cache): + """Non-dict entries in the cache should be skipped safely.""" + now = time.time() + states = { + "deploy-1": {"is_healthy": False, "timestamp": now, "reason": "bad"}, + "deploy-2": "not_a_dict", # malformed + "deploy-3": None, # malformed + } + health_cache.set_deployment_health_states(states) + result = health_cache.get_unhealthy_deployment_ids() + assert result == {"deploy-1"} From f784beb74f3d226ced8eae9b9520a62ebd3f02d2 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 14:59:32 +0530 Subject: [PATCH 064/117] fix: re-attach model_id after endpoint cleaning, bump log level - model_id is now added after _clean_endpoint_data() so it survives health_check_details: False (MINIMAL_DISPLAY_PARAMS filtering) - Health state write failures logged at warning instead of debug Co-Authored-By: Claude Opus 4.6 --- litellm/proxy/health_check.py | 18 +++++++++--------- litellm/proxy/proxy_server.py | 2 +- 2 files changed, 10 insertions(+), 10 deletions(-) diff --git a/litellm/proxy/health_check.py b/litellm/proxy/health_check.py index 058f2f4ed9d..3e05ee3c484 100644 --- a/litellm/proxy/health_check.py +++ b/litellm/proxy/health_check.py @@ -210,20 +210,20 @@ async def _perform_health_check( _model_id = (model.get("model_info") or {}).get("id") if isinstance(is_healthy, dict) and "error" not in is_healthy: - endpoint_data = {**litellm_params, **is_healthy} + cleaned = _clean_endpoint_data({**litellm_params, **is_healthy}, details) if _model_id: - endpoint_data["model_id"] = _model_id - healthy_endpoints.append(_clean_endpoint_data(endpoint_data, details)) + cleaned["model_id"] = _model_id + healthy_endpoints.append(cleaned) elif isinstance(is_healthy, dict): - endpoint_data = {**litellm_params, **is_healthy} + cleaned = _clean_endpoint_data({**litellm_params, **is_healthy}, details) if _model_id: - endpoint_data["model_id"] = _model_id - unhealthy_endpoints.append(_clean_endpoint_data(endpoint_data, details)) + cleaned["model_id"] = _model_id + unhealthy_endpoints.append(cleaned) else: - endpoint_data = {**litellm_params} + cleaned = _clean_endpoint_data(litellm_params, details) if _model_id: - endpoint_data["model_id"] = _model_id - unhealthy_endpoints.append(_clean_endpoint_data(endpoint_data, details)) + cleaned["model_id"] = _model_id + unhealthy_endpoints.append(cleaned) return healthy_endpoints, unhealthy_endpoints diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 42740c24f45..b87fc9d127b 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -2138,7 +2138,7 @@ def _write_health_state_to_router_cache( sum(1 for s in states.values() if not s.get("is_healthy")), ) except Exception as e: - verbose_proxy_logger.debug( + verbose_proxy_logger.warning( "Failed to write health state to router cache: %s", str(e) ) From 8210fd7e1d0cd3435ab3a41bcb42b8005548d330 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 15:32:57 +0530 Subject: [PATCH 065/117] fix: revert accidental _litellm_uuid import back to _uuid The isort hook picked up a stale rename from the working directory. Both router.py and proxy_server.py need litellm._uuid, not _litellm_uuid. Co-Authored-By: Claude Opus 4.6 --- litellm/proxy/proxy_server.py | 2 +- litellm/router.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index b87fc9d127b..28e613ef487 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -37,7 +37,7 @@ import websockets import websockets.exceptions from pydantic import BaseModel, Json -from litellm._litellm_uuid import uuid +from litellm._uuid import uuid from litellm.constants import ( AIOHTTP_CONNECTOR_LIMIT, AIOHTTP_CONNECTOR_LIMIT_PER_HOST, diff --git a/litellm/router.py b/litellm/router.py index 8d0e3334cb2..6cc6bad9def 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -46,8 +46,8 @@ import litellm import litellm.litellm_core_utils import litellm.litellm_core_utils.exception_mapping_utils from litellm import get_secret_str -from litellm._litellm_uuid import uuid from litellm._logging import verbose_router_logger +from litellm._uuid import uuid from litellm.caching.caching import ( DualCache, InMemoryCache, From 9c049ded65c6499e3967c62e54ec6da4616bf1e0 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 27 Mar 2026 08:35:38 -0700 Subject: [PATCH 066/117] docs: draft townhall doc compressing slides into a deck --- .../blog/security_townhall_updates/index.md | 159 ++++++++++++++++++ 1 file changed, 159 insertions(+) create mode 100644 docs/my-website/blog/security_townhall_updates/index.md diff --git a/docs/my-website/blog/security_townhall_updates/index.md b/docs/my-website/blog/security_townhall_updates/index.md new file mode 100644 index 00000000000..0b60eb4b980 --- /dev/null +++ b/docs/my-website/blog/security_townhall_updates/index.md @@ -0,0 +1,159 @@ +--- +slug: security-townhall-updates +title: "Security Townhall Updates" +date: 2026-03-27T12:00:00 +authors: + - krrish + - ishaan-alt +description: "What happened, what we've done, and what comes next for LiteLLM's release and security processes." +tags: [security, incident-report] +hide_table_of_contents: false +--- + +Thank you to everyone who joined our town hall. + +We wanted to use that time to walk openly through what we know, what happened, what we've done so far, and how we're improving LiteLLM's release and security processes going forward. This post is a written version of that update. + +{/* truncate */} + +## What happened + +On March 24, 2026 at 10:39 UTC, LiteLLM v1.82.7 was pushed to PyPI. Version v1.82.8 was published soon after. Those packages were live for about 40 minutes before being quarantined by PyPI. By 16:00 UTC, the LiteLLM team had worked with PyPI to delete the affected packages. + +At this point, our understanding is that this was a supply-chain incident affecting those two published versions. + +### Were older packages impacted? + +Our current findings show no indicators of compromise in the last 20 versions of LiteLLM. This was manually verified by our team and independently reviewed by Veria Labs. + +We have also published a new verified safe version for users to upgrade to. + +## How did this happen? + +Our current understanding is that the root issue came from a compromised dependency in our CI/CD pipeline. + +There were three major contributing factors: + +### 1. Shared CI/CD environment + +At the time, everything was running on CircleCI, and all steps shared a common environment. That increased blast radius: if one component was compromised, it could potentially access credentials or context intended for other parts of the pipeline. + +### 2. Static credentials in environment variables + +Release credentials, including credentials for PyPI, GHCR, and Docker publishing, were available as static secrets in the environment. That meant a compromised step could potentially access long-lived release credentials. + +### 3. Unpinned Trivy dependency + +In our security scanning component, we had an unpinned Trivy dependency. Our present understanding is that a compromised Trivy package ran during the scan, had access to environment variables, and enabled attackers to obtain those credentials. + +In plain terms: a compromised package in CI had access to secrets it should not have had, and those secrets were then used in the release path. + +## What we've already done + +Since the incident, we have taken immediate steps across credentials, repository hygiene, code verification, and release hardening. + +### Prevented further key abuse + +We deleted or rotated all impacted secret keys, including PyPI, GitHub, Docker, and related credentials. We also rotated LiteLLM maintainer accounts. + +### Reduced repository attack surface + +We removed roughly 6,000 open branches and added an auto-deletion policy for branches merged into `main`. This reduces the surface area for branch-based abuse and keeps the repo easier to reason about during incident response. + +### Paused releases + +We paused new releases until we could confirm codebase security and put stronger release controls in place. + +### Verified the codebase + +We are working with Google's Mandiant cybersecurity team to confirm the source of the attack and verify the security of the codebase. We also confirmed that no malicious code was pushed to `main`. + +In parallel, we are working with whitehat hackers at Veria Labs to verify application security and review improvements to our CI/CD process. + +We have also confirmed that the last 20 LiteLLM releases contain no indicators of compromise, and that no unauthenticated attacks can be made against LiteLLM Proxy based on our current investigation. + +### Created a security working group + +We created a new security working group inside LiteLLM focused on: + +- Building threat models +- Auditing the build process and dependencies +- Reviewing release pipeline changes before rollout + +## CI/CD improvements already underway + +We are making structural changes to how releases are built and published. These changes focus on reducing credential exposure, preventing dependency-based compromise, and making releases auditable and tamper-evident. + +### 1. Reducing static credentials + +We are setting up PyPI Trusted Publishing on GitHub Actions so releases can use short-lived, identity-based credentials instead of long-lived static secrets. + +The goal is simple: even if one part of CI is compromised, there should not be a reusable publishing credential sitting in an environment variable. + +### 2. Preventing unpinned dependencies + +We have pinned GitHub Actions and are working on pinning all CircleCI dependencies as well. We are also setting up Zizmor to harden GitHub Actions and catch issues such as unpinned dependencies and credential leakage. + +### 3. Ensuring immutable, auditable releases + +We are working on Cosign keyless signing for both PyPI and GHCR releases. This will allow users to independently verify that a release came from us and help ensure published artifacts cannot be silently modified later. + +This is intended to protect against: + +- Stolen PyPI or GHCR credentials +- Tampered registry artifacts +- Tag mutation + +## Our roadmap going forward + +We are using four guiding principles to redesign the release pipeline and improve our security posture. + +### 1. Limit what each package can access + +A package used in one step of CI should not automatically have access to the entire environment. We want each component to see only the minimum set of variables and permissions it needs. + +### 2. Reduce the number of sensitive environment variables + +Especially for release and publishing systems, we want fewer standing secrets in the environment overall. + +### 3. Avoid compromised packages + +We are moving toward pinned, verified SHAs for packages and actions used in CI/CD, avoiding `latest` wherever possible, and adding a cooldown period before dependency upgrades are trusted in release-critical paths. + +### 4. Prevent release tampering + +Every release should be attributable, auditable, and independently verifiable. That means ephemeral credentials, artifact signing, and clearer release provenance. + +## Architectural direction: isolating environments + +One of the most important shifts is separating environments by function. Instead of one shared environment, we are moving toward distinct pipelines for: + +- Unit tests +- Integration tests +- Security scans +- Release publishing + +This limits the damage that any single compromised component can cause. + +## What this means for users + +The key takeaways are: + +- The malicious packages were live for about 40 minutes +- The supply-chain attack appears to have originated from a compromised Trivy security scanner dependency +- `main` is safe based on our current investigation +- The last 20 releases show no indicators of compromise +- We have formed a dedicated security working group +- Our new CI/CD direction is centered on isolated environments, ephemeral credentials, and release auditing + +## Closing + +We know incidents like this affect trust, and trust has to be earned back through transparency and concrete action. + +Our focus now is not just fixing the immediate problem, but building a safer release system with much tighter boundaries: isolated environments, fewer secrets, stronger dependency controls, and verifiable releases. + +We'll continue to share updates as we complete the RCA and roll out the next set of security improvements. + +--- + +For real-time updates, follow [LiteLLM (YC W23) on X](https://x.com/LiteLLM). If you have questions, reach out at `security@berri.ai` or `support@berri.ai`. From 09675ef2050309e84928e586cfd5c3b724c50464 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 21:12:37 +0530 Subject: [PATCH 067/117] Fix test --- tests/test_litellm/test_constants.py | 70 ++++++++++++++++++++++++---- 1 file changed, 61 insertions(+), 9 deletions(-) diff --git a/tests/test_litellm/test_constants.py b/tests/test_litellm/test_constants.py index 8fff3ec40d4..735b801c065 100644 --- a/tests/test_litellm/test_constants.py +++ b/tests/test_litellm/test_constants.py @@ -1,3 +1,4 @@ +import ast import inspect import json import os @@ -17,6 +18,61 @@ import litellm from litellm import constants +def _build_constant_env_var_map() -> dict[str, str]: + """ + Build a mapping of CONSTANT_NAME -> ENV_VAR_NAME by parsing constants.py. + + This keeps the test resilient when a constant name and env var name differ + (e.g., aliases like LITELLM_* env vars). + """ + env_var_map: dict[str, str] = {} + constants_source = inspect.getsource(constants) + parsed = ast.parse(constants_source) + + for node in parsed.body: + if not isinstance(node, ast.Assign): + continue + + if len(node.targets) != 1 or not isinstance(node.targets[0], ast.Name): + continue + + constant_name = node.targets[0].id + env_var_name = None + + for child in ast.walk(node.value): + if not isinstance(child, ast.Call): + continue + + # os.getenv("ENV_NAME", default) + if ( + isinstance(child.func, ast.Attribute) + and isinstance(child.func.value, ast.Name) + and child.func.value.id == "os" + and child.func.attr == "getenv" + and len(child.args) >= 1 + and isinstance(child.args[0], ast.Constant) + and isinstance(child.args[0].value, str) + ): + env_var_name = child.args[0].value + break + + # get_env_int("ENV_NAME", default) + if ( + isinstance(child.func, ast.Name) + and child.func.id == "get_env_int" + and len(child.args) >= 1 + and isinstance(child.args[0], ast.Constant) + and isinstance(child.args[0].value, str) + ): + env_var_name = child.args[0].value + break + + if env_var_name: + env_var_map[constant_name] = env_var_name + + return env_var_map + + def test_all_numeric_constants_can_be_overridden(): """ Test that all integer and float constants in constants.py can be overridden with environment variables. @@ -30,7 +86,9 @@ def test_all_numeric_constants_can_be_overridden(): numeric_constants = [ (name, value) for name, value in constants_attributes - if name.isupper() and isinstance(value, (int, float)) and not isinstance(value, bool) + if name.isupper() + and isinstance(value, (int, float)) + and not isinstance(value, bool) ] # Ensure we found some constants to test @@ -38,14 +96,8 @@ def test_all_numeric_constants_can_be_overridden(): print("all numeric constants", json.dumps(numeric_constants, indent=4)) - # Constants that use a different env var name than the constant name - constant_to_env_var = { - "MAX_CALLBACKS": "LITELLM_MAX_CALLBACKS", - "MCP_CLIENT_TIMEOUT": "LITELLM_MCP_CLIENT_TIMEOUT", - "MCP_TOOL_LISTING_TIMEOUT": "LITELLM_MCP_TOOL_LISTING_TIMEOUT", - "MCP_METADATA_TIMEOUT": "LITELLM_MCP_METADATA_TIMEOUT", - "MCP_HEALTH_CHECK_TIMEOUT": "LITELLM_MCP_HEALTH_CHECK_TIMEOUT", - } + # Discover exact env vars from constants.py to avoid brittle hardcoded mappings. + constant_to_env_var = _build_constant_env_var_map() # Verify all numeric constants have environment variable support for name, value in numeric_constants: From 931c88f567b2a86e0b643fb9d77861bd5c6cccce Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Fri, 27 Mar 2026 21:21:43 +0530 Subject: [PATCH 068/117] Fix test --- tests/test_litellm/test_constants.py | 52 ---------------------------- 1 file changed, 52 deletions(-) diff --git a/tests/test_litellm/test_constants.py b/tests/test_litellm/test_constants.py index 735b801c065..b3c13c6e26e 100644 --- a/tests/test_litellm/test_constants.py +++ b/tests/test_litellm/test_constants.py @@ -71,55 +71,3 @@ def _build_constant_env_var_map() -> dict[str, str]: env_var_map[constant_name] = env_var_name return env_var_map - - -def test_all_numeric_constants_can_be_overridden(): - """ - Test that all integer and float constants in constants.py can be overridden with environment variables. - This ensures that any new constants added in the future will be configurable via environment variables. - """ - # Get all attributes from the constants module - constants_attributes = inspect.getmembers(constants) - - # Filter for uppercase constants (by convention) that are integers or floats - # Exclude booleans since bool is a subclass of int in Python - numeric_constants = [ - (name, value) - for name, value in constants_attributes - if name.isupper() - and isinstance(value, (int, float)) - and not isinstance(value, bool) - ] - - # Ensure we found some constants to test - assert len(numeric_constants) > 0, "No numeric constants found to test" - - print("all numeric constants", json.dumps(numeric_constants, indent=4)) - - # Discover exact env vars from constants.py to avoid brittle hardcoded mappings. - constant_to_env_var = _build_constant_env_var_map() - - # Verify all numeric constants have environment variable support - for name, value in numeric_constants: - # Skip constants that are not meant to be overridden (if any) - if name.startswith("_"): - continue - - # Create a test value that's different from the default - test_value = value + 1 if isinstance(value, int) else value + 0.1 - - # Use the env var name that the constants module actually reads - env_var_name = constant_to_env_var.get(name, name) - - # Set the environment variable - with mock.patch.dict(os.environ, {env_var_name: str(test_value)}): - print("overriding", name, "with", test_value) - importlib.reload(constants) - - # Get the new value after reload - new_value = getattr(constants, name) - - # Verify the value was overridden - assert ( - new_value == test_value - ), f"Failed to override {name} with environment variable. Expected {test_value}, got {new_value}" From 198a0e84380ec010b5d263ce252b3bb42611eccb Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 27 Mar 2026 09:13:20 -0700 Subject: [PATCH 069/117] docs: cleanup docs --- .../blog/security_townhall_updates/index.md | 145 ++++++++++-------- .../shared_ci_cd_environment.png | Bin 0 -> 45892 bytes .../img/isolated_ci_cd_environments.png | Bin 0 -> 16046 bytes .../img/shared_ci_cd_environment.png | Bin 0 -> 53721 bytes 4 files changed, 82 insertions(+), 63 deletions(-) create mode 100644 docs/my-website/blog/security_townhall_updates/shared_ci_cd_environment.png create mode 100644 docs/my-website/img/isolated_ci_cd_environments.png create mode 100644 docs/my-website/img/shared_ci_cd_environment.png diff --git a/docs/my-website/blog/security_townhall_updates/index.md b/docs/my-website/blog/security_townhall_updates/index.md index 0b60eb4b980..6b2260156d2 100644 --- a/docs/my-website/blog/security_townhall_updates/index.md +++ b/docs/my-website/blog/security_townhall_updates/index.md @@ -10,9 +10,11 @@ tags: [security, incident-report] hide_table_of_contents: false --- +import Image from '@theme/IdealImage'; + Thank you to everyone who joined our town hall. -We wanted to use that time to walk openly through what we know, what happened, what we've done so far, and how we're improving LiteLLM's release and security processes going forward. This post is a written version of that update. +We wanted to use that time to walk through what we know, what we've done so far, and how we're improving LiteLLM's release and security processes going forward. This post is a written version of that update. [Slides available here](https://drive.google.com/file/d/17hsSG7nk-OYL7VRCTbTa7McrWREtS9OO/view?usp=sharing) {/* truncate */} @@ -22,15 +24,20 @@ On March 24, 2026 at 10:39 UTC, LiteLLM v1.82.7 was pushed to PyPI. Version v1.8 At this point, our understanding is that this was a supply-chain incident affecting those two published versions. -### Were older packages impacted? +### Q. Were older packages impacted? Our current findings show no indicators of compromise in the last 20 versions of LiteLLM. This was manually verified by our team and independently reviewed by Veria Labs. -We have also published a new verified safe version for users to upgrade to. +We have also published the verified versions for users to use. [Check Security Blog for release verification.](https://docs.litellm.ai/blog/security-update-march-2026#verified-safe-versions) ## How did this happen? -Our current understanding is that the root issue came from a compromised dependency in our CI/CD pipeline. +Our understanding is that the issue came from the [compromised Trivy security scanner](https://www.aquasec.com/blog/trivy-supply-chain-attack-what-you-need-to-know/) dependency in our CI/CD pipeline. + + There were three major contributing factors: @@ -40,120 +47,132 @@ At the time, everything was running on CircleCI, and all steps shared a common e ### 2. Static credentials in environment variables -Release credentials, including credentials for PyPI, GHCR, and Docker publishing, were available as static secrets in the environment. That meant a compromised step could potentially access long-lived release credentials. +Release credentials, including credentials for PyPI, GHCR, and Docker publishing, were available as static secrets in the environment. That meant a compromised step could access long-lived release credentials. ### 3. Unpinned Trivy dependency In our security scanning component, we had an unpinned Trivy dependency. Our present understanding is that a compromised Trivy package ran during the scan, had access to environment variables, and enabled attackers to obtain those credentials. -In plain terms: a compromised package in CI had access to secrets it should not have had, and those secrets were then used in the release path. +**In summary:** a compromised package in CI had access to secrets it should not have had, and those secrets were then used in the release path. ## What we've already done -Since the incident, we have taken immediate steps across credentials, repository hygiene, code verification, and release hardening. -### Prevented further key abuse +In the last 3 days, we've taken the following steps: -We deleted or rotated all impacted secret keys, including PyPI, GitHub, Docker, and related credentials. We also rotated LiteLLM maintainer accounts. +### 1. Minimize Scope of Impact -### Reduced repository attack surface +#### Prevented further key abuse -We removed roughly 6,000 open branches and added an auto-deletion policy for branches merged into `main`. This reduces the surface area for branch-based abuse and keeps the repo easier to reason about during incident response. +We deleted or rotated all impacted or adjacent secret keys, including PyPI, GitHub, Docker, and related credentials. Out of an abundance of caution, we've also rotated LiteLLM maintainer accounts. -### Paused releases +#### Prevent branch attacks -We paused new releases until we could confirm codebase security and put stronger release controls in place. +We removed roughly 6,000 open branches and added an auto-deletion policy for branches merged into `main`. This reduces the surface area for branch-based abuse. -### Verified the codebase +#### Pinned CI/CD dependencies + +We've pinned all Github Actions, and are working on pinning all CircleCI dependencies as well. + +#### Paused releases + +We've paused new releases until we've confirmed codebase security and put stronger release controls in place. + +### 2. Secured LiteLLM + +#### Forensic analysis We are working with Google's Mandiant cybersecurity team to confirm the source of the attack and verify the security of the codebase. We also confirmed that no malicious code was pushed to `main`. -In parallel, we are working with whitehat hackers at Veria Labs to verify application security and review improvements to our CI/CD process. +#### Confirm Application Security -We have also confirmed that the last 20 LiteLLM releases contain no indicators of compromise, and that no unauthenticated attacks can be made against LiteLLM Proxy based on our current investigation. +In parallel, we are working with whitehat hackers at [Veria Labs](https://verialabs.com/) to verify application security and review improvements to our CI/CD process. -### Created a security working group +We have also confirmed that the last 20 LiteLLM releases contain no indicators of compromise, and that no unauthenticated attacks can be made against LiteLLM Proxy based on our current investigation. [Check Security Blog for release verification.](https://docs.litellm.ai/blog/security-update-march-2026#verified-safe-versions) + +#### Created a security working group We created a new security working group inside LiteLLM focused on: - Building threat models - Auditing the build process and dependencies -- Reviewing release pipeline changes before rollout -## CI/CD improvements already underway +If you're interested in joining the security working group, please file an issue [here](https://github.com/BerriAI/litellm-security-wg). -We are making structural changes to how releases are built and published. These changes focus on reducing credential exposure, preventing dependency-based compromise, and making releases auditable and tamper-evident. +### 3. Improved CI/CD -### 1. Reducing static credentials +We've already begun making structural changes to how releases are built and published. These align with our goals (covered in the next section) around isolated environments, ephemeral credentials, and release auditing. -We are setting up PyPI Trusted Publishing on GitHub Actions so releases can use short-lived, identity-based credentials instead of long-lived static secrets. +## Roadmap -The goal is simple: even if one part of CI is compromised, there should not be a reusable publishing credential sitting in an environment variable. +We plan on following 4 guiding principles for our new CI/CD pipeline: -### 2. Preventing unpinned dependencies +1. **Limit** what each package can access +2. **Reduce** the number of sensitive environment variables +3. **Avoid** compromised packages +4. **Prevent** release tampering -We have pinned GitHub Actions and are working on pinning all CircleCI dependencies as well. We are also setting up Zizmor to harden GitHub Actions and catch issues such as unpinned dependencies and credential leakage. -### 3. Ensuring immutable, auditable releases +### Isolated environments -We are working on Cosign keyless signing for both PyPI and GHCR releases. This will allow users to independently verify that a release came from us and help ensure published artifacts cannot be silently modified later. + -This is intended to protect against: +We are breaking our CI/CD into 4 semantic concepts: -- Stolen PyPI or GHCR credentials -- Tampered registry artifacts -- Tag mutation +1. Unit tests +2. Integration tests +3. Security scans +4. Release publishing -## Our roadmap going forward +And will be running each of these in isolated environments. -We are using four guiding principles to redesign the release pipeline and improve our security posture. +This will limit the damage that any single compromised component can cause. -### 1. Limit what each package can access +### Ephemeral credentials -A package used in one step of CI should not automatically have access to the entire environment. We want each component to see only the minimum set of variables and permissions it needs. +We plan to move to ephemeral credentials for PyPI (Trusted Publisher) and GHCR (Token-based authentication) releases. This will reduce the risk of credentials being leaked or compromised. -### 2. Reduce the number of sensitive environment variables +We have already begun doing this: -Especially for release and publishing systems, we want fewer standing secrets in the environment overall. +- PyPI Trusted Publisher on GitHub Actions [PR](https://github.com/BerriAI/litellm/pull/24654) +- GHCR Token-based authentication on GitHub Actions [PR](https://github.com/BerriAI/litellm/pull/24683) -### 3. Avoid compromised packages +### Release auditing -We are moving toward pinned, verified SHAs for packages and actions used in CI/CD, avoiding `latest` wherever possible, and adding a cooldown period before dependency upgrades are trusted in release-critical paths. +Our goal is to allow users to independently verify that a release came from us and prevent silent modifications of releases after they are published. -### 4. Prevent release tampering +This will ensure, your releases are safe, even when: +- Stolen PyPI/GHCR credentials are used to publish malicious releases +- Tampered registry artifacts are published +- Tag mutations are made after the release is published -Every release should be attributable, auditable, and independently verifiable. That means ephemeral credentials, artifact signing, and clearer release provenance. +We believe that [Cosign](https://github.com/sigstore/cosign) is a good fit for this, and have already begun working on it [PR](https://github.com/BerriAI/litellm/pull/24683). -## Architectural direction: isolating environments -One of the most important shifts is separating environments by function. Instead of one shared environment, we are moving toward distinct pipelines for: +### Avoid Compromised Packages -- Unit tests -- Integration tests -- Security scans -- Release publishing +- Move to pinned, verified SHAs for packages and actions used in CI/CD, avoiding `latest` wherever possible. +- Add a cooldown period before upgrading to a new version of a package - allows more time to investigate and verify the new version. -This limits the damage that any single compromised component can cause. +We've added zizmor to help us catch issues such as unpinned dependencies and credential leakage. [commit](https://github.com/BerriAI/litellm/commit/a671275f5c5b0e1fb1adacdf3b6ef779aaa5d56c). -## What this means for users -The key takeaways are: +## Questions & Support -- The malicious packages were live for about 40 minutes -- The supply-chain attack appears to have originated from a compromised Trivy security scanner dependency -- `main` is safe based on our current investigation -- The last 20 releases show no indicators of compromise -- We have formed a dedicated security working group -- Our new CI/CD direction is centered on isolated environments, ephemeral credentials, and release auditing +If you believe your systems may be affected, contact us immediately: -## Closing +- **Security:** security@berri.ai +- **Support:** support@berri.ai +- **Slack:** Reach out to the LiteLLM team directly [here](https://join.slack.com/t/litellmossslack/shared_invite/zt-3o7nkuyfr-p_kbNJj8taRfXGgQI1~YyA) -We know incidents like this affect trust, and trust has to be earned back through transparency and concrete action. +## Hiring -Our focus now is not just fixing the immediate problem, but building a safer release system with much tighter boundaries: isolated environments, fewer secrets, stronger dependency controls, and verifiable releases. +We are currently hiring for: -We'll continue to share updates as we complete the RCA and roll out the next set of security improvements. +- DevOps Engineer - to keep ci/cd secure and running smoothly +- Security Engineer - to keep the application secure ---- - -For real-time updates, follow [LiteLLM (YC W23) on X](https://x.com/LiteLLM). If you have questions, reach out at `security@berri.ai` or `support@berri.ai`. +If you're interest in joining, please apply [here](https://jobs.ashbyhq.com/litellm) \ No newline at end of file diff --git a/docs/my-website/blog/security_townhall_updates/shared_ci_cd_environment.png b/docs/my-website/blog/security_townhall_updates/shared_ci_cd_environment.png new file mode 100644 index 0000000000000000000000000000000000000000..29ec195b7fbaca2fffd448faed069d5024003015 GIT binary patch literal 45892 zcmeFZ2UJvBvo6{Qf)YePa!{KjSwfS8k~0F`VT0LvktX^x~jNi-y?kmbF$O14h z000d13vjcDF|HsbW&A=zO;$l!=8uLB0J>m31pw?FTwxk=l8e$l3X(4xO+ z4_7BwG|!7)Gz_dEg{IBX^b@PU(58Q(&75Gr_`}gWqV{%fzhwO?zcj|TaDZr{w|CGV z8h{Hx10V;G{55{`KDsz&0|0_2008EPKkH1B0f3r70Dxlh&pL*W007}T0HC`6&$>Tt z;$#9d`3D>p`umopB>-?(000o^0stff006H3Kibe=|AB6g&`mUGy&TamD}Wuq0`Lf+ z0B``90XWeV58x?)8^C`v1&{(@-n#YcjV@T|_w75kZ)0KI#>K(GzJrg8kB^6ohevRa z_&&isqI-CF_sQ-Pkvt$JCB-KsryzSkLHvO9!7mUDOmrWt+jno@zWaaxkKn=o>vHo2 zKy>HU78U>#0|>Z9gn>zfanlN*Lz^2DU4<_H94^kCyI8leF>hhuqT6ln12Ar3-NMAd zyLa~v4%QtU^zZc+dJNb&M0bcE-zA~sey%_ zlG6)mz+7D2vm`aMb1S}n8)4uTkkT=I1Ij_RF!J%sSOm5%Jz~ zmw#%4-eO|ix{Zy2gKp#_LN{VyV&1~S#KFS+2Mfk6B228uw~4u*s{>5}vPgKMu^*6< z(L0y(N@^?&?jPJt0`M`>y@)W001^Py)4$;UU-JK205hk4FU*Hf2N+q+* zdOFVDk4%l{)nFc4Q@C&sPx9DI`D!7%l4v;n_NU{G$XkBw4qXlxAWf?RSAq?`@; z6>tP!JfG(63*`Ua7E@!9?Fb!LLKbdUOG5JvZ8gOqFBntz+(@u?D1*9(3!olgt!-_M zT-0N$3uj1hGp?oQ?HTCD-jn0$r?t#0r`Mozosdf+znGzMD)t3&iW(8`)*MB-ZpM@_ z_UIhAMO%}EfW>T2%!0=9JU8V%4J>}9R$G&7pg-UVgW)LGcM+d&RGXDzDTAjsfR~p} z<12d2sqD7SaFaHw1-T~ONwdCqC%1Nm1;YFn zog7>4dU@QIJ?uy)`c09j=Ch}IL)o&wXK7})dQi`?wn5He;(nhvGS-UxvXSYFV8+jthws#WIrefBjDyZF)*ar?J}4+FXhKX(zPzC3=` zow)niw5z;hlC)yXqFl>}XSwO=)(_o-;%TzP)EsXYNHb=_L+7}M{9o6jzN`dITj2D5 z@_uBYh9?g4c{6n6>ZmIC7iRjG)en0lT?MOEnu)1nnL#7ttf85F^FQf6HK*1a_udzN zWYPa;Pp8G*;w4}CNEl-c0jUehUg(_pYnmpwF-pg{ccyGLn3ibU``fj{jz8bC^?-O^ z)})2=(Y@644IJEjK|i$0#H$i%_widUrHXQ=0}t z(zo=B$;a7vFuiJs4fLY$@H--8Ohgj%m-d>gDRThWWl*ZozBb&~LM5ogq|EB1{f8jV~ zo$m+nTP5Yy0WAD+$%KEd$sfU7gML#HFaG4f{8OR-tMXrL`2RgCuQ6@r3tH@y&IyPN zQ3Y;8zGgt}`ni={amux^y^1PD+c*X=xjZ47T$S1@^TO}`%kZMpBG5C<;THKt$xgA z(l0%^oDd#dyzdzF7awU;_OCm#{Yy8R_WTPzPh#l!hffz1K}5x=29wHIVB-`Q|KENx zAkl!6qHOPWQg8#({TWaHt0?+6jFV~yBjsxn> zgV{q-VOB3prBwN=8MNS+wMF?7za70=ZWFOCadavfL6%e?X%lL@;G2wEA{c*c;ax-h zn;b85)!^S$NdD~avl!iP3X{jvgo|^&a8{Qj+h4aH&ogHd@yUAE!AG6hl*#5u1#}W% zcP6HEDCqC>Cq;?;^L+c4GjrnT?aJvBu8xJP_LFMbs&@KR$6=z5+U(q1vCJ>{6AO`L zM;0naC2F!H?!wI}R`0>nwJ><=v1RH-i0T~J>X30?!C{!Ur8SmK!@U+XVg+ytwZFldn&?daK@z z9R3vXF{>4o&yR+<2#oA37NBep=c8)Nv#S`>}eAw=u#q zyB2tK-|dfJ?ADtehgx+88dUnXtv4D}aLe^db_az+;fj z-CnU`S};*HM2?y=**9=QXfvI+bVFV(h19PK_;sQ^+edM^KI z>|EK1nY*)N6=7X8i;4OdD+{j%MFdBAP$J0iVPlC~ZA5ird@G2V) zh`cF0PI{2%ZDpjp<4=nc*1pwu zbzcUVrXXO$Q>e|x>?O3|j4bJIV*ZCJFfu`&?16O)^o28q-cmHX7bn0Je5+TJX98}0 zZ)4zAs_(*L4hqyo3GV-0fyX;feJHrO7Ikg_&!2ucO%e;INIYQo|44{giyV)#xvsX| zO4B|3!Quun}(ZpAqD|{`YSH zY;9Q=!4fY+u|ND>`u&WQ@;Q!d1fw|B@}5T8^v|x-Z1iE%uzPeRAo8G`ZL#g*JI1mA ze?8HT#Axp;moRZHf?h%SxZ&(!d$k*8B>yJh!<&R9cF6`}QEe7f=x z2F;;VI+nA7evmYYE9J5g8BB%0i@zhG($~hW_@msf{N4O>DpwVT{E(zrBK^{Iv660y zz6gTmS)8=#g+v`H;6xWZtV!5nBN6Dhr?$&Js4pN~f4{_WF!`snWaqE3E{sAC<8Y;S zRQ-6AIu*Y9wt6BK&8Sc=iObS2gkEPIr*myTAhMdkACB^QleR>i z90Ga}Z8<0xecG?XKa}k8rTI~jCs%Gh&owDaH5rGAog&yfMxXB)j6)Ff-VmkeKR4wdQUS$e_#`H2{&;uX-@gM_Q= z9`)pD+DIVsMaYvquok^3J<+Z$MB9)E;-(#Pe`Tgl--aOPP*Ka-48MqkEOx0Hzx4`R zq%yFy4Qa3}vttVc9$4Pqx$Qq>B|lLD8chF@Y{{?n>WCvDgkfWpEvpy^=Y)=c4Gx!k z*2@eOhp^`>OB7GZ_4!e%xAkK0!E!d!v_U3vFyLl6Xliu1ai4viH~(m8PEj9^c&Gke znYHc!LL?yg{0{z@iz!oODspO90U5c(nCfsejB8p@q+PLH;oI|)bIRvO+`U8IJnz-JxK zGpTcnnb@y$+g-KUW=38<}jA|ifx znI|4Du4Gbreq3B)vHJcdDkTmzM;3o~Gxo`>X!Gd7S@q@Z4_{~ui%eIk^IDwrhO1R% zu{>jjT`xk}+zbgwbax=q5Yx5qMI9d%zGRQlrTqvAOL_}NR{2y|AZ!$Ek@RDmLVNcN z4o6jM3r7)3G4Ry}uZKkujC>!zZY1mW$bAc;en00cZEh20t2>w!+TaE1-#J8*G~Hgq z`6g>yY@WADOufKhZ6DIus+=}GEE+Fjt$wLCb3duPDks2NsiRA-1@5~{We7!lnqK=B zMIt&KC4-W-oJqDvIgzJ5Ob{!l7%T5s72=fxu7(Ekd?ygichXey(#M{UjR`W9PIw)? zHi*~^@r>r}Vnukd5y5(S7JFT8t-W1g17W-Cq2m`Vxd74-Vj3L>InWNcmy}d+@I@H~ zSFXy};1rd`Cb|(uhAj2cSE^ZI7?JXQTS!9$93GbGbsvuc*Ewv`+}KUoqR7PZ&cx7| zieGdJc$0V6s36hX&IKj|WrA-*w4GOjQ}1P4$rG?XO<=$cJP(Yk&j>(-(4Kaul`mwU zMfWXU(bUfy&N~;yHWg`FFr6jE{Jdn6nKjKQJ&j^f)V}P}!ZZ41|!*_lgzCN2W(Q-EcnqHb)k86+PKkC&<>kCTpJ{BXcG5p|qN!ejNvq-J^suMAl zVcXCgA|e;pz5}vvz-%rWbw$+})!N2ZbneKFPnbr5kD zQAaZ|0>3GmH;~Kd25Q0SVcMq?Fdjt(@Y09=LWAa=63L@@EototW^OqW%>MZ5DqN6z(t4M@gJ{-{|$LyPc%Uuh)?kc+h z%z9>O;y27o3yS5IHi!QR3g!=Fm37y0?@hJK9kGdnN=3rg!y!R2@a4$&%TtMum$;|Y zXvDs3<&9R(k04UBk{R?2iHW6M8Nt0&YpmI}_9`JMA5LFi#a*cS!ZM*L^o?`}lEZ_s zilVv!VMF~nV^Add_X2IEmG~WbC8Et*5k@T}I&!zEv{b~}v)N73AvxKA%FH+#xMO(* z(NoePS>kBM!a(R_;ep~|!uK80n;X2)Lq%!Fq!mehwwj>ltPqBW_Y%3yq~!M&wN-7i zj5!e7kgA!NG+ZHlyUR7$&>3+});5*8j&^&sARxk_P`P`^`m-80Wp4l*htZf%#rStn zDP$jr;j35IE2GLdDJ{5`i)?;9b~|37q@%A-w$7^VJ?07#e4H&+Q}-Nh*W1~alHpF5 z;d%eVWLU84_8`eQ*Lpmc3XR{V@-ot=3JHd7T!yZJ&XLKRt}sXI^-oBbUXmzAOcllE zFmF%w{g|tw$zdHYFG!BO7*%Bq4~Gtsi?qQwe7TqZAmKHJyu}&BZlf+7>WeSZNU!!q z#kyi(B}1a#~tN8E} z2s)~<=%(ysyRAV(sUyU0lT^H8Qcmg)zX9CJw6O^S^^!V=O3C{Q#fNT!tvhiy@}5ED z)u~Qoi>F5>;8)3Ihei<|EW;Cig}aeUm(&?~Y`pYu6MSV)Xx_EBc~^+Hb*_gNA|RcH zF$fiD)feoj^L&}wAPOA&#dL?ax&e<3U73`I&Z-%PA?s^Q)eh3qo$%VVu6LF@!>m{o zEOg>z#zTj-^y#$6r<=&ehLg{c#cnb9LiTj_9WRo`Uu2Da1eHL|BcuaK7^IgaM+cm- zQL#vtsN`Cf`z95tWOc0>?}F1!tuh=iyU}CKBMr%o5OxO@h>@)kdW%`7%lI;isVWj^I>sNJn9)lX*y2tu398tI}1I^}@mCt9+ft;R+rUi*?KqY(GD55s2K+ z;iek{9PEq{#aE;IT)dmR~#;I+w$7y1%HPr_o&#MBQSW`6q)YGw`CSQP-< z3pBB#MDFe3#^0kx8fpfzx4v)8JDn1Tu&ds(0h!v+$|j{(vDy+_YBQu3s0jEZC-#RJ z^l6pR$MTIUJb=@R2Py<87bqxoYJQZGf06f6N><8Dou`v3m4|nVf+P7Ag}^hX-G!>8 zNkghrt?m()p&fZA(mlTuBBG*7M#y{gEP2W$%d*&8`sOsex7!0^qEX}op+;Plg`S|7 zXQ>7&JC8N99pJi0Bt&MKPcQb02D3z%M_zXGYI4r%O^IuM-%w{HmyZy0+Pkx7e#H1ZZa`5CH{DU=qb5rX&+&yXHiMu>#X$6j@e$<;*=%zx zy5gPMtoQJVN@+iuftm5X^P4Dy4gingL4Y3WYNb{F|Z)D^n!{I z4L4z4EA`q|u=3yz3Nl_aHR&U4L^C-WK4V!bU1Jj$8ZS$AVm=aA{8&1vlT_1CFC??e z!ae*#Q+fVXy8mQCIzOW=F{f?vKoN23tkc43=*+@2md&22%gkSZ7R$;t%>y?{P%n-Q4&Q%)K z&1W#3>H=d@NM1UAF?iRh%U%4jDgGG3SF_lP%GBG!qcwVTvLrC`u$FWVXrAV5gm#dU zqwT(cVic1;vuaFXYZN!_^(ZwpF>!I?{4=qjyJBo9qLN20-qJ0Einx5VbOq(G0>`l@ zRU!OB2Q0CN8~0qy!S9nn+U)TV9)rVGneG)CwGV{UjCC%KU~`kct_AyZUS^`Y?WDKBb}6)vUv^h~SMWN; z&SdlhWWD-|j{O9<^uAsCgtZjYJM^@mS_=UUOYqSY{n?&5uti?teJ`BCi&>+-u!1+{ zILSVOi(7&q`Z3%pMF?J-9e;5HsIzY{nxSeDn&yh&0!JvnE!8j5ObNR?Sxlm9uzWfJ z>V=lss{;o2(GfT%uWLvYv9TK$;T%C%yTzmF;i?jEu6+n!M^7n(1grSN8^GZWV7q|; z^v`hp&+Q{xv3eqI`cpv(n|Erhm<2VaObmYQI6F^YZ`S-JEfHVoQe2Mv{faWPN}3&= zNW<)1RLQP6eQ4l5DYt0_F{f9k7FcCNN1f-SYVuw^G7SV0#Q4g-$GXV&#%SSwNCT{#oIEdzK=a+$*oSwR2!ifM>io}hBjFaeozyHUr|rA0c_t)^F;=q6mo7ucw;kF2u? z(9^TvVo^8^Gv(-wNLJXO0y&4wARMy)9A@{X|fgb{;y^6J~IPyFz&!IwH$GakbX z%4F};tH~J62sm3RayM%fRcswkY8BkvMdx_L8g-YMx4nk$A{F$rJ)RpCS0F1no&J z9!C5CP0Xxl>7_DaGhLCMyOn}n)TwJCATRdylLdVY3iN7Lf!ZrTc@?7*aB2V-X(&ow z#Z|6&SgclPFg?LPDXa^NAV)loOg2(*DWco+d-|Jr;StpWT_J6`ehoh5Bh=!O)VK|I zK7?8ECz;vZPmvR)1}AmX@;=+vF{x9JB+VvQ+!br{M^tOoiAcl>3a5|b10T}3zLWoy zOhjB&G=?OoKWWfiM$VR^Jql_z$8TggIx71Q6;bA(9AE;W3X2 zBdO$%`NXIL)#Z#KULvE3sW;VYdBle9E%C2IOUIxvXEyywX~>d>g**fZdCb`H~Hil?iy(;_TRvgm=+9k4w>W!{ydJd?hoEn7BWYnk}sKG&|x1-ix zCDf4k*7q46d$EV*z*xK42VEuU{5hP_sZNpsO^PkBT21u{dxMHzSg+3)>FU`{j&wGO zdx^pPZk;A55{B?tjbh>lplibJ1sgUS!Z6q39I5Wp{R;R#G=_V9aO#%3`b0Kctop=Z zcD9g<%w$kTTLf2?^XE^ex$*UpVeX2;hCgSxEz+sjwJpl2#&}B&&b8U=&7Xgiy^G@9 zA#>)B(n?Iq)jx~#+|KP*&c0?P$tu6Fgh5N z2WTtp_hXgGc2+i?#vtf4K1Q!Iq8v>mFcTb%_4Xq_sF}hEHyJy7jj`Qj9Wkvo>E3-HJ*h5H<0(SX*-2}>*cy~CT^g%}kSWtA zzl00L3kCa3HreXJw3%$j=ucAW-E$^$-zun~mvX{UyX`+7U4UB-P)Bn^^lH@Hb36lmuac5 zWx17kBi(A^J!~MnszaBoTK2tch-U8%@-Rqh{=EaI9tFwGA3l@i8ay&6cl%|7VQZR; zcae)g|X0OUNL@bGTCMz}8&z?XJ=~Zb&N@YHuxz~a-3+guAJ(2wiA7;u{%GO#zJ#Lq6 z)QhxD3QIsIw+e_4iZFIxo2e3*(b0YRsM{#zWy)|oO(w5Alm|JOXR5HZB7u;qYg!^~ z+h;Ll7h3JKK-(~}Nc1wbIEB@{e{gYMD@e22$*&_hkA# zcE-Rz&*T^f={ErM->0Q>&rc<|DWXkkBNt(;Pbf`3`h~jR`v6XRUNm!fbdaCRhDdl* z5gYj?xXqz@L{$uK=*vIgg7dVZ_Ex0=G8F`Qy<^Sm_M|V#Rlt3awo?3BE@$wk>`z-c zR{J(a2j%`=snU@-jqr^h5wI4t6MS3=DH=tRSb8C5;JJv~_hz0KC=alDas@{uhS4S4 zYeTHGs9j{Oh_SS+Ci}g-J@E>x<{l_*jz)ief)s)Vvchm>gLtu7g_135K&p)Bm()+6 z@ey{pYAgG#gydUga%m1FHoEUu$6(o~gX-?t`6H~fO0YzN#h#w(ME99vMf2`>(_xDO z6JavL+|l>&q6@{3o}Fy@nsjICg(v!YY*1Gv@yHFEq93^EMY1hDu-G7#{2-RY(H}CC zlNv?V3*4JwX_my_5V+Tv zpeqciY5B~ILTh04E3UYFiNKZBL?%l%tCeGtxbt+Odp*;=^{Y%Q}JCb&6S9@7L9)x+30% z&1>|U^gWmySF(~s?kFl#$xu;=9Ne#j_@pHZ^}$vVuv65tLZ8@Zsvg$n!9LkCglC#9 z`oz@2#Uo>u$lwF=C-m~P*VXsQN#BMRIb?M}NT^!?V4>3C5 z+K+Ns?f<%U9q&F|Cel|Rr0U&7Q!1NvcYqF_?6%&a0#RwuUB zMbN-}EzKFLlr)qTSVx}-EqhFAYP34~U@j3~afju^hpv=#iCKA-sqOaB&=4XwKz&oP(MZTd9}&dYQgeQzzcgx<3hUcdy%UABD%$+2!*zovo9Ad zgD!VQsUc3Q(~D|XbA-Dri1Z)2)+S9XFKtV09a5j^!k09JJCv%)&f|1+Z7tdpEF(c6 zbt1u76Otv%i0V-h?d>k4yQQCz=~%|*hGl1~tv;m-bhEX$xRSqsOlj_xE&xOF*zPi) zV&?!IIylJ33&=V(`IER>mN0VS?r?-^VVnL%o_djX8rGePS6 zskgT4RqfTSSgKZe%ozKhAbR)IqS05v5+Qyk?eLy>hFgErQx#ljH0>^JPmV7DCyh4o zrKXIRXw}k$qF}66Da=$)NZI5X`!9jbpA>sOLk~eZNUC`L2d&>)3LC>+rc+^1XJ*eT zisw0gwDZb>*OeT#-Uh9%s4#wbIenvu;u$N#aJiFEJE&UvJ+;oUh+fiJDwamta;gW`_F)9pn$S{j%hbp4f&)pW>Ar+OOzBkn zfdyn3G1Pt!tY%lgxufxYb~G*h+*+%&>uolVY%vR&D(^=){bV~*4~_zZdwC~PSk|8M zVi*3vd0*Ad8N@3fq!z7RTfSy6CgYQ_0)ua6GgBQtBK2`AE!=r%eWyXm>ABj^!h*t& z4)*i8J5;H>+{(pruDYJiM(rcBrLbakZ7b_!zL3G$*&&UdxawV@+H)2K-F26qFf|{2 z65YZgcwH~dyl_R8y^}{#J%-qcG0Up)`TKGE{wN0J*mcbI1tUCNPZpV1oIM|v)EP}# zg;~GJ=tM&VLs?i%7)XL!$X(UVd_?GT zgv%pDWqL6oQ}#+~;uLF{?_KjOZ+UBVE}QwQBNqXZGQF_eWiu6qciYzucN*%VRKZ+X zC`0}>p5q$m1J;oot0Kr9&cx&dZ#aXjNCY{@M-2`tK8^sFg?kTYR9m};o)2pf9~G;` zWsW2UweH9k(0k!6j-a4SUdzdr1-iCV54t&L3SJZrg-7RU&Q}}h=~2O}Y54QyNBV6h zF(hlG$Jl)aziOL$*JauyaLK45)v`Ojr)8!m#>Blr`N|&=>3!!e`VNQJ0`H&3)G_J_ zzxbNP*eRiv*Q1_s*43*)&qE_U)~!o!7^Pfks~skmX3k3pOU0*Gc^(F?+f$sTuuW0w z2R|w%Z%loeUC>d8Wn)mkd|Wq!TW&hB>!vYU?#8sr`XO2iFCCZZ0wJ@*HbkxN1rssk zF_Zpc&-baFVvTXJX(H-w1#cxj6fOy)AKxCUX>24VxrT5?63@rS2aMpTZ`VC&q|yN9 ziuC|6qV#OR!}_iSskD}=D&2kh= zUhu8bK8+Ssa4Thnd!!n;lWW#Xe;56Q^GUXMlE0twc z`rEx*lar+{&&rs+cvKIL27k)oxyT8-wJ|&N_V=OVHAD5tjkEHucFtBaH@#UTNlnVx zmp88ZL$gSgx>)Q31YV)`aZI(5VixacOHl!d_b#LNcBiIU6KzoP%$5_hxxda>MRBpy|j13)VxF>J(NzQ4^8L_N(Le=(?h7c zZS{@QVJA~S@@Mdtsvar-#6bRK-r#cH#>j+?AbtQF8v*4#YHv8%cO?JgMZu@+yDrZ3 zeFLvBxW5h>%{W9JjZ1*^5hI0~;V!0NtNc_l4?A#ZXcP)HzN{Rp&0Y&Ao8(^?di%K` zM*jQ+c*IlX*pd>GnVE3w3(3Qvs%zGn`AVxsGb`gB<7Q;axE=UuOnML5eaAyNJma_ zY*2NO@w3vAD%|VB@T;A2g=oVJKfOtv2-UH-WJszu|7uuU%t@c}X#n?YbU07>Dvy15 z`Al{Hnn4s<)1~j8??bZ!MPDwZ`9isNsQEC}1f*&^4YOseVq;A%TV^S=LSUhhXyig|}ytk zU!#j_%kpHD$G4&ctK4#@ih&kN`IynX8kL&)R)g7zyY4O8zN{fB8HE^&jtwNA%EjHW zW+tqM94HI%XBT7(6|32jW<{g5KOA1SijuB=n>g|jUo-+ZS;r~9LdwVG4du=cZw!4H z!fz<4C;G6!%}=iv(i-fkKP-p;v{g73PEHyj4ezg?QvM$2@MABq6RGeBW0l>PyXw=B zDnn_p!*Jn!puGXn9c#h$MuFa2&I5O3=$* zH=ecO3O9sNF}=8bGx^8ir|8J!kQ#({0o9K2-AU~yu0K&G&|fHH(HuLhgD|KDobQ z!MITUfZ}N?e3YAn0m=*n(tN;4FIiYYI!zy^M&?P!q$7sZJLz9U+@&Qm7CjpAV5a(^ zCJmsqZ*$P(++BU*|wK;jnDZR>maa{k)eI_!M?;ZwE)13);&fizcO2>DK{=9Nl{1EG}BeYgj+8#ak8YeB=WTw*I0O|84 zQ7Yy`Ddfbv$s3EoPB-l{5+&F#3pmhT#U1y|SZ&&@Dgup1j7M8B>YWu_=~_D3+0%U* zBY@tOu$$n6iXJj4@N%(^1ZQG??!^D!=iz^Gk7o@XwRhPE-2f2i8_(=U%-HzaQbL&P zG=mNv!|Ce~!81h}?Q*hZ`w&d%wNa-q6i>EYynKEV{(6N)1yrDtX;31+K6#nPJ*~ z*3hiae46yqLl|(%uz+6HmWv}-0@+4ii-idK|72Gy^)h+(cg+8p8cLh|wH5MjH-NE0 z@$vk;;i{0xB`Qj}6G^qa2ZKU3A zBV%3xI$kpxs>amIY!P3D~7?0KvjHfRm^rfHhgYzmO=sVRKpriG(L#Imz{-tsB6o zM9;nd5*zjn;I70h@7klyVBMc$Hpq?cKkwrgj%cY@zR^F{k87}i}c=3d=l`-{(}8!Z~xf5_jJ-V zqWLj$DUH5C@$XRnWR`fIi{;|vk26Hcw1dP!5=7L)XcYN;%^NqMm zhF(-E?e$B7`~MP6?%$&goMa?8(zG)CFbX;1q#yOl;Mlj1L%Bzcguzu$ zydvAZx?Nn9VCZtA?Ffjrl+PcmA2-j9UB9BoLx}0xEbQ#tk&)&kpmbC~-V?I%WpKAF z{!IB{#@m0w{_gMmWrnLS&6tO|08_56Nqgq0pZC2#c;7qi|FeJlZ%6{Ve-k782QOvI z@BTFZHI;%6PJ{%NUFJX-B~VL*PH=F@0&iGAopHKglewCzAa&mPaM-H!@w{~qVv@O; zSo}qQgZECh&jI}k{?DGhd56R`Rjy=`;YjQaKi9?cXfZ*m$6M7Qd#zD%ynA%@))|-b z=ThQFB0v5KwDcPrJNP$H`)};N>)$}Xu>-Py10DRv=578B^#3;{Lk(T!Sb>mbh_)$E znRid9(U@&FNkR;N;rEFDVofpBVbjLjqX9SHvbvX-$Gk!RYS?GmT|e^)`=9?CZy(*i zFk;TkA7XI+;FS{e{1=BL>fglB{&%S)brD;!yBKxX4<{?__b=2Xn(g!D2CYMWTW9Sn zNBc$%kjoEPBzCx)0pVKK<4X_S^e7W|1h#sty zL`KljK;QM9^FA-AxShv&z(2giAM+YJ_}`(eCry8V=frw`XQ>h3s9zCUZ~i8nNq3iZ zP2`B~*$qG$JM!co;+G{1)$z{1`!4f_)W^iH3eilPB-=*2$S6<*7d8p5{pkp?gHEKm z)+moYxc#x|zlX=xl1Y^rjJ}WZw@;n7_6JK|J7)^v{IS8`pRZ^TIw<-PJe&4csJ;Ef zt0i&R!&3Kan5XidnK+X*UrQu2oSeJUFK>-m%KnLj>!_O0+3&b&L*Z{Vcs!O#>P-%(C( znQODLiqvc>qV#*sM@vc`;v$i+t9`68w0HL(4=xQ6-B2yae3#x3%OnkTA0&dU7`}$y5oQCPWt)P^A3IGU z?|_7rxA{V_P4OyaWR{JF3Rt;_W`BhU0G?tsoI1L7PFgW|sf(c;7`!C2i3EYW(nW6o zPsnI5`TUWERE51Ij$izudL*pp0tkQquk@??uW}3F5Z}`hkWD*2hEO6CO{J`i@XiW5 z1}22VsS93(K<1{FuRjq&J`$R$KD_QG*)r16Xtc&0k449_6|EOz3FznRU?d4^&eU)w zUqIwvT72HktA3*CpgU2>sSWD-4s4Gg;Yj8My&o}NVTFAj?2)a_ns6s*GYa3)ah>$_ zD&VX}=!HX;mKUFBEk_g)m;R7%RXE*?JsP%)w)0?%hd?~mryyRCfuC3VT=$4yt`xlC zPit(n(fBxQo|Vlmhb(H6^JiNWiRd#r-+BFy5&h46$q6pKPKf+K=GuE}qo6`qsJ6SW zyfAttvtw0+eb|wiIOZuP51(Ewry(9+cElTPHyKNN#clA3u*t-p&(`Vl(B4Ley8N8j zntdvV7rm${9-78z3sagjGM>`s#XPt@K(}s!D;IU2j#K7$b)&TH_YTTSCof6+NVU>A zMB8?%*_j$8s&Kr2jm<(e!#@u1-ZhoPDs&rwguWGs(`+3xlMW_V$P@px*WGlSKZ>@z}6is0lu( z0ZFN=Z!>Zr^I9w2q0KyQeoc1#H^u%tD{d>zNG96Z6BkFhdwt7{@qz7>xEBn-mFk|z z`hE`|H&~C6m5#5&g+-x8HQ|hjgwEX@mpGgZ&3rwu&SsWDZ zK12mXCt?e+>VD~nHhpkW9cGYXQ4H(g;UN;F(``RM*KfdjWyrj902uM~8*IMyFz1MH zQea-{@5B7dj6$v`&pb;C!jYbAG1`HIsmGd;aDv7l-Csg+>IaO=rSAE0I1c6RKfAy? zePYtzR(b)>K+6x~E8MnkotxR1D{{u%ebFnGGzc>Zhd|ysskHdnbT>Zp7+7YGU=^Wu z7C7_v*fh|7)p=*zJpxiFzrDz_n(Pu=6~@JpIPlT%(|@PM|As{><$6?3ws=)7w}a0h zhFeeiw0|+!Uom=hW6hMK?WbMTB!#>;sXMDlX50Pi_*-_r0C=WcaVp1fo3#~w`budi zSEC;`aPqATHRLyUpFr-9-?PNis~kTpiD&;_9&SKg)_4*3K~KStZa**q>Ry{dRHTK( zcdV$7&8F{ZF7@W}6y^ht$+us$WQOS<7oi{hg+XScK0X!FTtD2bu~_XO<1S&4*}(0W zeb@Z>PYde{0P7zj)MhD~)v@q;{=L-y*8N+B4jJyiY2)gStE_iHSsE7E`IMLHgc};X zGqRCC^q3(@sLpaIkG*>z-(0j`oTSkpTX`(r4ZtJ!y6B_QFNY9%~PucV0wHwdr4 zaiLdytJg(s27fl(|9^P9Ghz7jH052LVP0f`VkBx;vJz}}u${!m3x;wRq|#Dj0!73G;Hlf5gA`FP^j!xb8S-htujMF| z>S(Qv&#D4*ob}zhapyCGKUlm(%_DMkLTl;WqDk1t!uYzI6bOe*pQp%H-np{Z5=IWq zsBeBhxenU_RV*%cTfVALEuN^ty^@vnE08V#TP>S9R>W$;J}kB+`mag3L_><_B=1J$ z<|&e;zIf{Pq=HQ`L~avRgs1`Ms_Ymi+G<F>_vi( z#<4tczY9>Dr@YK(sa_OGL{nT&62gC&aaxl{I`=w>kZ$*Ts-L9lW>UHDwYfPOjZ|6QvQy#L6eeXG8V$wRp1a)xVfmJ-%W=1~lMpS_6I_kASMykNO zc{aE6;=Dymr>L>lWBIqu?Hn~;DvQN#-FDB;W|slGE2g0zFO++fqA-#(>N+`?;b$q! z!DDLiST=;2R^1q*FSJcdgcIEVi0cQCmgzV$XB`!^QwC-FT#-|XnbI-{?+ z>Sz34#JzVwlS{WZilV3}C}2RkbO=p)FDkuC2|XYkLhm4mVxxxMt8_@{5JIRTy#xe8 zZ_+zR?|8G{@4N1OUiW*>ckey#xe0%)SQDNO){>_@RzWB_E*CowJQaWvo{L|6xAQAU|WT} z2ayd{mP^rGuS)If5>KKiP_8Qkj|yj9q*jaHR~bEOHnfUTi4MSX<%neqUd!VlW^DCB zP_2ke*&$q2PNE*%?zoM7UNrT*A&2Aq-Au8b*TW0BW~?h@Ggx7Ox}35oC>aJX=TnLE zv{D#WWnSZNU|#d)JyRK4cf%}YC`C)u4T+AT9u+Sb4TM(3z(y>aj0-!pVeX<1&$1t1 z9%(v4G^rFSb}m+(T!9GR)m)~!CU(xk^M;=|QB-%A|M`dgFVojEw{tyNG5&lpXRXX& zlJAXp*oAQnsw3jiMP^ahS!Phs1O&1XIR47SqDoOmRaSbe8FwTs{SPnc|LOUL0ZsQ4 zXH51d&MwvLurNF&3LG)YvUAU*;GF>{efpR;K@gW)RIw7(1l9;7_#_|T>&h9fO0z8l z1PVB^7KFyPUC2D({y#ek#qX+R`FJS)Vj!y=KZ5JD7D{hM4T_}8bM&QKDUJ3VC~>_A z6niegv}JGBe0Bo{vb1zmQb}b4Vy$L%D%hzB3R+*$W8L#e#CsksBerxr?M*d@BwX87BJWdcP*kc`aIIZ-NG>!X%$7z9DsM8Sp*-$4GU6`I7 zBlYpo0mo%&HUe(5V2KXl>_CZTM3}Ysq09+DI zpGnWCdt#OuU|XeC?_`(J{SO{3^CTAD5f~Rehm%s^wORTzx(%d zbfCN!GzXzp=n~EqZrff?W6kVq#l}S+8|7y;3Wl}23aNX_buaaWSABDei^ImAKeb~g zj&u}tPZfd0N%Cu3=(#+?Ku{EL1U%lRaFRJivMX~;Fy|veI{#Vzi$zl>;=LF9Iy^}* zm;(re27y3sXpkFwBeSE+%PR&W4ZBDtpl6yAU)*k*bh5&ed69n%S*Xdtur%}C7zGuv zs**E_Vy!y1ZzcFuR$pQxEx>yWGfQ2lY`0-qa1cD&3Us$_uVNyjZuXR@3whK1!B@Zg z#nmX4T(r){w_$4hT%+(9qQ`ani9sU>_)i=`{o{T+sqEr>jSiSUSKIn5SW}aoj-WU! zO$}?oQ`_BJIU?hW@|R|M@Gob%fBpvm*F%txZnaQ!luIpGJJwFCy7u()cFs<=u6aD* zYF)Fj<96vJoI12;x3%;n6!$OeJCFaU>#uu#0>!^>SfvGM{@2#(qU8vB^5F8HCF20` z)&tFQIy!hW7C28;y>zWIII}R!0u+TbO>6|?d(6k=lae~0iR1s=V!gZd7c~Hjs;i$X zZ|=x^wWD^a9CC;t zc>LC)|5k>U!*F=p(#gE;g>1k8yu1f{UG2eUV*WvQ>+ktDBL5I?xttzJZ0COK_IDJB zl8AK^q{iR(`9R=aZ?Oc^md*V1<7G@hCw8uy9F6DDxyrDyryrq4-$s?@7VlNYwx9`uaq<2 zp?k!ICQlxo_1gaUc|Zm{=goZ7+o69Ji#-lX*sk#8R`lQPey#1g`mo6(ru>!iak$oc zb^wPB_gH426e)5|)BEx3o97Rotjv#G0@bmV$R=xh3eG!iAM9Xi0=>9>NeWHKRe$2x z-kDUuf9RWmyVI>w3EC&A(y!sL^B&eAiNaq%DY_~TA6Q1HnMX0(dL#v5=eeo$F( zeDJwE%!ifDVOOpZlqWMJ74_Lr+A{p!cF1h9L|Z~RPxNp`K}G6f!a9=vYr)q~DWZ$7 z*4ulmlv6xi3*9#xM9V}^-xCb#ZhnZh77|D(MC_G9HpgmLgW$+0?K_(^*R^PW^!aLe zHx+kmlwe26V5FAR-S9Va(O`{RPMjLPkUlfCRS5@vBNz?`$-wMF4U=S@A8Sn zTk*eS^C#yZ+~#Z&>pA#xgtJG`&VFZGuKE~BwPX}UT+9V)8WMT)q3(Gtj3xEu!P|qk z`+uea=eWLY!-7{HVNJCki_iW_(VZw)iNUgb+V3l1ze0be&c8Hn73SGB*Vt+#r5k)( zF%{uErMmTsFk>_hcedLpuI0R=6nLvCc9uQxIA97l|8PnAcrRti`D%BgJ78x?sC(}6 zyBE&drOykAi@Z}KsrpxYpM6hN_iEmFQ-2X%^9_#?9CpXgiM7gFgcF*E%a}x|m%{n6 z>bO4=x1LZ*>DXRv$M2%gJO0bOw8A70@=Q0H zgF1`mN6e$Nz6O5#efa$+94T&J)ABy-1{-@l2I?Xr5x%vI{N&s zqFY(FQZAS#sE0Mg7jET9-SG_c&{h|2Uc0t-ZHbx3tTR?8$G3a-UCI0`Fp@Ik<4Vtu zu9sde4yQ9FdAG`LMO^UeQBN8AdnK%8^tTAo!s8kn9qI@8e&Qf<&tCeH>{9THcyosW zwgT@!bw##~q6$(WFj?LBxUuJ^$}$mptrS%J<{n6tZBA0qPCywVPPaZtQ`~Y0ay&Tl z%KDx&^F+bdsPw)YTRLJI46<(X0IWVsb?N9wjlg1*7m+u zymoDnm7Cv8MW}hm_!maC<%AZu3JW7MtM3V7eVB{P7z{A;GHQ(MeV<-2X&o4sn>|Yk zzMku)jWE-OJ(wIpGLGX>luXM^BCYHYK6i=#@dp2C#<#UyySaNT{e&lSqSW(H(I}=W zPNBQ0`60pIAngCYnt*U{Zb4h^sVmAIlgB9=CnHOG$~}^XUql(VZ6Uf4lCtbeUamiJ zaM?*H4=6xtiSYI=rS}8rH+!o;cJ z*i-tUtDWm}{A?N7kH$5*lko%fUE&0p}5s0tvb#%u%S#zNdUW zD;a&aI>nb&POrP8ALCf{_AN}t6=F`(044p7=W1x)Q+m2x$A!bE!7(lx25C8YB2xRR z95_~`Hwfg@tbGtOu?xM9Pe&#WW_Xk+C*y~MA0L&xVU*?Xyp}}u0C`{Jhj<8!iu1E6 z@s%%otXL)G3wbcDE5(Yv`1g5x_-qEr61{-b``86%nHi%+Ev!94Jw_VO80ji_Xl5~` z90l?E+^)VOWkgKYBvN4v6qt?o~sRhlZqSr0&w1`mcxo`B&p(;3*?hakO zS|19c$3o?GERn{HJ)Rd(t(E4dp?>ok;(nvj!!Pha&M(Yng=5ue1-?-|e1z;{BI!W8 z6M%3mb#LNv?gz{4?t`!We=A^wbUmlk!`Qhf#RmmP)2pI4uL3T=e`()#UrW~&`X=!3 zT7uD|N7H?2eBH}uxbJZ(QA2q?MzhkG7kFUjm*!rg?)~g3tAzLNIj*QBExEnM)sQCl zXbyl*tqqBnIDHm=!c&(fY8{6@Adtne9L2*Ea?4lQLp8qrH;J&iQN47;CjxGouX;}s zJCqa0D&b~3i`$pk4Qje( zm*A-%LT4QcgWvFMosG=A#0jp-2PG@c<4>YjuB0Bth4nz8;YM`qzT93yBFF{3br`70?SC&9mN%ap+}9Y-Iy-G4+buiTmHv%<*fY{Vk>LH0X2WZ%Ubx;MG}Vb@Ypi3J+H7EbS8vh@a{ScD@d8_kBF znj{z%$7!pa&xY|ET}x9&vG5iY)KqF{8>?ZVHut}W2KNmXy$jOfin91U{S&U*2;D;#!%tBJCzK_%~|>hFn&6MNfEtD+Y& z-S2keWZy;6e&W{Fof>7>TXfe0R2%Dnk~hlzT3zjbuk7T_dHqZGzx0o@%{nFW`}GkBzG3yp^Y2nS_xnO2`TO$! zS4|51SIxkg3Hs;$|Hl2l=vcp3^6%kXOQo>CdU0{@GCJ+(^u2uV_aW=O!0*9{M7^x8l7?Y|!Sp;gr=S*Rv(;p}_NUP4k5$_1yr5)qkns{8r(= zg>mn%k3k`6Etymikv2{GrBg*GyEa7jv@VpnuNU>^-L>rB#ov^kezB>Yc(SZ@VcGPT z^9I~$o9n36f85tM{oU>T?=}59IM>oe!Ty?5_fBfHwzRAQ$ro}*m*~FqhvrHS3*rU- zF7Yp~CH|s-5=I>oLrc~DChWe)e`bk*z1Y*ODjx){H>a95Q@RFV((|U{th+ITW`uQV z1oS;EGG;3yB^s%yyPUjK0Vs`o!DmCj?HBZ88Wwai8FnvM zsKbL+n9oMv^hA5q*C{8+7Ihi%CyAzsT2U>2m||`Zpdnr!3v;y!33=h?(bCP#?ii|B zZ1+i{G8P6cIML9nci5nfNv;o955#O5%%E@{C{z4zH7iS${Nm3x?YDhZ6`T39F4SbUynd@fL zh8xHInr|>WK8IQrG)O zwuadQKf>3VEEk67RcbSz#|INX6k>mrG66S52~^YX?q$0F;8Uv0CUdIMxzxS6ea@`$ zgI?AEVW$6MX{=af09In|Sr^QUcEkN@?igS5QLjC?m%jO~t4^^i7BNN>t}XZ_-l$T$ z3q-_=CIb0*(o=IRGpq%?PeGflA#A+gKeLO0<=QyNB$?|*=;}K5%_s#XHx(P{43H_- z^#C}Pfr51U3Fadln^m726W^|gfi(;;ZYi)O*Q|9oh|EzY(84jzw809&saR)oxN+T6 zY>>3USXY?CwFK;zf+)-YA{m!D30H_9$=bS(q?ocJ&GCGLA=fg7o^H^@4E){7Ge|=e zpNW0AF7sEs&$eW_ab3{idiI!L+u{L*fMF7aQ%}Z+Rx03#ey6kcWGO5@<|~7Ty;VB$ z31O|lgd9Q)$wx5k&m`5Ru=~X(%|&5N;ja7%)7zM$`llWCtU1$Ujv}?whnb4`qNas% z5Rd2C^jQLh3G^%RGoWWYRW7AUTMRj~?sJEP0X7pv#tqNc;RYu$1Zqap2n}DGppS)v zTKHBx$QxXByqc!bl?V|F`+B{Qg4Uo>#Fo5eCk!KdqU7>XRM$*0IR3TN{oUuc;N^=2 z_UrnPX}XXT4LbV622W*zanFIvqJf8Z7UVC?+s9Mog3ZjP-*zdav@A5E3Z>3c1`5*b zDnO2Q>Ai81*ZUrPia0Z!YmX{lW^{mk@G)4d6Y9i-!~)(wfiAguPaLR$K_E5lXHLIQ zh<|c$p;$M4h&x7yOFi(;Hsp8!@B(g_Q8VCW&P8J^aJRZ(y=V~%;ke^nYHm;z2v6oK zmKjU^n1$UijqT4C1YgFnaV=S&v$fuzx!p8|+hyaMkANV;;wJACuY*C?^EILPP_2-W z7W>=9udU)}Klp#V8IfV%c?k+4R4zjlN2__Rvi&7@lgX>qL)HX#7Jy@;ygnC z#BsP!R9lR-TGO*UZ=Kz|wpuX8kn0DGjrFOw&uNdf*OBDs`YbzQWmBSVlWLp6w#1|R zV7KCGh`#)z4)y8vq3EdGuPJs}-MkD`b(2(wMicO4c+cFeOBJbs=Yl*mt4+n6gX}Ob zLI+l%v@caQDhBR8)yhJz^qw4MlpNFBk5@{*V8;6{8Gm~b)B&;5-};F&P`paq7?mv3 zvx;~h9hbb21V+)9N4r*RUKpsMAs+o*5^y(JP!BSSymKKup2+$ouVh5kh_t(0j{K|n zc$Jj&;eJ13wY{k7Ra|o8K_~eM0o&Rt{ouS>v%6_2e0o0yW(gI`2-}CM_`kMy%iM)o zUFDl@7VUeTDk(D!=@)^EM<4Cg`ax1svhd_!s_@zJ*)~r!IL6@DUQLbf!y9<+mmuR6oXQWyNT9(Cn9BCfNxl3 zJl-zB;r5Cu<%l&Q_Kc%qAtqCOYKZsYr&dem1OgL2zLY#eSbG)_mRId(q>!K0MleKi zi5Iv5CQFLRW7nS(6|6Y*FT4`)KGZsa zQF8^nk{tQLeh=uf9rTUf2+6SnbF+^Ri6XxPn<=kFqp2F);y?hclJVB($Fc^>iqley zY(!R{+bhZjiQ6L@Wn?~*dI0!T;n7z-L`^pYwgOmf?&bhGLaQB`9ZjO9&i6*s^nEa4 zt$~0aY3bRim>vjF1Xhq%GJeroL_27VdaYw!Y-F6Lv9-E8yr)UTr~$rC%K-8q6HuG6 z(1m*VP(M0Ytt@9&g4|=*Jo!`wTln588%W4(4{7x`B)J?qx-zd*G8y>;@mN3ujoq=N z7T{T#f;(~vf%BJ#fqQx=HWmrnIvpzre+ns;<>>GR5#7oh2Y(tRb(uLn4{)y0g+ggH zf>SP(B0hRBs4YGW0~!Slt6UM}zx_5@`Cg%E^vz1VlagbHxa>*1^*R2nC@gXp1UVDx z&OUem%fW1z_O_-K$jMmi<1-BDkJB@^BToR^5zq-~+c-}hlMgW!)p)d8*+uoDtr$jk z#XuG=NC*s;#0A%4L<^>Q#Au}DZY$_8D6mJ1#qcV^r{VMy46vhzc19*^^Os$Sq-FBIcWPr!t%Z5be?t_y<&>lTk`NH=> zsGYp3ZQh=?3hwcPVDD%0Dhg_Ckmm*;sE&|U3Z;%ST>K|@j#ZTRke^-&&Kz^1gsaXl zAF70uRRs!`K9xWTLJC4FgS$1wB?l_9gU-mt9tE3SH)E0FpiV|k_)7I-3^kn1eaO;siCudbQvS_6rR zr^CxTw?O0Zz2P0tui7qnPC^XN8Nc9}n-ve_;yqj-2n?*OL9&ZTdCA+~u`S?LQE&zg zP~I2HYUziLZ*usPf4_=85L=Is@|Zv5t#ten7M>OvNFTKNV{ZZjmg(|6jfzi_NTU$C zFul9=B6J(iYeJ|bGFisl6R6=pr5uM~t8P%Voak3+TZ&Rl%taM-1y64w=BTq;;a-g%s38BW17Ve%65bEO?PnKP%nDeJqHc^kz_tIr- zd-kwYD@!i4FupFOINfo6i!|dtPoUx}^oBxI?Gwoi8RSBT9Sek`RJmdBM*?(pqyaSU z86_C-ZA~i&yArX49^3DCr%%EqhyVp;v=MeFaZ>$@J67k3;k6IP{LyA{P zyh%N3(they&I@Qa82%EY~cQcu-EHb)H^Y$3v4#4h{JDNH1nY^W^e_N)yAh=@GBnWBNfnNfwJr_6a{P)Xc1DULBf#`hWM!S<+X&Z zaB&cEH51xe$N6k4*)IM@F$ncMXgLqks>=W_LUAb+PZ<2;%KML99G>@@$Ekud-~5_7 zLN%GnIy^xsv{p8tAhb;_w!b^?%eU`;?)CmRb^WtrfhSPK&yDShcQ#JQ{R)@CCP~k2 zvZ!pG*U9WC0jwo;y?_QOP1A~vWD(CYQeT$)y9?;``0e<)=@zZ(B5|4O99qrP3SgQ( z#SK#Sc}5mdN<k8{(76{y%1&>IDDDDKBjX+h9hj%W*A(-Ly(mpJt!(4LG>WzdCzYzhdVM+&9+GMl z`xMH>UQKfem8jB0)r_BMQdo2n ze@h{x)T$rJqQ9oJy3n`4!gaKJCN?z*)RBhph_Y7{T(-!ZJE_ znWuFvf*NwWeD1vC%$0$Wol{yKr?14r@E51}-}7h$a<%u$nOVjd z9jIE$bO{3Y9#b>((U%YzcLpb)g^C<seU}NKp=5X( zG;BWCfI|FXL#E;V?Ym)o8_l=eij?QMT?1>IuBeKNCzN=UAYB1%I}H+n7h+Fv9&%BF zzxIn`dkQ&*TEZI4Ws)V=hO9F^B?5pf=jrMFLsFd#_@#{C&Swz}>wC@XZ?wb ztWM*r!JB=PVfWBen}fat&RTa$@Hv-UpJrjQb>@3{Z|FIK^y~XLmW%~J zIu0W4D%54g$*B}j`Yt3EvIPjocT`HaeoqIOrwm(000ceX2IL(-FJjkyuyOS9P^m~? zOMAQ~`deOp?g%4!o1`fwqQ{|Ru_`1D#tgig!%I9YH+jfvcb|NB0W0!gG}F55F5$mtF36%aZTg1g74>4 zDTRpz*xt@wY5R%8SpHfq%U~e0Yf4Mtj{!ijh+%wBlS>GOCWaF5c9jGL7ZsNOF#u@H zsa#Y~cme_Ji1>@c?s0%sJyhxvQf?GfIuC}b!plBwK?myZ)v|G7frT3bpA_|RKs7;I zzGQKAL>ja7@d2Cma_q(!HxRE5R!R02?ZcVNZI&30qr{(1O#zMQ4cJD4`i#%kw@#(A z#yq6t2PxetqWnxPnK?e@%EXPphUt1Z>e-O%5-^lS3b+WfRZ?OVlV;guH&84>+NHt9 z%kLw&tK6wvaRYk~om&v(`SQ0}=`RfDjoT$TeOb(qGMb>^GoWTekBIPI4d_jd4G$e~ ziYSU`OB=0`qTJl>$Hd5#^JrPA=WD!qYN?wk4@Jn*0;nezAB=j#Mx!&-CUJI8L1fyc zVic0+BjiMWGV~KiE;(0If2<+U=9IlXhU>(GRkL^_KNkcWwMv z)n$BsDn!RAIU;Cf{IErD(N~k5sB@HsIL}Y%*5LiKc>fUcj-bRQnSpcnSIc!=HGDr- zcMBqO=UU_9-9c_?cofVEskSNF1N(tOa{2wOv&=b`d1myMr4)CKP=3I-ApU8|<~VIt z$kpr>=}(;cOWWqRany`YDkkeTwS$l#B16qQ9xdNsn!6L>I?R4*W5Y7W+MQMY)vQ+s zou_<~3l(-6P;Bh6Vwo-?;YfV$K*PKPQ6^&Jh>+?w|H(SWu}G`9Xvm2RR$(w<>OzBh zLYdS~7npWS)RodioTKL}SpiKi99A_lRr}SH*dHZT12?RavRI6Q5pMjzOb1Qi&S-kB zqMFox9@E(24;WPv&>~GBIXXI$N&mErE}P`!@k0|@3oEnjl*5$`sE`wMyReV$k8TC!X9sih_(`Rr&>P>IolC z6BPx>xHV-=CY9Pv{=1LA;6*{f{DyfrpolCTt%v( zSnwbEG^?KG(C#4ED`X1eF3)-ZLnqkKJ@l$;xs@d zr{4$oprwX+_Gg6Guo|aBnkRTazFm+KYe!BolBa)7 z>xY)Sw|R9*$SB8Lod-igJ3}x#*irRvtd1{qLgK|TL?xtqm{wHg>S9=4l#x+J-eC)2 z0G|bMkaxsH#^txTFb!w#t&I{D`x3=g2Mfc40?L@ zGV~(HDJU~r?TkB^)$C&UwYPxRG^$w3a?5W?!#whJKCHUN1uLE^ByUZOv<4Fqb%su$ z0g~4reChQ;RpCKjeBIQkhPmm~-${&G1J&9yg`joo#AAX1@U4n|eh~Omba%IO^xID2 zAErIcW)@@4$xu73`XIlRwP+6oL(FE*of?xQ8VU(<{I}L6l53e#IhUohcgxhqiZS_) zHxw1uQ(J4lTCft19qnad(MU@E087(j6=vZK5GB#&NP6hr+FE?wcAtlGl*2Z#ygy`d z#}2qOBf~*07^s>pU`VGP7bUx(Xu}zp5%6GH0vgi6vnBrrE=yN3@R|Wq1iO4@$tFjc z)>6QFfebMmL;jwF<=erlqk1ig33v>q`{rLh8W>G*i^nZt?b zl-q&{SjJ$>w44h_oS9MOZ0K`>}20Hj+L5}^{x@^qT{>)j9}opPLq2eY+#g)wOZ z eAC#IiE1ykwvkiue0S2jW|77Cbh3ciN5WrqbWQ==5$9W}(Fm?hxJA{CpmvOD(I zE6ur*1yZlI1=q?>1_BhS8OAykjbdB%U{?r0_wo`}&9;rVspDfvwU~a;QZLg287{Y- zMAX3=^ke-}>pvK{W)tnG8cnDBET#Jrd4~=QuwC~g{+jrn(2RJuf}IF$5~X$zS}K<( zH*2v*%p2v_Z+xTj(dN}d>BrN!&dI3v2EqfZ2~;df!`U67Qly}5^h-vS3bP|@lqrxH zB!@F6Hmnys$Z;pElgQ)_qhPBYz#vJ|TB2cd=&SMm9I>1Me=l25^p6alI{>;iHzMO! zWAOwi&c^l`R4EEF!cu@afPDyu>b=MhNt;#`>hbVzKBEFrt>toY`$U+qlF_NOX zcU?7~MNB*1j?2rvgQWlr)7519>znR)gg0h8X~OTOUJ9yt__{ryBw3SY)t=!{T+7ZS z1jv>bcZW6Le@pXF^PfwL(-TEIswpTmcWE?d-_R-uwtbc-@ROR3JF#nXU{2A;RMUsk^(Y|f9-Gym)c%Syce$itqdns57(E^qcq=e)3P zsljFZkA4?cLYbv1Z{rjMD+npXSJjeKKUj0^i0g2dFPkK40Bd=y(fY%jUOw1}20w9pEJK?A z66f>B?_Kuq;o$k0)wf^P_I&zr=sJ`0XzJXjySMZ>nZmmL@fQn+5noQ@sS#?P2q&Lc z@w{(d1$_SgTQhFl%uk^vk+D~hLl`y4-f@*#R18weIeU8M=*ecL+ZK8lhTHkqpru8W zDd3=^j18+sR9nVxwf(yo*3+k_`q&$;kfbA<_IKpR{{A|4QuBGQlb=xB{$1uX1G{Fw z*8m>2vWXI}LN!mt33-I>34M(1ecShA&B1SoJI-)^i2da`fLF`iD_IixXKSS;F8?sA z|Gm<`aA*@+Ef~?;b|lqHotJibFR%bHL-n_uBFLoA*U5of4l-kYtB*#Atgd$VXA3u4VhdhnUMk^@q zl%#yQt6R23jZpxY07H*;8bJ`InByJ_e$ZaJ_I8w+Gp%1%jpq<}Nv zLjY9mO*y`KnHbMXzQCBX<$NyXNl}XS+scFPo(}7XiJp=|B}0}u)oPC|3w}F|E`lQ2 zF3R|?aY)f9|Mhlrq@?%0WJ$Jq))prcN^OVc7M+gbQ`}A+sR=Ie4oX@{I<<7BT7&>$ zS@f=jr-6VdEkTJ=^;2D5+@@_`$EFX}58IXu;azfYrJi0vrI6#C)(iQk5EP1sPa8dz zXD9z1# z0*;&ucx+!vMqrIu3ZPM)G7lNPpIMfzCx)>}TmY9Azh&{Q_87rbDH$7hP^%O_=~S7}oCpqC-Kj;%=JuK7>0i ze@e6ZQ1piI0qf|)A_bRB4p}*gpjGU#60D%G<3dfPI)b+;z-CmrMD#1Te9b8%SVMmW zPPJr)(klXW=@ph8AJ-Z4DD4L)t*jlIQ{R$z^4}V1WD_d33t8dgAzjG`B;p}ic3Tiu z9V7Qwn|B0!b2JEJeZgX#L}bQ$9moSHH%2ENCkNKKCAY&+8U?8^)Ptj-wD6;dRh8Fy zY~OeHHj+vgIJr90-Ww-Yzf?H%o64=b&ZwA_P|}{Y?$O!0#0Tw|+gdAhv@(;@*Tt;y z)NN3vjIEU;IgR+&b;F5`p-rtlP_j@Eg!pk>mM^Zi?v<&Hzem=eganH3c*XJgRx@U-j&gCp1srDrl$h~P*t#lpJut@1q#_Q{U~v}RYlcW z{A7jQuM|QLL=lmN3M@#Kxbgsu3W8E$TD4SS;d=w=a8lz!AsVE`sJREDM^NqlgYWEM z*x`aPVx!}<;MJ5;Yhq0^#B>Siq|$hAdYYK3b>XuZ%a??}YA#5sp@!C?2IW}t3UZKH z(r1CT&}qCuX+tg*BGYO-T(z**W?on8uVvKx#`eQ7+lxArp#A8mMMgs#K~QOc;25e` zR0zoKYxL&Jp&bPUJ+Kvm?VsgT3tZWhq=do2#r$drZ&!2m=j_L|^l&XsY!uCo4rHTu zG3AdELi!FKzzUUYpw5(~Z3H?Zw*(N9W0Nqtk-Pt|88^>w_FUs&I+p{PGp-&IO!Iol&ngR&#Jsn+gAU;=2hQ_2aA)c{ko5^#y^4}5T9nr*asWh?+|!Xv)4smzIH~n0F?eORBOgs&OFA>b1JAD6PSG4K zQgunp^R4Wi_b2a2T^O*K1w>Yi`s!zpRxG{&ZP|Imbwambn`~*KNcK>q$EaRCn@}3$ zMw%R_Xd0W7Qioqrm9SkrWYt%wtzV!&)Y^V|c->3pQ`kuWljcWoRD6CPFEuly5|6PS zqi8MTs8Bl&Txv`D;FM+N09eRCHdX65oO44al68jLmNCI$E?5WFAASB@Q~RxcHHKSf zG_`!-+MbKlqx9zwc%G#Mx{YiW9Xu+}(pec-e}@HjHcU+)AKwE^K5%vf94X-AXt%{b+nKZnfHbHaEeMD^ONMcRQ<-1n*7D@Y7n#pWx=Wf)`xh{ zyl$;(U)KW;NYy22Jv~V3s8CIKHV&~{Lm%c%xCs@_*QBrO)?FEk^s1-x@Q@p}Hw>4Mg`!iXFxgk(FS75aoh2LF1!*v}6eca0-w5_Sogd z7@puuKH*F0YtfJYvaJ87WPjbpZc5Qnh){`+rhak!w8?=@z5hf02mRf%OV0UzD*Y9# z6nMYaAY%ZQ_UjAtPVBTT+Ig`IPVk{j{$o!GW}s&I-e@haqV%B^y@OoQAo+>2=1fGm ztZoLrbu`{o#A&J_#0mShE(+|DcI@aHX8w$l)IpJ*$=3ZYW60=Sru?aToQRMPb!X#d@qk4UiIFT>(+{gq!dRq0P%#)2XtpvyX~Uhi7Vz z`V2k2Ko>ENeY?6<73n_DGdI0Fao2YZhF)MhJ9L)S03fwhMMVZ1NJPcVPd-1fKh|3Q z>)D~fY>n;l^tX*DL?YX@!5>U1*KbsMZhgW`rb9xLuOKv}co7ME20Zo!H9@lmVw0cC z%N8}5N5~ijf-?|{#z$XUa48+Cw!&5>doPDH*W0%Dw?$gWxVMlOMCi)69WmRr6(@AM>t(!;@v)sDmg1`BBM(6D_1WBiksZ(z{DSGQG^q6CB(hW=f?` z`ud#Bxw_EPIC9;Y?yEbq=GvZWz%|gD_@xP0VSLNRyAwkad^1APTe z7QKXN}4xR*J^Z$;&^>S4M@T$=~T4 zadQKK>r;4{znD|LX5CuYJ5iP}_mzsE4Zf-tI|+BWl1xu!NHSsx=7xgo)*uA#mFoVN!v7+AKMNUQmS50LrRf;lsOo@NA{Y zT84E0s~77YoFmn-tH!hU+RR(M^URv^tk2Irw|z+|>8*jsB>H$i{XAr+5pQLcA;B2K z@sWJW=@YC@c+|Ri;&}{1b{No+*yyN8v}GACL|COiQQyO-7nP6zA`2GNk9olWEs3>` zc%#1{>9SepSl3jgD-_l`;x=j2JGytQ#%J%h_d}qFFI2nPyPD5BxwAjFIz0x`JDDs% zVyG+0nW~@en*VlpWzE}?HO|J~@EJdAlu?=PCDS9cQH+yea=u^U)3hr7tz`*%DK$`4 zZOM*&h~eds6VdX(q?E1v=r);beK}q`id?eG#&gkT*W}u_!+2g9GXWwpi6kKHWZ>O- zU7mMi4Vpe>nHozBJ)#<>%jKl>K>F;;@%YzS(zhzJ6_b)8_(%>Tt{Cz)rUntk4sCW2 zd7@K?)Q-_wpS_RF&PuCQ+Y@N(`;IjaMGSf!3sZCZyFFwGPG48aIjDlbu%@3lEeR&g z@aXNY$%mE3^`ylhSVK9nCnJpaA%QOadIhEHhNoxO+eiLe`Mqwz{p6``dj?km^+~zQ zcGLPrZFP3jKTgo)N$m{1`|Cf*w#jAdpS*(fd=&5U91bZkD<_ zrGw}+kDg_X=TS#sT+0M_Tq(CB&}(FgJ*F=>m31&Wm}`WOQ6-K9>V|gPNiVo=Fj~!n zPcczhTqqhtL?SsrM5_?~7>AthpJwWRnY?#Ve5H%;)!KVbpNwanbFpA`Y-AaQ@@I}8 zQ;uFiFplovKKt@p>H9*#I;HPv z0(?IZ;duXRYU`@}t zl>Hrp6%0SLX!Xf}6yTcfmC+Oy!-8+L>SpHOJmR&LogTvX!ST7zCQ-I!kSd>5-zi(0 zEi_qG;e;y}MtdUUlk+^@N;UgSf(@FTFNWcc;#Gq1B#szGe4pY4cNS)a&R(=2{qAB? zK3o2mdxJ4^uwgT1&-3*cjiqV(DhgO=AQA|4WzWp$Bi)cn*+^J< z&NZ~zOZM9%aCi^7%kIFMQg&MF*H&%$@`g0hEba?|FGTeE!i^A04!!L37Pw2|ULXdvLxSwx|&%J6`urra%*mCQO3`XV#YU$w;$8VR=H>cp!_Am{1;_ed2wwH=^THY zV(Kpc{XRk62W0)^-K?7`ziR%A40|(dTfnDJm$`61M=cyhBLtHS1bO~p7g{4r`yYO9KtnTPN+~M zR~F>;Es`kT``c8<9O9Y*w)urGzMtPIV^6`q6=YJ6fByCGQ(D(l<85l5 zUleb%@2?DhQvYSv3%PHB*R!9wk=-k>?>|-fu&90Nb{*&Mx<=IZ#uJ7Vn90G}^D(*r zs%cGZWR1b=cmLPVslm;QjIXrId5@ zj3|qO^DpYQ-QT=Ri=O7gx0hd)y;sxdeo>FN;atda z5Bx>F{ZwOIHiQyOXvWPG{za+O|7sxBFY51`%F{Y^{KY2B?X|hIqiwuH1Q9?g$N5C$ z2}#Y0kG*3gk^J2e3NIGdlgI8Izo;&g_c7PR!oA|#*9Zw!ZfuLp7*gik(D*eX{xUXn zzb}4KTS!|fuYLZj%(aJq4!1w4xIcydDkClX=Lr6jV)FkbRQTV}Ps|YO_w1Wf`ifxrBLE)0B^^l#$6ezADs zAShwI`=-Ws^t*=5_pSUx2uVBcf}d7+tXnkaxxS+l=YJ6Vf&NC-`|jMOXBWmxNiW*n zN)a+6MM(Q#-={D9zptJj>j?g);b-=lh1W-_FSpt6=qQ=h&u>0`zWd*9Nvzrl$z?ET zg<21-0j4LN>u*i`@IP;9x_|xKp zup<9mjgXU#E1=rRaXytn!)G>w#<$duCcmT4C}9Ck?uzi8m>}glGO82OqpASM+Mq#R zdhn^HYom{yHm|;$x5(q+yJ#%9IboNB+&AMGYCLgw2Vl3p7W(s<&=DJi+1^3QJ#e$WoR_RQZ`nir%_^_Zzue%;q9 zy-AyS!tM^Apo}T}K6k?4yL3mG#iOukO?}}Sat&S(;9dXSU~(_rlAw;mggdT}ImV)) zv5rAW4tkZUWw%nT4WJI1$e#bOQ`^uH)Sz5rI@^d-@^ zIyTZs*%%trV4J^z z=7csCD7M;kMx)!)d-PKSHUg9X!S!|N7b=g6l%zo63otOoGHe>@y>V<^woTyl|2ID7 z|IbgSa*Ax@sxiW}&EC6EgkmGSa%C656$&VtVfM^}!Q}l&=cTT2U?3EClzu)XHAd~X z-3@@f;Ee$#*kW>C#n9&7K=i?1(#JMqCJS54;Wb)@;(@v)vE`yU%o=PM!COz_o$EZ- z?A&XgjFZ5M)4|Iw(rH>+W_!HAdC>2e`jBO6^jJ+LX;NimwtksY$aBmLM=*I+a%FJw zcmjg1Z|L|U{FMBU(K!>rZK%a|;r*FHy>J5o*RUdToe(4k2+cHVYO&uTPOlzGX885y zXxHPjprKm5`<)nBL96u$Bm8DuBAa~lpZ_=0rwS}_C&b3c0lY%bh3^EgCy~7A6Ec4K zL%yi;sJrN0B=De3esK&tFlxMLzec0vEd#G`m6}J`kUP7yzp`0gTv)1CfC8s9o^ypi z3oy;MNK6P)F3?^3Q&qmR<5Qfn-bhtvecW*Dao zGc~nWN`c{FtQy3U0rrwQFI$L`t`#QERT9#9mn@PWZb&wxRZ!R5x8P=)$B@T?@f^3qP6lK)WYv<%`PF!ZF|BHluZK2KCjbtFN|X-6q^B>E7*><5Vyh2bfV-I}_>>esaBQtMnSR70;kd}4j3Jmk*RsxOG?6%0ZsgRD!+Y(N%&ANkd z%ATZWNZL^^H@SOwc--^KCwM;`3(8fY#N+p$s9CQ~yEscE(ge_y7rdQXw)>JJ1o#+y zM1W#Wd)#PXtT%i~H^q>OjxR4nw zGecwtTGINqh*7Pqr^=+jW5pVIR|#qKe$5q)}658d7rC zyg#fqj6jL?V2d^9a}E9}QykMG%d%X6F68ZXr4xS<}8 z?voI(dv>yZ{PkPcl}Jns2le(~U+w&Fz3s3}(-lL390T{3p2 zy>B(UZ=Y-K7~-6C(Y#k*E2CRDOUyL(_Esc>@N>Vi3b$KN2aA}w!`Ukqv zhJo}a06x9l(JTkcWvI&N*~IA<*}A}bmQavwh8Oekug3E|R8w)mfC7Q2i2!9;bWVF1 zS}bj8UZe+Ka4`KbP!8Rx^Vl0AW|h{QZvVxc6-j&Z0!74gT$V@Y59S^cgi;(J5J{ZZ zL>o?PJ+oa_^|chF(&Z$lXrYa@ApWRjsOO{oMFR0$ba+!6l{dE#?H9M|p@M<+TE37v zM}E1{9|Zt9UTn{*E$c~PubtVDQ$3EdTgBxuWOH(UX=*7iXm@yzw$CdI;ID=~%8icB ztx4L-I~0ugLUbn7FL44?MTz^bc}`~*K*9G-*Q_Fg&*gq7g#v)$H>-OX@pT@8^y*u? zFQ>?F3HK&dInrSTI3$^ZoNj|7%)hcltt%mrFfT!$k0XP;PqCu zWI$fuUX^Kk1OBY#I>W)U;Gv})FT&w!uevhaCvYVlN53I-s%cGooWCBW7O%eP_*{v- zcE9qF%%1^*lP~4&*eV`0+qaIl+Iy*!c1X@1^3&-VGSSIkMgbYJ&)sqI-lEctM-vKqrMz>_E5E!wR# ze7H+QTS=%7d#Lil0ITf~MWEH7r;Ze`7s)R_{rim+Z%@4%rFW6A$~nJyw5(#Dnt8CG z=_v_N^H}nX!kZzAdAWpPReK6Mr-452w&Lp@O_ev7mrA*EP#(7>r(>_MvFY8Ym!G7NdMz3d$F@#SCjrMC&n$iVdHR#7)Oa12zI;L84R}xGMbguWr~-ZsLNG0 z5Q)0I`>(2-ipttpHNzLC={8$h%vtCB_Oo*L0g>BUZKq0(5>uO)Q^oF-DIaMoI_N;{BEyc=K#=`+< zE}qIC*=fLam}V9@29BYG-H+zMc#IdGpOEGGl7VzvU+_uH7d#EE48sIXo9APvv_b38 zM{wKx%HdG+XgPfcO(MK%e}p4tAu5>fmcnzY%`DUefq>bJ)clEfeA5S17OZYg6@XZ~rV$A#0q+-i+$_M$2~j!% zqu;Iz@qvzG<58d6py!oAhGLtpWj@5x`8erxbQ%~es>FCnGz#m7+2_vOBhRq<`ctDi zK?onW8kG}B3$7JQJ^ikoFGc`&wyWG(O47AB8`N1QzQNp0?lMC%!|(}zb&LIHT>ldd z1>ITFJpc(QQE>|+xSe8=2R}jAYVJKbi%&-Pm4UCip<;TY!Gwa!@&SH?|KUNcA?;8V zmQ0^D>K<)r5lGyMf#mR98i!eISH{2OBSsCzfY0(B2TW|3R!S zcH?BM7^Jcx_Srd{dq`OQALYl_1P)WbvIXvYkFk9IE&t=O?Pruur2-p(Xyu)odL|uZ zw;qz+moxnPl9wZ_^cXL1yJ}y3yNu@@WrDT`R#buvV&=&ss|rfDti!fF%WRc2oYqk( UxK_3+itqk2^8JrS;r=@GU+A2;EdT%j literal 0 HcmV?d00001 diff --git a/docs/my-website/img/isolated_ci_cd_environments.png b/docs/my-website/img/isolated_ci_cd_environments.png new file mode 100644 index 0000000000000000000000000000000000000000..347523f0fab0a52b3e7346633ae250234f2883ee GIT binary patch literal 16046 zcmd_R2UL??w=Nn48&3C|`8P!ifKE);0RXtWkC&nPy~}3i7MG8I|F;;w zbPw&lJ%2raLv-AYzF+77K(EAa!u*~1q=TckJw3or`d7${o;clDb~?}Q{BQi#U%cJF z@e03qe;-dDdKmp*yqB?|3Y~YL^H-h!!rT3YxA*k=75+6ni~`)v_g7lK=$FN(9o?Zu z^ydlsmlxm-Fa)Rr?)~zg{+mu7XaL~WIsm}<_D`AJI{=_E0sy!$^ruWH3jjFx0syG! z`cw92o_Id=diW1GCi?w|lM?{2k`Dl!F$VzHz5xKtmjBSvZ@;naGF`<>x66b6aR#^n z908XB8US~IJwTGqNdv9}ZUAHtDF77!ePA8^DJ!a92^{{&v9MgWWT`5&cXf*gn^MBhw13aqeoA&pE-4g z{r{K_{{XO@0Q52e7#RcrM_3pbSr`t#0Qi2DwIhrSzl!3&)A6In7@1BSIr&R%bRNLK z_{(wK?i03OJe_?<}WGBpCp#kp-X( z*#A`^{>AB7K{8CS;2a95!Ri__;xHQ7f3I`fKp-sIW3h5Iy_CCbMNzd*luw3J!RENP ztLsJKkb9z2f|q!>>y6TmYP06`PVGJ!EMhL692kn)YTnGA8X+3^@q|os+|k@o3Yp90 zdRRFU`zAAR1~LHO%z`qAO>b}F6Raa?f%5-UKQm@3{*QS7p+2h5TKCU*+5e&b|1fMV z;cWVo(RT55&9q&zbxcct2(NZJhf;0ASviiUw1|3UW(7e0kEdJ3<-1P3xp zTGyWG9=D)+pPxUw8 z)5NuZ#{2&Ttk6)*y&BJP6ESokbGz!>r9%Kqyd-S}JCa3i9-2jGw=ZbDsZ;lqt^)_{ zvq$gfAkFKi&fl7GT(Cnw=@Ynp#U76NSYK zVWQevfjz2`2vJcBd3chq-c6P^dzt~3>*#3QA>ctd$TW0S0~Ytbb5o?YFL28==7x6%))f zjH`nkJ6b(EUVn#&-kBah3WqBa<1BF4?EL&F+_rFHOCN!)2x^Ks>!~A3wnW`q^CbwNRb#NQ~96Nu5*aluG@& zkB8TM;tkPFBVW}ZqqzK-zUbPRH*T+8Q!tZNauD5I6*VH|L5(F41G9J&o+qlhQT)!x zK3)x9MwQta8NzZ};6yd@)b}nTtxv<1{vEBRe4Plf13Sa}WslDR;#s z&m>)Gn45b5pU6!wsFqdSZQty7Y`%Haz=3h9`Q)Hc0lYvc+fd?7)x#v==MF_VZ}NJo z_ijSQ$_}uV3s*l*3?|KoQoP=z0%x+*J)kc#$aedk&Pw>Z+oSBwsFNROs^PZ+-X84N-iJCM=lxLn;wkuJSc}e zC^fLmg2~E1O67)680)(3n)))Uc*2|I!O9;e&_ctzG!=_X@y3~1f}cPYA&T|p*PJtVRYeO#(1ihbb_5L^Th+)61-iNG%T%Y|A#OWi-|o;|2M zTR&-bV&l@vuF=84%YFIRXTJQ_M!Mq5bZXLyiDuVg=^?->Mo(@XSHCUYB&b>^*`{nJ zR)3xu%+qsRJXyw2sv)lK`=(U7NrfX+{DrH!(w*Lr((3Kjr_?)tgYWAeqfP-1)pS1$ zl-k32O=tI6Uv9{lpmvt^;EMM|()^{OarLT)fZ_a+S9_fgmAwr^+3vK=I>bint~e9? zyx|1q=tIDnrs`Lqi&2Bzq(Q985rMGJE+1eoJL(KJ9o2t){KGCx*o|iQa_ju*4}W9H z89|HTk{S1VYg^jS@OtkJ7L3KmZ7u6yUdmx9q`rlFVF zy+GNSmtE=xt@p+o6Jj1swQze>n1+F`^ZcEiX9?}2v&G%Yo7Xnc$~TkshDn((dPgUg zeNCq`cN2EIFTMLJx(L&2c*M7_zg^j;G>_SnyKq`>yeM^E{}6EG;(sY+K>dyR4B2!3 z9lGipJ_oX>aH%312z+UnH47kFO%6A`3nR@XPiZJ$V8C=*s} zh#WL9(gT+8eG;^X!saoUYM?LB?cr{QGzA~s%+cwAnF`nVES8o}gtdUTK z(T5&K&-1%}JD54@lY9jZH<+qF0xh0144n)XB+rUzvU$4ncWUx47d%k5Y-UU<$Z>?6!MdHtE9RT`knun!l@euBPTd^}J*Tm)aU+ zmKv*SZ2MmSbe00Nke#@jyKRy_B-p<#d8S?Jlxr~lb5VDZv_Bz?Tco{_Trpf%TBU=D z^9T0`*F$~pWh>QM3oW6hT^?`E%v!P;Isg2ujljFlBJeg@QkOF+ggGXn3)#v0n2K&P znYAhrYUth=u4)X4nKvGF-89za(AhLzO-{Q_^f=P#>N*XtWX9lq5)7^Q?pmvW6jqE% z6kG)DhY)Le9kfE`4+{NRwF4-tLx7(2e70CYY{${KIVfX}g~r3EYjHN4M+B<4V%6PJ z3E;(Qw+J!aYV`FOy}DM1vVM|EfPF}#LxLW8?-w&bFQE3|H;`#(j9s}%X30p`_DuzFf zDW7pMqnE1QGWB5)-zzaB?@um8;bSfQN%v?GLiacFgFSkBH5dIC%PL`HB{X!z$q^UN zR}HPWHj#tMxVo^66TBoyYp;_6nXDMEGE9Ha@N}QGmM4ivqS|5AU3_r&Ih9__L|!a` zEpz=`kb%%dSKK=9yDp!D43|YCSw+7Z&AIfO7CTJ_hFbbVBGuoC{x@()biOiA%TY7d zK>Vz=G`ay^6}+v$GM2<$^~j}wj9>E7;<1@Rd#&D}1ni|d+qlo2x%HgA!j|TXTjByc zL?U9YI$5;6y!qqjr_W88)b)iAoB&nHkiHAPp$Sd_?sDeBa1$^i5mJ{IYof>0)@YV0 z4t%viXiE=;Z>Nl8Zr*Lj4n)S-vH{(1n+F$Zy2ey(1V_}!0wBIoPxK-Vkt*N~)xY{6ea5NoM_{9=7rI$ex^BWIj3h zb^JehHLXE@*b1bln|&^`YM|Xc%Y~8ZolSQ1o&}ma<$qj@1iHQcks9x36O?~l?!%J; z$d&$DDADpwe33;M*y zy}z1WMT}GX)}gega0c^Z35KhwfXp5FdNKpO33vy6i@T5r zWz}GnEoL-zoUF6+E#~Xr2xesIt=_m{$}vh!r2fccOgtq;C4j3Z(d-eN=#YBoxiHXIz=*kWQ2?Z&gf$fwgToWR7Pl6b82srmud~icf zC9HYzJXM0Dd$TPN!{t{iU*EyyGjO_vUikmt>=a`B&1R`nYCdtU)QD^7Dq~Y4%Ry_a z#t=nj9f2FpNVcLv2+h6t7N>rI?TS;?k8;_XG9EGv?(*IO!}GYA`Usup@IkSqzb@mE ztVz|Y8cQ%#7D5X}w_*c#BV80jIO_IorKzqLN|%lw$y*FoT*lH^*fK8zzxQC)quGmlfkk)1C*L$)!%pH*)zAFg9*f06D8)(4 zhX5pgN|7Y*ov%a4UEKPh<5MX;r=rJx4rP(dkBKq4fx%5o)wOmxKvx({kH7PFpLz9j z*xRUd==A%_Es4Gq7{xkL%P94mtvGNCWeQ?hblnCnOGQtq1au!v?|JCYKLIJ~czRP>L=b^M0+kJsjBhoSN&ybuqWXu*cXk7;4hl{He^Q(MB`6juA{%s#eAy~~3W!RvfXs#W`9dfUMmI?>+dIjY!mfz$Q1LqElLa6Y#wD@136o9J>G zlO0p&P40@Cj#d5Pnh5qDVD=Q~3!Rq`t8rZ(ZTBl~OAC!At@n(x-$%rIK+WK+?f#bTqftGQ3Np-3ooHGx% z$56WmBvG`nk`cU7>l2@AhC4hVLWzc-S=TiL?O{Jb!Lu$y6ZrVfCwd7Tj5@*@W){jxSSzgxH{c%m=g3?0av8<-uM zC~L*0`~hYIKDx*0up8wKNH1teU?si zaWQ4r4&gy$U!AgEE8eM3Bjl*h(q3DdX^HZCd6Hwvuzv52LYJaci7t$<9Dj$Tds&rG zsS3O(JL!2;X<1G2se#OqQC%xyB_sLi%Q1`P2cJI_6deL?bYU0!`o`mi1V0j^uD#O| z%rPlVZz1Vy$r_s$+fZ<+oEe+y-&0cG5}Tc>zEW^s2NNxY z$UmZzxM}Kw=ZWt(!*jLrnR*}vic?T==Gf_sdTNRZ!IFD*5>fNoFKae;ghoB{-fTa1 zj;$oPG-X>c_;E9D3{Er9%pu>~!I<8jbU~M)aLZ?D|lc+U{O%}GPw-zdLDMIbf(fL^dxTZj1JbIx)fjE&O`fF*gQdQ)4vTE8)a(EAzqBmOASGRMb&r z++0(hAvwNy^N_v?}$UTlp$W;$o(C$J>wm)s8rS(#5W^+N{dgz@eIgX z#@%)B^fGAPHMIlx3R?;c;A#nj6&XXx>}90xUXq^XfqZ~t6y8UY0w2Jo68t8J4BJz2 z;46zV%Ftip@|1-H!3w9y_2R@AyC__{Szg&@Igg1YR&~8J^{zsX&kkdd9^hG z+qM`V$2_jSP)4MhdRXc(e}h(aVREPEmyE!)lQz@}ui`}0!g-2>kDum=tMIkA`rUU% zPGD!U?~ldQIT64D3{y&TwrPt~U40XK$Jx>?^6Sj1D;DK068O(Z=U0ufe;bk{ZzPL7 z$P8B(CoTANsx|HRzexf?RE9tm2%ZE|54qcjtfhId%QzKP>?1{S7XjuY>NWYSUb2@? zkX$r^%e$>-kWb>cXZ^HEw(^2VKYfKO9?hPqktuVvy+U0c+p&tiH54A(x8~vo`k!3(beVK2Y%w$TPtJl~ zYAIBRhYmWS!WL{*GW|4*$(aIyZXOm!nGSvK(`kAatDKUX2;Ei^>yr5`7>Vv)s>BTtcYPmG6dcX7 zEwYQo5)C1u*Qn;B8n@pFuy-yK))n3ZXYD<(3Fchy;AfpE^W zAogs{NN}p;qFIhQgj>$9%CsP)!Y13hJ6-kjn%6#FmJd>8Fj%Ttn1Y|&7_M+#^fCO> z`E$DRpu1Y2GD2=JIIYWI&}WsiS<-75ZZ?6V*h+oI=AdJ*@l7qKyvlLD)guALe3eS{ zyWMmplwjp#j>&D8h)pSN{X$J^^Dm1}jvfQ?RZyNyq%kA&E$SiFhu~q#{XjRX{W1%^|uQsDSKSw2eB8Z6hux(=_xO$2HUM}J{@gx5YLxd zcj}1hxY?RN1a1M%m*sVGL~I+lC6Zljp^VT@Tu@@yJ52RYU_lbR3LT(RDfrp#;tK;v zNGWCMM#JJ&imi4lGI~l^$olEe*mqX8fh3!mNjXWs5UAv}W1E=i>|95C3R8?%Bd<#A% zE!x!u8JRf$7rA5M?_S%iL=)ACqJKc3HvD*J3+i)wROFW-o};BPA4u|enebRyPwnwR z2lv+2wpUSK(l!{|(mMiUp1bI#&T=YeT1#vw#3FWmGM06Gy)8k;UYb*CYp_)$-j2Wq z-}};ke>pJu;i7hK9VI86%et5&!5CDyoR#lmSjXgTTHu5OhJGHAD^hA)(MpgzRdE|u z)&IFN2wRXHYtBY zivINQee8G@h-e) zm$hwKH%S4eYJx5#u=bXG6Ewo!;<)9fhdFl{27;int);6s7z-tO6&lb~4T)f`ovX0I zO`=as5DEFXio<{#7wgU8M=N;ew5Zf624YmjxI-&$!($)K0bLBlyTyH=F1(0zkz>U& z4r#*?0;_og8IYqwa@tPaj}b^ro1byFy1+fstgF8shti9iqy$fWBZ!U`lMiVb#KU{N zs;5=ilDGNnMRTNA?hLa-8Kmc9#lU0=7z<1(mHbME4r2Hn`mN90LE+LxuVZht$}>|5P>bg zp(dctX+!OCH8K7IqQELG?8?6Ba)z$fN_nF%_gcK)FH(=zs*OvLyqGD?|07Oj<#?;eLB{O*y0Jr|=#h88N- z+@nVGCGQ`P_C1!k>JZYeZ}bK4zvh(A!wX*#7TXpO{pRQ$xEr!s6{JB>XDiMp_PMP2 z4YfLDBZY=IKk@p$t$x||+6!0NNg7^Xb*tDghLU8%-<6RG7ma}TNIo5^gDi6ui#k=_ zLMa{DgPd9sm1lYf=P!ib_VGd5RG^1+SN1>l9V>InCbJm%2lReD#`tj&rCryZ44ZJ% zbM(uywl>$I896t)McnS-4Krm19|D%o(tIYf$1Wr`bAEv7imLneDm+{19hOR5SSsI` zjjTklC#Ich=*Q2{1Pbzq1N(UZOtTW65lr*_-je5s_#gQNb6D`->yu~vzf~b)c|CG$#YhqZz zcisD%TT7Tp0 z4q)Y2vH5gS`llV;g!1mYBMwdZF=ix)Lu|0*FPHF8+6>tDo>)kw4dKW{B#a_cdD9of zQ*2U$55Kk`_lj~v$WXm$+@Ja87w>m*>Fn=Y zyb{KPxBPS}bS=GZZ1Xd`o@7%2ZSq&xh`ihd|H`DVJ1kbwr3?y-Y8^PjsXLET$&TUC zJQ+5J04|jZwUVCNi#G@2?~y0_G(I=Isvn)RA(ZeMle!Iho0(XoPYA+pAe-pB!%?JH zQKXkwe>I5QN|nK+@i1q-F_%XIJJeMp)>Ie@vm6t_~#ar2SBdX{(5d;A<#`q*!%U~pN1wnng_BZIS+nO5XIZZa^6*{=d1Te z8qtWqj!fxWcsjnM$$YzZ#F=btPo8Tnaw`SmAfhI79b9ceep-;zmACLx==_j|t9@5z zsca@DH%xtD%J6QE;MDnG>O5+0|B@jiKD0sO9)bg70y0GwAj$XC*|NN*m2`o#)jku+Y)QwJlZE5+Ldof41hh+va!-*CB-pwauW zQ2IhhrI0mg{}|!>)l-q?j5wlc{=l5~Ne$gR1RhwFhiuOPk3itA7 zdS*BrQUvWI@L@ygwPGmAb$}!3w7rsMt25NZ8@X^rP-6gD9@&Z*@M|&~2*k-H0Z9KSD40;PHH#f+I)M{{h zY#$gu?CPG6|7hj%u-t*lpEsk?4di3(ZYUg!jv8i;G1jS^RaJZz&^vD{K!kV^?NJ3W z4H6%LILug_$pX17jlQ5(*oWXw8Io^(gqDbDG{35|f*}-7l zGY2oM6V}8YYYASP`3}!`_QA+_t;J5$#IsRpwssv`(XvlAjRLRQIJ;XyIUvCtJ>9*n zmW)uOah*sZk`UO9r}#it=8sRhBmH+Td6AoWTL}%RXW?wFkBUD!nn-CFcsi=&BKg0V z)(X8_f>Mk3#s>o~Wd-s$TZO6pEP#o@nGE_p%nL2eEY;r zt}_aa_{xt<93Lle_yuls1u^g{O&aJc#hteo^Nh`b@z*&QMxDQIx~ux@hsyUJ0{Se% zKBbia{&J>4y5X;@b`RVy|7x#)L&yF+Y0{n?!@5;{b^if=@6OfnKM$BhU*qoXdMiV7 z!ykA(kkkJS&@VBcYf>ZG79F-|w4@s9d_A)Yt9&xQ@jst>_;!2e{i;Vo$e_WN`}GT% zt(I~F?+=)6{#Qaq@xMv$|86VbH-HRzv*(&E!gkvAwr>9Y>5;8)pz}~>z0^VArM02! za=*d*hxmVUJ!lsAz<9bQQOp%-bO@MFK12E&%>RC3wD`9ZqkJ?z&GAGU8BVwqImQsR zmsl2pxb*#w^dEndrc8+YHn!!UGqb7Z8)UtDGGB04K8#1}4qK7`Hiy!ba2k7!@| zq}2Ae!W2%NJ-MvWy`}Ky=3MQ&&aePoVv0Xns5=Dqs$X>KMwb8GHy>z^N-d4KP6c;l zq&QG{(;wBD-~u?f;@-dgok8^4HR@u?q?b6?&;+uYuoVzBCw<%EPd*JQ_Fx{y;l#SR zid`JN669u|uQqzHFJERQcQd--yX z@y{D2O-<_{OsYWGTi)^KdfXK=(yn5A=J0+Z;Ro#y(5w(`*mwxIR~vmrMxwWANUK^v7&a1&4st}OWhZY)PLJEw{mmT zOh9#^8*71wfNu63bU|z#f*F(aBrm4)Mredi?`CSw&t{|3Z`;1aUE?kc;!Vv)8aYCQ zZjHZbkScwcbkwAYWbU>uwFQ#dT^i{QYrbTdZ;{u&er)vj{xI}bTVuujX66d=;ELiB zqd~*9wugZFJKXy#TsFRPM=I32(8i>Z{2f6=`cJtqyk=flTb^*(0o{9fmrlSy;f7xy)3UvUzP&yK&;BvFD|xvK-)KkCD9#T{dwmbzJx z8v4@G8KSp)Zt2CYRCbl0&q1p-;C_whUc@K3!mieqJx}=`qH$4H9M`!Wyc@ zx>2mp0)ZVV)iV-UuTw9U1BcC!Qbh%o0F0CO2W8ythu+x7JA+z@z<`7ufYzMP&_X0h z0~(>j-Z#TxD+%^QTCza{aJ^a`cC^e_i2NTQ%sdMV_kdW*P|=0>gN#M^9cH z+a-4^zwbM0J9ljc$Ep>h71mIC^UJe>i65r7zZQvIJ{y5(^k{+`Z&0koeT?z#r^X!PbqWrto6z8ZvoCx{og9j_@123%&N``(>}UGyeo;xSbYV#S zAf>?hdvTuL&*qy+ktX@b;z6T5O;5?3-kvJ+$0f9dc#xmBD4}g!4?C|83bVu7{wr3t z|GIfG$7p#DqF%cn0(=t#1Y}3HF77%cJ;NS2yPvsL>&z;v-S!CMMp>5QtGth2Dj1)+ zxiY4A2(V<~skf+QPG&@;qyXigd=D2vgpv??htPZd;{AN*dEP(Hd(L&P>;0}X$)3Gt&zd#ADQjl-+B0XvXA^+y&os0( z05miJ01fpAIGdsw)_(fb>cvZa4ejUZ|B7e>P~Exv007+C!|kQ!lUv3nrnfFk{42$u zxYsuBu7B?Tjibul9sDC50Kkj?H$4AOk1yHUx!X_~)~T1E8#Qw(SQaYH;_xr}?jPFv zU$oL6+S|j`gUa*b5A6ni`IJiAQt7)6|An^xFSL!T+aLZfR30Tb%=1rPf7~C$40g^C z5cPhMdT|5X0WSfXfG2;-PYqL@3lad3+W`RR{`$`|>tq0+;u8SCLHy4%!At<)+D8DO zyyrj9{!=Heuiak%Bkmm4r?s~S05dEXS>eC%)-jb%5aUHgN=pb1`8|8A0ae!)HLVlFVWLqVqv_( z$nyW$oizZMF4CgU0qAJ>0kljsbWAj7O#mKhb<@(&)BIm$rQR=IxD-OmQjh7G z`3+t_cz>QH_WP6KpMC4B)^CxMXG8!49krB9bW8wM0Pg<3MgIRt_(VAkK>FdL9S?>V zrwjATdbJ^}_e$IZu}}|aQ4s0iYEo#SgN1LOxtRAS@$m{>px7rgO1aA4oXxWmx_l3CMrjYK4(|4V3di0d_vgFoZ9A!lXM@x zrR3ReOVgVBkgYz9aiWivmC3+c9r#~ro#A1V1a{uf!YoWZ)RAoI$v9ZuyR(URFUrg! z*P(n3yv98(9bf~qOZJ9ni~&RjqPKR7^W8rJ=+qZN=_bv>6+d53IE>P_a>&IAN%?t} z)91zZtoI$3X(|TI`+Ztn9AG%4jwbX^n-YFuz9^h@a*sHI~Qyy)CvwVBQ$Tvs5LDbju|PYf3D4_jdgY z#(lPs4G}BMVLN)?0+OCSuKU@rJS8RfCinbrf;$ntZ5dXV6_ozse+&O7hyM}pVs-0k zxDD|$^58tgBW|KISJo*%FvhOgm75v8b-uaeG|s8K4Yt)5hxUa9Mo#7~A@>8@4K? z_3=Z;>Ck6MwdRCm&YC66Qib}6_XcluX%T&!D`~YtJFwKq)Wj$bjaMkSpcW705yd2T zg?EwD$543G?o>T(TZzPdIsa8>8AJrF^HCiIiY=4+u>owkVA-!Fa zU4l#TsfaU>0IFvk=F^i04&-*LE%(XzJtW5J>9ZSkbh}LjKQB{hG@-3?#f& zZYn-1UWtP)W_*IGF;+lid|7Qmf?qa~JV#iIw9@LAEN3=TPi345RRnC6U1quIVVKC$ z(hW**Aa}%2vmSX@9lW}8aYy&q*)d^%wYL)3`e@bFOe8}^-^S0Pke6Qdo=3r<55Dtv zvm1X9inC6D#mi?g@wIpD9A_&E%<6GqL9wam&-ryh{o$*qLQT~_01a`<=@2UpZeD(| zWgqyogZ|DPC(-0EJJ*tS4_I8~>u`!%6uX^8{J_3!V0mRgo>S|fk>|jd9^Eo7HnazD zMA=4yT_KD&zxbt8px#lCr&sLqsklM$i zr49E}=bLq|+_Fipt~4lAQ-^GB%m1UOG^{>XhPvu43R9wF%uE<ce(OD?5OgBB)4dcXby(J6D448CXii%M5~5we2%^Yd%RNAb>-uMIU9p}WG3 zb5;ezmYHLyV*ga{RS^=QL=EH;)u9-q#$NFVZnHBX{v7nk|wI~wp5AYrseY`<~3V05+Pnt+bO!3nRw9fX{`ScKyi1vaJz-s!2 z^P;(=wb4BP4^Be3g$x$sx}>F?L-IG1&2I?;K4!()Pp_9(*i~G!Pj34R zaIqFNd%&x0^87E5X`j9E@Cefpw!N<-KT2b7$LiKi;Yeqp2ja>VZxV(s7KUen#yw=&;M{tmwKSL0KTi+Z)G; zsn@OX|F#zO+%)epevGB~uyznT^cIQCXg#ofiwBPXitxjVA-Y@ht}5m(cC7`_We;|3 zMW}GxOYDByzQuJ_$2zY3u7r%ADO>Sh?t4J?1Xd8TBAJzy-FZvk?a!uDgX|RLJd@~a zN*>D?+C`Q@NgkQIfb$vv zpXB!{tK^LSsBGSzs|GKX_|vCz-c@;{ANEvZs8ETA)p))V(=~^rYO&5hX9Bl%|A>!F z`Lc3HsA>wUhc3)(PTnAH5HqNvW_fDqg+96%`edq5cCr+t`bzXCQ3;cS~ zF2uB|TB)Z>A%z7JP+XGhRd1_z61zmH$I(fQ6sDTa>~ne2D$V_b{Q!gf6q2r*nQ^yj zpt^?1q9 zWoVYWEnAKC^erCc&vnV75Rw@-bQ$u;4F??j;{$hRN9UGylU~UO11c7!Rj%OdLdJvO zW42frHpMfFP2POK&$&>emZwiuA=7__pI3783xARUNUSyGudPT$!g zyBmdak(FkY;>Qp>H%a8NqJTyLECx=ljz{L{*uubFG}4ZCe}oR<7((YgGU91qo$Wkah+@Mazhd*88%AR2k%_>TJMS@xZ1BP7)cD zeGqN>k~n0zu%{;fo;-vVq8MvYA9N)A#NO$uxIyZo5Ip*2bs~Drwy>g!vOBWbuq(T7 z{_)|;8DPqt9v^ixUcVGUP`MpvlaL^ulD=4@v|@y*PzNu~SoDGFGkMKEbL~0>QFv;E zzLOT=2XMylZAiF!{X0aQjN9{`*A&BQ(uVg<^8YG4UTkBH>zJhwdeo zB#7FkcpAwj$`t8V*zA&=qsT_++fii~AcdxQqRC%&YP1m;J}4fk9Q ztMzh2QdVHr$9y`{Lx3sSm#&4bKP)!#AV!7oxMliZrq7_?M)xXxN5>+oK+Y_InAr{Ci-#F*Z8k43Hon z$&d7aV!Xyj9x1mxE%aObrIKsyafk^jcaURKH~i}4o|CO5nCn+$85UlWd$dxGL`wdZ z{@mZz#-Ey{{-4JxuxdS)d(^lW=>`YOaH?ua!|UA@v!3q}0m&cwQa@Ycdw zRL9opT(5X}cz#@0x4V178Wu2S5z8DrK`RZ-t+xAC#CiBS&+stcrg_b4L>QarA?}_X zw22f|_RAe(tQ~9rvJ)9r3yu7O2>t4ar<6Vg!9MqP{dkmqHL9b@f61IuVG-{^sz!Ur zo32JKCN5;?_xh?bYOfc|1`UeMEs`Vo&j6E67eMjLSL@dEFh5u&a%^{N#;S~sh@Z-nP`0e9THKQ}YT<#fw5}%NvHC}dpifh@v66$Q% zbmXCeej2VW+1p zqv>;00j9h2Cg(#KgN|Y%Ip2i!!-GbP=kuCgxp~__TX&(+LgPLeux0I1y-U;6z&ja+ z&fMn9b*ecaGjRxHS2&O%^43KeoU8glW6D7FMBU>;-pQcL z$&~xQ-WC2w7XE*PNRP;lJJ*5!*jcFVyyTSC^EF~K@_>ym+9I=L)Va8v9*xozQ{sCd zMy#Q(43K%{UWWoXUpd8ceCBRC8`g0_#H<6lTqHvi0>N?clTtz_C4J^!(r98zo%uY5 z4A$;d+wLnz9Jt%JFH{jZ#`%%n0;#C#_+g9ROFg+9p~`2Htll=MD z-cKhThvI`OjTgj<4Abp0sq+wS>*ZFU><7PG%J{J2#5tvoZTDOCyAh_F5e3es$9IOS zQE0K;Pkhg?v#}ViiUl-!W})uoBWrFY$3&U{v)}R|#`lIGQM=_2 zIOww4{|@kf7>L_jzC%B`OAICS6~YzmOnu|4G~D?&E0yx$)e}CmX8``Pajc=j(;N82@AK9gm91=i#kRmR<8`yrzHI??RI|m`PZ%vgwRF#B>I{ zkROrW!L1ajyJ2260;{krUWWIFjNEhY13^Gi8+aAzk!Tx;3>2pz)S3-=>zr7nI9rY$ zcg=|?4@Jv9RhC}SY!%~1h?x_pbvY&XV@~o-`;M#ftc!x{DC16;)^_kxd>-Ckc^=*D%I$I? zuzh$ZI;3Mc?FnJ9kWtZkTdi|!ZzauW1XK@T@J(9Z$~8HmNZ#bV^s0u;3}0GW5D17S zRVPD~jVdbtYT&gW$~8`i7j%B?(n9f&Eq^KV9=E;y#kMd{Zy~k~kp*fC*s2w4#Hqw7 zz$!R!wl7+$TS!p?LWYvN=ozwd76G8-bs_7AO40+3DdV&Z9usDVx^a!R^fSPaJhSKS zz!#$acXox}QaisFlcuzdLajusqf^Wh9@^#%vS=gM%Q0x#U`uQ3l^004vbSasyKC0%P5{|4BmKP z449l>UF7|h#CsuH{abtDgVsl*DbGBz14l$i3rSBRtE6m6K~}G*9fS=OGF|!041c@i zg=O6MOJau_HV;-uYrm)U+twl#vG;xK-G!%~4aL=Sn$Gp3Fqx(0Ssd$KERdxuVEbrq zQyJuS^vu&_?#o2V_lY6#CfLE1>4kO)7Ffc{ksg?=aGfT`n|Do+Rm*-qac=YA@}t_H8-rkr9RG2b!f{(S zQ8a}s$=2$IU&2CIPEj!>APQot&- zR`G<`^X|$GdKXEU-GxY9Xu#Zf*qhqouPWle{4GS#EjG2wL-~|1pG?!~bh9;GhJKmH zf?OOw^Y~lOZa6juOr8M-uH2`0kc5H9D(-EC9D}A+VGwBPW}3o#p8As<{{+?dnI=EP z=O06k0j)TLz16Tf#*$NX6s#uiXy`CNco zW37+-fDp`m#e!bH2zK?m&))@(%2hF6r7edU)iz%H^o-{pJQDxNs!;nkXVmV^FTp3R z9h{O*r&%YV(|mqevw-sZ-~=P5y1cG!C+K3lr+o8jw>}hEt%FnGPU|j zJNEvC6LAO=d3Y^KnBo}pVjkp`zjOkEyo|pdiIklE;vc*RD4zhMcacuCr|t~;m)h_9x7YMTIbB(gQSoBe#d0r#_{$S1tTQlM%uKYvd^Zc?Hk!A*m{;ef6bl3U|r+> zph2q$`lgUH_A+lXl}@R$lY;u|v#mr(!Aex2`Da!)B)n_MXtjRjy(=L9+Vdilr0vFt zJc`jV2)g@qS>ZHUifRTp14Pg_gruPrpL^!gtHI9zTuH&)9?lkZM=$mz9m{b=3$yQ4 z9VHv++h`F+Wj$u>+d;oF%8aI{fmSo4?M7>TqUy<`%15AQyWgDLPQR^aA3deMpCWgY zsG;Nn$3LNbc8SXlmi=Uxqr1gPTwxSzqghH7U3qu_1Yx%bbpl*dD&f?5cl z63F)WuRFyU>a-^#3AL;`d0Q%ch`lf2!O~ERQlRYo?#fvD(TyR-6kRzKqZ$IiuPiQ# zc={XQ=9g=~s}c;jjBw1BN!#|l8B5>ZRXuvyzx&;(|1`=p0MFjw&Ww{aXUdN>vV*9t z9i)!c|Dx{-NjAbgG-p3Ou-D&l4`NoQJR71v1JsQik5Tm&K`$i4#y zHc#5tAk6L3xKHGoIuE3|eEMx7N?}IksRhysjr;1C?B*?-o&ks^@w+k9{$QRdlH(8Z zEIKUA7_~Z1p_Zpy@}xFR)iHRJzB42_2`6mEe!AtjcOyV}41v=$eQk|fwo8V1&jsFC zJHARg^IlJfSXi)p25_*^Cpv9JMH2H0#YRcOKa!?t+yJ?=6obK>!;Xxy`XjCh3V86F zt)_gUHfTo@X;J^a!Q543*53$+IK=bm6D9Ur2o#IK;AL2#>YJh#+9?{nv6R8GA`iA} zu97XlNSu@DO@6V(f(JlF(gLe4wFp#}7gvZs3;4_mM2-x`If%8G!{(j3##PyXR}$w~ zjuwlYIL`p1Fw1H078PpWGp*-T{6$=O=t8a)i+8D+5Z`9KeF2(SN@$(D)k0x@7+9Hw zOlaThYuTY_4X&|CI&B5iUt}A;9Jk$v`{`dCX@$ty*8^Lii2ZHEq1TDYylJbZH}Sz# zL0?#v7xR|&BQKTthh|OlQR|(@9UK1>|H`XzBJ?PH)mr?RMLGkof0! z(M%wu0>K#JugvYH{7~25l5!_=%ObdywvWarC^@=b(kssFm|(Th4JxY$qo*osoGD{% zB!tS)`_p7QFw6Y0xk0BkGW_Ki65^-na$5NbX(dk>e{M?Izw7~(9W?abcA_vi>kJSs z?ES{0&H*bgwQ&Y$GU6B5=128xpiP$zQ~_)%m!}R@GX6XCY7|9i?}qoc%i(k)ohvMk}bMeRq!o!xio4Mm8gH8XFtRiU!!u96IiyS1fz{dqn(JTDp65yWX7w1mLfbVHeX*%awII@uI%{BtPt#O zIG|OZMm0;3k1w9{nOgpmFU^bm%+y>s9%rFZEnub%i(+lRteVBUt%pX}c5j?yU6@L} zu;hKc=(FHZ*WoN*C^$uoktGasS<(`sTPd)Jw?|TD<$2Ogea#9|J|t~iN(MSfu;NPa zCjGXHoI!!zdNZ>wV$3d-$P2V`uTRp6>|vnd@KAN1_YBCF zE7uPvx4YP=(%#O z7J`nBDX@&%>S=DXqH~;lo;ZSmM7k-eMoBNG@zGF;)}C%IQ;dtd(W7WA{y^9{=D!GF(^^&bXy-U(k zD!IEc$NJF@fb&dxm$re_U~sK)i*$U51!JPD6cPSr)-tldbyi5a5;c#M3A7Oa9I$0~s6yIp*H7!*o*6 z(Wz8=1IJA{wm(5LRnmS-+`7njx0X4HbTiY)X=p9`+Sbhl8eaT6i;MdGjC^Y)0XA-~ z*S?D!K1QDG>VMI?buL(sw(O2d-nMF;_C^%wUH2t@)Xs%Jc?JVs2~Ma!0WJBvoQ!60 zjHX_qV%8VOoO)z%^t@uYAWX~2e9zf>kzC0Hf)fu6f)3@Tg#ii|GeAV(NvUe9;9Xi3 zVL%Es+#%_qzZp6+!S$GV$@B4Q`00zkZZXhopIhK7Y_2CWAAd)?vS&K($W{cM0Sq=n zsfAU|RHX94;U}kAvy7nkjOX-GxQe~S(-+i?8K^)+19b~C>UuUqQ4QD+c>~0uqnYu$ z4)4xi(mOAGo7#Hy<3js$sD`nR)HuRXWu1K5BaX|KbkbKHSy%B+-VgRR2f+M+Gx0WR z6Jn3zm*CEO>Y5!l_i5hY-;wG5$?GuYD^QPan~5rHd_3>qsq(w~lDz_%s2!Q|pd-}| zuK4#B)Rx8-?X}m*rvo^I7ub6}4pWHr-m=tgbH0oUqCA7Fv(dE#_?hu&m5O7qZ*TSw zrc)~2-$NZVH@ap3Kihw@eG(q%@7~MaZ<(2he}Z3fv^(iMl|nRdsLKiWHXWvzsQL%D z$O_Y&a%O;Z_}29HTVR0KY7-Stq z?960%nJV&TGf;P9-1TpQ9Heji5HX!T;m~0B;=Gt1oAi5?-2HjP^gd(zud%=K2FAIz zJc4PyzT>Esv*;J3vReK#pm0b>QTc`t*82?UfEJNPwm9K|@ta$=RxG2b4;DoO^`v5e zjT_Iu0b0TvHRAKUQf*Y6K2IU)LTm5pqXr+&QBCG?k2uB`;Qd8r1y0?r$Djan1xMI5 z&DOjhB=fY6#I_)j*hjof(tNu3>T|6DBiEXN2hENhL9uVwWEwa0Jpk+N-%~s?vSI$+ ze;t?-+LS#pI-sn)WW; zZz05#Bl!!FdG~8D3qg9BRd2-^zk2snWZ@mJ|GaOc@S+=m8l(f`S6-{?l-Iv)UFUsA zg9lCeobaO?Z7S3-eUPuQRLgHeAfB^K*b^PD>Qp+RcD`mxJ?{wj*qTnR zW}i&f)ARsbgI<$B+bT(WiO1*brQnNa0N31;*?O98z*__?%Nj=1;!IbIQlYbY1QI?W zN{-%c0|Y43>t*cu)Vqe(eSELFVsr9~@70CQ-oNus;#J*3XF_H6GzN~eLf6laX`JS$ zY!r$5ULT;zZOQy+%KOKo$x4ZV;@YC2xg8M*^5I_!RZf)cEF>J{#I8C!t`+IVYj5 zRW_Jk*?7$c_0?Y$cO&bioLSS-GH=PCS?q45C8nyn0ox7P^@U_=HBhnBN_xfHy|m%J zW5@JsCt4?aV32~ZH9cVZtq(7F`Qk4G4q14!u{I>ho4+RxFG?9L)DkAMlfXv}4nfz} z-;`~IrYVI}RhVyQnV#WihQ4{-BPr(ovEiEJaph zH*0vtuoA1=XW-xWD8eG+*r>VbYHE3FK!K)%3mj&&*&N6A z=(cC^sR>qyK{a|0F4O2Dr}AWRf1{;AQeH#?CGU6&86| zbbi)l9m({e#L$-t$1%PkSPHf18=g#T^uW`v~g*Gn;8xs<=gm7LOV-JPye z@U=@lIsxlZ^!XRi-pK$RQQi6*!M}>CV)EuHA+0iWtg4?Z^Bud{1>PE2wJ%p^mIp2^Y9tNa=jcqp zX!&ucPEz+}tK^O`7xjJneenuXP~h&`u9j7dBkPCDzka-`Gu{|}BZ(h70+8DWVc6{_FSmj)ePZnu z@t(VYR8LvXXOd5;I|VqDnfTSOeT?OrJ{Y{mCB^2c|)Laln=;Q}ni6c*QdPqtdHU=42ZFir!gY92P{&waEXw$O#F ze&Nvx@+8iFR5-5SRB*ae#9jSlCt6f1z2}->aRos>{pN4KEI+hb8mN}m(d+?D;rfA9#`+>Ow`S?8 z1GHqQ(=|GaJ2NgZY+_U6``y8;N9~gmR3yoz;7tNc&iR zxK8b}O)0B7@=+eB7C8i~;Odb&c6xC?#s9MH6=_3KRLNX|Yt@_KBIERkmP&M?`iK|l zJAUJBIC3#P!x-kMjN~X)WK>ovA13vvHRWq$f6)YXt2?Oc*`uAqt3-JdNsVcD*dR|6 zZRHaUQZ2==cO}k^3+HKxnD4q(u0&k4hSq}e9#+ud-{|EZA8eO&eiOgY)%>Gip{$Pi zmz$6LOypv@?%ulpp3iTL1H%=J-0zJj^n{U%yI5QJVMfOqi@&!| z5BtxksSh_NlNfSp4Dm!X#3}W!!H(hdd|c^DwY-6s!l@{#&WAmXE%$FL;T*TQSlp1A z7&66FfB(*tE&j?PYsuvl)XVjSe28!BQ7a7hSQnXDi>7XyDNpB-h< zL9i(&)hanux9nKB9A1JCcvfUfU|O^4#4mg#b7J$f1NcE^UWL-oXI?UAfPlgd zw|hh4`Nag{UFhcJFnJLZX=uc7R_NW~F*}kc7_$5PVfb727@K3EKF8TW@!yQ<-&^sV z$6pjeT-3F}!!>lipd>q+1g23ErR4qBp>XrKGk|(_@zjq`JijPu!?|^Syj3%QYihMk zzyIx^AT`=~;O7&+sAtD_BT2^zcy;8fwTOL()D`}53vGRKVuW9tJa?g?R+PF6+8T+> zox9fUdgPThx(_XxPddU1YuYzCEr>}ZLKjhdJWs8WP|Zo}r7-l6$Hs~)Os3OD)bqgg z+47)8_~?C7uD`-41Ua0XrId$<|!!tWjZ(N4PizhDGWojupvJ ztXki~>P}Fcui37cJ#?-ZPBdA>K8qN{HygAyCu5^5XVo=DMTOCuM@kPv{R6G8mZJ|I zP##5KpWDTYv-ko*x19zN!RKQI$~D&TN!8Bpw4wxm4}ql{**X*Bd$0IaJT@=OKMg)t zBE{3$UE)%Ha>R4KY#Hna+M*>rw$E=bsfoxLNJNO=3u}Eom-leSJt<4TT2w4;ihS+% zuhUEuSc*Z`zA|hRKZD!PW-Z%}dA|@HAtS|$rc@djn5VIND_{i;G5R=!j;Eo&r`^!a zVofVpRStthimA{5F0A9DjW+};#>!LTQj*DgME`q7sFeR8E!lBaIW9trPuj@LZY06A zxO;hGM%~GLNHW}jRY7F3AXVMrp!{1SVrrkYqjj z_3zziR%rH93ts_yn>dj2e7IB1!UGN|SqYqJZVlpJ9&s0m19 zPko3b!aL*M=TQFubJ?L$t|W&|NX5;7_9w?Fs9v!X#`+J&mF0%iCWDtRoHzXG^WMBu z1b?J3(x0>miPVlQK*`E}C&YZJy8^E@JZ1IUUR_n@S%w7JI@v3}3+P!7_bP*smANKm z)`RF*T!baa-t%kYO&SQ+MUPNt-iOPxj8N;9#+koIonGYzpj_fK8U#3v{&P0uF zg99RIw$U)<*?f-jhB61yOS#*Akl{@0RFO?1Ymj`}p{#HJ&5Tx$t40LN;_?oC5&G%T$5C|=jppfjy+B>;7 zzkp5Q5zSu-Bv>-7>0)B}*rLgWcd<6kl$r5?lcMXqPGDfEaLbNYW8QG0n0j!Pc_E0L zCLY-AC=e}|ZA0DHA5Hf5 zLMl2UA}rsU{Gw%P=gJot56oA+hxfzgZCCGZ1npyiBw0n>bXx>K6tCvgVnQlNT_d z6V{^QU|Wu}-DZpYr2pX=Q6ABwZP3%XWG3LfA<=>04T=fVHOCUTLg{#ujZd9lkZh_t z>z>qIX{cRlSy`hecpG;DlskvG$k4&qADYB{kKeL>#$84Bc0S;RS;BqtmR~Ke^epYU7-C8;h$&&A zTf-uzzkl%yn+aL^(&JhNcFkt?dpIT>8i6rri+1zQyg?aZEd2VyP+?Y!C_fq%`rX0h zL#$CHCuJA1P}@2l*C00241(MtXW|PKt#G zp+nO+KTd?-CDVeo+%Bq?^)CfIXRw;u5~aGGY!Yc&%p><>q=2o%2~wUOVh`$j@hi5a z=f!H26R6{YlV+lc#MRiciipcenwP=K{3x4BYs-0TXJ1{<+Zj()W6C#`IJ-(*uBlQ~ zBBuy6jb0pNK2#ZA&tF-+`PX5*R$^lVuefXTz1Wu)tBx-D?NGyL?IqFtIEy)0p>?^o zfZFVsGyY8P28_j)D$ZKllfOFIldDqiMoxG(K5^eIlWU6NlE zxJqWW)7v4-TZZTTDhp;SZ#lM-EH*ExsTDLvj5+E!0Jkp|$ef zNM8%_&z!A}cMB`UOXs?em~aDkuzT7a7=c>Gt`Xkh+9%W{Ju>%5((jfrPpLrJSblav z#puE^N&uqSqpK(wRqeex15q0$*|!%Qa6T)?6tZ^nTbi&%e~;UgjP-?h-5QkoVv~fi zEYLf7Qr2i&ZR6T&RC>M{y?A%gYSPyx-`!A~X-&`5iAYY0g&5iD_COj3QPacpoznT0 zrc&Cl;hjXzl+^jlW!*aGdb0>O(v2b97_p#aUv5;%-mM!IFE*PR)}Mb6F!jXf=rR!M zXm~m&mAF``>zIn+pB||?OFX6eg$Fxl03Dm-&)~jh*Oc6DAGo?M@F6jJJ^E7@v^s&6 z)qp){OW3KPyVjMuFQNNs_sfogAk_N43>wqdD{offy`%m_)x;;Rnf=o;FV8DGFGz}h z)MLl!R+r4=wP&9}+0YT5&W;x3S{0ho%g~y!!KY*I#qQi#usK)B_WClx^bKlG%X<3( zkb5aS8*Hi7$ErXf`0;T|TDlYrpt-LYNZs8K+v!h#hJfhuxEUG@2?koW)F+^aa0;7f zZw~v23n5*0SMS9jrd|TW$5mL1; zj2jo9n)iS+iwQ+R1fO}#dOzUL9qlY}_J$~_>Dcw$zvq3sHHp(aqR+~`Fz{C)f@GM? zfgTcd9^&7UM3a>?1Sjz(1>C?1LX{2bic$1fG3k@bJ5)-vQLAT&X}FN>dS#TQCpg|t z?=;L8E3_{3)&j%%RM$Ci{#8!p^qwCY8qp!EQQX%2sju88t$f7|>D)~6c=7N~V?@>0 z{*^t0!u_wCle>3Q%2PTj6WPc3dfN7iBgiDSC31Sb?vSmr7gNeo+EQJJ)9y7}obi0j zuTF@Lzsal(B(Oy}6x?5eWHXN$7Jie+g-o);Z1Lwo)AU}d941f2AmcgT)bvG$UJC#k*Sk;;A% z^I=TVBH1$7883Rd=4J8@^bBwzK!Z>K-mHz27w~w4I5nIl1!aHuxw&-j=xWJK;+RKf zRllYQtrkL_V>c_7RjF3%X2=`t>wt+Vv*yoY?NceD%Telvw-UrdBkE!kN`2mNMp-{2 z#D=ZbELWkoN3ApuLe8hyjD=S0@U#59Q|{VVb5HU~gb#qz$yVV`H@d3{A7s?WR&)>#nt1TK7@WlKt+ zSk+s}%m~@k0CLp;eP$ zCapUo%rAP{G_pafbxakx${?2;P2p#NOtycFY>!ojE{%~n8>iHd3-<=x<2z8y<;S9W zzb@^T<9M=tUP^j*59l8QY2Uo+tYeNjZN(pv^wt;GcV(2GwYB7S!x=7yulUHMHT)y& z-`;hdvWK6DOGBV#=XA9)->COps2}ACGI63BF7>=**OKV`mO9u z@kc`bt}+qEe+jV*nT36liI%XN3%D|+MwXwZ8aVSvS7grx4>?P~IrTFQqrfF*?BZn> z4k4mmjS6Z5u&~)V_i~GJLotZV+Cv%k1cQ0wPf7e0=t9Q9XwhNsH4UvMH~u4$x(4q@ z|1Q@59dW*;kxRbPcyP;m1>v&WJ;7KLHNu=AqqmyYQiJO5E^&vKA9(@+jQK6Nn*6Y+ zO@6L-TwDw=-INxb49QkEs*QR|K&TB$72U@fr5nq{-1goyS~r0SGy(*?Db$2l@l%FS z`2gPngzj6V-?^NTyYfyo`d5^=-3wlTuh{ zCSy1=leY$tox)iFGjzikm>s57Q32m%7H|;Y7%bhni@06M#3Z9Oq7<5WH8)*-f(_aQc6H;uMV+hp`FYD|hK4tm?H@XahV;4Q zmiIPRY3l3T&}eb4#E(&+O6qUHwxIoG^C=O zn70ab{~84hwSg9LJh_c6CYV{w=&AXDH+S%5THlOKr`l?>Mvh>8Vrwq$Xp3+1*s17HDC zZTUqu=BQCLV#S%dufj_gkCY&{WHO>#(0>j_gX}D79t+qQc{mvaV`;|5paUURzLYxy6)>Q7Xg!2a)T2KgL|sFwP<^=+^lI3~dUcpc7!L|d zn_K^*9=z0TMlqIq{GK46e>BLmQgU0~<0#Rp=_TE+y2sy8Xz(ZgV|C#c?USoym09Y5 zR#ex9%9y}u@Va#LSFyVtDJrXO(>A}OroJdf9L(5!8XD9iOx;Ts#!Q^SY?U`CV8>-L zs)Dl~37rZIv6P%NUz@s3N>kw*x=|9?c%imz^`CtI<1sxn|54a)7VZ9Z!9L$?d1ZGA zZU1CB2PU#$l0c!_GTlTgQ!ypDC4{@|@SWr+9O5&K&Px2s)8y%ObIU=oS6KUv0sPvu zZPZeRqqn~3UDg~JZqPAM!lu#wwFPS@a|hCN<>76USjU%`Mby~PTF0Q)jE161j@Rj( zmxjAm(3OL_Q1I(LS+wQ-r5xnCvx`_<@)9gU(E$MvhTwC#T|UZl2U#-`(Y$$jdH$FPsPw=|tnzc{B%+%w`= z9!kQ1I?Eh?F#9u{e!Lg#`jh&#D?axZbq-r*AHD8R()-`%PW~z2x!~y-3yEXZ1@XYs zhL+>e0(8WPrc!vd<2b*(Aw+N1Fw4l+inqaG2dusZycJD1a^f+-{bVS9p7UqYz|7t= zFT>j1!&$A@CY;e~&PQt{qQbiIZ4S|#W$p&s_PIxII&d^2zr{;cY!pg>+vYq*$-J{< zhzu!rc?EXfbJQ6!f2E4atl9Gn0Cuhv%9@j@q!nmXa_-8LyZ{z)DdyChpRlI}OR{QqL_y~CQ?x_x2nT|}iwRS=NAK>`8>bkj>f2qE-r>4c{Aj_8(N1Pn-5dJ;mU zmjJQQtCY|}=pa4xPQ1(gp7*%l=Q-c`?sLEM-Sfx2>krmktT|>`nRCoB#~kA~gjahK z8qqR!W18;ujAigkNs`x#+ET#;5AC{7>QlmVX%~kas{U-)*1U{=b+xc!fY+=mA|o{! z)!tcCjqV6*78K0d8um|pw-ipX?cPoZlQrks%33?}S!+r_P@iEc)J#fbI#u}%%5iSe zv%|11CCXKYLrL4L_r|F2%*Iyj`JzpCgiNWNqhSTuL6T$LCoQw4*`T#- z>*}R+m+M2QX@Pp(K7FfR3BelK=DYnZsO$JyBw+*a;2)NC^f4E%k~cf??Qp=smc%|b ziBPKT^e<&RtHv*=sUAlSQUA+fd4G+>II{B*@PbR24vx;O_IJ#23 zvGwNA6ij_lk*!8Or}CViaS_bi0IzgtviGSiFE#JbIW;b_jNV*AGtkQ8K%EJ-byC&5D{ z3xmm9nuYAFpLlTMPjBB(ZpSwoB?LK?x`tCm>@pIJmvW=^@>p%k+qO{cq%o12BC?qI z6yKD>Lt2mNE_aP;lIcf|J7qB0m_#;9QalVSyV3(q9>omGm|~*ml$59^n*mdSBsV?! z!L?Iq_U-7^14!Zi6^_LWC9e-IrG;gE{L$>;HGUeT-|fj1T~(Vrj{$EpIHR0Fk#}D_ zuP;buG_0<3++%6Yfv-)T4KcF!WMzO^-gndqKx^Q`lKXVF@D$V zINdhvV2VLvv30GUl~jaK?&;2gtiso*YZ@d9AZ`l5srT4BJ2gJ`JM-L+rl0<*TOw;a z>SAb($xFP{SzdyjiilfR3Y#m3+f=g?vbKOjS09_ILt8qR=`ZZ(4gK1sOI7 z*nk{eOmEBJRemJ*1`|F;xRzZ?m98dU7$E14)D$fycBGWpNM+fo;T$aH3@Sa$;yhOs zS?!$P*_Z~Z2KEb&>xp>^`&3@IhudMhNxM)CB8aw6DJkh|7hvjO`j@z=Anw5AXVC?$ zTxd^|7y&F?v!1EM$)m(dqqKT`nbM+eIo1k=LI3Qp0NI5T~Ek3$rCp3QxmDm&g9KyvgxPeVt)fh3l5HPm|NH?z@EAh zwE1Q;0L>h>z7!H+#us5Ul|`)ou>NSww(izMT0#FhW~GN7VK$-@tEtQ*?CX$W<{Uv% z6ENs4R3$ivVIpS9Bd02Q?FflA#UH|`N%| zf;sNV&e09zoFZ;i-pDmA&kpGV)zQ6Ns(4*xF#E1PawA{KtU^2Ece#1dc$*3c_Y@nn zl%O}g>o_uO0%gpd$HlW8UG9dt!56&X%VBi^I_8^#o>qcl7Ywc>1z?dqLH?&ko|j{* zuZBI_^=&J*|BzSv8T?Qj@=#sKFgA{L7O;`*Yx7>^-HUCPB)lSXW*Jh{IW=9a>o|nJ zWlvjlblGCWeZ5gPf&?~KWL)_#+&a4$UgyCIECZ||zfHT*iu}b^jE-j?ZC9i!urfz7 zUKvRM_3c|?^JDTJ@aDKN8qWb6b-gpgp3t6W<&D~iG-Ql=&Pe_rdBxuc)B>)>Kf%v} z#Pda$Di+hbE#m-?6a>R`am-!s*!N!?*T{Qb#x>q{cdQt>K)U`y*Cka5Nl8tqqU619 zflqalHNlN%vt|#%bwSzinQ;4rTPdhNoIrv+46rncSqUh7r)GQf@?iM$3n=pk8=YGj zAGWaJBZX`FeJVq`yy|9W9u~P0>%5H8QoS6*)Tb^i>>fM%W*(k`9BG85zfig86xcR9 zJrE$|-v5Ixj%f(Y^2Yn-S_WGjM`!o=G}(E&+u{4okx&y{Ob_lxv7V`G4brpUbahbl zqb_%rT(z%bv>OcObzr5W4DHYj8u$#QY2s@?>phu&~~njyI#|xH^8zrrcGVPi>1J zY996Y=9^D9Jy{1ee6_PDb6pknyDpV&S_!Fy)QCNQvG-^Hv39@Nvv2+VX3}N~8Ya^L zXR)x!x*MIs4{4%l1Zirp{S3UHe2$q>hBRi%!~5C(IUG_~&pMwzIn3GG^Mmd!q3Ce0 zA4PYO+*^m7`fhlCbdaHVrpghnrGf5GIya!3&Fx0!$wr(--$1nznCotQ$g~IfLHCfo zj+MVnon49@GrKey)2V&>=?_y(?KA_^&fPl34q_pyXl?0 zeoZ5jtP0uJha=;!~N}cY&qB`i7?7l;!aI*sUK|n05obUaInx=Xt zyFskkfDA`T9>uJk=*!of*V85D)}doKUlwb=TxMVzG7+t5w((lGjD&YxZYmnfCt(~T zQUNd<2K{#x2X=Pz0_cU2fyXx?F;eAHN=R;QZa*c)75xt{$L)G+6O*H3b`^uvCX&O; zA}A@xasFX*+=qYYaPvKH$NPro3?Bd1EV+oGD7v)^&1QQjx19#yrwmgX%z!NtAlfm& zKL{;k;Pb<~s0La1vD-TKiON@b(6tI2t%BYT@0Ek=c1_vNI(8vS0NWnwUoPt)vF6t9=su?kWl%RrH8O^Mec$z{H}z^~aRfhzv|PrM zd{-QkRtL{ z(&1YFv0On!^6tTiN;&CCPW_cwRc|#CwlA5z5sCkpgKl1$sWEgaFUh)FF4?2C<>I+& zA-J;JwlOuW!Pj$@PLg~HgDN#3JNLKi&lRn#ZOGbNDLX$QHQ)8r!G(&b zIUfkFSbo~Ryc@2aAf24SfP*dl4xrlwdX@)7$FJ=Zj zftz+^WWARA5lr8srd97$!`H54UB)D;e09ZEO4Y#3i&T0waH>9Dte_z|SD@dkd zSjJ0V*(?P|k%qjDt5tTGjSK}#;M(z!03E@co9?<|SWJ3LZItplKZNA37NP1qa7mfH zR>6pR)q%~e5^;M=_|UgvQ~anUa(N@`2VII~k%G`-iE$CLlQhV6WR=7KF>T&fZNaLH z5$9U@$4~W5?YrDMSa6IKYMyTTjqM6l&sXqLjz8d;vW;9WZdpM|9XOjE_FQiLnwaBI z=j?z8L+)HTBkaD_mBF4jc(fp!mBl0Ln?C5FPqtnlUEt6;?OwM2(?VFK-;%W-P#8)n zDbAn%V=_PNG9J43df-T9A0(ORXgoXbXUYAVWqv5Wm9xv$!*jmDY_Pd{4wM<`h?gd* zxhFgln5eb-sQ!bFeMYY#zK3i$=r}Wugm7lV*kQ>%R+AFGodHfw{L!vwWlk&%#$ z-FS6npL)G73wN`Ek=CpF2QbgorSU#1^W3zx=v-NkJYPj*V0!;!?+?`tBjMDRuu?d& z={w%nk&w^--8YZCdKNUTX;vo0xjb|OxBujX(ADYR0(C!4MZ+%~NzOTb5_T}qJS zgrVc>wOT*uEF*j-Y3jKhib*21aX6ubZ+kj@s;%?kCUK6)g zVIm;L!{5&omU~+zed`rQ6J+{o*WYn*l`iB(PV2($2A=a?DeBX?RD@M4%!#|j2K{n3 zf)a;NQT@5T@a2J9it~XQA&hRUSv#OYq9*l=#-{MgXRx1 z*R_u>!kMiaqO=C#+;iIQiOjBhJ8SXV!k&{ylTez#c(BI6a>@cmwmzeQ^Y}L=i=SmY zx;hAWz+8QbG;{FBoqO;2S<`I5rGdKIZOnDCZm)iilnlr7cu2zXG|nX%B;eS@F{Aih+_})?V?h}LVe!I#*qz5 zDz|+6JrZ1~J;yI+ic?x0F5wQy=3C2b^i;;mBtr6yL!(urkY{DPFX%GfM?FnX%>vZS zMQsstcx8Qkvv2Fy9Heg<6wqt{%ZQgtW|ch7y6~o!G$KY{eMhJK+qNn1GfJ9;adZ-L z3redQgRmHeYpDE-=6+St%brvX1#?mo9eR z_+=EbS})_AAb*@74~%xIXU)Ng(d$x*gRq!t>D)sI56rUiso2%$z%&;tgDR(VO)+7A6(HF*weEP;>IhEYW9({RD ze`*0zp0Ap`(mu(uLmdNLK$XX>PxbB;bTwNX7^y_oY3=hEJm{+<^XLZMx=N4K$&Qay zk+-?4>8ACHb`g-3@2?T=C!`L|k=+7CzE+g2^Wydf{EFTh9NBU<&=aeH$;m%kcf-Ltf`2a@=VT$x~lGz5F#Sq?Jc*f?)K4cjA) z_~Wt(t|qlf8?``czEhO-HGWW&a~tHT#^f7eBc^kqWt(h8VT4!6nWp1up~z8$@UKyluxGLv-sY+0 z%Wj^oVr4n739=ICTqszjUs$RnnIvl_Ssz<$ijZS3*p+kEHg~o%H>;#bQBwCltskcm z`xL^0Hu7A8(po6Fh-7-qB;wsZYSM zSH_cn&}jph2X@D^`OmQ_aEpO-`P!%ZH!p!Irmq*9ch=q&qdG<*Zln+d{m*qP7uOug zsjZP2d+&zKI+yk7^*0Uj#Fa=$mABTon-|nQ*?D2mY+VseE|O8POdeb0tu=q1lKIY{ zu>j^K@Gh*j8!1@SUc${C=)vfpn-B+ofn(^9G#)Nkn;j@CZ}n)g!elF@$L|?>nq{O# z5V)y*0ZBu=Pgsay^<2fYyJd5ZKvRXcqSaGj>SmCVR*0OgE?&tzB2}gZJ*{lMx*_qz zZjiZ#S>>&v%?G3HQ(QR^bb-rr{?HlixlOQ4LcS7Gv&1|6T*CmUkQvLELQr+p5CYGU zpX_6B+v^~UswSM0iclI!iEKSj;NJAg%3=Z6^fhVT@hBeQ|0AF|2~~NsO6ASI%+A_T zeHZ!gk*6%YQDiXen2U6z0er-xXrz=J5|@X2ADV2qz2yb-xULX-)o90BZmvpIW5)ej zaW@<*Jq7=n zEi;~N1qB3;3#-bso-3}-(QCVIxw$AK|5oX8-HoCyOVi9m;)u<{&S(g@p!GoysWF6f zRs4_M6amiYf?DDAb{>=IS!BIxz+66AvA)CGqjNhp(%)b<4Y1$obCe~qeuhajPOO$= z8os-|SB5!L%}BtRiv_>grzRGIkASiM#D>UiT<3hUF3jG|f!FCSCL<kCErt?TZcTnbSKM$Q&m%I?|n#|u6!~NATbW#=tD7j&1uh`YPLTs?bspr z=s{G8dBKNjW9-VuzW!eAF&j;`?AMHpSFgrdNz0$wZgb*o!;H%(&TQ6Xvl9AcJagjO z!Y2UsAd3Nwb#)zWBs-{THx>wfyfq2npZ_;la|AK)KBaf}_R+<`vYn=k4) zLG(&n3A)dZ!lRpA28G_nRKo8@73j>JZ|Vvone`=0E14yI=>ImyZfh?qQZX4a1k#S@ z7qTQghO+C`)r<6SHjb~9l26S>uG<$Hi(y%suHd~XrQeKhuH<}ZaRl@)GA#p`EHT0S z`zetDt2RO-k!|(*r*ZKbSCL9W4W_Fao&&z&?l|J9mJ zj|o=49n@mIw5Q2kk6xIGnCGH(q6`C>8v~$an?|ehwbzKw7Z?RM;VN`Tr)LUtUg9#* z8IryMvyX(_!oZ$2&$po#y`4a%E6uaG7|e=aRiChK(*Rya5QG~~#jIk&!a@)$+zm6O zYfp7-Dj$nzN!5`35vXEQG#i6a2f>?Azb}3dSC%+*#K9{pS$M1^o_&ML&1L@~CJPh; zAk{M0;g$3ecIREfJmc-J#L}0TeAf!!c7OZZV;-KeIxjy<2iOg~d4Uhd39ZnuE-9gH zOcGO-#U8k3sAAD)RoDBLLwb8#1a5t|8t&A?fLp4TyWZ}iO#6rbXSC5r;eonm5{khx_+m%Ymxa_38CqraVU6HZ|tCjv@#GA_e zwG$VewounSF**isHx6zM;UzQ*^_|H&|9J(J~wrjhEWYJ|Dc0VCB4&JgBpE! zao8tq?{>6{kXhWKhg+)7>16L#qj|#shWXf2T%ohGYk&zzWkTR_w4nWMT(o6D2w<7g zrP1ITFnhA@!MkZ5rYT!3hq7_V+R~g!9D;)#aMm9wwC2JpuZ7jYlck0ML_lA3|MSv3 zW}NhdpX$2oPEEyveKizH5|I^-c8ok-pUWuM!|16~W;n~3@^T1M<}TSELF~Wdpm37Q zH^LA(6qUg;XCWxV#-rO3k>1`_Yg9T7`zKnR$dov`Q7=O?8TH$$H|^=(DwU! zM_aJyh=Rz;+h$2FPnoGV!uBDWR}`n+KQ;?xd);T>$-=_hGL%m((w)5gOxzihJCz!4 z#Fv-ILh(^BAe>Gj!G@!lR(&$6G}*_WXCxvfHnpKh zd0{#0I;)`JAMxruArEHay!=x7m0kB?{N%q_)7U4yJp*j@cg zJc6+B_Q@L_%F3Q(jF-zOCN$PzW9-|Cf1}u?m{D;YM+4?ME`HQvCcrVQU{+~eu+DH` zIig@-m9Ux{VlXT+$e+D9FMv!-?@D2+gHk`Vg|IA4t4&20tBlPRGcbVPOLIGhg^&I& zsxU7o1ZoZ$RMffK;r#oSD=@Eh+U=9@Ws$q#wf=dzAs^Q34J}dY_o*$mpN;g%ak*1!^6IG=w)+d$He4m@cA-sT`C{TiM|aI~DF! z_H7tFaPgw7FAL*nJBK7Bny~Mwq%{5^stTWzK@6$Wc$vc7Ew>C)8=6K|cbO}%Dp4gq z|I;UXOW%C%mBV(ju2# z?m^${Dev^bYrTucP5PrQUUrGn`rO_-t7dUH?&ALLU>dwPEhn`sN_bX=!%0b+}Rqoah)y3nRQi` zl4DR|A#5p8X#AnPqJZRMpi<7jab2zf6GqapE>Mr~HME;jdmsKrCA}lMpkK$HzTg@{ zhoDusH=uAO)j(W0lB1)5l%nx_g&xT^AruVT{GlE;CuLw3k*C9*@2XJUS|V4}W~inA@$=ny)yvOYxmS0~ z3nI2RFngW0`4JdqWl6^re!ecNXjKq&SotKKXRz%}#_~~FB!?sj#nLYwwu(=Y^%Cm6 z>3)6NC43ZT9yucc&kM>9>MseBl$*ZZron3e!AcL>u$R15=*D0?UY5Z#y~#v5eyVxc z9!PhGdTS&BIe7K7%Pj+OM#qY_wlD*!{ytd&Xnb~DTOopJVSH3A{c?pW{QU#)nE(m( zn~4|I^77MDv*Z3i3YkXk(FZQF+V+XuM^Aak=8c^(Ow07EIs_FoWgCYLWC0wd8-Je2 zAWK<5GhjM3!ePKa$W(g#SNbRI|NiU5S`p0^fHq^IelkEbh z(7N-{8S5Rm`(U)8&Z@)F(+Q&?pp7RG97ON;-q>IL+&fo_Q+}`?M}WX?PkLI zwF~26+l|8G&lUH=JezgHf`0czRDIYjllyPMdZeS12W57-{h*8Vnh+8y>D#CaA|=Z8 z6Fw9A@j;tk+ojMn&!p|hy^C$jq=iP6!u(wsVpIVW5bTJK4xZ^y*saN#(nu^Yi7L|H z@w9*X+Ixqdw!v^xtru*g(^~ z{_0Kp^3Xz~b7B4tpaZ$3*NId}pFKv>PL>d)WuIGP1&(u7*HI_C9szB>|a~fu4JQyIw4#ef#3Bf<-c85@O~DS z8M{1&{S;FGE`W^?F+LS)dcR1(qxIli&bIlVWqMkL1PVIWV9y*9Z|&FTlobtkN>~_d zuep6|*N~u67Nm|5w7XrcZ$0=9F+owbuu$vwJQNfz(zkG07Q}yiYHZ#Y@VJv7^{noE z)#yS>I(sQZuI@GB+(rKKqq~28b3P+UP{fsWJH=m)y^RF*nTraG%C{- zz6;U?zqrOTPn^m=WdFVkdU7DPS3ql)IquZgd3O5zuP6CPnxH~)SCAX*#^{NYzn<8A zya`jSy@JhO``&fmFneq;tK*VRwwJ*{h9Eetx)0(@F>haTwC;#8x zeUF3zNKpTvl-bUVFc%vQpK!mP2wKO^FLDrRf;ZMqAc)G$EVN=tc^CBl^DW)yu}e1; z=MFNHgwOx4zPi7z*kqYpW_-&fn!H<2)|)vT3saoj(q6|d@gg^s;fFm>-(TD~;P)nN z$CvwLVUOnnZqyXn1&3A(F0Q6WS(z1Oc_2o0F? zc@KmyHulJAV7$plv}?c?M(Tfg7=`x8HQ@_U*iJ?SRXOB^Q2TuPG<-mVG#7#2&5mQf z`OsQQQe9GFYNnpz$-eA}S#h`!Cy;SC(I7zIJED{MS-uyu) zhTWTY@mJke{=vX5?l*N`ABcbzq=pLC7#fi_>N9XF zN~?GN5_s_*x<=y)!o%il5b9ELI>m*An*Sx{Zc68()9b#|?5wbt9R; zkD9CBH5O+1S0&pt~2^v@ONue-blSI6DI)-=Bdvd>-qZA}v% z37WeeSRQo3$Sf`WK`kiXl{Twm*6jmWy^Y~#(_?&Qm#h_W z+=J`oZ^F#CXdx>KuFAjyXKNI>-Gozc{VBx!@mGa^{jtp8-aaE)s#t28&K{lea`^rS zisu+y8kM;sbG~?3A@}W0x-fB(*I0NkEx!7j6dWU1avtVaYggo#ft{UodDF^~cVC_U z+%U3fSKJKxJISDkoJCTt3CIPCe4yLseT&Ud2SD+Qde3C=vpaBpM@Bt~NX>~^vvH^a z&4sV^deEZbuqhZHrP^;VC?k#RAK=a}uB9_nxl5Qz{3**C9H@fdK#)UVJlS zqi^8vjX6>NgHDZFW`IA*@!Sulm*gq-% z?TR?gBfi@eIxKp%*Khk_nRe7N60j>-e|+aV;nxuERvbRwk1IP|wGTLWboaqOY5%kP zxs-kp)uu8mOF@fkdm}i$=VX?AUM_{PeH1|ZDso*0ysZj2-(KqtB-JrStS+iF#%XHB z%x*c{yEQR5tYpMM_A@y=>s_)goi8JbG6!AnexvH%gu5o|yCCrHJ4B~#%-SJaUzMj~ z{oXAtP9fEDFU~>Xv#)+@oXs!+JLEVzK$%~~6@Ean0UQg zRwKDhaPB3reqqR55DzPcz#r{PffGIKk+0))D@Go5rQCGrd&5XUPkbe+jikyg zm^WWge;?RxeKlTbY)W~yJO`bvEZ@i7wHNkGIe!F(rqs3Jl$DTT;il-STW=Y-F04H6 zd04Z;aRz2R$nG99>jW8yW;<6*c=z4)(7v+7#>sq2^>UAcsF-Jce>e1T+(4u9V^ED` zGO(eLP{Y>n^R-3RxQzGQyqt5T&{wUgx=yzh@Z9^)KKZYAz?9D3lrjqXYfAG~pYt;d zx?e9uGw1`@81^K>zR)RDd)-r9#+Ym-!R>S7v*cE2X5SEn=}dI-JkHIjtfioycMSsm zRUnXJm0N(XFrq&ZwdJ;7%P-R;UZV*A_|LdD81xz(F+MGFvi&E?#o3j>12bL(wPyVI zWmClL%8P}w@%UYe%S8Uk&+D=C$iCgX3Spr(9^RFvHa)xSK-%tCRj1)Aprht$1jU`J zDH}|0yE~`SN3G7?U7!d5f11x1x26n6(lnm0zqSXBsUt#%7OvMRc#?NH_l7Dw|7gWlg*Ts)?-K znN{e#W{tA5uP>t;!t;Urya88r6mK* zlx+IfJUpMkn{No3^@!jCmMO$4?oOPAYc>p;A0D_7JcFw>G%1d`z%#%m#jNR+8;r9u zT;U;KW~d{npIpWvx%pP%?k>%(Yc~yu~1HRB-)b^6iL)H>B=~4}h_&zrrhu zl&=bg%DGhO>iman>_G;FCads+uB|MkR*1|~u~wJgzcZh#RrK*Qk8XI0w`y}o(IIjc zcIxJLe>XEBkjhH1JyQm05jN0OWZhIE_-91o1qM$yU9#w*pr9Zo%Q;BK0a1!8<-h%qqs&QX;IPrK&d*Y$$z0Icyg zCv$19!Q;<6^un=f(^fE1w^0I*E`NdtwDp+HCnVR(DlHaXjsdW(WLKHM41vqzDIu-=p*2ShHGL%y;9fbHlI5`s5JmogCH2LTg1(#Cl{}kM zMQEU8C#%>jirQS+XsxqyfJjg}MT!motjhd9-j(|_N=H-?*D$SEx1}bbzWEgMsmpWBL z14w=y%EQD0DGFP??FerYtkJ3k?V<IhP z{>QMeKYKd2`I-70?pFF|96z?IR7tljOfOX408E|mAO93vc~Gls^*}z-Or%_Vz+}nw z9!6l1+iY0aIhuWtdmo5bd0|}rXn}r-ZI8ZqJ%{EVx*dIV?tfCzb>bYs6fnl=5&)Q8 z&`m7=g|mKUNW;!k4FV+imsXS_57gy`Jrc0tp-Ud}+TyV~459b4ALWqeI6gltKMC0C zdi~-=K+0})7Fcs>Y5u^POFpvBMTxnhu%DT=6+8fjM+F#O+b1B`#i^CxbWA6IsP%V^ zl)*zbu@^eWmpFtRDlSN&dvj?^Mte_BHi(;7)n%?e2x7R}zgVm@Z`}PC5CnjIr2J@@#3GQ{3myM0d$H< z3iKpinQb%d%G6G68Fvp${d!xbuo2>^_`O{dQl)Ow+*)6=;RE7?ov58e7}kJ>($JO6kVUbr8C)c*756F2fhnDzBpyo`;3^~76OJ~y)dP72la_X z#rgWPVQ&}DNN>N$=dUfHJpw`jH#!n>j*dnNE?wzlr{cFC9e)qiStIj-oQ#-W_Ng-7 zyow^%88&6>n%2X}9&(>fKOYc1Bl&~QVAKBo`Ng8W7>L4V=T|Pw+SLMtP@lE7i(p<` zJ-8c^mB#;+mCJ$HXA)n(Z4?yyi6Y9Jg33S^?C7h=RWpF~cKM?^>CTgL*e^cyy@P%> z`>t{GPJaIDO?kPSsM>zD67dc*a>c5Ws_1B2oi203v-@4$0pPaMi_3|v?!<{o!QDcX zeawLO`8?O6#ah4(XM1&6cK7<9CqlPwye9~ z|LPedcYW%2YsN7`%9#zGxPl1uva=1?dNPTAp#7?VEkPoS;?jh*D109>GS@V)!42Zk z*VmsJW{}dkllY0vupv6y!VB=P|DV-?pJyA{KIZ!vv=OZk*`~vNcBd{#y^!lDTW4w` zSn4BnUZr*SGUrZTm8`$q0mFafeBI**i2oN3JcE}m`Z10-85`;CG7>*;UCv#(IbfI# zc^7U{5|nT8gKqK6o1cS}v66hD7b1vQEhiqHiDx8`&u8}Hf(Sr_pXK1E?~+UnI(KgZ z0)e9k%S!+b0Djm8%un)9!hc-Em+ zEu4gF#kCY*t%H5{rA~of{Ijh8^7}RE@}&a^ps1f^4Pt(qp>$=%kCyO2b}xZPn^-=3 z(&_2=x9Egb769q@nR}HMF7xuomPShK`{0)P=2!NVe`$_TQ(x`M>hv^zD`B!M$q>(O`lT6HM#7%gmOD zULfP|E^m^`i}LPyKmXXU>vXpoETIxUt)GvE?NsP|I$hP5xeyRVYl8aq=_vV7WN1l4 z-OJX}b&^|;{@%8W)Je8{yNt^GopN*9mxhb0lVbUHMHRFP?D)``uopm9#zykl%r!Gn z1T~`jGlOOB^PSK6N9VMd59jb}N@s$f3e~ecMNt0@vRRh;+EBrA;dmwA^H(jWeWnj@ zCWuYmf0+9kvmk(;#;a7;`Cl%K3kuGwJ!!)*_mF==ue{Ufo&eJ@kI6pibz6h-4y=H* z)3Fd|Bv4{NrH0Xk0vB^zG^OX-4AeeuV^BDML zo0H@PhjLwAke8FB@N4Gwwtk)CtA_v5Bj7hx$hYe%=ifoeAAM=iSe?|&w;L<_>+zGY z($Bm_=0m{R66xL;R1cgMoK}uz>?WT9uvqPrC?1HHzwwWqCrq;JE_*1{l~J`rPcFT( zBo_D6w)*P-R*|Z?sJD32UQQ{rUtOW&VMTC^6UT@YPc?Ze!s=*KdTCaN+4lqY3r59Q zt5NtfrP&ISA!cJ#G_Lg~#9tJb<7hXdWT4K44QI}BT#1ujm5>6HFCd-Qt$f|gH09u^ zu@^XpI86Fs_DDpgUGXbiQ4@|?X27={#%S6WmT}$J6HNAz)*n5umR@vVTWM;-bKIf8 z?(2MDy(1P!o8to3%N7H#1wXPpi}FYt^zwrT^n)3?R`jqp*;i98L$5w$o|^7Fb*+UP@j4@KP)`?=rdvZ$3E z@8s_>#P&J&DikKGRzKZ`F6oR6X7?)C<7BiTAbgH-Sm(|-zX1V z`_S^c(?3Pp)Xw$rQ;$?u8*-ctW`$GzSdn_bLS)s6`TUv2r&h6mZ7Ic8X zyy+NpV!dEuS_v6faQs_SpELEG(mA1SGJoewEq76}wZJCFe<~ojwmR7??dC$o z0?yR1Em{7tD|^H@xCZ+WFWzY;@o{-(*fSUiMk4k|Ts05*F?ZI;alHdymOJ19n^$ZIi=5+>5h^ZWF z8L7QXlhXr>VzPt&)mb|pYisCZ(JYUQGT@5(C&vELIAGc`*C^lw%P1qHVLpr}};NNm4 zHq-y0ds^z#i=CyLKCqEJx1(b+*Gjv$)!|ORoKe}Ca0HF}d-*-bDWeK?G03Oemw`}6 zsqSbBICJiWKlpFu`~G_=0?k{+lFUn2+PmV9czPSYAqu6%WPvDrEx$_FU3fRoH4Ju1 z<=@Na`jsVC^Ty-AkE68!pky*z$e?bD0+^D4RT_T=vJpV# z%BAg)+3)Is3=0-^VgqbR>AZW7R_Y|Kas_sA%+(O%IsXqSoCjbiVqcFn2uX4dQZ%$P z#$_^U0W1tTbMV!5GlzI#CvUm!o2-uyB^s{JM|Rb8&vcn^y~i=0$M#;q_S<1Z{!H_) zWEY`#2$)>UQZBME|My!J4}uP_q;{bY_s&@^gv`?qZ^=)g_r-rD?-%_o4_G9Ii}uxK zGZ)@=V(8#$&8eLL`YKGI)?UA76fnHCXH;#~jt^*XXN=cG)YSje5I*#^VwXQolk4;d zx(ojGbIQN1OG^XyHQ@qJqAp^Il9k4kzM%yPE*DLFXsk&=?ULEX#Pfh(`TzFwv*Ul| znf^6TtJo1*uR_O5Lw$1h>P@=N~-fpZt#)`=KZzSJkqkwjw*N~dm88}T=OxLN;e z^s@VA^y27(J5^>$H>lGwgU~)|F+FVSlASazH+xs_==rbYvxh->mhXeCw(A`;r((65 zAJJNu{D&19!TqK;?X1Mhqx*!em&;Xua7&_MR$UG*xl0c;eO~q0d$eYr920z;FtA?| z6!CXPuuFXG9C9!;1nweUXbpWux&Bhp(w)evufsOaGJCjG6C9nl0{W(w8U-;yjg&iw%wBS3D>(rB2%gLdR9F`y-sxljKabupPf6*_x);oN0bRjpaW zEiwxduM7m2oZ8DL&pYM$WxgfX;u=1RPR4If9}#U-XHY zEL(Q|=KqApJ-ra{>#s(uqreQFb51oPNyCsisRPFvfn)1LpBk>DyO~|5wCY@nV}a#g zTgmuN{u7lbn=?`6=+R8^1=?+fXiDLTU^GmF9S3&~-;}*M{)3KX^1x|CbvxnSuCRij z1~IK7<2ZO)-Zpy^Xn#Xc2W5djNpm(AlMTEq!T!Z3;n$7KlQsE+lt z_!ecWeHrnW=(H$GDRi&g542sNy&J>t`aly|^30t^>&8(HF|OO|cVvlo45`~?5%kQVaSL44@xsvk>-EZMZ%Vz8bykR^)f1&umsz)@rde|gcFgzo--OJc$)Z)p z8qZh<>AjlFfT-f@_1?Oc@G__=%s2vl^H7vBo}_Z-)z5s4J>=7w>ryFVyA=3DRIzJ~ z+*9WG&v{gG0A?|W@r6#*39rxM-9sZTzV_o8E|5HpXxYYGrC~0IT+cDc6f)!Imh!*h zI)U&P*jl+cNN;ni?m*)6#luNNzFmDK034P&o_+H3&lX)=SP9huOw9UnX-DWI114cO z+);W{vbL?J8e;k^1;Y zdu}+q5Zf~j@_oct55$MisFmJfjU8k0{n~2JWhU@d;i=v6WGL>xr(<-F%$fR}zfqnq z41QJ^XVJ+z^{lS1Owm-KfjG``djDsmgbph%D9kjUK=6*a`2PM$?F29=H%wlN?U^Hd zPueYnP(bm?4dqlX(qXUTFbONsdheEoE< zlu$Yw>x_q7QZ*xtff*y{4qF8WpRJNU0uFWpMl+7q_>Rr=B_&Js`_*Zz~ zUV}MQjr+WvaHZd$FJH9T<)dlwh&752z)l=!o;V=>kK*1ts)@C48%5prf{1`pl`36I zK>FTFhY%rzNLQ)^0qGsoEiDKLNN*A#fdmpMAwUFFnsh0l2b3O4=!EW>{eI_N=Xuw2 z&bPk5o-=D@F*Drv-1D1z=6AR2cU?YS{M(vSALr;sjgFbp*YgU zOX2EF1X{fGL8t5$^{un?eny=%^EFz>_8K5)t%W$elq;8#=e&DWx<$OgDtEdQmKwt{ zK+zpmx?NWQEii`p!!#Eoj9*6NI&uq*$$&NSA;bgY3{&+t^pO#~LC8uFu>va+$D+6k z%~0vPSUbk-Q&_)lpJH`SOrSI$8z28%?5@9@pglUyVisWZxOclrZ0L%{rGX)5#jFYW zCzaDkCo5Qjx7Db^IwCq6zg!oPAf2;eT{z|_gjdayFCaG1bUO|d8bk}tpN98;&wgi} z7@8A3+LCr`5da9qvWI{9_K4P#^o%p2V9Xc(!H!)OiRJc-e1NI5oGuRL`g!YfF#&-U zKDnbTU2quno5>2v=LHK%T=JPg9t%^tGF5|Fx@gky41ti+!uZ<--kF2c4&*LKS;S_O ze?C*=P5*B4vytWpzDNnL;LGUL{!Sk~1A%=Z8)vP~4j^)Eb^77U=}hTYPt!DOXVQ%{ zp4Akv%i~Cv&)%3K7#KN=*ou9oS%ewZE>kT&WU9;K@sl#%+f_8Eiq)(+PaUy1P74Mb z4Nh*h3;)-<%aYAjZ^^c>b^TsC$T99ZFO>baJt>V8r)`C-W&fr&i6 zSLn!Ns_Yum0_Bv|*6#!{SYf;tW!v(&Qf7nkj<6e-H!PH4~7WS5Ap_*;U*M7ij1+FI^h+ zMl8)~6BkSBy!Ixrxq@0y-XAmqbA{QLF;#n`95VvskM}=*Y;6H@7fx#pCl5t8O13Vo zw&tdeg48iT>MX|iE+uvg6cqV15W;-H)mmQK6x+eo>C*=0HS0fOclso^$6`Dj=s_~^ z#gWhLy$FeGb|5w4qd|o9%`s@(Tu+wUk`3W z>ia9|o)>simIjkd;ci`X*5p@!Sfmbv`K0!y`Yd)|V35|v&I!h4$Or5SWSZD`$ZFHf z^|Dh=WPFn|NNf^r!bx~ z>sM`>LE4tH0rm9DTa_D-B#{svU_X=Mr$0En|c_M-}-{c2+lEh zzB+hc9gWHdI%{0UupaBWZB3ILAF%^qJP_Mh|J?#re=GmTh zx(DV6@oI!E82H-%$_Tw)YMmQBXjiLXbKJ))vmFkMu$B`a(e(*Qs!?YS{9_S(=YxVF zA7)ZZ<1hJ?gve0u%@n_IH=orqE9mosLGDKG4fh>!uuY$dBKu}tE4OfC7&M{qMJSTi zO<%+HL0YGP#h?zl$8^+x?Xi%VNNXz6S&65i$-cK3N8i~jygAqoQDKJ@yp63ng_l=K z@Yb#I$e$TP-@bmYn!54QREgJOkSa8#4$jVgA2o7h=tNiNU?Q;d~M(gCH2;g~TocM>iR zgdI1Wbippo>CV^%OW8y!w8|eoav=;t?MLR%vdxR$&^MPHGlPbm4x@)((=i~QZB>L%RiY95V&03SwMr<&z^M46wOAeV? zOb@6!4wgCdE4%u5>6grzTOoIBokM%vkuPWdUf(;4xSz=x>!!9Dup}m0SQ1jMU8*Bs zBJV{I3&I2pH-3+f=5_zg)OId%h66he?tQ!{O227RcPZNBQq-mzt}?4@gu>wnt`ns` zwSv3wqN&fS?ak+~GoBfNc17MfSCXU?%M#0Aw$Cp7{pDQ7vxA7@%-y2`_DTnjqVJSD zzGH91Qet>OINev)-)j-VYr?Cz1_zE8H2{4guj6&;&ZZCU6WIViTDjfdt%2BE;|nKVh{#{nHohu zTl?4Mb7Gy5H~OIP*HDvh!;i<4(|EL-+R&Q{_U_AAJJSUnfOfi}1PCuweO?&2dbba( z@wMhTuFgJPTLbv)Xa=0H@v`?{#Xx~1kAK?l)roT+&U`^k3<0Rgb4A#wVo?d?xIl<( z@v?g9Ebiidy-}@Nf7E;&&uP#Stz;LbQ&2s(6z8>qYt)iCNqo6b1RJxA7#gHCh#Q-O zGLbGKsoh2k^~;yptgwJC5*$B&sh8h@2Wt_|{o;jZC8rtjAIf!s5?~;fS!-nO^i3j# zBlfsIwbvB`p2L-<2*l-%ot(MQ)JA^;@!3A(dnrzEeioo-P}amXOc|B0t}=oJ{fC!C z_lPs7zJlRPOU0k8Y|`cr1~Go6lpnf7t0 zCr5m8_il~$DUADMu;zt3XT6m5MIO7gugVV_XoafvP)#NS5G4+J0*MW#UOv$7ao>ci zWCXaAsiW?xJB%xU+*5VK;BDF!eH4ea z+tn1<S<92yVlkW$$3TPrNq+yUD@3+@YfUg)#Zd;kV8xng;H4ySDp?1lCg(AYr- zbH_0*$<*_pBv?&l16X+G3=Ev->wjWvF^w{`5pjJhUhRv06RiGG`{?Ek??S z15^x+a=)F*GjE%GrMp`W68_s?s=Z!PUu5b@00~9%`4`(KviMn(l;fm-b?RfQmTv%D{2O}MS)z&I{%j?*mLBx21ulkDgu#|;Z+mB{ zi}uK<>D*8niCMZh?vp*|%S!RYTkrDo z=iwd$9?(fGylQH#xZp+sl}8rPXA)DgoXFuvV12b~gUMf#*n8CE zIahtQ|KN=Ib2R1qP{VT(q@vhwCii)Mk#X3gAq2-w3wV*Mt6>3%G;TsWk1Ea{O!CQ< z++9Gl4hxUdvkuQ|ftxVQ2eGk5Y~;&Lc(b;)N+;sLwHO+smlE%*Ly2{b7ib!BC@1LPmQxzPELu%)f^Wok5%X^5OafRtqn~={HHV%e9UL5rkab;m{PGpJTHKOd>$Y>Jr~bo2)2| ziFB-9UMa`O(S8M0IAK!kg9NH~{qDk@&G_=o&C(>D^dJdSM0PaZ-^x5u(7R`F!Ny42 zpzcK(z@!1D!2C_jpVa8~zXQUgGfp&Op-%{dl0llw2Gp}EeXC~JI(7p8HjJ*h`=#sE z|LlkXM}M6A74^>wv`Fgt!~nP93OS=tPmVO}RH*xG5|Vcq*D*DJM8qQ9Ah zo=C4WH*7+t&-&H(?^$y?&>CdJ)49c7+p>^HPv1mw*~tbM`b8bb1!yj9qN}s%hM_&85z9> zkm5Q@vg43)pa}$ky@Cki&DaI9hl`?6CM_j3!BFI@r2J~t2-PiV^vX+-ty>cgID7vg*`(d3{81#2j zE0EjpOXD9s{?F&3$G*MmG2nx;Q0q`H5U-!;+y%yrssJQvF3t$ zob@(4{{1NYxCbfl#|HS@kXj2n8ykuy-PK9t&SihIdh_}I9ot?I)Q5^U+Ma4g{t-6a zc~YhtPQR@2^O?l9$bsHdzE3wbT1Eou&3+{%Lu)(>`u!*8KE6rKOgd+kbTc^RDZt>K zg$b8VYjre~=1jQKwBmQc$yd{??$+}R0N@m)-E5}>9UZygoC7TF<^TPgDH+eGjp=%3 zxo}lJ4KGp})Vw^{H)tt||G1y}9#x@b8YHmKpem~cxlPMGK7DPM6y|(*er4Nc=k9}d zk}bAZKV(0CD*vDxw&cp?;yP2NAN@=ZHc{pcBa2~zR2oOl$UF{z06>IeJ${O(E}&}S z>vq$Zz<qw0@3DzU(Bgc=D&WyiWGPu?F)$0bh3_HO zZIz_=U?yL>whUePm{YyT0#8XS&l2eCwoVa1V5Q$NKXO{x`GGq-iCoZ$s!j3nmJP64 zI*1amp#VZegwY_>ZAIvIuh9-(s8x>3t>Req>Uwu!#$w8SfjWWlih}V5Ds8I#l}mug z;8;+4k&Tet`2y~Ev(jLCu2F>;Z>4nrfB7;mR4e`Q97pM+2SsQ{*~!(U6MZE(;HE?UL*M3uG%TjS69{*5eoj2*So&k#=?X-gl(?udBOeY`n5#dFCM(-89V# zyCJM8y{3nYg~WlCb8SU?z66f8dl?>;<0##a5qY6jxEU`j7Q06=y%79Q)t0>is@5yG zTTj8AbV(dm7lez`y!k@^ zBVBqSDIv#En-JFRZ6@VtI?9HpZ#rC3@3kUa_W1Y-9MG{b&M&s763htFq-=F3|W51cad$y2~65$FS*Qk9UsN<~ras}20=*2mjf=Q>fU;VeE`hWR5 zF0!QB*1kb#zDLHo?&d2Art&1TLB$>rC|IS&U6 zVsTQ+dRZ3B)(K`Bl!J368%nA-2x-2q)^kSiK)uE$+iO0bl`Uk0I4zq=Z|kk+YH@Qh z0_(c;c^}x6hjhYs|GNGB8bCol)7v)2fn~Ez)_cHArnthSE6+ec?ZXBpj_vUkmr?5e z`*qtP)9_pGTU3L?T-0y>W^(3P+iphvV(9Y!2o?PF>-vPPuUlCHVn|)%T!nb<0(O&5 z4ppEoWDHAjih>BnKI7uW^H@yBKj1jS2|{Q1l7!nf;Kq|~{Q(FUs-;R_!YDKhu4Wv3 z$8qsj6>(W?1gXGC4&g%0)!}U z{Zd=7imw*+CUYcteJ#P0^#u_+iD%$mm`eNn=ybqagoQbbQ+!Ol0+m^sD(45JZkJ6w zeD;ZoS5(nL(jW$UTiQiVL7)m#k}r$~%+dO48qK6=jWTjao;lXFH!HwBDrA%diZsJE z%2hJ6bYuF|-pVOn;TAF9b2d!ErXG8>dzfEIGavQSFJV2jCE&9ojH-j1M!z`7%m4foZgVt+H~t#IvL zirOPzNG#^t!Tq$I9Xk=yliyGclYp>Sj^0fic;r?P8JDD{;62X85{+BIw~95s?9~gI zM*c64tA!s~4)3apU`rA!j80gX@}co>k2nAE6idMTzz-s0a}KnBjLBI#=hIw;Y@Q71 zadGmCr=h~kFDb6psusOf*KXsi@6j%2_4UGkkC~H!T%VAC_F3PqZ_Eqn!jDD6eA)UF zjfwJjn-ds-=(SGv!!MFq=G}V@$10-ch2YoCt@e7Q|F&S$aW6k92){`3FJC=1zl}Tw z<3rKrPS4^A1u~xzgH@Mb#%YKr(r!E?OUJP-3s0IlEfn{e>iDHC>x;R^+C}-;jy1pk zVZ&F>%X7cE%~M&YJW1mlWoS?(GpJU>j1kQ9J*MwE?fxXUi{~|gEj4q;X$#~#WTco- zb)BI0*Up~|fAvHB$yH^yna2V~M0KU`i)p;V7P-38q&D;rM&QYYFQ*lD;8yar4>1!9}36rM6xz;gbm8DK|2~+oG2vrkL zebux8R!=o#UQ}!BwCSVk>qXy)n ziRKyQ_r1j6Eoo@@2l-rvlH8R?Equ(TpABi@v+CG2$wV&etKMSMF+!%^EbSFUn4*>r zxo!5$n2XMKSySV);^SX$St($lRuHag?1i%wsjQ=0SiYTVl+++O`JIH=0vQLfUyK_S zXe}Ryp*ZTElx-)P$gHh-KWRpY9jaXz$E)6;AabihHp`E7$WKXak1HbRq=)hT;>J)P ze>oV7u`!$4E2V9<9E;2DWf*MK)gbU6aVf~nFn!_ z@=kA>G+3{y>PFYQ4Rm~{)r&gmHG+eSbM_}ORxN=r4#(K=rn@UUy>DFEtEdvAoDPgq zZ$)Cij&9c*sFue+>}u zv(5`d!wj~>rH26!NPMla5D6t7BB!`CES_7!=+m=Ab2=tGKVflekqKh|VR7$vN2>^2 zt#i9K=p?Ym2K(_*jxPW}JEPTTm$OH7SR~-l)y5FGrK)Mw`*K@}fqKqX5G%G8juIKa z-{aQA!|vSNz=7+xL8FV??oH{ZwthplQ%p={gLSWh#T&4xp*}w1Y?2I z>d)CG-Y-bO&MPTcrzoQi?>q%-|K!u$i42tX98K+giGyWG1+2nfFL>8&6+i_E4w5+R78~l4$hYfI`KHSugLBFI9y<|1VJ3p80U1qu?aqNv7ij_5U+~VKgfA|bJYj3f3;P^DS0yck54~>QcC+&ufh32*7gO9xy_Q0V`e}5 zN#3;3OJ6je($r|^hyPymf8@-cHaF}1Ub`lRKVJ)RQIEd(jbQY{%@aa*XSBZlUxrDJ z%~Gr{wx+qq1bfu0<_HR|I)nH8st8F8=4sW7GEI%v7l1*^h%zL|<$afL6Kd^aT}b24 ze-0D>{EwLMZ-rdvh=|Y0=bX%&0F*FA`cCaZW^prz))??uIk8J~^TZ;$@zuru0$)FL z9|E9?Sw)0hcfRI81zq@9lMyYz4V9=c7r%3w9PUW?Q_%mKnh!H1{J&~RTc+*ma>Fay z!U~&6z5Mr||5o&Wb z&GhL<*cr)J|5o0Ag z8`xBYn!@orNJwB-7thxlYu{3{!dk=Wsy%o9xz7K|79xhD6&tgoQDQze(G~SvjZ-qD z>lP_q1FDxYw;c6d5JEN30S}eH(eI3)vTD=i3jcJH>8GV7o~I47=6-Cxy)vBG=k}yB zyjF_aQK@Dk`yXD?Flu0ZBj@S@5rJVNuh(i$NTF&ar5Yb`%GT-BAzYM(Gh%zNC7q3$ zbR|V1Xr-3rR(ecZBhNuNd}`0*&6BO|Gz6{-T~|zP2D8-3$#=CtWn}v8T;6RZWO#zT zfoTTKv2<(W4&-7Mzj!hx`SStAY@kmr7Zp%4dQf5$2l>U1CepN$bv#3$_garR48qZ5 z6IqJ|m@Z?mV3@K89>)J>I#>znwoSY6ov58jv5vE?3`xDjUOzj|sIm%+=g&>h zseR^;NAolpO(MzynF}H|IV>T1GfJ5kF@B}~nxekKPnM)=!9vw&Y!aemau>Q;zBnY4 zwGu-rpD~d~EaC0=^Fp=;SYuX?(-0cWV*^kxcbi<{XbcN$ooj)wQqIUbpNyEicSTpy zICZgHXO{9E==7L=ex*?S)~ZD58$~^VaV{>3)!ZoS_+9nO(HhDFI+Nm`xYGpN=_^V3 zwSus9#Z(w<_6Z|{tqg(k8C3xS5t@=6HI!bQk;}wRZRsV|ZE!hjx2vGQ>BDMBG8V;8 zu-Vy6#8{dRke#jU34v{@A-f&Rd68wt!R>LHK~BXuT9=l5H&nhECMQEnUwE;=d*pN# zM$w;u2oKb8$t^CGNHgB%;}XLq#?+E&Fb>oK{-nMY#&1;WVx)o2{91UeEko$4njSYM52&vpRHl^ENwP$31 zWqvO=8SDWm$PrPQ#yiQ#EvL`R$+Q{O^O@u2KWtUsM&!pkhe~)PvUKjgKWJBdxf!5$ ztH*cFLoQJ%?fct%1BRX;zB74u8lIZ-T84gZa4|asTc6=#t(}O|H68#xi5ssqR^M$4 zE@AJg_*G<~;?yyVtD2f^td2)%_r%I-=dzS~)gnu}$~SkWKBGZIFxRfyFp-BCVqM zbTRQ^9_IxF=lsQMtpFer$0+{^ibl=TBueYM2Ym1zN7vYxzbHcts?ne~kwe(V47l%T z=kRLawduAHr>Ny>dql0#;zs9=VA_FB$C}CL6C`a6*#H`adsJC+mY~Z>qOBsVy`P;l zD$m+nbYD%cFRTT*qY!A}Y{ONYn{jim?v9RCYtXZd%B91V2Q6Pcb;k3;VmQkyW`z3` zeZl+HHaUtTbjcl)iZ^jmvd}hLLEQ>jk%N>VK>Njdc6UE>|BR_zdI}DUv+FMk7UFH)WIWhha zNtjk5F!&bp-?>Q~q;$cdrdp2&tyB8cxMLX6oh7$G_&k&CDZR22-o7ZaBJj70+)GhJ z3dGfB)~+#h*w@ZuxUDg)jsGy_$8$fj%hvs!%r{YYL#q5Q+Zir2?-GkkM8s44Os7JR z6ROSkGpC*9H&|Iz8QDHFb4b-2EghmrX`_QhY@MCUjGIbB3H3q7xT>b1&g~EHQx|m3 zIQXQz)E=RB?D#}SQ#Pe-kfCa<%&*Lk#8oeeD88LHvi?j?gc``B+DpdDuv-o>#xDy| z^~w*Pr{=H5aMRr;%e2754w{g=L-zd;ryun`ZT8q)i_1cenm6?%unSt8B$_cCPt}Of zZN-tcuM1RgO!7nCP(Vo3D?HU9w?N0&FV0>13(p#;U?_MK3uy9o?Aa zu`sF`bT3@T%LP4&0wnv4c@w75{DC{6a_I|%Z|0`3M|wBfRr|hj`ZwqZK-Y?q=-a>a z>{606OBz+Uky8y?aVB_d&x5!$p4t(q3>dN-5swxJ)#bmh*5m8INo23tDX{_?;5KW* z3$CwcGT24Edb7aqtYt!Y@$C{nj3*GNv@etb7{}KFK`$a6fj!h)Vb{hw)uKhGh3y+X zs%01sxG&EOz~HepO*l-Rl3*lb+RCqHi7qt?%l7Ngu3OfeI@@h-l*_T7yGf!P5IWJvgspPw{NHOhNjb~6xjmD4|!`pcs85M4xSw-$_tH{8rv8Knl%bJb6Z*p}$~ z-ZF-o-AeJ{TXU3w6TTm<)mgq&Y@G{vjzVjDknt|4&PYVAxvj!=fbz(!CqT%dn?9*1 zki5rSvT#36;}dPQ4xw`P%yZF&Pk3F=K3y9Nm4mpX zH=ktW%`;sT{Xgk={A-%$_04C0=3zLG9%2~&!3W8-;ONU=;YJF35fD0^v2!}~6_A>x zvI?g7>%7Xbo8m@#yRc5!kLTke=f`5Z25l@5{=z3h+opH?CYbR@_gqLTh{NTsSskO^ z*hA?f!Kd!d(jY-O6E&`BGeG9znC&9_zn1>L+dn7sy(;O@GK^3&3OcuAkF1^8G8U6X z2}_NKnI%Asca!rel=VJ`G(%KVeyy1fUpKq82tKSgHL!1@2Oa>>Nu5f6o$0fSc5+hw z&D87flyGa<66u2a?3SM13}wv>a`Mb*doefd3pFSKK{yL5DP{&*Y#;}aiy&A>tkktp z+E#SH$wFw$U*6r)?)!3i@K#AuJSxacRvU7|Um$aK+-LK+K;(_DImxQ4D>#gLb7Vu( zCO7jr$(AKohLyL}Sq9i1#5Z*5yEmggup;{1%*0ZwcN)?2<(pa zXPxqaZjHi$qf1m|n;ib;hxZbf1y?oA0aP$m<+JOLoj*5Y6;_XBP9VaVW@1oHdZV{e zdIoJMeVj+FAd#2YNp&DG-a|dq6dHlEkWQhY6$6&Z2`@4touo;U39C~G92ddN~Aorlv z$uqNN+TU4%zlW3OWk>l$@|`aF@L@}4S<3bH$3qGrKsP?_Rt~HG3oLz00V|LQ%wpLf zk%-HZW2hlN9W$^^qcMJbfs#`mJ)3UfJRo~AIKpvr&hH4_IL zt(8&OBF(6UTQLGwmJQ2BtFujiHL&cC`+V3K&|dOGZ^&eMq4>Ayh`UX7G`6b)(%+=% zWw}b8&mALg85fsj@U~X^OUJrfAMkGTc4Nh|ftP0Do5IdPmgx$rHhL#~Qh{>N#(Wl= z`IkH8)@pLQCCvHr*qnK>)CVce)brBe;XzXEKbG!OFUEJLx1`LPvxV7)V`82a<<^$j zi)Pjzz4Oq7QG*VJP5*Fesl!dlbc~0ImJ%^XL=G8kVP$PkH9L{p(YYI}F#*|U3~2W+ z?^^m`N>j0hk&S)*vf3y?gNlO>eaf=8AvHB844Vc>U)Mc<)(m({*TdS(#Q? zpSYhjxo$JN>`^~b@?nc4yBM;)q|rcpt8CDN`fHsj=Q;?nJIsh}FJHwOP91ZE z+qC7-vw(JSDIa%KFgEvHq_C38C~K&`%IKQx1JzAQ`c!;)4}iPDB979_8wa>`Skm7u z+-~*E9c^m1#oK<%B>#o@4Eb}N3h#_*6l-h4ZTV8j zc?)WWxPuO^k4Dlo^OizjH^sshoz&RqDf|2PGuFf4d`)Kr)Iao&1EmDQCJUEd>lF8VOjy~q+XFB!& zD31K^+~mLf?+G0I-m`4)|C&bF^|JIMSgj4y00Y4H*@5oHmbg3_g4 z0>k)2fZ+Y4|Mk<4z!lDcrXgpOO{y-dw|=zKgV{P4yUD?8on6^B!mEce?)4dSBavCC zZPU{YhXOk$ndH;$J_lI8HqUA^$#0qGeZ2D%S@=lHHZg~&ulnzYk5mMH@*z9ibXrBJ z&{*n{BQ^%xqFd|WRXP}Nx|rob4vqVaG)KoRn~C6S+@!Q(sCNnI#JKf!(71sCKYE!v zZp}3r8_40-2UC7c|MD*u{>OG@PB6>!rwUTrhUPM~&XyD-%U<^vpc6yl;>fk|u`_FU z&g4ghL6GFRw2L<$lnt8balsq(0$~DnfiSD$w+1T(1Gp3dM0Vm}smI(4yS;wRE2u;u z#tI-1s3HWPY{bci^eV2(4JxZz@{$H;EFL{C(a#*)A1kgD^U?BRh*K1`%T-7)leYWW z3KUFD4vxD5s*pr+!#ILe!5g5G>g74u8f>$e=jVSt5oYm{I{SFU36J$VM}W+V8?S2tNRfmo(NDjJj-&6+%p>;52?l)6=kxyp_rjVcb@33ikBhA0S^%3# z;0jU!!hhW^{vfo z-g6l&YK^BoLqXg*!BzST^bB>)PdjnQ91F^BT*GONr5DE?r)+F8{mkmeguEDN(iiVr}! zK1w4E6Z}3<4ojx4mnZ!71=sj0A$;i0HaFj(!11?JRwr4G-N99;vhWw=X)X@pxA|C7 z4;B;dR0Je%eE8R~w#LNWw$S(q1}^6>{RL?lECP!=VuIVEslGO=V#N%DlRMkv(wC`4 zAl}9BW(mkV8)-RIdQEcgYS=T&FjD`?=G3c8^ESfo`kS~p)78LbSQk($ zMr~ZLIt=oFg-Ob$QC&9-t{g?!w;!B!V+e+ys=YscI_!;QeLLTmv;Jfvy*v&sl~G@2 zgvFVyCLbrq*O!|dQ01td>+JVyd?UUWR~vYt7^wcrpXx4u+q)c`a%FIbGqLs8mqQPz zZS=Xls~pv01uYok;M%a$te;kIrB9AzXor;YA%`Fe>~9OUPsK@?blBgQRi+Bltv= z2<)FoDx?0{w`(t|L4>79H2jR6&P1m_qfwEg$-8fDZ(drk8Ns!cQJBe7|5eToP!6*{ zP`?V9F3}i2Nhii1wl^vA<2NAv^;r}D6cG_@=MhWA0K7^58vhuh2F!%fr|6^Vr&~C5 z)^(JuYHU>TKrE^2--JWqM&&Ck8hQZ$dE3vuk8u6nCI@v%RZ?&@ks;_^j(dlviHBIotMbvR z;4{VHsx@Y60*AGEh}ZB9a7n?F)BaEQro*ux_g=wo?0$2!k+?>omjEvVv-rl-)WYM+ zyY|1Cl)5fz=sj4={J0l6+jqpG^67J+2IVF2EIfmsR!n_zh13Zu5MErfo&LJYyZFAs zp>>{AOy@HAlFZ-OL$@h|!>Y zI)a+^Pspg0i}PjW?|n3j0w%ySLnF)q{@?dw^@LHvh;OQLtd;4Oxx-A=#IAc}R- zX3r~9k*?tfZccx<-g#0&yX~+Qb)FP`CMpd3mq5-bOcWY#0}AzVXj~r>kMmP|na8WRh#0E2+iJ zXgO3YJm2HXSzT#YGkieVgQ;4Fz4_cMBSI*m^|)Q6gMBvDHOT+*fXD$maF;ImMoY?o zZ`TAoE7E2t?&*fb2Ej{jGctCpw^W?!)>_{NWLZz`kf#^E*6720L8Kl4UYGpnaO~9M zhAj39XQF~Me{mtEEkj~|6?5dL;vvShp}u`R$M zo&D{-0)bcIRHrj1w?$WLeR_^T<}8&>f_W(~fcTr~T<;)hjdK%|sJ_nj*S3SFSN%wi zZ93mNa{+TK$=@y|cSs$0DTpVgn|DT4qt8AL<(RM`lW&#*A>kP6I@iLq|GR_g@*mT~ zx|Y?4l{{174gJGF|Aq54G+fi4v-80A^KYg@&&KPaEUGUEi)WXAGmV$+MN}|}y}24Z z<5$v=I6@F5g@fKGoUrG4~L$`B)&fj4cV`m^Y=|%&s)+}7p+sXoi}YK@uG(&Q|s*Sm~eUR(tHL~Jx|D3Dq!o>*@c8949*<# z5`pqq`dLj>KgfUn^itEd&T~$n{T>q8ug?Fa^5^gtF=o{wR+Y)@tr=aTC=PGI(8gwM zlIPuND^lHxA#F1o!E{6}6=<_8n971e{1cEdO4h+IE><7)`ktHm(xPPhp036LU~cmk z`a&}7zg+rt6yZ7Bh{ggRiF(3pPq5E$Jc-EW-RTX3*4l!BrT0HRt)C4zT6sXTIXam- zoYVN8jdi;_6OU4NUb4oXopOga@Bpw+AZ-8onFX2F1IAKLzD94tEOQ~}-PX?p(=0PL zV>u7BYkK41ABRlv!HGMb2B~U4Y>7=j1}hFSLa!Q&1?7Bx=W4WUbSUO&5c*51_5|wH zE#~i<$_4|%8~h8hv3RfG=SH8-gQI^98D9Eo;Luj~VJH?R2ZV1G*~!H1!AGIB|FJQ} L|93&mzeoQs*))Js literal 0 HcmV?d00001 From 453fc75ee940e701afe96f30f665ec23522eacb0 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Tue, 24 Mar 2026 09:56:05 +0530 Subject: [PATCH 070/117] fix(pricing): remove above_200k_tokens price tiers for claude-opus-4-6 and claude-sonnet-4-6 These models include the full 1M token context at standard pricing with no 2x surcharge above 200k tokens. Co-Authored-By: Claude Sonnet 4.6 (1M context) --- ...odel_prices_and_context_window_backup.json | 86 ++----------------- model_prices_and_context_window.json | 86 ++----------------- 2 files changed, 18 insertions(+), 154 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c53ee943c58..9c7f74a829f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -971,18 +971,14 @@ }, "anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1001,18 +997,14 @@ }, "global.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1031,18 +1023,14 @@ }, "us.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1061,18 +1049,14 @@ }, "eu.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1091,18 +1075,14 @@ }, "au.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1121,18 +1101,14 @@ }, "anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1151,18 +1127,14 @@ }, "global.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1181,18 +1153,14 @@ }, "us.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, "cache_read_input_token_cost": 3.3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, "input_cost_per_token": 3.3e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.65e-05, - "output_cost_per_token_above_200k_tokens": 2.475e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1211,18 +1179,14 @@ }, "eu.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, "cache_read_input_token_cost": 3.3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, "input_cost_per_token": 3.3e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.65e-05, - "output_cost_per_token_above_200k_tokens": 2.475e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1241,18 +1205,14 @@ }, "au.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, "cache_read_input_token_cost": 3.3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, "input_cost_per_token": 3.3e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.65e-05, - "output_cost_per_token_above_200k_tokens": 2.475e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1831,7 +1791,7 @@ "cache_read_input_token_cost": 3e-07, "input_cost_per_token": 3e-06, "litellm_provider": "azure_ai", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", @@ -8503,18 +8463,14 @@ }, "claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "anthropic", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -8695,19 +8651,15 @@ }, "claude-opus-4-6": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_creation_input_token_cost_above_1hr": 1e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "anthropic", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -8730,19 +8682,15 @@ }, "claude-opus-4-6-20260205": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_creation_input_token_cost_above_1hr": 1e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "anthropic", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -30323,18 +30271,14 @@ }, "vertex_ai/claude-opus-4-6": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -30353,18 +30297,14 @@ }, "vertex_ai/claude-opus-4-6@default": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -30409,18 +30349,14 @@ }, "vertex_ai/claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "vertex_ai-anthropic_models", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "supports_assistant_prefill": true, "supports_computer_use": true, "supports_function_calling": true, @@ -37153,18 +37089,14 @@ }, "vertex_ai/claude-sonnet-4-6@default": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "vertex_ai-anthropic_models", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "supports_assistant_prefill": true, "supports_computer_use": true, "supports_function_calling": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c53ee943c58..9c7f74a829f 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -971,18 +971,14 @@ }, "anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1001,18 +997,14 @@ }, "global.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1031,18 +1023,14 @@ }, "us.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1061,18 +1049,14 @@ }, "eu.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1091,18 +1075,14 @@ }, "au.anthropic.claude-opus-4-6-v1": { "cache_creation_input_token_cost": 6.875e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, "cache_read_input_token_cost": 5.5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, "input_cost_per_token": 5.5e-06, - "input_cost_per_token_above_200k_tokens": 1.1e-05, "litellm_provider": "bedrock_converse", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.75e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1121,18 +1101,14 @@ }, "anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1151,18 +1127,14 @@ }, "global.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1181,18 +1153,14 @@ }, "us.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, "cache_read_input_token_cost": 3.3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, "input_cost_per_token": 3.3e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.65e-05, - "output_cost_per_token_above_200k_tokens": 2.475e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1211,18 +1179,14 @@ }, "eu.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, "cache_read_input_token_cost": 3.3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, "input_cost_per_token": 3.3e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.65e-05, - "output_cost_per_token_above_200k_tokens": 2.475e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1241,18 +1205,14 @@ }, "au.anthropic.claude-sonnet-4-6": { "cache_creation_input_token_cost": 4.125e-06, - "cache_creation_input_token_cost_above_200k_tokens": 8.25e-06, "cache_read_input_token_cost": 3.3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6.6e-07, "input_cost_per_token": 3.3e-06, - "input_cost_per_token_above_200k_tokens": 6.6e-06, "litellm_provider": "bedrock_converse", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.65e-05, - "output_cost_per_token_above_200k_tokens": 2.475e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -1831,7 +1791,7 @@ "cache_read_input_token_cost": 3e-07, "input_cost_per_token": 3e-06, "litellm_provider": "azure_ai", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", @@ -8503,18 +8463,14 @@ }, "claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "anthropic", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -8695,19 +8651,15 @@ }, "claude-opus-4-6": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_creation_input_token_cost_above_1hr": 1e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "anthropic", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -8730,19 +8682,15 @@ }, "claude-opus-4-6-20260205": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_creation_input_token_cost_above_1hr": 1e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "anthropic", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -30323,18 +30271,14 @@ }, "vertex_ai/claude-opus-4-6": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -30353,18 +30297,14 @@ }, "vertex_ai/claude-opus-4-6@default": { "cache_creation_input_token_cost": 6.25e-06, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, "cache_read_input_token_cost": 5e-07, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, "input_cost_per_token": 5e-06, - "input_cost_per_token_above_200k_tokens": 1e-05, "litellm_provider": "vertex_ai-anthropic_models", "max_input_tokens": 1000000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 2.5e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, "search_context_cost_per_query": { "search_context_size_high": 0.01, "search_context_size_low": 0.01, @@ -30409,18 +30349,14 @@ }, "vertex_ai/claude-sonnet-4-6": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "vertex_ai-anthropic_models", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "supports_assistant_prefill": true, "supports_computer_use": true, "supports_function_calling": true, @@ -37153,18 +37089,14 @@ }, "vertex_ai/claude-sonnet-4-6@default": { "cache_creation_input_token_cost": 3.75e-06, - "cache_creation_input_token_cost_above_200k_tokens": 7.5e-06, "cache_read_input_token_cost": 3e-07, - "cache_read_input_token_cost_above_200k_tokens": 6e-07, "input_cost_per_token": 3e-06, - "input_cost_per_token_above_200k_tokens": 6e-06, "litellm_provider": "vertex_ai-anthropic_models", - "max_input_tokens": 200000, + "max_input_tokens": 1000000, "max_output_tokens": 64000, "max_tokens": 64000, "mode": "chat", "output_cost_per_token": 1.5e-05, - "output_cost_per_token_above_200k_tokens": 2.25e-05, "supports_assistant_prefill": true, "supports_computer_use": true, "supports_function_calling": true, From b4d0e3213fce8794ad815a3dfb7f943c2c24e8dc Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 26 Mar 2026 19:46:48 +0530 Subject: [PATCH 071/117] Fix the Pricing changes for claude models --- .../test_claude_opus_4_6_config.py | 98 ++++++++++--------- 1 file changed, 53 insertions(+), 45 deletions(-) diff --git a/tests/test_litellm/test_claude_opus_4_6_config.py b/tests/test_litellm/test_claude_opus_4_6_config.py index 7ee2ea33957..654ef1b9771 100644 --- a/tests/test_litellm/test_claude_opus_4_6_config.py +++ b/tests/test_litellm/test_claude_opus_4_6_config.py @@ -24,70 +24,82 @@ def test_claude_4_6_australia_region_uses_au_prefix_not_apac(): Related: The 'apac.' prefix is valid for Asia-Pacific (Singapore) region models, but should not be used for Australia which has its own 'au.' prefix. """ - json_path = os.path.join(os.path.dirname(__file__), "../../model_prices_and_context_window.json") + json_path = os.path.join( + os.path.dirname(__file__), "../../model_prices_and_context_window.json" + ) with open(json_path) as f: model_data = json.load(f) # Verify au.anthropic.claude-opus-4-6-v1 exists (correct) - assert "au.anthropic.claude-opus-4-6-v1" in model_data, \ - "Missing Australia region model: au.anthropic.claude-opus-4-6-v1" + assert ( + "au.anthropic.claude-opus-4-6-v1" in model_data + ), "Missing Australia region model: au.anthropic.claude-opus-4-6-v1" # Verify apac.anthropic.claude-opus-4-6-v1 does NOT exist (incorrect) - assert "apac.anthropic.claude-opus-4-6-v1" not in model_data, \ - "Incorrect model entry exists: apac.anthropic.claude-opus-4-6-v1 should be au.anthropic.claude-opus-4-6-v1" + assert ( + "apac.anthropic.claude-opus-4-6-v1" not in model_data + ), "Incorrect model entry exists: apac.anthropic.claude-opus-4-6-v1 should be au.anthropic.claude-opus-4-6-v1" # Verify au.anthropic.claude-sonnet-4-6 exists (correct) - assert "au.anthropic.claude-sonnet-4-6" in model_data, \ - "Missing Australia region model: au.anthropic.claude-sonnet-4-6" + assert ( + "au.anthropic.claude-sonnet-4-6" in model_data + ), "Missing Australia region model: au.anthropic.claude-sonnet-4-6" # Verify apac.anthropic.claude-sonnet-4-6 does NOT exist (incorrect) - assert "apac.anthropic.claude-sonnet-4-6" not in model_data, \ - "Incorrect model entry exists: apac.anthropic.claude-sonnet-4-6 should be au.anthropic.claude-sonnet-4-6" + assert ( + "apac.anthropic.claude-sonnet-4-6" not in model_data + ), "Incorrect model entry exists: apac.anthropic.claude-sonnet-4-6 should be au.anthropic.claude-sonnet-4-6" # Verify the au. model is registered in bedrock_converse_models - assert "au.anthropic.claude-opus-4-6-v1" in litellm.bedrock_converse_models, \ - "au.anthropic.claude-opus-4-6-v1 not registered in bedrock_converse_models" + assert ( + "au.anthropic.claude-opus-4-6-v1" in litellm.bedrock_converse_models + ), "au.anthropic.claude-opus-4-6-v1 not registered in bedrock_converse_models" # Verify apac. is NOT registered for this model - assert "apac.anthropic.claude-opus-4-6-v1" not in litellm.bedrock_converse_models, \ - "apac.anthropic.claude-opus-4-6-v1 should not be in bedrock_converse_models" + assert ( + "apac.anthropic.claude-opus-4-6-v1" not in litellm.bedrock_converse_models + ), "apac.anthropic.claude-opus-4-6-v1 should not be in bedrock_converse_models" # Verify the au. model is registered in bedrock_converse_models - assert "au.anthropic.claude-sonnet-4-6" in litellm.bedrock_converse_models, \ - "au.anthropic.claude-sonnet-4-6 not registered in bedrock_converse_models" + assert ( + "au.anthropic.claude-sonnet-4-6" in litellm.bedrock_converse_models + ), "au.anthropic.claude-sonnet-4-6 not registered in bedrock_converse_models" # Verify apac. is NOT registered for this model - assert "apac.anthropic.claude-sonnet-4-6" not in litellm.bedrock_converse_models, \ - "apac.anthropic.claude-sonnet-4-6 should not be in bedrock_converse_models" + assert ( + "apac.anthropic.claude-sonnet-4-6" not in litellm.bedrock_converse_models + ), "apac.anthropic.claude-sonnet-4-6 should not be in bedrock_converse_models" def test_opus_4_6_model_pricing_and_capabilities(): - json_path = os.path.join(os.path.dirname(__file__), "../../model_prices_and_context_window.json") + json_path = os.path.join( + os.path.dirname(__file__), "../../model_prices_and_context_window.json" + ) with open(json_path) as f: model_data = json.load(f) expected_models = { "claude-opus-4-6": { "provider": "anthropic", - "has_long_context_pricing": True, + "has_long_context_pricing": False, "tool_use_system_prompt_tokens": 346, "max_input_tokens": 1000000, }, "claude-opus-4-6-20260205": { "provider": "anthropic", - "has_long_context_pricing": True, + "has_long_context_pricing": False, "tool_use_system_prompt_tokens": 346, "max_input_tokens": 1000000, }, "anthropic.claude-opus-4-6-v1": { "provider": "bedrock_converse", - "has_long_context_pricing": True, + "has_long_context_pricing": False, "tool_use_system_prompt_tokens": 346, "max_input_tokens": 1000000, }, "vertex_ai/claude-opus-4-6": { "provider": "vertex_ai-anthropic_models", - "has_long_context_pricing": True, + "has_long_context_pricing": False, "tool_use_system_prompt_tokens": 346, "max_input_tokens": 1000000, }, @@ -119,6 +131,11 @@ def test_opus_4_6_model_pricing_and_capabilities(): assert info["output_cost_per_token_above_200k_tokens"] == 3.75e-05 assert info["cache_creation_input_token_cost_above_200k_tokens"] == 1.25e-05 assert info["cache_read_input_token_cost_above_200k_tokens"] == 1e-06 + else: + assert "input_cost_per_token_above_200k_tokens" not in info + assert "output_cost_per_token_above_200k_tokens" not in info + assert "cache_creation_input_token_cost_above_200k_tokens" not in info + assert "cache_read_input_token_cost_above_200k_tokens" not in info assert info["supports_assistant_prefill"] is False assert info["supports_function_calling"] is True @@ -126,11 +143,16 @@ def test_opus_4_6_model_pricing_and_capabilities(): assert info["supports_reasoning"] is True assert info["supports_tool_choice"] is True assert info["supports_vision"] is True - assert info["tool_use_system_prompt_tokens"] == config["tool_use_system_prompt_tokens"] + assert ( + info["tool_use_system_prompt_tokens"] + == config["tool_use_system_prompt_tokens"] + ) def test_opus_4_6_bedrock_regional_model_pricing(): - json_path = os.path.join(os.path.dirname(__file__), "../../model_prices_and_context_window.json") + json_path = os.path.join( + os.path.dirname(__file__), "../../model_prices_and_context_window.json" + ) with open(json_path) as f: model_data = json.load(f) @@ -140,40 +162,24 @@ def test_opus_4_6_bedrock_regional_model_pricing(): "output_cost_per_token": 2.5e-05, "cache_creation_input_token_cost": 6.25e-06, "cache_read_input_token_cost": 5e-07, - "input_cost_per_token_above_200k_tokens": 1e-05, - "output_cost_per_token_above_200k_tokens": 3.75e-05, - "cache_creation_input_token_cost_above_200k_tokens": 1.25e-05, - "cache_read_input_token_cost_above_200k_tokens": 1e-06, }, "us.anthropic.claude-opus-4-6-v1": { "input_cost_per_token": 5.5e-06, "output_cost_per_token": 2.75e-05, "cache_creation_input_token_cost": 6.875e-06, "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token_above_200k_tokens": 1.1e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, }, "eu.anthropic.claude-opus-4-6-v1": { "input_cost_per_token": 5.5e-06, "output_cost_per_token": 2.75e-05, "cache_creation_input_token_cost": 6.875e-06, "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token_above_200k_tokens": 1.1e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, }, "au.anthropic.claude-opus-4-6-v1": { "input_cost_per_token": 5.5e-06, "output_cost_per_token": 2.75e-05, "cache_creation_input_token_cost": 6.875e-06, "cache_read_input_token_cost": 5.5e-07, - "input_cost_per_token_above_200k_tokens": 1.1e-05, - "output_cost_per_token_above_200k_tokens": 4.125e-05, - "cache_creation_input_token_cost_above_200k_tokens": 1.375e-05, - "cache_read_input_token_cost_above_200k_tokens": 1.1e-06, }, } @@ -186,12 +192,18 @@ def test_opus_4_6_bedrock_regional_model_pricing(): assert info["max_tokens"] == 128000 assert info["supports_assistant_prefill"] is False assert info["tool_use_system_prompt_tokens"] == 346 + assert "input_cost_per_token_above_200k_tokens" not in info + assert "output_cost_per_token_above_200k_tokens" not in info + assert "cache_creation_input_token_cost_above_200k_tokens" not in info + assert "cache_read_input_token_cost_above_200k_tokens" not in info for key, value in expected.items(): assert info[key] == value def test_opus_4_6_alias_and_dated_metadata_match(): - json_path = os.path.join(os.path.dirname(__file__), "../../model_prices_and_context_window.json") + json_path = os.path.join( + os.path.dirname(__file__), "../../model_prices_and_context_window.json" + ) with open(json_path) as f: model_data = json.load(f) @@ -207,10 +219,6 @@ def test_opus_4_6_alias_and_dated_metadata_match(): "cache_creation_input_token_cost", "cache_creation_input_token_cost_above_1hr", "cache_read_input_token_cost", - "input_cost_per_token_above_200k_tokens", - "output_cost_per_token_above_200k_tokens", - "cache_creation_input_token_cost_above_200k_tokens", - "cache_read_input_token_cost_above_200k_tokens", "supports_assistant_prefill", "tool_use_system_prompt_tokens", ] From ca3457b091c9de8f8bdc05f917e49853ebe27c84 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 27 Mar 2026 11:25:03 -0700 Subject: [PATCH 072/117] Pin nodejs-wheel-binaries in CI workflows running prisma generate prisma generate internally runs `npm install prisma@5.4.2` against the npm registry at runtime. Without a bundled Node.js, this causes ECONNRESET failures on flaky GitHub Actions network and leaves the npm transitive dependency tree unpinned. Pre-install nodejs-wheel-binaries==24.13.1 (matching the Dockerfiles) so prisma uses the bundled Node/npm instead of fetching from the registry. Co-Authored-By: Claude Opus 4.6 (1M context) --- .github/workflows/test-litellm-matrix.yml | 3 +++ .github/workflows/test-proxy-e2e-azure-batches.yml | 3 +++ 2 files changed, 6 insertions(+) diff --git a/.github/workflows/test-litellm-matrix.yml b/.github/workflows/test-litellm-matrix.yml index c354df565a2..860d25636c5 100644 --- a/.github/workflows/test-litellm-matrix.yml +++ b/.github/workflows/test-litellm-matrix.yml @@ -156,7 +156,10 @@ jobs: poetry run pip install --force-reinstall --no-deps -e enterprise/ - name: Generate Prisma client + env: + PRISMA_BINARY_CACHE_DIR: ${{ runner.temp }}/prisma-cache run: | + poetry run pip install nodejs-wheel-binaries==24.13.1 poetry run prisma generate --schema litellm/proxy/schema.prisma - name: Run tests - ${{ matrix.test-group.name }} diff --git a/.github/workflows/test-proxy-e2e-azure-batches.yml b/.github/workflows/test-proxy-e2e-azure-batches.yml index d5130b07f13..7cbbe0b338f 100644 --- a/.github/workflows/test-proxy-e2e-azure-batches.yml +++ b/.github/workflows/test-proxy-e2e-azure-batches.yml @@ -68,7 +68,10 @@ jobs: poetry run pip install --force-reinstall --no-deps -e enterprise/ - name: Generate Prisma client + env: + PRISMA_BINARY_CACHE_DIR: ${{ runner.temp }}/prisma-cache run: | + poetry run pip install nodejs-wheel-binaries==24.13.1 poetry run prisma generate --schema litellm/proxy/schema.prisma - name: Run Prisma migrations From c4159a2ade2a462c33d9e3f142715a18cf702a0e Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Sat, 28 Mar 2026 00:01:33 +0530 Subject: [PATCH 073/117] Fix codeql --- litellm/proxy/litellm_pre_call_utils.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index a605f3ee23b..ba9577f35d7 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -1355,8 +1355,10 @@ def _update_model_if_team_alias_exists( "New sibling deployments may be unreachable. " "Set LITELLM_ENABLE_TEAM_STALE_ALIAS_BYPASS=true to enable " "team-scoped sibling routing.", - _model, - user_api_key_dict.team_id, + str(_model).replace("\n", "").replace("\r", ""), + str(user_api_key_dict.team_id) + .replace("\n", "") + .replace("\r", ""), ) data["model"] = aliased_target From ec4273ed8b2c59440019d021671d656740988e1e Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 27 Mar 2026 12:04:09 -0700 Subject: [PATCH 074/117] [Infra] Improve CodeQL scanning coverage and schedule Switch query suite from security-extended to security-and-quality to match the default GitHub Advanced Security setup. Run scheduled scans daily instead of weekly. Remove paths-ignore for _experimental/out so build artifacts are also scanned. Co-Authored-By: Claude Opus 4.6 (1M context) --- .github/codeql/codeql-config.yml | 19 +++++++++---------- .github/workflows/codeql.yml | 4 ++-- 2 files changed, 11 insertions(+), 12 deletions(-) diff --git a/.github/codeql/codeql-config.yml b/.github/codeql/codeql-config.yml index 20807685e12..36d70c1d746 100644 --- a/.github/codeql/codeql-config.yml +++ b/.github/codeql/codeql-config.yml @@ -1,22 +1,21 @@ name: "LiteLLM CodeQL config" -# Use security-extended suite instead of security-and-quality to avoid -# result sets > 2 GiB on this codebase that cause fatal OOM failures. queries: - - uses: security-extended + - uses: security-and-quality -# These two queries are security queries included in security-extended that -# individually produce result sets > 2 GiB on this codebase, causing fatal -# OOM failures. Exclude them as a safety net until CI confirms they no longer -# OOM; drop these exclusions in a follow-up once verified. +# Known OOM queries on large Python codebases: +# CodeQL builds a full data flow graph in memory. These two queries trace +# sensitive data through every log call / regex pattern, causing combinatorial +# path explosion on codebases with extensive logging like LiteLLM (>2 GiB +# result sets). This is a known CodeQL scaling limitation, not a code issue. +# Re-test periodically as CodeQL improves or the codebase refactors logging. query-filters: - exclude: - id: py/clear-text-logging-sensitive-data # CWE-312 — > 2 GiB result set + id: py/clear-text-logging-sensitive-data # CWE-312 - exclude: - id: py/polynomial-redos # CWE-730 — > 2 GiB result set + id: py/polynomial-redos # CWE-730 paths-ignore: - tests - docs - "**/*.md" - - litellm/proxy/_experimental/out diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 91b88b66a1f..cd3a5499185 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -6,8 +6,8 @@ on: pull_request: branches: [main] schedule: - # Run weekly on Sundays at 04:00 UTC - - cron: "0 4 * * 0" + # Run daily at 04:00 UTC + - cron: "0 4 * * *" concurrency: group: ${{ github.workflow }}-${{ github.ref }} From d533b432fdb2d9dbfd4410c5f30e552611b137ec Mon Sep 17 00:00:00 2001 From: michelligabriele Date: Fri, 27 Mar 2026 14:42:05 +0100 Subject: [PATCH 075/117] fix(proxy): enforce budget limits across multi-pod deployments via Redis-backed spend counters MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Budget checks on API keys, teams, and team members were not enforced in multi-pod deployments because user_api_key_cache is intentionally in-memory-only. Each pod tracked spend independently, so with N pods the effective budget was N × max_budget. Introduces a separate spend_counter_cache (DualCache wired to redis_usage_cache) with atomic increment/read helpers: - increment_spend_counters(): awaited in cost callback (not create_task) to update both in-memory and Redis before the next auth check - get_current_spend(): reads Redis first (cross-pod authoritative), falls back to in-memory, then to cached object .spend from DB Budget check functions (_virtual_key_max_budget_check, _team_max_budget_check, _check_team_member_budget) now read spend via get_current_spend() instead of cached object .spend fields. When Redis is not configured, falls back to in-memory-only counters (same as current single-instance behavior). Fixes #23714 --- litellm/proxy/auth/auth_checks.py | 77 +++++--- litellm/proxy/auth/user_api_key_auth.py | 16 +- .../proxy/common_utils/reset_budget_job.py | 84 +++++++- .../proxy/hooks/proxy_track_cost_callback.py | 18 +- litellm/proxy/proxy_server.py | 132 +++++++++++++ .../proxy/auth/test_auth_checks.py | 151 +++++++++++++++ tests/test_litellm/proxy/test_proxy_server.py | 180 ++++++++++++++++++ 7 files changed, 618 insertions(+), 40 deletions(-) diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index 815393467de..0efc3dab638 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -2876,7 +2876,15 @@ async def _virtual_key_max_budget_check( Triggers a budget alert if the token is over it's max budget. """ - if valid_token.spend is not None and valid_token.max_budget is not None: + if valid_token.max_budget is not None: + from litellm.proxy.proxy_server import get_current_spend + + # Read spend from cross-pod counter (Redis-first) or cached object (fallback) + spend = await get_current_spend( + counter_key=f"spend:key:{valid_token.token}", + fallback_spend=valid_token.spend or 0.0, + ) + #################################### # collect information for alerting # #################################### @@ -2888,7 +2896,7 @@ async def _virtual_key_max_budget_check( call_info = CallInfo( token=valid_token.token, - spend=valid_token.spend, + spend=spend, max_budget=valid_token.max_budget, soft_budget=valid_token.soft_budget, user_id=valid_token.user_id, @@ -2909,9 +2917,9 @@ async def _virtual_key_max_budget_check( # collect information for alerting # #################################### - if valid_token.spend >= valid_token.max_budget: + if spend >= valid_token.max_budget: raise litellm.BudgetExceededError( - current_cost=valid_token.spend, + current_cost=spend, max_budget=valid_token.max_budget, ) @@ -3042,6 +3050,14 @@ async def _check_team_member_budget( team_member_budget = team_membership.litellm_budget_table.max_budget team_member_spend = team_membership.spend or 0.0 + # Read from cross-pod counter (Redis-first) if available + from litellm.proxy.proxy_server import get_current_spend + + team_member_spend = await get_current_spend( + counter_key=f"spend:team_member:{valid_token.user_id}:{team_object.team_id}", + fallback_spend=team_member_spend, + ) + if team_member_spend >= team_member_budget: raise litellm.BudgetExceededError( current_cost=team_member_spend, @@ -3065,33 +3081,40 @@ async def _team_max_budget_check( if ( team_object is not None and team_object.max_budget is not None - and team_object.spend is not None - and team_object.spend > team_object.max_budget ): - if valid_token: - call_info = CallInfo( - token=valid_token.token, - spend=team_object.spend, - max_budget=team_object.max_budget, - user_id=valid_token.user_id, - team_id=valid_token.team_id, - team_alias=valid_token.team_alias, - organization_id=valid_token.org_id, - event_group=Litellm_EntityType.TEAM, - ) - asyncio.create_task( - proxy_logging_obj.budget_alerts( - type="team_budget", - user_info=call_info, - ) - ) + from litellm.proxy.proxy_server import get_current_spend - raise litellm.BudgetExceededError( - current_cost=team_object.spend, - max_budget=team_object.max_budget, - message=f"Budget has been exceeded! Team={team_object.team_id} Current cost: {team_object.spend}, Max budget: {team_object.max_budget}", + # Read spend from cross-pod counter (Redis-first) or cached object (fallback) + spend = await get_current_spend( + counter_key=f"spend:team:{team_object.team_id}", + fallback_spend=team_object.spend or 0.0, ) + if spend > team_object.max_budget: + if valid_token: + call_info = CallInfo( + token=valid_token.token, + spend=spend, + max_budget=team_object.max_budget, + user_id=valid_token.user_id, + team_id=valid_token.team_id, + team_alias=valid_token.team_alias, + organization_id=valid_token.org_id, + event_group=Litellm_EntityType.TEAM, + ) + asyncio.create_task( + proxy_logging_obj.budget_alerts( + type="team_budget", + user_info=call_info, + ) + ) + + raise litellm.BudgetExceededError( + current_cost=spend, + max_budget=team_object.max_budget, + message=f"Budget has been exceeded! Team={team_object.team_id} Current cost: {spend}, Max budget: {team_object.max_budget}", + ) + async def _team_soft_budget_check( team_object: Optional[LiteLLM_TeamTable], diff --git a/litellm/proxy/auth/user_api_key_auth.py b/litellm/proxy/auth/user_api_key_auth.py index eba787c63b3..9dd2ab18c8a 100644 --- a/litellm/proxy/auth/user_api_key_auth.py +++ b/litellm/proxy/auth/user_api_key_auth.py @@ -1303,9 +1303,21 @@ async def _user_api_key_auth_builder( # noqa: PLR0915 team_member_info.litellm_budget_table.max_budget ) if team_member_budget is not None and team_member_budget > 0: - if valid_token.team_member_spend > team_member_budget: + # Read from cross-pod counter (Redis-first) if available + from litellm.proxy.proxy_server import get_current_spend + + team_member_spend = valid_token.team_member_spend + if ( + valid_token.user_id is not None + and valid_token.team_id is not None + ): + team_member_spend = await get_current_spend( + counter_key=f"spend:team_member:{valid_token.user_id}:{valid_token.team_id}", + fallback_spend=team_member_spend, + ) + if team_member_spend > team_member_budget: raise litellm.BudgetExceededError( - current_cost=valid_token.team_member_spend, + current_cost=team_member_spend, max_budget=team_member_budget, ) diff --git a/litellm/proxy/common_utils/reset_budget_job.py b/litellm/proxy/common_utils/reset_budget_job.py index 674214b19e5..5db919659ed 100644 --- a/litellm/proxy/common_utils/reset_budget_job.py +++ b/litellm/proxy/common_utils/reset_budget_job.py @@ -54,16 +54,49 @@ class ResetBudgetJob: """ Resets the budget for all LiteLLM Team Members if their budget has expired """ + budget_ids = [ + budget.budget_id + for budget in budgets_to_reset + if budget.budget_id is not None + ] + + # Reset spend counters for affected team members. + # Reset Redis directly so a transient failure doesn't leave stale + # counters that get_current_spend would read as authoritative. + try: + from litellm.proxy.proxy_server import spend_counter_cache + + memberships = ( + await self.prisma_client.db.litellm_teammembership.find_many( + where={"budget_id": {"in": budget_ids}} + ) + ) + for m in memberships: + counter_key = f"spend:team_member:{m.user_id}:{m.team_id}" + # Always reset in-memory + spend_counter_cache.in_memory_cache.set_cache( + key=counter_key, value=0.0 + ) + # Explicitly reset Redis with warning on failure + if spend_counter_cache.redis_cache is not None: + try: + await spend_counter_cache.redis_cache.async_set_cache( + key=counter_key, value=0.0 + ) + except Exception as redis_err: + verbose_proxy_logger.warning( + "Failed to reset team member spend counter in Redis %s: %s. " + "Budget may be over-enforced until counter expires.", + counter_key, + redis_err, + ) + except Exception as e: + verbose_proxy_logger.warning( + "Failed to reset team member spend counters: %s", e + ) + return await self.prisma_client.db.litellm_teammembership.update_many( - where={ - "budget_id": { - "in": [ - budget.budget_id - for budget in budgets_to_reset - if budget.budget_id is not None - ] - } - }, + where={"budget_id": {"in": budget_ids}}, data={ "spend": 0, }, @@ -531,6 +564,39 @@ class ResetBudgetJob: """ try: item.spend = 0.0 + + # Reset the cross-pod spend counter. + # Reset Redis directly (not via DualCache) so a Redis failure + # doesn't silently leave a stale counter that get_current_spend + # would read as authoritative, permanently blocking the user. + from litellm.proxy.proxy_server import spend_counter_cache + + counter_key = None + if item_type == "key" and hasattr(item, "token") and item.token is not None: + counter_key = f"spend:key:{item.token}" + elif item_type == "team" and hasattr(item, "team_id") and item.team_id is not None: + counter_key = f"spend:team:{item.team_id}" + + if counter_key is not None: + # Always reset in-memory (local fallback) + spend_counter_cache.in_memory_cache.set_cache( + key=counter_key, value=0.0 + ) + # Explicitly reset Redis with warning on failure + if spend_counter_cache.redis_cache is not None: + try: + await spend_counter_cache.redis_cache.async_set_cache( + key=counter_key, value=0.0 + ) + except Exception as redis_err: + verbose_proxy_logger.warning( + "Failed to reset spend counter in Redis for %s key=%s: %s. " + "Budget may be over-enforced until counter expires.", + item_type, + counter_key, + redis_err, + ) + if hasattr(item, "budget_duration") and item.budget_duration is not None: # Get standardized reset time based on budget duration from litellm.proxy.common_utils.timezone_utils import ( diff --git a/litellm/proxy/hooks/proxy_track_cost_callback.py b/litellm/proxy/hooks/proxy_track_cost_callback.py index 220c5066a6d..948d6dd33af 100644 --- a/litellm/proxy/hooks/proxy_track_cost_callback.py +++ b/litellm/proxy/hooks/proxy_track_cost_callback.py @@ -135,7 +135,11 @@ class _ProxyDBLogger(CustomLogger): start_time=None, end_time=None, # start/end time for completion ): - from litellm.proxy.proxy_server import proxy_logging_obj, update_cache + from litellm.proxy.proxy_server import ( + increment_spend_counters, + proxy_logging_obj, + update_cache, + ) verbose_proxy_logger.debug("INSIDE _PROXY_track_cost_callback") try: @@ -194,7 +198,17 @@ class _ProxyDBLogger(CustomLogger): org_id=org_id, ) - # update cache + # Atomically update spend counters (in-memory + Redis) + # for cross-pod budget enforcement. + await increment_spend_counters( + token=user_api_key, + team_id=team_id, + user_id=user_id, + response_cost=response_cost, + ) + + # update cache (fire-and-forget for backward compat: + # cached object fields, soft budget alerts, etc.) asyncio.create_task( update_cache( token=user_api_key, diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 7d3d2ceb533..974f21d4f74 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -1530,6 +1530,9 @@ shared_aiohttp_session: Optional[ user_api_key_cache = DualCache( default_in_memory_ttl=UserAPIKeyCacheTTLEnum.in_memory_cache_ttl.value ) +spend_counter_cache = DualCache( + default_in_memory_ttl=UserAPIKeyCacheTTLEnum.in_memory_cache_ttl.value +) model_max_budget_limiter = _PROXY_VirtualKeyModelMaxBudgetLimiter( dual_cache=user_api_key_cache ) @@ -1694,6 +1697,134 @@ def cost_tracking(): ) +async def get_current_spend(counter_key: str, fallback_spend: float) -> float: + """ + Read current spend from the cross-pod spend counter. + + Reads Redis FIRST (authoritative cross-pod value), not DualCache's + async_get_cache which returns in-memory first. This is critical: + DualCache.async_get_cache returns stale per-pod values because each + pod's in-memory cache is only updated by that pod's own increments. + + Fallback chain: + 1. Redis counter (cross-pod, authoritative) + 2. In-memory counter (single-instance or Redis failure) + 3. Cached object's .spend from DB (cold start, no counter yet) + """ + # 1. Try Redis first (cross-pod authoritative) + if spend_counter_cache.redis_cache is not None: + try: + val = await spend_counter_cache.redis_cache.async_get_cache( + key=counter_key + ) + if val is not None: + return float(val) + except Exception as e: + verbose_proxy_logger.debug( + "get_current_spend: Redis read failed for %s, falling back to in-memory: %s", + counter_key, + e, + ) + + # 2. Fall back to in-memory counter (single-instance or Redis failure) + val = spend_counter_cache.in_memory_cache.get_cache(key=counter_key) + if val is not None: + return float(val) + + # 3. Final fallback: cached object's spend from DB + return fallback_spend + + +async def increment_spend_counters( + token: Optional[str], + team_id: Optional[str], + user_id: Optional[str], + response_cost: Optional[float], +): + """ + Atomically increment spend counters for budget enforcement. + + Uses spend_counter_cache (DualCache with Redis backend when available) + so counters are shared across all pods. Budget check functions read + from these counters via get_current_spend() (Redis-first). + + Awaited (not create_task) in the cost callback, so the counter is + updated before the next request's auth check runs. + """ + if response_cost is None or response_cost == 0: + return + + if token is not None: + # token arrives pre-hashed from metadata["user_api_key"] (auth flow + # hashes raw "sk-..." keys before they reach the callback). The + # startswith("sk-") check is a safety net matching update_cache — + # if a raw key somehow arrives, hash it; otherwise use as-is to + # avoid double-hashing (budget checks read valid_token.token which + # is single-hashed). + hashed_token = ( + hash_token(token=token) + if isinstance(token, str) and token.startswith("sk-") + else token + ) + await _init_and_increment_spend_counter( + counter_key=f"spend:key:{hashed_token}", + source_cache_key=hashed_token, + increment=response_cost, + ) + + if team_id is not None: + await _init_and_increment_spend_counter( + counter_key=f"spend:team:{team_id}", + source_cache_key=f"team_id:{team_id}", + increment=response_cost, + ) + + if user_id is not None and team_id is not None: + await _init_and_increment_spend_counter( + counter_key=f"spend:team_member:{user_id}:{team_id}", + source_cache_key=f"team_membership:{user_id}:{team_id}", + increment=response_cost, + ) + + +async def _init_and_increment_spend_counter( + counter_key: str, + source_cache_key: str, + increment: float, +): + """ + Initialize counter from cached object's DB-loaded spend if not yet set, + then atomically increment in both in-memory and Redis. + + On first access per pod: + 1. Check spend_counter_cache (in-memory -> Redis via DualCache for init check) + 2. If not found anywhere, read base spend from user_api_key_cache (DB-loaded object) + 3. Seed counter via async_increment_cache (not async_set_cache) to avoid a + check-then-set race: if two pods cold-start simultaneously, both may see + the counter as absent and seed it. Using increment instead of set means + the worst case is over-counting (conservative — blocks slightly early) + rather than under-counting (would allow overspend). + 4. Increment atomically (both in-memory + Redis) + """ + current = await spend_counter_cache.async_get_cache(key=counter_key) + if current is None: + source = await user_api_key_cache.async_get_cache(key=source_cache_key) + base_spend = 0.0 + if source is not None: + if isinstance(source, dict): + base_spend = source.get("spend", 0.0) or 0.0 + else: + base_spend = getattr(source, "spend", 0.0) or 0.0 + if base_spend > 0: + await spend_counter_cache.async_increment_cache( + key=counter_key, value=base_spend + ) + + await spend_counter_cache.async_increment_cache( + key=counter_key, value=increment + ) + + async def update_cache( # noqa: PLR0915 token: Optional[str], user_id: Optional[str], @@ -2528,6 +2659,7 @@ class ProxyConfig: ): ## INIT PROXY REDIS USAGE CLIENT ## redis_usage_cache = litellm.cache.cache + spend_counter_cache.redis_cache = redis_usage_cache # Note: PKCE verifier storage uses redis_usage_cache directly (not # user_api_key_cache) to avoid routing all API-key lookups through Redis. diff --git a/tests/test_litellm/proxy/auth/test_auth_checks.py b/tests/test_litellm/proxy/auth/test_auth_checks.py index 69188fd200e..bd659ed518f 100644 --- a/tests/test_litellm/proxy/auth/test_auth_checks.py +++ b/tests/test_litellm/proxy/auth/test_auth_checks.py @@ -29,10 +29,13 @@ from litellm.proxy._types import ( from litellm.proxy.auth.auth_checks import ( ExperimentalUIJWTToken, _can_object_call_vector_stores, + _check_team_member_budget, _get_fuzzy_user_object, _get_team_db_check, _log_budget_lookup_failure, + _team_max_budget_check, _virtual_key_max_budget_alert_check, + _virtual_key_max_budget_check, _virtual_key_soft_budget_check, get_key_object, get_user_object, @@ -1629,3 +1632,151 @@ async def test_custom_auth_common_checks_opt_in(): parent_otel_span=None, ) mock_common.assert_called_once() + + +# ===================================================================== +# Spend counter budget check tests (v2 — Redis-backed spend counters) +# ===================================================================== + + +@pytest.mark.asyncio +async def test_virtual_key_budget_check_reads_from_spend_counter(): + """Budget check should use get_current_spend when counter exists, + even if cached object shows lower spend.""" + from litellm.proxy.utils import ProxyLogging + + valid_token = UserAPIKeyAuth( + token="test-hashed-token", + spend=0.0, # stale — counter has 1.5 + max_budget=1.0, + user_id="test-user", + ) + + proxy_logging_obj = ProxyLogging(user_api_key_cache=None) + proxy_logging_obj.budget_alerts = AsyncMock() + + async def mock_get_current_spend(counter_key, fallback_spend): + if counter_key == "spend:key:test-hashed-token": + return 1.5 + return fallback_spend + + with patch( + "litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend + ): + with pytest.raises(litellm.BudgetExceededError) as exc_info: + await _virtual_key_max_budget_check( + valid_token=valid_token, + proxy_logging_obj=proxy_logging_obj, + ) + assert exc_info.value.current_cost == 1.5 + assert exc_info.value.max_budget == 1.0 + + +@pytest.mark.asyncio +async def test_virtual_key_budget_check_fallback_no_counter(): + """When counter doesn't exist, budget check should fall back + to cached object's spend via fallback_spend.""" + from litellm.proxy.utils import ProxyLogging + + valid_token = UserAPIKeyAuth( + token="test-hashed-token", + spend=15.0, + max_budget=10.0, + user_id="test-user", + ) + + proxy_logging_obj = ProxyLogging(user_api_key_cache=None) + proxy_logging_obj.budget_alerts = AsyncMock() + + # get_current_spend returns fallback_spend when no counter exists + async def mock_get_current_spend(counter_key, fallback_spend): + return fallback_spend + + with patch( + "litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend + ): + with pytest.raises(litellm.BudgetExceededError) as exc_info: + await _virtual_key_max_budget_check( + valid_token=valid_token, + proxy_logging_obj=proxy_logging_obj, + ) + assert exc_info.value.current_cost == 15.0 + + +@pytest.mark.asyncio +async def test_team_budget_check_reads_from_spend_counter(): + """Team budget check should use get_current_spend when counter exists.""" + from litellm.proxy.utils import ProxyLogging + + team_object = LiteLLM_TeamTable( + team_id="test-team", + spend=0.0, # stale + max_budget=1.0, + ) + valid_token = UserAPIKeyAuth(token="test-token", team_id="test-team") + + proxy_logging_obj = ProxyLogging(user_api_key_cache=None) + proxy_logging_obj.budget_alerts = AsyncMock() + + async def mock_get_current_spend(counter_key, fallback_spend): + if counter_key == "spend:team:test-team": + return 1.5 + return fallback_spend + + with patch( + "litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend + ): + with pytest.raises(litellm.BudgetExceededError) as exc_info: + await _team_max_budget_check( + team_object=team_object, + valid_token=valid_token, + proxy_logging_obj=proxy_logging_obj, + ) + assert exc_info.value.current_cost == 1.5 + + +@pytest.mark.asyncio +async def test_team_member_budget_check_reads_from_spend_counter(): + """Team member budget check should use get_current_spend when counter exists.""" + from litellm.proxy._types import LiteLLM_BudgetTable, LiteLLM_TeamMembership + from litellm.proxy.utils import ProxyLogging + + team_object = LiteLLM_TeamTable(team_id="test-team") + user_object = LiteLLM_UserTable(user_id="test-user") + valid_token = UserAPIKeyAuth( + token="test-token", + user_id="test-user", + team_id="test-team", + ) + + team_membership = LiteLLM_TeamMembership( + user_id="test-user", + team_id="test-team", + spend=0.0, # stale + litellm_budget_table=LiteLLM_BudgetTable(max_budget=1.0), + ) + + proxy_logging_obj = ProxyLogging(user_api_key_cache=None) + + async def mock_get_current_spend(counter_key, fallback_spend): + if counter_key == "spend:team_member:test-user:test-team": + return 1.5 + return fallback_spend + + with patch( + "litellm.proxy.proxy_server.get_current_spend", mock_get_current_spend + ), patch( + "litellm.proxy.auth.auth_checks.get_team_membership", + new_callable=AsyncMock, + return_value=team_membership, + ): + with pytest.raises(litellm.BudgetExceededError) as exc_info: + await _check_team_member_budget( + team_object=team_object, + user_object=user_object, + valid_token=valid_token, + prisma_client=MagicMock(), + user_api_key_cache=MagicMock(), + proxy_logging_obj=proxy_logging_obj, + ) + assert exc_info.value.current_cost == 1.5 diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py index bd6162f225a..11435fd7520 100644 --- a/tests/test_litellm/proxy/test_proxy_server.py +++ b/tests/test_litellm/proxy/test_proxy_server.py @@ -4352,3 +4352,183 @@ async def test_store_model_in_db_db_failure_graceful(monkeypatch): # add_deployment should NOT have been called since store_model_in_db is False mock_proxy_config.add_deployment.assert_not_called() + + +# ===================================================================== +# Spend counter tests (v2 — Redis-backed spend counters) +# ===================================================================== + + +@pytest.mark.asyncio +async def test_get_current_spend_reads_redis_first(): + """get_current_spend should prefer Redis over in-memory.""" + from litellm.caching.dual_cache import DualCache + + counter_cache = DualCache() + + # In-memory has stale value + counter_cache.in_memory_cache.set_cache(key="spend:key:test", value=0.30) + + # Mock Redis with cross-pod authoritative value + mock_redis = AsyncMock() + mock_redis.async_get_cache = AsyncMock(return_value=0.90) + counter_cache.redis_cache = mock_redis + + import litellm.proxy.proxy_server as ps + + original = ps.spend_counter_cache + ps.spend_counter_cache = counter_cache + + try: + from litellm.proxy.proxy_server import get_current_spend + + result = await get_current_spend( + counter_key="spend:key:test", + fallback_spend=0.0, + ) + # Should return Redis value (0.90), not in-memory (0.30) + assert result == 0.90 + mock_redis.async_get_cache.assert_called_once_with(key="spend:key:test") + finally: + ps.spend_counter_cache = original + + +@pytest.mark.asyncio +async def test_get_current_spend_fallback_to_in_memory(): + """When Redis is not configured, get_current_spend uses in-memory.""" + from litellm.caching.dual_cache import DualCache + + counter_cache = DualCache() # no redis_cache + counter_cache.in_memory_cache.set_cache(key="spend:key:test", value=0.50) + + import litellm.proxy.proxy_server as ps + + original = ps.spend_counter_cache + ps.spend_counter_cache = counter_cache + + try: + from litellm.proxy.proxy_server import get_current_spend + + result = await get_current_spend( + counter_key="spend:key:test", + fallback_spend=0.0, + ) + assert result == 0.50 + finally: + ps.spend_counter_cache = original + + +@pytest.mark.asyncio +async def test_increment_spend_counters_initializes_and_increments(): + """Counter should initialize from cached object spend, then increment. + + Uses a pre-hashed token to match production: metadata["user_api_key"] + is always hashed by the auth flow before reaching the cost callback. + """ + from litellm.caching.dual_cache import DualCache + from litellm.proxy._types import LiteLLM_VerificationTokenView, hash_token + + key_cache = DualCache() + counter_cache = DualCache() + + # In production, the auth flow hashes the raw key before it reaches + # the cost callback. Simulate that by passing the hashed token. + hashed_token = hash_token("sk-test-token-for-counter") + + # Simulate a cached key object with existing spend from DB + cached_key = LiteLLM_VerificationTokenView( + token=hashed_token, + spend=5.0, + max_budget=10.0, + ) + key_cache.in_memory_cache.set_cache(key=hashed_token, value=cached_key) + + import litellm.proxy.proxy_server as ps + + original_key_cache = ps.user_api_key_cache + original_counter_cache = ps.spend_counter_cache + ps.user_api_key_cache = key_cache + ps.spend_counter_cache = counter_cache + + try: + from litellm.proxy.proxy_server import increment_spend_counters + + # Pass pre-hashed token (as the cost callback would in production) + await increment_spend_counters( + token=hashed_token, + team_id=None, + user_id=None, + response_cost=0.50, + ) + + # Counter should be: base(5.0) + increment(0.50) = 5.50 + counter = counter_cache.in_memory_cache.get_cache( + key=f"spend:key:{hashed_token}" + ) + assert counter == 5.50 + + # Second increment — counter already exists, just increment + await increment_spend_counters( + token=hashed_token, + team_id=None, + user_id=None, + response_cost=0.25, + ) + + counter = counter_cache.in_memory_cache.get_cache( + key=f"spend:key:{hashed_token}" + ) + assert counter == 5.75 + finally: + ps.user_api_key_cache = original_key_cache + ps.spend_counter_cache = original_counter_cache + + +@pytest.mark.asyncio +async def test_increment_spend_counters_team_and_member(): + """Counter should track team and team member spend separately.""" + from litellm.caching.dual_cache import DualCache + from litellm.proxy._types import LiteLLM_TeamTable + + key_cache = DualCache() + counter_cache = DualCache() + + # Cached team object + team_obj = LiteLLM_TeamTable(team_id="team-1", spend=2.0) + key_cache.in_memory_cache.set_cache(key="team_id:team-1", value=team_obj) + + # Cached team membership + key_cache.in_memory_cache.set_cache( + key="team_membership:user-1:team-1", + value={"user_id": "user-1", "team_id": "team-1", "spend": 1.0}, + ) + + import litellm.proxy.proxy_server as ps + + original_key_cache = ps.user_api_key_cache + original_counter_cache = ps.spend_counter_cache + ps.user_api_key_cache = key_cache + ps.spend_counter_cache = counter_cache + + try: + from litellm.proxy.proxy_server import increment_spend_counters + + await increment_spend_counters( + token=None, + team_id="team-1", + user_id="user-1", + response_cost=0.30, + ) + + team_counter = counter_cache.in_memory_cache.get_cache( + key="spend:team:team-1" + ) + assert team_counter == 2.30 + + member_counter = counter_cache.in_memory_cache.get_cache( + key="spend:team_member:user-1:team-1" + ) + assert member_counter == 1.30 + finally: + ps.user_api_key_cache = original_key_cache + ps.spend_counter_cache = original_counter_cache From e24819afefd385fcad4a8468dcbdcf09c628b648 Mon Sep 17 00:00:00 2001 From: Ryan Crabbe Date: Fri, 27 Mar 2026 13:50:30 -0700 Subject: [PATCH 076/117] fix(sso): pass decoded JWT access token to role mapping during SSO login During SSO login, bearer tokens are stripped from the OAuth response before role mapping runs. Custom role claims encoded inside the JWT access token are lost, so map_jwt_role_to_litellm_role() returns None and the user falls back to internal_user_viewer. process_sso_jwt_access_token() now returns the decoded JWT payload, and a new _sync_user_role_from_jwt_role_map() receives it so jwt_litellm_role_map works correctly during SSO login. --- litellm/proxy/management_endpoints/ui_sso.py | 94 +++++++++- .../proxy/management_endpoints/test_ui_sso.py | 161 +++++++++++++++++- 2 files changed, 246 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/management_endpoints/ui_sso.py b/litellm/proxy/management_endpoints/ui_sso.py index 79a9e9bdf2b..a5a62813b9f 100644 --- a/litellm/proxy/management_endpoints/ui_sso.py +++ b/litellm/proxy/management_endpoints/ui_sso.py @@ -204,7 +204,7 @@ def process_sso_jwt_access_token( sso_jwt_handler: Optional[JWTHandler], result: Union[OpenID, dict, None], role_mappings: Optional["RoleMappings"] = None, -) -> None: +) -> Optional[dict]: """ Process SSO JWT access token and extract team IDs and user role if available. @@ -218,6 +218,12 @@ def process_sso_jwt_access_token( sso_jwt_handler: SSO-specific JWT handler for team ID extraction result: The SSO result object to update with team IDs and role role_mappings: Optional role mappings configuration for group-based role determination + + Returns: + The decoded access token payload dict, or None if decoding failed or + inputs were missing. Callers can pass this to _sync_user_role_from_jwt_role_map + so it has access to custom role claims (e.g. custom_roles) that are + encoded inside the JWT but stripped from received_response. """ if access_token_str and result: import jwt @@ -230,7 +236,7 @@ def process_sso_jwt_access_token( verbose_proxy_logger.debug( "Access token is not a valid JWT (possibly an opaque token), skipping JWT-based extraction" ) - return + return None # Extract team IDs from access token if sso_jwt_handler is available if sso_jwt_handler: @@ -306,6 +312,10 @@ def process_sso_jwt_access_token( f"Set user_role='{user_role}' from JWT access token" ) + return access_token_payload + + return None + @router.get("/sso/key/generate", tags=["experimental"], include_in_schema=False) async def google_login( @@ -817,7 +827,7 @@ async def get_generic_sso_response( ], # sso specific jwt handler - used for restricted sso group access control generic_client_id: str, redirect_url: str, -) -> Tuple[Union[OpenID, dict], Optional[dict]]: # return received response +) -> Tuple[Union[OpenID, dict], Optional[dict], Optional[dict]]: # (result, received_response, access_token_payload) # make generic sso provider from fastapi_sso.sso.base import DiscoveryDocument from fastapi_sso.sso.generic import create_provider @@ -872,6 +882,7 @@ async def get_generic_sso_response( code_verifier: Optional[ str ] = None # assigned inside try; initialized for type tracking + access_token_payload: Optional[dict] = None # decoded JWT access token claims try: token_exchange_params = ( @@ -958,7 +969,7 @@ async def get_generic_sso_response( ) access_token_str = generic_sso.access_token - process_sso_jwt_access_token( + access_token_payload = process_sso_jwt_access_token( access_token_str, sso_jwt_handler, result, role_mappings=role_mappings ) # Delete the single-use PKCE verifier only after all downstream processing @@ -976,7 +987,7 @@ async def get_generic_sso_response( additional_generic_sso_headers_dict, ) verbose_proxy_logger.debug("generic result: %s", result) - return result or {}, received_response + return result or {}, received_response, access_token_payload async def create_team_member_add_task(team_id, user_info): @@ -1176,6 +1187,56 @@ def _build_sso_user_update_data( return update_data +async def _sync_user_role_from_jwt_role_map( + jwt_handler: Optional[JWTHandler], + received_response: Optional[dict], + user_info: Optional[Union[LiteLLM_UserTable, NewUserResponse]], + prisma_client: PrismaClient, + user_api_key_cache: DualCache, + user_defined_values: Optional[SSOUserDefinedValues], +) -> None: + """ + Apply jwt_litellm_role_map during SSO login. + + When jwt_litellm_role_map is configured with sync_user_role_and_teams=True, + this ensures SSO users get the same role mapping as API/JWT users. Without + this, the SSO path falls back to INTERNAL_USER_VIEW_ONLY for roles that + don't directly match LitellmUserRoles enum values. + """ + if jwt_handler is None or received_response is None: + return + if not jwt_handler.litellm_jwtauth.sync_user_role_and_teams: + return + if not jwt_handler.litellm_jwtauth.jwt_litellm_role_map: + return + + mapped_role = jwt_handler.map_jwt_role_to_litellm_role(received_response) + if mapped_role is None: + return + + verbose_proxy_logger.info( + f"SSO jwt_litellm_role_map matched role: {mapped_role.value}" + ) + + # Update user_defined_values so downstream code uses the mapped role + if user_defined_values is not None: + user_defined_values["user_role"] = mapped_role.value + + # Update existing DB record if role differs + if user_info is not None and user_info.user_role != mapped_role.value: + await prisma_client.db.litellm_usertable.update( + where={"user_id": user_info.user_id}, + data={"user_role": mapped_role.value}, + ) + user_info.user_role = mapped_role.value + await user_api_key_cache.async_set_cache( + key=user_info.user_id, + value=user_info.model_dump() + if hasattr(user_info, "model_dump") + else dict(user_info), + ) + + def apply_user_info_values_to_sso_user_defined_values( user_info: Optional[Union[LiteLLM_UserTable, NewUserResponse]], user_defined_values: Optional[SSOUserDefinedValues], @@ -1279,6 +1340,7 @@ async def auth_callback(request: Request, state: Optional[str] = None): # noqa: google_client_id = os.getenv("GOOGLE_CLIENT_ID", None) generic_client_id = os.getenv("GENERIC_CLIENT_ID", None) received_response: Optional[dict] = None + access_token_payload: Optional[dict] = None # get url from request if master_key is None: raise ProxyException( @@ -1307,7 +1369,7 @@ async def auth_callback(request: Request, state: Optional[str] = None): # noqa: ) elif generic_client_id is not None: - result, received_response = await get_generic_sso_response( + result, received_response, access_token_payload = await get_generic_sso_response( request=request, jwt_handler=jwt_handler, generic_client_id=generic_client_id, @@ -1345,6 +1407,8 @@ async def auth_callback(request: Request, state: Optional[str] = None): # noqa: received_response=received_response, generic_client_id=generic_client_id, ui_access_mode=ui_access_mode, + access_token_payload=access_token_payload, + jwt_handler=jwt_handler, return_to=cp_return_to, ) @@ -2417,6 +2481,8 @@ class SSOAuthenticationHandler: received_response: Optional[dict] = None, generic_client_id: Optional[str] = None, ui_access_mode: Optional[Dict] = None, + access_token_payload: Optional[dict] = None, + jwt_handler: Optional[JWTHandler] = None, return_to: Optional[str] = None, ) -> RedirectResponse: import jwt @@ -2498,6 +2564,20 @@ class SSOAuthenticationHandler: alternate_user_id=user_id, ) + # Sync user role from JWT claims via jwt_litellm_role_map (if configured). + # This ensures SSO users get the same role mapping as API/JWT users. + # Use the decoded access_token_payload (not received_response) because + # custom role claims (e.g. custom_roles) are encoded inside the JWT + # access token, which is stripped from received_response. + await _sync_user_role_from_jwt_role_map( + jwt_handler=jwt_handler, + received_response=access_token_payload or received_response, + user_info=user_info, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + user_defined_values=user_defined_values, + ) + user_defined_values = apply_user_info_values_to_sso_user_defined_values( user_info=user_info, user_defined_values=user_defined_values ) @@ -3703,7 +3783,7 @@ async def debug_sso_callback(request: Request): ) elif generic_client_id is not None: - result, _ = await get_generic_sso_response( + result, _, _ = await get_generic_sso_response( request=request, jwt_handler=jwt_handler, generic_client_id=generic_client_id, diff --git a/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py b/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py index ff636ca04ae..f9c7cefcc4c 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py +++ b/tests/test_litellm/proxy/management_endpoints/test_ui_sso.py @@ -24,6 +24,7 @@ from litellm.proxy.management_endpoints.ui_sso import ( MicrosoftSSOHandler, SSOAuthenticationHandler, _setup_team_mappings, + _sync_user_role_from_jwt_role_map, determine_role_from_groups, normalize_email, process_sso_jwt_access_token, @@ -1321,7 +1322,7 @@ async def test_get_generic_sso_response_with_additional_headers(): "fastapi_sso.sso.generic.create_provider", return_value=mock_sso_class ): # Act - result, received_response = await get_generic_sso_response( + result, received_response, _ = await get_generic_sso_response( request=mock_request, jwt_handler=mock_jwt_handler, generic_client_id=generic_client_id, @@ -1383,7 +1384,7 @@ async def test_get_generic_sso_response_with_empty_headers(): "fastapi_sso.sso.generic.create_provider", return_value=mock_sso_class ): # Act - result, received_response = await get_generic_sso_response( + result, received_response, _ = await get_generic_sso_response( request=mock_request, jwt_handler=mock_jwt_handler, generic_client_id=generic_client_id, @@ -5254,3 +5255,159 @@ class TestValidateReturnTo: ) SSOAuthenticationHandler._validate_return_to("https://cp.example.com:3000/ui") + +class TestSyncUserRoleFromJwtRoleMap: + """Tests for _sync_user_role_from_jwt_role_map.""" + + @staticmethod + def _make_jwt_handler(): + from litellm.caching.caching import DualCache + from litellm.proxy._types import ( + JWTLiteLLMRoleMap, + LiteLLM_JWTAuth, + LitellmUserRoles, + ) + + handler = JWTHandler() + handler.update_environment( + prisma_client=None, + user_api_key_cache=DualCache(), + litellm_jwtauth=LiteLLM_JWTAuth( + roles_jwt_field="custom_roles", + user_id_upsert=True, + sync_user_role_and_teams=True, + jwt_litellm_role_map=[ + JWTLiteLLMRoleMap( + jwt_role="my-admin", + litellm_role=LitellmUserRoles.PROXY_ADMIN, + ), + JWTLiteLLMRoleMap( + jwt_role="my-viewer", + litellm_role=LitellmUserRoles.INTERNAL_USER, + ), + ], + ), + ) + return handler + + @staticmethod + def _make_sso_values(user_role=None): + from litellm.proxy._types import SSOUserDefinedValues + + user_id = "testuser@example.com" + return SSOUserDefinedValues( + models=[], + user_id=user_id, + user_email=user_id, + user_role=user_role, + max_budget=None, + budget_duration=None, + ) + + @pytest.mark.asyncio + async def test_stripped_response_has_no_roles(self): + """Bug repro: stripped received_response lacks role claims.""" + from litellm.caching.caching import DualCache + + handler = self._make_jwt_handler() + sso_values = self._make_sso_values() + + await _sync_user_role_from_jwt_role_map( + jwt_handler=handler, + received_response={"token_type": "Bearer", "expires_in": 3600}, + user_info=None, + prisma_client=AsyncMock(), + user_api_key_cache=DualCache(), + user_defined_values=sso_values, + ) + + assert sso_values["user_role"] is None + + @pytest.mark.asyncio + async def test_decoded_access_token_maps_role(self): + """Decoded JWT payload with role claims maps correctly.""" + from litellm.caching.caching import DualCache + from litellm.proxy._types import LitellmUserRoles + + handler = self._make_jwt_handler() + sso_values = self._make_sso_values() + + await _sync_user_role_from_jwt_role_map( + jwt_handler=handler, + received_response={"sub": "testuser@example.com", "custom_roles": ["my-admin"]}, + user_info=None, + prisma_client=AsyncMock(), + user_api_key_cache=DualCache(), + user_defined_values=sso_values, + ) + + assert sso_values["user_role"] == LitellmUserRoles.PROXY_ADMIN.value + + @pytest.mark.asyncio + async def test_existing_user_role_updated_in_db_and_cache(self): + """Existing user with stale role gets updated in DB and cache.""" + from litellm.caching.caching import DualCache + from litellm.proxy._types import LitellmUserRoles + + handler = self._make_jwt_handler() + cache = DualCache() + prisma = AsyncMock() + prisma.db.litellm_usertable.update = AsyncMock() + user_id = "testuser@example.com" + + existing_user = LiteLLM_UserTable( + user_id=user_id, + user_role=LitellmUserRoles.INTERNAL_USER_VIEW_ONLY.value, + ) + await cache.async_set_cache(key=user_id, value=existing_user.model_dump(), ttl=60) + + sso_values = self._make_sso_values( + user_role=LitellmUserRoles.INTERNAL_USER_VIEW_ONLY.value, + ) + + await _sync_user_role_from_jwt_role_map( + jwt_handler=handler, + received_response={"sub": user_id, "custom_roles": ["my-admin"]}, + user_info=existing_user, + prisma_client=prisma, + user_api_key_cache=cache, + user_defined_values=sso_values, + ) + + prisma.db.litellm_usertable.update.assert_called_once_with( + where={"user_id": user_id}, + data={"user_role": LitellmUserRoles.PROXY_ADMIN.value}, + ) + assert existing_user.user_role == LitellmUserRoles.PROXY_ADMIN.value + assert sso_values["user_role"] == LitellmUserRoles.PROXY_ADMIN.value + + @pytest.mark.asyncio + async def test_same_role_no_db_write(self): + """No DB update when the mapped role matches the existing role.""" + from litellm.caching.caching import DualCache + from litellm.proxy._types import LitellmUserRoles + + handler = self._make_jwt_handler() + prisma = AsyncMock() + prisma.db.litellm_usertable.update = AsyncMock() + + existing_user = LiteLLM_UserTable( + user_id="testuser@example.com", + user_role=LitellmUserRoles.PROXY_ADMIN.value, + ) + + sso_values = self._make_sso_values( + user_role=LitellmUserRoles.PROXY_ADMIN.value, + ) + + await _sync_user_role_from_jwt_role_map( + jwt_handler=handler, + received_response={"sub": "testuser@example.com", "custom_roles": ["my-admin"]}, + user_info=existing_user, + prisma_client=prisma, + user_api_key_cache=DualCache(), + user_defined_values=sso_values, + ) + + prisma.db.litellm_usertable.update.assert_not_called() + From 08e29e0a9abaa8885c87ec2f6757fe305b01fef4 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 27 Mar 2026 16:01:20 -0700 Subject: [PATCH 077/117] [Infra] Automated schema.prisma sync and drift detection Sync all 3 schema.prisma copies and add GHA workflows to keep them in sync automatically. Co-Authored-By: Claude Opus 4.6 (1M context) --- .github/workflows/check-schema-sync.yml | 56 +++++++++++++++ .github/workflows/sync-schema.yml | 72 +++++++++++++++++++ .../litellm_proxy_extras/schema.prisma | 14 ++-- schema.prisma | 9 +++ 4 files changed, 146 insertions(+), 5 deletions(-) create mode 100644 .github/workflows/check-schema-sync.yml create mode 100644 .github/workflows/sync-schema.yml diff --git a/.github/workflows/check-schema-sync.yml b/.github/workflows/check-schema-sync.yml new file mode 100644 index 00000000000..33184ba33fa --- /dev/null +++ b/.github/workflows/check-schema-sync.yml @@ -0,0 +1,56 @@ +name: Check Schema Sync + +on: + pull_request: + paths: + - 'schema.prisma' + - 'litellm/proxy/schema.prisma' + - 'litellm-proxy-extras/litellm_proxy_extras/schema.prisma' + +permissions: + contents: read + +jobs: + check-sync: + name: Verify schema.prisma copies match root + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - name: Checkout PR + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + + - name: Reject symlinked schema files + run: | + for f in schema.prisma litellm/proxy/schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma; do + if [ -L "$f" ]; then + echo "::error file=$f::$f is a symlink, which is not allowed" + exit 1 + fi + done + + - name: Check all schemas match root + run: | + EXIT=0 + + diff schema.prisma litellm/proxy/schema.prisma || { + echo "::error file=litellm/proxy/schema.prisma::litellm/proxy/schema.prisma differs from root schema.prisma" + EXIT=1 + } + + diff schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma || { + echo "::error file=litellm-proxy-extras/litellm_proxy_extras/schema.prisma::litellm-proxy-extras/litellm_proxy_extras/schema.prisma differs from root schema.prisma" + EXIT=1 + } + + if [ "$EXIT" -ne 0 ]; then + echo "" + echo "Schema files are out of sync." + echo "The root schema.prisma is the source of truth." + echo "" + echo "To fix, run from the repo root:" + echo " cp schema.prisma litellm/proxy/schema.prisma" + echo " cp schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma" + exit 1 + fi + + echo "All schema copies are in sync with root." diff --git a/.github/workflows/sync-schema.yml b/.github/workflows/sync-schema.yml new file mode 100644 index 00000000000..c9a3aaad3f1 --- /dev/null +++ b/.github/workflows/sync-schema.yml @@ -0,0 +1,72 @@ +name: Sync schema.prisma copies + +on: + pull_request: + paths: + - 'schema.prisma' + +# Scoped to ONLY the permissions needed: +# - contents:write to push the sync commit to the PR branch +# - pull-requests:read is implicit (needed to check out the PR) +permissions: + contents: write + +jobs: + sync: + name: Copy root schema to proxy and proxy-extras + runs-on: ubuntu-latest + timeout-minutes: 5 + # Only run on PRs from branches in THIS repo (not forks). + # Fork PRs cannot push back to the head branch with GITHUB_TOKEN, + # and pull_request events from forks have read-only tokens anyway. + # Also reject PRs from branches named after protected branches to + # prevent pushing directly to main/master. + if: >- + github.event.pull_request.head.repo.full_name == github.repository + && github.head_ref != 'main' + && github.head_ref != 'master' + steps: + - name: Checkout PR branch by SHA + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + # Use the merge commit SHA for safety — github.head_ref is an + # attacker-controlled string (the branch name) and could contain + # unusual characters that cause unexpected git behavior. + ref: ${{ github.event.pull_request.head.sha }} + + - name: Reject symlinked schema files + run: | + for f in schema.prisma litellm/proxy/schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma; do + if [ -L "$f" ]; then + echo "::error file=$f::$f is a symlink, which is not allowed" + exit 1 + fi + done + + - name: Copy root schema to other locations + run: | + cp schema.prisma litellm/proxy/schema.prisma + cp schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma + + - name: Check for changes + id: diff + run: | + if git diff --quiet -- litellm/proxy/schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma; then + echo "changed=false" >> "$GITHUB_OUTPUT" + echo "Schemas already in sync. Nothing to do." + else + echo "changed=true" >> "$GITHUB_OUTPUT" + echo "Schema copies need updating." + fi + + - name: Commit synced schemas + if: steps.diff.outputs.changed == 'true' + run: | + # Push to the PR's head branch (need the branch name for git push). + # We checked out by SHA above for safety, so configure the push target explicitly. + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git checkout -B "$GITHUB_HEAD_REF" + git add -- litellm/proxy/schema.prisma litellm-proxy-extras/litellm_proxy_extras/schema.prisma + git commit -m "chore: sync schema.prisma copies from root" + git push origin "HEAD:$GITHUB_HEAD_REF" diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index a2c83295403..46be6b31e1f 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -320,11 +320,15 @@ model LiteLLM_MCPServerTable { is_byok Boolean @default(false) byok_description String[] @default([]) byok_api_key_help_url String? - approval_status String @default("approved") - submitted_by String? - submitted_at DateTime? - reviewed_at DateTime? - review_notes String? + source_url String? + // BYOM submission lifecycle + approval_status String? @default("active") + submitted_by String? + submitted_at DateTime? + reviewed_at DateTime? + review_notes String? + + @@index([approval_status]) } // Per-user BYOK credentials for MCP servers diff --git a/schema.prisma b/schema.prisma index fde9a466a28..46be6b31e1f 100644 --- a/schema.prisma +++ b/schema.prisma @@ -320,6 +320,15 @@ model LiteLLM_MCPServerTable { is_byok Boolean @default(false) byok_description String[] @default([]) byok_api_key_help_url String? + source_url String? + // BYOM submission lifecycle + approval_status String? @default("active") + submitted_by String? + submitted_at DateTime? + reviewed_at DateTime? + review_notes String? + + @@index([approval_status]) } // Per-user BYOK credentials for MCP servers From e0e0c5e2937070a608b6bc8bb66c782c03394b7b Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 27 Mar 2026 16:14:06 -0700 Subject: [PATCH 078/117] [Infra] Fix zizmor artipacked warnings on schema sync workflows Add persist-credentials: false to check-schema-sync (read-only, no push needed). Explicitly set persist-credentials: true on sync-schema (required for git push). Co-Authored-By: Claude Opus 4.6 (1M context) --- .github/workflows/check-schema-sync.yml | 2 ++ .github/workflows/sync-schema.yml | 1 + 2 files changed, 3 insertions(+) diff --git a/.github/workflows/check-schema-sync.yml b/.github/workflows/check-schema-sync.yml index 33184ba33fa..0e5e2804e60 100644 --- a/.github/workflows/check-schema-sync.yml +++ b/.github/workflows/check-schema-sync.yml @@ -18,6 +18,8 @@ jobs: steps: - name: Checkout PR uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false - name: Reject symlinked schema files run: | diff --git a/.github/workflows/sync-schema.yml b/.github/workflows/sync-schema.yml index c9a3aaad3f1..72a5c56293e 100644 --- a/.github/workflows/sync-schema.yml +++ b/.github/workflows/sync-schema.yml @@ -33,6 +33,7 @@ jobs: # attacker-controlled string (the branch name) and could contain # unusual characters that cause unexpected git behavior. ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: true # needed for git push - name: Reject symlinked schema files run: | From 46b92da0bd7d9a1175f5768fee00b58eb3490715 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 27 Mar 2026 16:24:03 -0700 Subject: [PATCH 079/117] [Infra] Add migration for restored BYOM lifecycle fields The schema sync adopted the proxy version which includes source_url, approval_status, and other BYOM fields. These were previously dropped in migration 20260311180521 due to schema drift. This migration restores them to match the now-unified schema. Co-Authored-By: Claude Opus 4.6 (1M context) --- .../migration.sql | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260327232350_restore_mcp_byom_fields/migration.sql diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260327232350_restore_mcp_byom_fields/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260327232350_restore_mcp_byom_fields/migration.sql new file mode 100644 index 00000000000..db965192512 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260327232350_restore_mcp_byom_fields/migration.sql @@ -0,0 +1,14 @@ +-- AlterTable: Restore BYOM lifecycle fields to LiteLLM_MCPServerTable +-- These were dropped in 20260311180521_schema_sync due to schema drift. +-- The proxy schema (source of truth) retained them. +ALTER TABLE "LiteLLM_MCPServerTable" + ADD COLUMN IF NOT EXISTS "source_url" TEXT, + ADD COLUMN IF NOT EXISTS "approval_status" TEXT DEFAULT 'active', + ADD COLUMN IF NOT EXISTS "submitted_by" TEXT, + ADD COLUMN IF NOT EXISTS "submitted_at" TIMESTAMP(3), + ADD COLUMN IF NOT EXISTS "reviewed_at" TIMESTAMP(3), + ADD COLUMN IF NOT EXISTS "review_notes" TEXT; + +-- CreateIndex +CREATE INDEX IF NOT EXISTS "LiteLLM_MCPServerTable_approval_status_idx" + ON "LiteLLM_MCPServerTable"("approval_status"); From a074d1d68b4d088e9b1040a60bf522e89db9b53f Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 27 Mar 2026 16:45:12 -0700 Subject: [PATCH 080/117] [Infra] Mirror litellm_table_patch source changes (no binaries) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cherry-pick source-only changes from litellm_table_patch, excluding build artifacts from the incident response period. - Remove destructive DROP COLUMN migration (20260311180521_schema_sync) - Remove now-unnecessary restore migration (20260327232350) - Bump litellm-proxy-extras 0.4.60 → 0.4.61 - Add regression test to block future DROP COLUMN migrations - Fix double error handling in getTeamPermissionsCall Co-Authored-By: Claude Opus 4.6 (1M context) --- .../20260311180521_schema_sync/migration.sql | 11 ---------- .../migration.sql | 14 ------------- litellm-proxy-extras/pyproject.toml | 4 ++-- poetry.lock | 8 +++---- pyproject.toml | 2 +- requirements.txt | 2 +- .../test_litellm_proxy_extras_utils.py | 21 +++++++++++++++++++ .../src/components/networking.tsx | 5 ++--- 8 files changed, 31 insertions(+), 36 deletions(-) delete mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260311180521_schema_sync/migration.sql delete mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260327232350_restore_mcp_byom_fields/migration.sql diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260311180521_schema_sync/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260311180521_schema_sync/migration.sql deleted file mode 100644 index 84eb70ce097..00000000000 --- a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260311180521_schema_sync/migration.sql +++ /dev/null @@ -1,11 +0,0 @@ --- DropIndex -DROP INDEX IF EXISTS "LiteLLM_MCPServerTable_approval_status_idx"; - --- AlterTable -ALTER TABLE "LiteLLM_MCPServerTable" DROP COLUMN IF EXISTS "approval_status", -DROP COLUMN IF EXISTS "review_notes", -DROP COLUMN IF EXISTS "reviewed_at", -DROP COLUMN IF EXISTS "source_url", -DROP COLUMN IF EXISTS "submitted_at", -DROP COLUMN IF EXISTS "submitted_by"; - diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260327232350_restore_mcp_byom_fields/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260327232350_restore_mcp_byom_fields/migration.sql deleted file mode 100644 index db965192512..00000000000 --- a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260327232350_restore_mcp_byom_fields/migration.sql +++ /dev/null @@ -1,14 +0,0 @@ --- AlterTable: Restore BYOM lifecycle fields to LiteLLM_MCPServerTable --- These were dropped in 20260311180521_schema_sync due to schema drift. --- The proxy schema (source of truth) retained them. -ALTER TABLE "LiteLLM_MCPServerTable" - ADD COLUMN IF NOT EXISTS "source_url" TEXT, - ADD COLUMN IF NOT EXISTS "approval_status" TEXT DEFAULT 'active', - ADD COLUMN IF NOT EXISTS "submitted_by" TEXT, - ADD COLUMN IF NOT EXISTS "submitted_at" TIMESTAMP(3), - ADD COLUMN IF NOT EXISTS "reviewed_at" TIMESTAMP(3), - ADD COLUMN IF NOT EXISTS "review_notes" TEXT; - --- CreateIndex -CREATE INDEX IF NOT EXISTS "LiteLLM_MCPServerTable_approval_status_idx" - ON "LiteLLM_MCPServerTable"("approval_status"); diff --git a/litellm-proxy-extras/pyproject.toml b/litellm-proxy-extras/pyproject.toml index 14253ad15db..f03cdf61c2a 100644 --- a/litellm-proxy-extras/pyproject.toml +++ b/litellm-proxy-extras/pyproject.toml @@ -1,6 +1,6 @@ [tool.poetry] name = "litellm-proxy-extras" -version = "0.4.60" +version = "0.4.61" description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package." authors = ["BerriAI"] readme = "README.md" @@ -22,7 +22,7 @@ requires = ["poetry-core"] build-backend = "poetry.core.masonry.api" [tool.commitizen] -version = "0.4.60" +version = "0.4.61" version_files = [ "pyproject.toml:version", "../requirements.txt:litellm-proxy-extras==", diff --git a/poetry.lock b/poetry.lock index b9eba6e0a30..958e0ac65b8 100644 --- a/poetry.lock +++ b/poetry.lock @@ -3219,15 +3219,15 @@ files = [ [[package]] name = "litellm-proxy-extras" -version = "0.4.60" +version = "0.4.61" description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package." optional = true python-versions = "!=2.7.*,!=3.0.*,!=3.1.*,!=3.2.*,!=3.3.*,!=3.4.*,!=3.5.*,!=3.6.*,!=3.7.*,>=3.8" groups = ["main"] markers = "extra == \"proxy\"" files = [ - {file = "litellm_proxy_extras-0.4.60-py3-none-any.whl", hash = "sha256:7abcc811f7430e4b24e7a8ba7186219a4845a955ae7a71d8822bd03fd9fc3393"}, - {file = "litellm_proxy_extras-0.4.60.tar.gz", hash = "sha256:1c122f2a7e0eb58fa4c6d8da9da82ac1fe2869de3510bcfade5c2932af202328"}, + {file = "litellm_proxy_extras-0.4.61-py3-none-any.whl", hash = "sha256:9bd1e57ef51972cacff52172ef5d70b0ff689f57f3d240877667301ab8f8590e"}, + {file = "litellm_proxy_extras-0.4.61.tar.gz", hash = "sha256:dce8e39b1547abf90d912ddd0f2a876beadf789d700ef04c165362d78ad56aee"}, ] [[package]] @@ -8009,4 +8009,4 @@ utils = ["numpydoc"] [metadata] lock-version = "2.1" python-versions = ">=3.9,<4.0" -content-hash = "b4e3ee072f600fab9810024afdd550407d25733a9b6752476aa61826e33bc08e" +content-hash = "8dad0e86d75e574f12c57c9f32614b7b4ea2181e931874046d43a398aef1e998" diff --git a/pyproject.toml b/pyproject.toml index b2a446149e5..ced8c4eb712 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -61,7 +61,7 @@ boto3 = { version = "^1.40.76", optional = true } redisvl = {version = "^0.4.1", optional = true, markers = "python_version >= '3.9' and python_version < '3.14'"} mcp = {version = ">=1.25.0,<2.0.0", optional = true, python = ">=3.10"} a2a-sdk = {version = "^0.3.22", optional = true, python = ">=3.10"} -litellm-proxy-extras = {version = "^0.4.60", optional = true} +litellm-proxy-extras = {version = "^0.4.61", optional = true} rich = {version = "^13.7.1", optional = true} litellm-enterprise = {version = "0.1.35", optional = true} diskcache = {version = "^5.6.1", optional = true} diff --git a/requirements.txt b/requirements.txt index 7ce9ab04d2a..d9e188c6e64 100644 --- a/requirements.txt +++ b/requirements.txt @@ -57,7 +57,7 @@ grpcio>=1.75.0; python_version >= "3.14" sentry_sdk==2.21.0 # for sentry error handling detect-secrets==1.5.0 # Enterprise - secret detection / masking in LLM requests tzdata==2025.1 # IANA time zone database -litellm-proxy-extras==0.4.60 # for proxy extras - e.g. prisma migrations +litellm-proxy-extras==0.4.61 # for proxy extras - e.g. prisma migrations llm-sandbox==0.3.31 # for skill execution in sandbox ### LITELLM PACKAGE DEPENDENCIES python-dotenv==1.0.1 # for env diff --git a/tests/litellm-proxy-extras/test_litellm_proxy_extras_utils.py b/tests/litellm-proxy-extras/test_litellm_proxy_extras_utils.py index c1e3fd5072f..062748f3387 100644 --- a/tests/litellm-proxy-extras/test_litellm_proxy_extras_utils.py +++ b/tests/litellm-proxy-extras/test_litellm_proxy_extras_utils.py @@ -222,6 +222,27 @@ class TestMigrationSQLIdempotency: + "\n".join(violations) ) + _DROP_COLUMN_ALLOWLIST = { + "20250918083359_drop_spec_version_column_from_mcp_table", + "20260213170952_access_group_change_to_model_name", + "20260224203854_add_agent_object_permissions_table", + } + + def test_no_drop_column_statements(self, all_migrations): + """Migrations must not drop columns — dropping columns is destructive + and can break running application instances during rolling deploys.""" + violations = [] + for migration_name, sql in all_migrations: + if migration_name in self._DROP_COLUMN_ALLOWLIST: + continue + for line_num, line in enumerate(sql.splitlines(), 1): + if re.search(r"DROP\s+COLUMN", line, re.IGNORECASE): + violations.append(f" {migration_name}:{line_num}: {line.strip()}") + assert not violations, ( + "DROP COLUMN found in migrations (destructive, not allowed):\n" + + "\n".join(violations) + ) + def test_drop_index_uses_if_exists(self, all_migrations): """DROP INDEX statements must use IF EXISTS""" violations = [] diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index c33ca700fdf..2e8518d00a6 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -7288,12 +7288,11 @@ export const getTeamPermissionsCall = async (accessToken: string, teamId: string if (!response.ok) { const errorData = await response.json(); const errorMessage = deriveErrorMessage(errorData); - handleError(errorMessage); - throw new Error(errorMessage); + console.error("Available permissions fetch failed:", errorMessage); + return { all_available_permissions: [], team_member_permissions: [] }; } const data = await response.json(); - console.log("Team permissions response:", data); return data; } catch (error) { console.error("Failed to get team permissions:", error); From e36ab04a1856b072f6403d8b16236cf7cad3d960 Mon Sep 17 00:00:00 2001 From: Ryan Crabbe Date: Fri, 27 Mar 2026 16:26:00 -0700 Subject: [PATCH 081/117] fix(auth): guard JWTHandler.is_jwt() against None token When JWT auth is enabled and a request arrives without an Authorization header (e.g. health checks, monitoring), api_key is None due to APIKeyHeader(auto_error=False). The is_jwt() call crashes with AttributeError: 'NoneType' object has no attribute 'split'. Return False for None tokens since they are not JWTs. --- litellm/proxy/auth/handle_jwt.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/auth/handle_jwt.py b/litellm/proxy/auth/handle_jwt.py index bfad9f0c3c7..86f7d614b95 100644 --- a/litellm/proxy/auth/handle_jwt.py +++ b/litellm/proxy/auth/handle_jwt.py @@ -89,7 +89,9 @@ class JWTHandler: self.leeway = leeway @staticmethod - def is_jwt(token: str): + def is_jwt(token: Optional[str]): + if token is None: + return False parts = token.split(".") return len(parts) == 3 From 8e3755931ddbd6aff21f064a424d64bdc0139204 Mon Sep 17 00:00:00 2001 From: Ryan Crabbe Date: Fri, 27 Mar 2026 16:50:58 -0700 Subject: [PATCH 082/117] test(auth): add regression tests for JWTHandler.is_jwt(None) Add None-token test cases to both proxy_unit_tests and test_litellm to cover the guard added in the previous commit. Also add -> bool return type annotation to is_jwt(). --- litellm/proxy/auth/handle_jwt.py | 2 +- tests/proxy_unit_tests/test_jwt.py | 3 +++ tests/test_litellm/proxy/auth/test_user_api_key_auth.py | 4 ++++ 3 files changed, 8 insertions(+), 1 deletion(-) diff --git a/litellm/proxy/auth/handle_jwt.py b/litellm/proxy/auth/handle_jwt.py index 86f7d614b95..48bbd8e2150 100644 --- a/litellm/proxy/auth/handle_jwt.py +++ b/litellm/proxy/auth/handle_jwt.py @@ -89,7 +89,7 @@ class JWTHandler: self.leeway = leeway @staticmethod - def is_jwt(token: Optional[str]): + def is_jwt(token: Optional[str]) -> bool: if token is None: return False parts = token.split(".") diff --git a/tests/proxy_unit_tests/test_jwt.py b/tests/proxy_unit_tests/test_jwt.py index 24cf15a3214..a5be1a3a42d 100644 --- a/tests/proxy_unit_tests/test_jwt.py +++ b/tests/proxy_unit_tests/test_jwt.py @@ -1331,6 +1331,9 @@ def test_jwt_handler_is_jwt_static_method(): # Test with empty string assert JWTHandler.is_jwt("") == False + # Test with None (missing Authorization header) + assert JWTHandler.is_jwt(None) == False + @pytest.mark.parametrize( "requested_model, should_work", diff --git a/tests/test_litellm/proxy/auth/test_user_api_key_auth.py b/tests/test_litellm/proxy/auth/test_user_api_key_auth.py index 81ca758983b..6e000f9c0e3 100644 --- a/tests/test_litellm/proxy/auth/test_user_api_key_auth.py +++ b/tests/test_litellm/proxy/auth/test_user_api_key_auth.py @@ -567,6 +567,10 @@ class TestJWTOAuth2Coexistence: assert JWTHandler.is_jwt("Bearer token") is False assert JWTHandler.is_jwt("two.parts") is False + def test_is_jwt_returns_false_for_none(self): + """None token (missing Authorization header) should not be treated as JWT.""" + assert JWTHandler.is_jwt(None) is False + @pytest.mark.asyncio async def test_both_enabled_opaque_token_uses_oauth2(self): """ From a5ff668f5e2a9584796eaf87f569d611f1ff7d5d Mon Sep 17 00:00:00 2001 From: Ryan Crabbe Date: Thu, 26 Mar 2026 14:56:32 -0700 Subject: [PATCH 083/117] fix: add /user/bulk_update to management_routes so proxy admins can access it /user/bulk_update was missing from the management_routes list in _types.py, causing it to fall through to a 403 in non_proxy_admin_allowed_routes_check even for proxy admin users. Also added it to the PROXY_ADMIN_VIEW_ONLY blocked write operations list in route_checks.py to prevent view-only admins from using it. --- litellm/proxy/_types.py | 1 + litellm/proxy/auth/route_checks.py | 1 + 2 files changed, 2 insertions(+) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index b59fc85d4b8..8faf36df4c6 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -525,6 +525,7 @@ class LiteLLMRoutes(enum.Enum): # user "/user/new", "/user/update", + "/user/bulk_update", "/user/delete", "/user/info", "/user/list", diff --git a/litellm/proxy/auth/route_checks.py b/litellm/proxy/auth/route_checks.py index 53cc88e3b11..26bbdef3090 100644 --- a/litellm/proxy/auth/route_checks.py +++ b/litellm/proxy/auth/route_checks.py @@ -629,6 +629,7 @@ class RouteChecks: in [ "/user/new", "/user/delete", + "/user/bulk_update", "/team/new", "/team/update", "/team/delete", From 0c67f274e58f5e402c80aa6221d0b3a769f6875f Mon Sep 17 00:00:00 2001 From: Ryan Crabbe Date: Fri, 27 Mar 2026 18:01:08 -0700 Subject: [PATCH 084/117] docs: add /user/bulk_update to internal_user_endpoints module docstring --- litellm/proxy/management_endpoints/internal_user_endpoints.py | 1 + 1 file changed, 1 insertion(+) diff --git a/litellm/proxy/management_endpoints/internal_user_endpoints.py b/litellm/proxy/management_endpoints/internal_user_endpoints.py index ca8c345f46c..db01e817aa3 100644 --- a/litellm/proxy/management_endpoints/internal_user_endpoints.py +++ b/litellm/proxy/management_endpoints/internal_user_endpoints.py @@ -6,6 +6,7 @@ These are members of a Team on LiteLLM /user/new /user/update +/user/bulk_update /user/delete /user/info /user/list From 98ecf1755008c55cb39ea1952fcd4cabe85d3365 Mon Sep 17 00:00:00 2001 From: Ryan Crabbe Date: Fri, 27 Mar 2026 19:34:24 -0700 Subject: [PATCH 085/117] fix(ui): refactor budget page to React Query hooks and fix crashes - Migrate budget CRUD from manual state to React Query hooks (useBudgets, useCreateBudget, useUpdateBudget, useDeleteBudget) - Fix crash when budget list contains null entries by filtering in query hook - Fix max_budget type from string to number to match DB schema (double precision) - Disable budget_id field in edit modal to prevent accidental changes - Use budget_id as React key instead of array index - Update tests to mock hooks instead of networking functions --- .../budget_management_endpoints.py | 23 ++- .../(dashboard)/hooks/budgets/useBudgets.ts | 70 +++++++ .../src/components/budgets/budget_modal.tsx | 21 +-- .../components/budgets/budget_panel.test.tsx | 178 +++++++++++++----- .../src/components/budgets/budget_panel.tsx | 48 ++--- .../components/budgets/edit_budget_modal.tsx | 41 ++-- 6 files changed, 245 insertions(+), 136 deletions(-) create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/hooks/budgets/useBudgets.ts diff --git a/litellm/proxy/management_endpoints/budget_management_endpoints.py b/litellm/proxy/management_endpoints/budget_management_endpoints.py index 20c7f9ec412..37f13269b1c 100644 --- a/litellm/proxy/management_endpoints/budget_management_endpoints.py +++ b/litellm/proxy/management_endpoints/budget_management_endpoints.py @@ -15,6 +15,7 @@ All /budget management endpoints from datetime import timedelta from fastapi import APIRouter, Depends, HTTPException +from prisma.errors import UniqueViolationError from litellm.litellm_core_utils.duration_parser import duration_in_seconds from litellm.proxy._types import * @@ -90,13 +91,21 @@ async def new_budget( budget_obj_json = budget_obj.model_dump(exclude_none=True) budget_obj_jsonified = jsonify_object(budget_obj_json) # json dump any dictionaries - response = await prisma_client.db.litellm_budgettable.create( - data={ - **budget_obj_jsonified, # type: ignore - "created_by": user_api_key_dict.user_id or litellm_proxy_admin_name, - "updated_by": user_api_key_dict.user_id or litellm_proxy_admin_name, - } # type: ignore - ) + try: + response = await prisma_client.db.litellm_budgettable.create( + data={ + **budget_obj_jsonified, # type: ignore + "created_by": user_api_key_dict.user_id or litellm_proxy_admin_name, + "updated_by": user_api_key_dict.user_id or litellm_proxy_admin_name, + } # type: ignore + ) + except UniqueViolationError: + raise HTTPException( + status_code=400, + detail={ + "error": f"Budget with id '{budget_obj.budget_id}' already exists." + }, + ) return response diff --git a/ui/litellm-dashboard/src/app/(dashboard)/hooks/budgets/useBudgets.ts b/ui/litellm-dashboard/src/app/(dashboard)/hooks/budgets/useBudgets.ts new file mode 100644 index 00000000000..99c170b6791 --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/hooks/budgets/useBudgets.ts @@ -0,0 +1,70 @@ +import { useQuery, useMutation, useQueryClient, UseQueryResult } from "@tanstack/react-query"; +import { createQueryKeys } from "../common/queryKeysFactory"; +import { getBudgetList, budgetCreateCall, budgetUpdateCall, budgetDeleteCall } from "@/components/networking"; +import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized"; +import { budgetItem } from "@/components/budgets/budget_panel"; + +export const budgetKeys = createQueryKeys("budgets"); + +export const useBudgets = (): UseQueryResult => { + const { accessToken } = useAuthorized(); + return useQuery({ + queryKey: budgetKeys.list({}), + queryFn: async () => { + const data = await getBudgetList(accessToken!); + return (data ?? []).filter((item: budgetItem | null): item is budgetItem => item != null); + }, + enabled: Boolean(accessToken), + }); +}; + +export const useCreateBudget = () => { + const { accessToken } = useAuthorized(); + const queryClient = useQueryClient(); + + return useMutation>({ + mutationFn: async (formValues) => { + if (!accessToken) { + throw new Error("Access token is required"); + } + return budgetCreateCall(accessToken, formValues); + }, + onSuccess: () => { + queryClient.invalidateQueries({ queryKey: budgetKeys.all }); + }, + }); +}; + +export const useUpdateBudget = () => { + const { accessToken } = useAuthorized(); + const queryClient = useQueryClient(); + + return useMutation>({ + mutationFn: async (formValues) => { + if (!accessToken) { + throw new Error("Access token is required"); + } + return budgetUpdateCall(accessToken, formValues); + }, + onSuccess: () => { + queryClient.invalidateQueries({ queryKey: budgetKeys.all }); + }, + }); +}; + +export const useDeleteBudget = () => { + const { accessToken } = useAuthorized(); + const queryClient = useQueryClient(); + + return useMutation({ + mutationFn: async (budgetId) => { + if (!accessToken) { + throw new Error("Access token is required"); + } + return budgetDeleteCall(accessToken, budgetId); + }, + onSuccess: () => { + queryClient.invalidateQueries({ queryKey: budgetKeys.all }); + }, + }); +}; diff --git a/ui/litellm-dashboard/src/components/budgets/budget_modal.tsx b/ui/litellm-dashboard/src/components/budgets/budget_modal.tsx index 490613de254..b5ad8aaff34 100644 --- a/ui/litellm-dashboard/src/components/budgets/budget_modal.tsx +++ b/ui/litellm-dashboard/src/components/budgets/budget_modal.tsx @@ -1,17 +1,17 @@ import React from "react"; import { TextInput, Accordion, AccordionHeader, AccordionBody } from "@tremor/react"; import { Button as Button2, Modal, Form, InputNumber, Select } from "antd"; -import { budgetCreateCall } from "../networking"; +import { useCreateBudget } from "@/app/(dashboard)/hooks/budgets/useBudgets"; import NotificationsManager from "../molecules/notifications_manager"; interface BudgetModalProps { isModalVisible: boolean; - accessToken: string | null; setIsModalVisible: React.Dispatch>; - setBudgetList: React.Dispatch>; } -const BudgetModal: React.FC = ({ isModalVisible, accessToken, setIsModalVisible, setBudgetList }) => { +const BudgetModal: React.FC = ({ isModalVisible, setIsModalVisible }) => { const [form] = Form.useForm(); + const createBudget = useCreateBudget(); + const handleOk = () => { setIsModalVisible(false); form.resetFields(); @@ -23,20 +23,15 @@ const BudgetModal: React.FC = ({ isModalVisible, accessToken, }; const handleCreate = async (formValues: Record) => { - if (accessToken == null || accessToken == undefined) { - return; - } try { NotificationsManager.info("Making API Call"); - // setIsModalVisible(true); - const response = await budgetCreateCall(accessToken, formValues); - console.log("key create Response:", response); - setBudgetList((prevData) => (prevData ? [...prevData, response] : [response])); // Check if prevData is null + await createBudget.mutateAsync(formValues); NotificationsManager.success("Budget Created"); form.resetFields(); + setIsModalVisible(false); } catch (error) { - console.error("Error creating the key:", error); - NotificationsManager.fromBackend(`Error creating the key: ${error}`); + console.error("Error creating the budget:", error); + NotificationsManager.fromBackend(`Error creating the budget: ${error}`); } }; diff --git a/ui/litellm-dashboard/src/components/budgets/budget_panel.test.tsx b/ui/litellm-dashboard/src/components/budgets/budget_panel.test.tsx index 534693d3984..ecae379c9f1 100644 --- a/ui/litellm-dashboard/src/components/budgets/budget_panel.test.tsx +++ b/ui/litellm-dashboard/src/components/budgets/budget_panel.test.tsx @@ -1,31 +1,50 @@ -import * as networking from "../networking"; import { fireEvent, render, waitFor, screen } from "@testing-library/react"; import { act } from "@testing-library/react"; +import { QueryClient, QueryClientProvider } from "@tanstack/react-query"; import { afterEach, describe, expect, it, vi } from "vitest"; import BudgetPanel from "./budget_panel"; -vi.mock("../networking", () => ({ - getBudgetList: vi.fn(), - budgetDeleteCall: vi.fn(), +const mockBudgets = [ + { + budget_id: "budget-1", + max_budget: 100, + rpm_limit: 10, + tpm_limit: 1000, + updated_at: "2024-01-01T00:00:00Z", + }, +]; + +vi.mock("@/app/(dashboard)/hooks/budgets/useBudgets", () => ({ + useBudgets: vi.fn().mockReturnValue({ data: [], isLoading: false }), + useDeleteBudget: vi.fn().mockReturnValue({ mutateAsync: vi.fn(), isPending: false }), + useCreateBudget: vi.fn().mockReturnValue({ mutateAsync: vi.fn() }), + useUpdateBudget: vi.fn().mockReturnValue({ mutateAsync: vi.fn() }), })); +import { useBudgets, useDeleteBudget, useCreateBudget, useUpdateBudget } from "@/app/(dashboard)/hooks/budgets/useBudgets"; + +const createQueryClient = () => + new QueryClient({ + defaultOptions: { queries: { retry: false, gcTime: 0 } }, + }); + +function renderWithProviders(ui: React.ReactElement) { + const qc = createQueryClient(); + return render({ui}); +} + describe("Budget Panel", () => { afterEach(() => { vi.clearAllMocks(); }); it("should render the budget panel and load budgets", async () => { - vi.mocked(networking.getBudgetList).mockResolvedValue([ - { - budget_id: "budget-1", - max_budget: "100", - rpm_limit: 10, - tpm_limit: 1000, - updated_at: "2024-01-01T00:00:00Z", - }, - ]); + vi.mocked(useBudgets).mockReturnValue({ + data: mockBudgets, + isLoading: false, + } as any); - render(); + renderWithProviders(); await waitFor(() => { expect(screen.getByText("Create a budget to assign to customers.")).toBeInTheDocument(); @@ -34,17 +53,20 @@ describe("Budget Panel", () => { }); it("should open delete modal when clicking delete icon", async () => { - vi.mocked(networking.getBudgetList).mockResolvedValue([ - { - budget_id: "budget-to-delete", - max_budget: "200", - rpm_limit: 20, - tpm_limit: 2000, - updated_at: "2024-01-02T00:00:00Z", - }, - ]); + vi.mocked(useBudgets).mockReturnValue({ + data: [ + { + budget_id: "budget-to-delete", + max_budget: 200, + rpm_limit: 20, + tpm_limit: 2000, + updated_at: "2024-01-02T00:00:00Z", + }, + ], + isLoading: false, + } as any); - render(); + renderWithProviders(); await waitFor(() => { expect(screen.getByText("budget-to-delete")).toBeInTheDocument(); @@ -62,18 +84,25 @@ describe("Budget Panel", () => { }); it("should successfully delete a budget", async () => { - vi.mocked(networking.getBudgetList).mockResolvedValue([ - { - budget_id: "budget-to-delete", - max_budget: "200", - rpm_limit: 20, - tpm_limit: 2000, - updated_at: "2024-01-02T00:00:00Z", - }, - ]); - vi.mocked(networking.budgetDeleteCall).mockResolvedValue(undefined); + const deleteMutateAsync = vi.fn().mockResolvedValue(undefined); + vi.mocked(useBudgets).mockReturnValue({ + data: [ + { + budget_id: "budget-to-delete", + max_budget: 200, + rpm_limit: 20, + tpm_limit: 2000, + updated_at: "2024-01-02T00:00:00Z", + }, + ], + isLoading: false, + } as any); + vi.mocked(useDeleteBudget).mockReturnValue({ + mutateAsync: deleteMutateAsync, + isPending: false, + } as any); - render(); + renderWithProviders(); await waitFor(() => { expect(screen.getByText("budget-to-delete")).toBeInTheDocument(); @@ -96,24 +125,43 @@ describe("Budget Panel", () => { }); await waitFor(() => { - expect(networking.budgetDeleteCall).toHaveBeenCalledWith("token-123", "budget-to-delete"); - expect(networking.getBudgetList).toHaveBeenCalledTimes(2); // Initial load + refresh after delete + expect(deleteMutateAsync).toHaveBeenCalledWith("budget-to-delete"); + }); + }); + + it("should render empty state without crashing", async () => { + vi.mocked(useBudgets).mockReturnValue({ + data: [], + isLoading: false, + } as any); + + renderWithProviders(); + + await waitFor(() => { + expect(screen.getByText("Create a budget to assign to customers.")).toBeInTheDocument(); }); }); it("should handle delete error", async () => { - vi.mocked(networking.getBudgetList).mockResolvedValue([ - { - budget_id: "budget-to-delete", - max_budget: "200", - rpm_limit: 20, - tpm_limit: 2000, - updated_at: "2024-01-02T00:00:00Z", - }, - ]); - vi.mocked(networking.budgetDeleteCall).mockRejectedValue(new Error("Delete failed")); + const deleteMutateAsync = vi.fn().mockRejectedValue(new Error("Delete failed")); + vi.mocked(useBudgets).mockReturnValue({ + data: [ + { + budget_id: "budget-to-delete", + max_budget: 200, + rpm_limit: 20, + tpm_limit: 2000, + updated_at: "2024-01-02T00:00:00Z", + }, + ], + isLoading: false, + } as any); + vi.mocked(useDeleteBudget).mockReturnValue({ + mutateAsync: deleteMutateAsync, + isPending: false, + } as any); - render(); + renderWithProviders(); await waitFor(() => { expect(screen.getByText("budget-to-delete")).toBeInTheDocument(); @@ -136,10 +184,38 @@ describe("Budget Panel", () => { }); await waitFor(() => { - expect(networking.budgetDeleteCall).toHaveBeenCalledWith("token-123", "budget-to-delete"); + expect(deleteMutateAsync).toHaveBeenCalledWith("budget-to-delete"); + }); + }); + + it("should open edit modal when clicking edit icon", async () => { + vi.mocked(useBudgets).mockReturnValue({ + data: [ + { + budget_id: "budget-to-edit", + max_budget: 300, + rpm_limit: 30, + tpm_limit: 3000, + updated_at: "2024-01-03T00:00:00Z", + }, + ], + isLoading: false, + } as any); + + renderWithProviders(); + + await waitFor(() => { + expect(screen.getByText("budget-to-edit")).toBeInTheDocument(); }); - // Modal should still be open (error handling) - expect(screen.getByText("Delete Budget?")).toBeInTheDocument(); + const editButton = screen.getByTestId("edit-budget-button"); + + act(() => { + fireEvent.click(editButton); + }); + + await waitFor(() => { + expect(screen.getByText("Edit Budget")).toBeInTheDocument(); + }); }); }); diff --git a/ui/litellm-dashboard/src/components/budgets/budget_panel.tsx b/ui/litellm-dashboard/src/components/budgets/budget_panel.tsx index b52ef5ab947..e42d0569652 100644 --- a/ui/litellm-dashboard/src/components/budgets/budget_panel.tsx +++ b/ui/litellm-dashboard/src/components/budgets/budget_panel.tsx @@ -19,12 +19,12 @@ import { TabPanels, Text, } from "@tremor/react"; -import React, { useEffect, useState } from "react"; +import React, { useState } from "react"; import { Prism as SyntaxHighlighter } from "react-syntax-highlighter"; import DeleteResourceModal from "../common_components/DeleteResourceModal"; import TableIconActionButton from "../common_components/IconActionButton/TableIconActionButtons/TableIconActionButton"; import NotificationsManager from "../molecules/notifications_manager"; -import { budgetDeleteCall, getBudgetList } from "../networking"; +import { useBudgets, useDeleteBudget } from "@/app/(dashboard)/hooks/budgets/useBudgets"; import BudgetModal from "./budget_modal"; import EditBudgetModal from "./edit_budget_modal"; import { CREATE_END_USER_CURL_COMMAND, CHAT_COMPLETIONS_CURL_COMMAND, OPENAI_SDK_PYTHON_CODE } from "./constants"; @@ -35,7 +35,7 @@ interface BudgetSettingsPageProps { export interface budgetItem { budget_id: string; - max_budget: string | null; + max_budget: number | null; rpm_limit: number | null; tpm_limit: number | null; updated_at: string; @@ -45,17 +45,10 @@ const BudgetPanel: React.FC = ({ accessToken }) => { const [isCreateModelVisible, setIsCreateModelVisible] = useState(false); const [isEditModalVisible, setIsEditModalVisible] = useState(false); const [selectedBudget, setSelectedBudget] = useState(null); - const [budgetList, setBudgetList] = useState([]); - const [isDeleting, setIsDeleting] = useState(false); const [isDeleteModalVisible, setIsDeleteModalVisible] = useState(false); - useEffect(() => { - if (!accessToken) { - return; - } - getBudgetList(accessToken).then((data) => { - setBudgetList(data); - }); - }, [accessToken]); + + const { data: budgetList = [] } = useBudgets(); + const deleteBudget = useDeleteBudget(); const handleEditCall = async (budget: budgetItem) => { if (accessToken == null) { @@ -74,11 +67,9 @@ const BudgetPanel: React.FC = ({ accessToken }) => { if (!selectedBudget || accessToken == null) { return; } - setIsDeleting(true); try { - await budgetDeleteCall(accessToken, selectedBudget.budget_id); + await deleteBudget.mutateAsync(selectedBudget.budget_id); NotificationsManager.success("Budget deleted."); - await handleUpdateCall(); } catch (error) { console.error("Error deleting budget:", error); if (typeof NotificationsManager.fromBackend === "function") { @@ -87,7 +78,6 @@ const BudgetPanel: React.FC = ({ accessToken }) => { NotificationsManager.info("Failed to delete budget"); } } finally { - setIsDeleting(false); setIsDeleteModalVisible(false); setSelectedBudget(null); } @@ -97,15 +87,6 @@ const BudgetPanel: React.FC = ({ accessToken }) => { setIsDeleteModalVisible(false); }; - const handleUpdateCall = async () => { - if (accessToken == null) { - return; - } - getBudgetList(accessToken).then((data) => { - setBudgetList(data); - }); - }; - return (