From 5e2df556d8065de5d8baf66e391d089570d88b05 Mon Sep 17 00:00:00 2001 From: shivam Date: Tue, 9 Jun 2026 17:10:15 -0700 Subject: [PATCH 01/54] fix(cost): store cost breakdown for /v1/realtime sessions Realtime cost calculation computed totals but never populated logging_obj.cost_breakdown, so spend logs and the UI Metrics/Cost Breakdown showed no input/output cost details. Co-authored-by: Cursor --- litellm/cost_calculator.py | 10 ++++ tests/test_litellm/test_cost_calculator.py | 60 +++++++++++++++++++++- 2 files changed, 69 insertions(+), 1 deletion(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 88029615ba8..cf8fa602be5 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -1567,6 +1567,7 @@ def completion_cost( # noqa: PLR0915 custom_llm_provider=custom_llm_provider, litellm_model_name=model, data_residency=data_residency, + litellm_logging_obj=litellm_logging_obj, ) elif call_type == _MCP_CALL_TYPE: from litellm.proxy._experimental.mcp_server.cost_calculator import ( @@ -2494,6 +2495,7 @@ def handle_realtime_stream_cost_calculation( custom_llm_provider: str, litellm_model_name: str, data_residency: Optional[str] = None, + litellm_logging_obj: Optional[LitellmLoggingObject] = None, ) -> float: """ Handles the cost calculation for realtime stream responses. @@ -2533,4 +2535,12 @@ def handle_realtime_stream_cost_calculation( break # exit if we find a valid model total_cost = input_cost_per_token + output_cost_per_token + _store_cost_breakdown_in_logging_obj( + litellm_logging_obj=litellm_logging_obj, + prompt_tokens_cost_usd_dollar=input_cost_per_token, + completion_tokens_cost_usd_dollar=output_cost_per_token, + cost_for_built_in_tools_cost_usd_dollar=0.0, + total_cost_usd_dollar=total_cost, + ) + return total_cost diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 82a4a60bf82..2b60cfc9ccd 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -385,7 +385,65 @@ def test_handle_realtime_stream_cost_calculation(): ) assert cost == 0.0 # No usage, no cost - + +def test_handle_realtime_stream_cost_calculation_stores_cost_breakdown(): + """Regression: realtime cost must populate logging_obj.cost_breakdown so the + spend logs / UI show input vs output cost (issue: cost_breakdown was None for + /v1/realtime even though a total spend was computed).""" + from datetime import datetime + + from litellm.litellm_core_utils.litellm_logging import Logging + + results: OpenAIRealtimeStreamList = [ + {"type": "session.created", "session": {"model": "gpt-4o-realtime-preview"}}, + { + "type": "response.done", + "response": { + "usage": { + "input_tokens": 100, + "output_tokens": 50, + "total_tokens": 150, + } + }, + }, + ] + combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results( + results=results, + ) + + logging_obj = Logging( + model="gpt-4o-realtime-preview", + messages=[], + stream=False, + call_type="_arealtime", + start_time=datetime.now(), + litellm_call_id="realtime-cost-breakdown-test", + function_id="realtime-cost-breakdown-test", + ) + + total_cost = handle_realtime_stream_cost_calculation( + results=results, + combined_usage_object=combined_usage_object, + custom_llm_provider="openai", + litellm_model_name="gpt-4o-realtime-preview", + litellm_logging_obj=logging_obj, + ) + + assert total_cost > 0 + assert logging_obj.cost_breakdown is not None + assert logging_obj.cost_breakdown["input_cost"] > 0 + assert logging_obj.cost_breakdown["output_cost"] > 0 + assert ( + abs( + logging_obj.cost_breakdown["input_cost"] + + logging_obj.cost_breakdown["output_cost"] + - total_cost + ) + < 1e-9 + ) + assert abs(logging_obj.cost_breakdown["total_cost"] - total_cost) < 1e-9 + + def test_realtime_stream_combines_text_and_audio_token_details(): """Realtime response.done usage with input_token_details / output_token_details.""" from litellm.cost_calculator import RealtimeAPITokenUsageProcessor From fa9664eced1b2ea5f0ce027f9fff83d28e2fb070 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Fri, 26 Jun 2026 09:58:12 -0700 Subject: [PATCH 02/54] fix(ci): exclude deleted files from ruff format check git diff --name-only includes deleted paths, so a PR that removes a litellm/**/*.py file feeds the gone path to ruff format --check, which exits 123 with 'No such file or directory'. Add --diff-filter=ACMR so only added/copied/modified/renamed files are checked, matching the pattern already used in test-litellm-ui-build.yml. --- .github/workflows/test-linting.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/test-linting.yml b/.github/workflows/test-linting.yml index ff6c40ac9ae..d0695e49268 100644 --- a/.github/workflows/test-linting.yml +++ b/.github/workflows/test-linting.yml @@ -54,7 +54,7 @@ jobs: env: BASE_SHA: ${{ github.event.pull_request.base.sha }} run: | - git diff --name-only "$BASE_SHA"...HEAD -- 'litellm/**/*.py' | grep -v '^litellm/enterprise/' > "$RUNNER_TEMP/ruff_format_files.txt" || true + git diff --name-only --diff-filter=ACMR "$BASE_SHA"...HEAD -- 'litellm/**/*.py' | grep -v '^litellm/enterprise/' > "$RUNNER_TEMP/ruff_format_files.txt" || true if [ ! -s "$RUNNER_TEMP/ruff_format_files.txt" ]; then echo "No changed litellm Python files to check with ruff format." exit 0 From 09f611e7b50d93ea224ab51053619ed67def4f7b Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Thu, 2 Jul 2026 17:03:05 +0000 Subject: [PATCH 03/54] fix: include realtime transcription cost in breakdown --- litellm/cost_calculator.py | 18 ++++++++++-------- tests/test_litellm/test_cost_calculator.py | 16 ++++++++++++++++ 2 files changed, 26 insertions(+), 8 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index bfdce95d89a..9d9564a7110 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -2564,7 +2564,16 @@ def handle_realtime_stream_cost_calculation( input_cost_per_token += _input_cost_per_token output_cost_per_token += _output_cost_per_token break # exit if we find a valid model - total_cost = input_cost_per_token + output_cost_per_token + transcription_cost = ( + handle_realtime_transcription_cost_calculation( + results=results, + custom_llm_provider=custom_llm_provider, + litellm_model_name=litellm_model_name, + ) + if any(r.get("type") == _TRANSCRIPTION_COMPLETED_EVENT_TYPE for r in results) + else 0.0 + ) + total_cost = input_cost_per_token + output_cost_per_token + transcription_cost _store_cost_breakdown_in_logging_obj( litellm_logging_obj=litellm_logging_obj, @@ -2574,13 +2583,6 @@ def handle_realtime_stream_cost_calculation( total_cost_usd_dollar=total_cost, ) - if any(r.get("type") == _TRANSCRIPTION_COMPLETED_EVENT_TYPE for r in results): - total_cost += handle_realtime_transcription_cost_calculation( - results=results, - custom_llm_provider=custom_llm_provider, - litellm_model_name=litellm_model_name, - ) - return total_cost diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 4ba64f13b69..3a5d43b0300 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -637,6 +637,10 @@ def test_realtime_transcription_duration_cost(monkeypatch): ($0.017/min). The .completed events carry usage {type: duration, seconds: N}; cost must equal total_seconds * input_cost_per_second. """ + from datetime import datetime + + from litellm.litellm_core_utils.litellm_logging import Logging + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) @@ -667,17 +671,29 @@ def test_realtime_transcription_duration_cost(monkeypatch): combined = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results( results=results ) + logging_obj = Logging( + model="gpt-realtime-whisper", + messages=[], + stream=False, + call_type="_arealtime", + start_time=datetime.now(), + litellm_call_id="realtime-transcription-cost-breakdown-test", + function_id="realtime-transcription-cost-breakdown-test", + ) cost = handle_realtime_stream_cost_calculation( results=results, combined_usage_object=combined, custom_llm_provider="openai", litellm_model_name="gpt-realtime-whisper", + litellm_logging_obj=logging_obj, ) # 90 seconds at $0.017/minute. expected = 90.0 * (0.017 / 60) assert abs(cost - expected) < 1e-9 assert cost > 0 # guards against the duration branch being dropped + assert logging_obj.cost_breakdown is not None + assert abs(logging_obj.cost_breakdown["total_cost"] - cost) < 1e-9 def test_realtime_transcription_duration_cost_resolves_model_from_litellm_name( From 659127bd0da2de6dc582d6c6cdab949cbaf5e943 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Thu, 2 Jul 2026 22:00:24 -0700 Subject: [PATCH 04/54] feat(guardrails): add unreachable_fallback fail-open option to headroom guardrail Reuses the existing unreachable_fallback flag (already implemented by generic_guardrail_api, akto, vigil_guard, repelloai) so headroom compression failures can forward the request uncompressed instead of blocking it with a 502. --- .../guardrail_hooks/headroom/__init__.py | 1 + .../guardrail_hooks/headroom/headroom.py | 89 ++++++++++--------- litellm/types/guardrails.py | 2 +- .../guardrails/guardrail_hooks/headroom.py | 10 ++- .../guardrail_hooks/test_headroom.py | 63 ++++++++++++- 5 files changed, 121 insertions(+), 44 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/headroom/__init__.py b/litellm/proxy/guardrails/guardrail_hooks/headroom/__init__.py index 9b7934b7705..d2b8a979261 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/headroom/__init__.py +++ b/litellm/proxy/guardrails/guardrail_hooks/headroom/__init__.py @@ -34,6 +34,7 @@ def initialize_guardrail(litellm_params: LitellmParams, guardrail: Guardrail) -> guardrail_name=guardrail["guardrail_name"], event_hook=_coerce_event_hook(litellm_params.mode), default_on=litellm_params.default_on or False, + unreachable_fallback=litellm_params.unreachable_fallback, ) litellm.logging_callback_manager.add_litellm_callback( # pyright: ignore[reportUnknownMemberType] _callback diff --git a/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py b/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py index 4badb48e2eb..575b4ca55dd 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py +++ b/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py @@ -214,6 +214,7 @@ class HeadroomGuardrail(CustomGuardrail): guardrail_name: str | None = None, event_hook: GuardrailEventHooks | list[GuardrailEventHooks] | Mode | None = None, default_on: bool = False, + unreachable_fallback: str | None = None, ): self.headroom_api_base = (api_base or get_secret_str("HEADROOM_API_BASE") or "").rstrip("/") if not self.headroom_api_base: @@ -223,6 +224,9 @@ class HeadroomGuardrail(CustomGuardrail): ) self.headroom_api_key = api_key or get_secret_str("HEADROOM_API_KEY") self.headroom_model = model + self.unreachable_fallback: Literal["fail_closed", "fail_open"] = ( + "fail_open" if unreachable_fallback == "fail_open" else "fail_closed" + ) self.async_handler = get_async_httpx_client( llm_provider=httpxSpecialProvider.GuardrailCallback, ) @@ -257,6 +261,21 @@ class HeadroomGuardrail(CustomGuardrail): if expiry > now } + def _handle_compress_failure( + self, + messages: list[dict[str, object]], + error: str, + detail: dict[str, object], + ) -> list[dict[str, object]]: + if self.unreachable_fallback == "fail_open": + verbose_proxy_logger.critical( + "Headroom: %s; fail_open configured, forwarding request uncompressed. detail=%s", + error, + detail, + ) + return messages + raise HTTPException(status_code=502, detail={"error": error, **detail}) + async def _call_compress( self, messages: list[dict[str, object]], @@ -273,67 +292,55 @@ class HeadroomGuardrail(CustomGuardrail): headers=self._request_headers(), ) except (httpx.ConnectError, httpx.TimeoutException, httpx.TransportError) as e: - raise HTTPException( - status_code=502, - detail={ - "error": "Headroom compression service unreachable", - "detail": str(e), - }, - ) from e + return self._handle_compress_failure( + messages, + "Headroom compression service unreachable", + {"detail": str(e)}, + ) if raw_response is None: - raise HTTPException( - status_code=502, - detail={"error": "Headroom compression service returned no response"}, + return self._handle_compress_failure( + messages, + "Headroom compression service returned no response", + {}, ) response: HttpxResponse = raw_response if response.status_code != 200: - raise HTTPException( - status_code=502, - detail={ - "error": "Headroom compression service returned an error", - "status_code": response.status_code, - "body": response.text, - }, + return self._handle_compress_failure( + messages, + "Headroom compression service returned an error", + {"status_code": response.status_code, "body": response.text}, ) try: body: object = response.json() except ValueError: - raise HTTPException( - status_code=502, - detail={ - "error": "Headroom compression service returned non-JSON response", - "body": response.text[:500], - }, + return self._handle_compress_failure( + messages, + "Headroom compression service returned non-JSON response", + {"body": response.text[:500]}, ) if not _is_str_object_dict(body): - raise HTTPException( - status_code=502, - detail={ - "error": "Headroom compression service returned unexpected response shape", - "body": response.text[:500], - }, + return self._handle_compress_failure( + messages, + "Headroom compression service returned unexpected response shape", + {"body": response.text[:500]}, ) compressed_messages = body.get("messages") if not _is_object_list(compressed_messages): - raise HTTPException( - status_code=502, - detail={ - "error": "Headroom compression service response missing 'messages'", - "body": response.text, - }, + return self._handle_compress_failure( + messages, + "Headroom compression service response missing 'messages'", + {"body": response.text}, ) filtered = [item for item in compressed_messages if _is_str_object_dict(item)] if not filtered: - raise HTTPException( - status_code=502, - detail={ - "error": "Headroom compression service returned empty message list", - "body": response.text, - }, + return self._handle_compress_failure( + messages, + "Headroom compression service returned empty message list", + {"body": response.text}, ) verbose_proxy_logger.debug( diff --git a/litellm/types/guardrails.py b/litellm/types/guardrails.py index 889e029b902..8d7d7311fad 100644 --- a/litellm/types/guardrails.py +++ b/litellm/types/guardrails.py @@ -697,7 +697,7 @@ class BaseLitellmParams(ContentFilterConfigModel): # works for new and patch up default="fail_closed", description=( "Behavior when a guardrail endpoint is unreachable due to network errors. " - "Implemented by guardrail='generic_guardrail_api', 'akto', 'vigil_guard', and 'repelloai'. " + "Implemented by guardrail='generic_guardrail_api', 'akto', 'vigil_guard', 'repelloai', and 'headroom'. " "'fail_closed' raises an error (default). 'fail_open' logs a critical error and allows the request to proceed." ), ) diff --git a/litellm/types/proxy/guardrails/guardrail_hooks/headroom.py b/litellm/types/proxy/guardrails/guardrail_hooks/headroom.py index 3186c9fc612..fd962b8ffa1 100644 --- a/litellm/types/proxy/guardrails/guardrail_hooks/headroom.py +++ b/litellm/types/proxy/guardrails/guardrail_hooks/headroom.py @@ -1,4 +1,4 @@ -from typing import Optional +from typing import Literal, Optional from pydantic import BaseModel, Field @@ -18,6 +18,14 @@ class HeadroomGuardrailConfigModel(GuardrailConfigModel[BaseModel]): default=None, description="Model name forwarded to the headroom /v1/compress endpoint.", ) + unreachable_fallback: Optional[Literal["fail_closed", "fail_open"]] = Field( + default="fail_closed", + description=( + "Behavior when the headroom compression service is unreachable or errors. " + "'fail_closed' raises an error (default). 'fail_open' logs a critical error and " + "forwards the request uncompressed instead of blocking it." + ), + ) @staticmethod def ui_friendly_name() -> str: diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py index f5ce6cedf64..954485d6b90 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py @@ -6,8 +6,9 @@ Tests cover: - x-headroom-bypass: true header causes guardrail to skip compression - missing or empty messages are passed through unchanged - response-type input is passed through unchanged -- /v1/compress HTTP error raises HTTPException +- /v1/compress HTTP error raises HTTPException (fail_closed, the default) - /v1/compress returning malformed JSON raises HTTPException +- unreachable_fallback="fail_open" forwards the request uncompressed instead of raising - CCR: headroom_retrieve tool injected when compressed messages contain hashes - CCR: async_should_run_agentic_loop returns True when response has headroom_retrieve tool calls - CCR: async_build_agentic_loop_plan calls retrieve endpoint and builds follow-up messages @@ -806,6 +807,56 @@ async def test_apply_guardrail_transport_error_raises(): assert "unreachable" in str(exc_info.value.detail) +@pytest.mark.asyncio +async def test_apply_guardrail_transport_error_fail_open_forwards_uncompressed(): + guardrail = _make_guardrail(unreachable_fallback="fail_open") + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + side_effect=httpx.ConnectError("Connection refused"), + ): + result = await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert result["structured_messages"] == ORIGINAL_MESSAGES + + +@pytest.mark.asyncio +async def test_apply_guardrail_http_error_fail_open_forwards_uncompressed(): + guardrail = _make_guardrail(unreachable_fallback="fail_open") + mock_response = _make_compress_response([], status=500) + mock_response.text = "Internal Server Error" + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + return_value=mock_response, + ): + result = await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert result["structured_messages"] == ORIGINAL_MESSAGES + + @pytest.mark.asyncio async def test_apply_guardrail_missing_messages_key_raises(): guardrail = _make_guardrail() @@ -875,6 +926,16 @@ def test_init_raises_without_api_base(): HeadroomGuardrail(api_base=None) +def test_init_defaults_to_fail_closed(): + guardrail = _make_guardrail() + assert guardrail.unreachable_fallback == "fail_closed" + + +def test_init_rejects_invalid_unreachable_fallback_value(): + guardrail = _make_guardrail(unreachable_fallback="not-a-real-mode") + assert guardrail.unreachable_fallback == "fail_closed" + + def test_bypass_header_case_insensitive(): guardrail = _make_guardrail() From 00dffcd075292b816356c9e0efd83c7f4f02baa6 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Thu, 2 Jul 2026 22:17:52 -0700 Subject: [PATCH 05/54] fix(guardrails): catch httpx.HTTPStatusError in headroom compress call litellm's async httpx client already calls raise_for_status() internally, so a non-2xx /v1/compress response surfaced as an uncaught httpx.HTTPStatusError instead of going through the guardrail's status_code check. Caught live by running the guardrail against a mock headroom endpoint that returns 500: unreachable_fallback=fail_open silently failed to forward the request until this fix. --- .../guardrail_hooks/headroom/headroom.py | 6 ++ .../guardrail_hooks/test_headroom.py | 67 +++++++++++++++++++ 2 files changed, 73 insertions(+) diff --git a/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py b/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py index 575b4ca55dd..fe6d6bf7051 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py +++ b/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py @@ -291,6 +291,12 @@ class HeadroomGuardrail(CustomGuardrail): json=payload, headers=self._request_headers(), ) + except httpx.HTTPStatusError as e: + return self._handle_compress_failure( + messages, + "Headroom compression service returned an error", + {"status_code": e.response.status_code, "body": e.response.text}, + ) except (httpx.ConnectError, httpx.TimeoutException, httpx.TransportError) as e: return self._handle_compress_failure( messages, diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py index 954485d6b90..fdf96c44f47 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py @@ -8,6 +8,8 @@ Tests cover: - response-type input is passed through unchanged - /v1/compress HTTP error raises HTTPException (fail_closed, the default) - /v1/compress returning malformed JSON raises HTTPException +- /v1/compress non-2xx surfaces as httpx.HTTPStatusError (raise_for_status), + not a status_code check on the returned response -- both are handled - unreachable_fallback="fail_open" forwards the request uncompressed instead of raising - CCR: headroom_retrieve tool injected when compressed messages contain hashes - CCR: async_should_run_agentic_loop returns True when response has headroom_retrieve tool calls @@ -807,6 +809,71 @@ async def test_apply_guardrail_transport_error_raises(): assert "unreachable" in str(exc_info.value.detail) +def _make_http_status_error(status: int, body: str) -> httpx.HTTPStatusError: + request = httpx.Request("POST", f"{FAKE_API_BASE}/v1/compress") + response = httpx.Response(status, request=request, text=body) + return httpx.HTTPStatusError( + f"Server error '{status}' for url", + request=request, + response=response, + ) + + +@pytest.mark.asyncio +async def test_apply_guardrail_http_status_error_raises(): + """Regression test: litellm's async httpx client calls raise_for_status() + internally, so a non-2xx /v1/compress response surfaces as + httpx.HTTPStatusError, not as a returned MagicMock with status_code set. + A prior version of _call_compress only checked response.status_code and + never caught this exception, so it went unhandled instead of blocking + the request per fail_closed policy.""" + guardrail = _make_guardrail() + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + side_effect=_make_http_status_error(500, "headroom internal error"), + ): + with pytest.raises(HTTPException) as exc_info: + await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert exc_info.value.status_code == 502 + + +@pytest.mark.asyncio +async def test_apply_guardrail_http_status_error_fail_open_forwards_uncompressed(): + guardrail = _make_guardrail(unreachable_fallback="fail_open") + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + side_effect=_make_http_status_error(500, "headroom internal error"), + ): + result = await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert result["structured_messages"] == ORIGINAL_MESSAGES + + @pytest.mark.asyncio async def test_apply_guardrail_transport_error_fail_open_forwards_uncompressed(): guardrail = _make_guardrail(unreachable_fallback="fail_open") From 53d2331c700ce1772c45806b6e56a49534498778 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Fri, 3 Jul 2026 19:13:30 +0000 Subject: [PATCH 06/54] fix: catch litellm.Timeout in HeadroomGuardrail to support fail_open on timeouts async_handler.post catches httpx.TimeoutException and re-raises it as litellm.Timeout (a subclass of openai.APITimeoutError). The except blocks in _call_compress and _call_retrieve only listed httpx exception types, so litellm.Timeout propagated uncaught and bypassed the unreachable_fallback=fail_open path. Add litellm.Timeout to both except clauses and add regression tests for the fail_closed and fail_open timeout paths. --- .../guardrail_hooks/headroom/headroom.py | 6 +- .../guardrail_hooks/test_headroom.py | 60 +++++++++++++++++++ 2 files changed, 64 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py b/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py index fe6d6bf7051..c218cac7aa2 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py +++ b/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py @@ -8,6 +8,8 @@ from typing import TYPE_CHECKING, Any, Literal, Optional import httpx from fastapi import HTTPException + +import litellm from httpx import Response as HttpxResponse from typing_extensions import TypeGuard @@ -297,7 +299,7 @@ class HeadroomGuardrail(CustomGuardrail): "Headroom compression service returned an error", {"status_code": e.response.status_code, "body": e.response.text}, ) - except (httpx.ConnectError, httpx.TimeoutException, httpx.TransportError) as e: + except (httpx.ConnectError, httpx.TimeoutException, httpx.TransportError, litellm.Timeout) as e: return self._handle_compress_failure( messages, "Headroom compression service unreachable", @@ -368,7 +370,7 @@ class HeadroomGuardrail(CustomGuardrail): params=params, headers=self._request_headers(), ) - except (httpx.ConnectError, httpx.TimeoutException, httpx.TransportError) as e: + except (httpx.ConnectError, httpx.TimeoutException, httpx.TransportError, litellm.Timeout) as e: verbose_proxy_logger.warning("Headroom: retrieve failed for hash=%s: %s", hash_value, e) return f"[Headroom: retrieval failed for hash={hash_value}]" diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py index fdf96c44f47..09422be04ab 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py @@ -24,6 +24,8 @@ import httpx import pytest from fastapi import HTTPException +import litellm + from litellm.proxy.guardrails.guardrail_hooks.headroom.headroom import ( HeadroomGuardrail, extract_hashes_from_messages, @@ -1187,3 +1189,61 @@ async def test_async_should_run_agentic_loop_detects_responses_api_output_format assert should_run is True assert len(ctx["tool_calls"]) == 1 assert ctx["tool_calls"][0]["arguments"]["hash"] == "b573993006976af767214fac" + + +@pytest.mark.asyncio +async def test_apply_guardrail_litellm_timeout_raises_when_fail_closed(): + guardrail = _make_guardrail() + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + side_effect=litellm.Timeout( + message="Connection timed out after 10 seconds.", + model="default-model-name", + llm_provider="litellm-httpx-handler", + ), + ): + with pytest.raises(HTTPException) as exc_info: + await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert exc_info.value.status_code == 502 + assert "unreachable" in str(exc_info.value.detail) + + +@pytest.mark.asyncio +async def test_apply_guardrail_litellm_timeout_fail_open_forwards_uncompressed(): + guardrail = _make_guardrail(unreachable_fallback="fail_open") + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + side_effect=litellm.Timeout( + message="Connection timed out after 10 seconds.", + model="default-model-name", + llm_provider="litellm-httpx-handler", + ), + ): + result = await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert result["structured_messages"] == ORIGINAL_MESSAGES From 01dfbf7ebb5a037e4de3ec3698aa1ba783a3f389 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 3 Jul 2026 19:40:33 +0000 Subject: [PATCH 07/54] fix(guardrails): address review comments on headroom fail_open - Prevent fail-open from registering user-supplied hashes as valid for CCR retrieval; _call_compress now returns (messages, compressed_ok) so apply_guardrail skips hash extraction and tool injection when compression did not succeed - Remove Optional wrapper from HeadroomGuardrailConfigModel.unreachable_fallback to match BaseLitellmParams typing - Add fail_open tests for non-JSON response, missing messages key, and empty message list paths - Add regression test verifying fail_open does not authorize attacker-planted hashes - Regenerate dashboard API types Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../guardrail_hooks/headroom/headroom.py | 25 +- .../guardrails/guardrail_hooks/headroom.py | 2 +- .../guardrail_hooks/test_headroom.py | 123 ++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 1067 ----------------- 4 files changed, 138 insertions(+), 1079 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py b/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py index c218cac7aa2..37863e0e356 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py +++ b/litellm/proxy/guardrails/guardrail_hooks/headroom/headroom.py @@ -282,7 +282,7 @@ class HeadroomGuardrail(CustomGuardrail): self, messages: list[dict[str, object]], model: str | None, - ) -> list[dict[str, object]]: + ) -> tuple[list[dict[str, object]], bool]: payload: dict[str, object] = {"messages": messages} if model: payload["model"] = model @@ -298,19 +298,19 @@ class HeadroomGuardrail(CustomGuardrail): messages, "Headroom compression service returned an error", {"status_code": e.response.status_code, "body": e.response.text}, - ) + ), False except (httpx.ConnectError, httpx.TimeoutException, httpx.TransportError, litellm.Timeout) as e: return self._handle_compress_failure( messages, "Headroom compression service unreachable", {"detail": str(e)}, - ) + ), False if raw_response is None: return self._handle_compress_failure( messages, "Headroom compression service returned no response", {}, - ) + ), False response: HttpxResponse = raw_response if response.status_code != 200: @@ -318,7 +318,7 @@ class HeadroomGuardrail(CustomGuardrail): messages, "Headroom compression service returned an error", {"status_code": response.status_code, "body": response.text}, - ) + ), False try: body: object = response.json() @@ -327,13 +327,13 @@ class HeadroomGuardrail(CustomGuardrail): messages, "Headroom compression service returned non-JSON response", {"body": response.text[:500]}, - ) + ), False if not _is_str_object_dict(body): return self._handle_compress_failure( messages, "Headroom compression service returned unexpected response shape", {"body": response.text[:500]}, - ) + ), False compressed_messages = body.get("messages") if not _is_object_list(compressed_messages): @@ -341,7 +341,7 @@ class HeadroomGuardrail(CustomGuardrail): messages, "Headroom compression service response missing 'messages'", {"body": response.text}, - ) + ), False filtered = [item for item in compressed_messages if _is_str_object_dict(item)] if not filtered: @@ -349,7 +349,7 @@ class HeadroomGuardrail(CustomGuardrail): messages, "Headroom compression service returned empty message list", {"body": response.text}, - ) + ), False verbose_proxy_logger.debug( "Headroom: compressed %s tokens -> %s tokens (ratio %.2f)", @@ -357,7 +357,7 @@ class HeadroomGuardrail(CustomGuardrail): body.get("tokens_after", "?"), body.get("compression_ratio", 0), ) - return filtered + return filtered, True async def _call_retrieve(self, hash_value: str, query: str | None = None) -> str: params: dict[str, str] = {} @@ -421,11 +421,14 @@ class HeadroomGuardrail(CustomGuardrail): return inputs model = self.headroom_model or request_data.get("model") - compressed = await self._call_compress( + compressed, compression_succeeded = await self._call_compress( messages=messages, model=model if isinstance(model, str) else None, ) + if not compression_succeeded: + return {**inputs, "structured_messages": compressed} # pyright: ignore[reportReturnType] + hashes = extract_hashes_from_messages(compressed) if not hashes: return {**inputs, "structured_messages": compressed} # pyright: ignore[reportReturnType] diff --git a/litellm/types/proxy/guardrails/guardrail_hooks/headroom.py b/litellm/types/proxy/guardrails/guardrail_hooks/headroom.py index fd962b8ffa1..71aa243069a 100644 --- a/litellm/types/proxy/guardrails/guardrail_hooks/headroom.py +++ b/litellm/types/proxy/guardrails/guardrail_hooks/headroom.py @@ -18,7 +18,7 @@ class HeadroomGuardrailConfigModel(GuardrailConfigModel[BaseModel]): default=None, description="Model name forwarded to the headroom /v1/compress endpoint.", ) - unreachable_fallback: Optional[Literal["fail_closed", "fail_open"]] = Field( + unreachable_fallback: Literal["fail_closed", "fail_open"] = Field( default="fail_closed", description=( "Behavior when the headroom compression service is unreachable or errors. " diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py index 09422be04ab..7f412c008ca 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_headroom.py @@ -926,6 +926,129 @@ async def test_apply_guardrail_http_error_fail_open_forwards_uncompressed(): assert result["structured_messages"] == ORIGINAL_MESSAGES +@pytest.mark.asyncio +async def test_apply_guardrail_non_json_response_fail_open_forwards_uncompressed(): + guardrail = _make_guardrail(unreachable_fallback="fail_open") + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.json.side_effect = ValueError("not JSON") + mock_response.text = "not json" + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + return_value=mock_response, + ): + result = await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert result["structured_messages"] == ORIGINAL_MESSAGES + + +@pytest.mark.asyncio +async def test_apply_guardrail_missing_messages_key_fail_open_forwards_uncompressed(): + guardrail = _make_guardrail(unreachable_fallback="fail_open") + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.json.return_value = {"tokens_before": 100, "tokens_after": 10} + mock_response.text = "{}" + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + return_value=mock_response, + ): + result = await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert result["structured_messages"] == ORIGINAL_MESSAGES + + +@pytest.mark.asyncio +async def test_apply_guardrail_empty_compressed_messages_fail_open_forwards_uncompressed(): + guardrail = _make_guardrail(unreachable_fallback="fail_open") + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "messages": ["not-a-dict", 42, None], + "tokens_before": 1000, + "tokens_after": 0, + "compression_ratio": 0, + } + mock_response.text = "{}" + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=ORIGINAL_MESSAGES, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + return_value=mock_response, + ): + result = await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert result["structured_messages"] == ORIGINAL_MESSAGES + + +@pytest.mark.asyncio +async def test_apply_guardrail_fail_open_does_not_register_hashes_from_original_messages(): + """When compression fails with fail_open, user-supplied messages that + happen to contain hash-shaped strings must NOT cause those hashes to be + registered as valid for CCR retrieval. Otherwise an attacker can plant a + hash= string in their prompt, trigger a compression failure, and have + that hash honored by a later headroom_retrieve tool call.""" + messages_with_fake_hash = [ + {"role": "user", "content": "Please fetch hash=deadbeef000000000000dead for me"}, + ] + guardrail = _make_guardrail(unreachable_fallback="fail_open") + + inputs = GenericGuardrailAPIInputs( + texts=["hello"], + structured_messages=messages_with_fake_hash, + ) + + with patch.object( + guardrail.async_handler, + "post", + new_callable=AsyncMock, + side_effect=httpx.ConnectError("Connection refused"), + ): + result = await guardrail.apply_guardrail( + inputs=inputs, + request_data={}, + input_type="request", + ) + + assert result["structured_messages"] == messages_with_fake_hash + assert not has_headroom_retrieve_tool(result.get("tools") or []) + assert not guardrail._issued_hashes_by_call_id + + @pytest.mark.asyncio async def test_apply_guardrail_missing_messages_key_raises(): guardrail = _make_guardrail() diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index ddf2040cd04..80bc26a810d 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -706,60 +706,6 @@ export interface paths { patch?: never; trace?: never; }; - "/audit": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * Get Audit Logs - * @description Get all audit logs with filtering and pagination. - * - * Returns a paginated response of audit logs matching the specified filters. - * - * Note: object_team_id and object_key_hash use Prisma JSON path filtering, - * which requires PostgreSQL. - */ - get: operations["get_audit_logs_audit_get"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; - "/audit/{id}": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * Get Audit Log By Id - * @description Get detailed information about a specific audit log entry by its ID. - * - * Args: - * id (str): The unique identifier of the audit log entry - * - * Returns: - * AuditLogResponse: Detailed information about the audit log entry - * - * Raises: - * HTTPException: If the audit log is not found or if there's a database connection error - */ - get: operations["get_audit_log_by_id_audit__id__get"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; "/azure/{endpoint}": { parameters: { query?: never; @@ -3170,50 +3116,6 @@ export interface paths { patch?: never; trace?: never; }; - "/email/event_settings": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * Get Email Event Settings - * @description Get all email event settings - */ - get: operations["get_email_event_settings_email_event_settings_get"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - /** - * Update Event Settings - * @description Update the settings for email events - */ - patch: operations["update_event_settings_email_event_settings_patch"]; - trace?: never; - }; - "/email/event_settings/reset": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - get?: never; - put?: never; - /** - * Reset Event Settings - * @description Reset all email event settings to default (new user invitations on, virtual key creation off) - */ - post: operations["reset_event_settings_email_event_settings_reset_post"]; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; "/embeddings": { parameters: { query?: never; @@ -9964,240 +9866,6 @@ export interface paths { patch?: never; trace?: never; }; - "/project/delete": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - get?: never; - put?: never; - post?: never; - /** - * Delete Project - * @description Delete projects - * - * Parameters: - * - project_ids: *List[str]* - List of project ids to delete - * - * Example: - * ```bash - * curl --location --request DELETE 'http://0.0.0.0:4000/project/delete' \ - * --header 'Authorization: Bearer sk-1234' \ - * --header 'Content-Type: application/json' \ - * --data '{ - * "project_ids": ["project-123", "project-456"] - * }' - * ``` - */ - delete: operations["delete_project_project_delete_delete"]; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; - "/project/info": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * Project Info - * @description Get information about a specific project - * - * Parameters: - * - project_id: *str* - The project id to fetch info for - * - * Example: - * ```bash - * curl --location 'http://0.0.0.0:4000/project/info?project_id=project-123' \ - * --header 'Authorization: Bearer sk-1234' - * ``` - */ - get: operations["project_info_project_info_get"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; - "/project/list": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * List Projects - * @description List all projects that the user has access to - * - * Example: - * ```bash - * curl --location 'http://0.0.0.0:4000/project/list' \ - * --header 'Authorization: Bearer sk-1234' - * ``` - */ - get: operations["list_projects_project_list_get"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; - "/project/new": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - get?: never; - put?: never; - /** - * New Project - * @description Create a new project. Projects sit between teams and keys in the hierarchy. - * - * Only admins or team admins can create projects. - * - * # Parameters - * - * - project_alias: *Optional[str]* - The name of the project. - * - description: *Optional[str]* - Description of the project's purpose and use case. - * - team_id: *str* - The team id that this project belongs to. Required. - * - models: *List* - The models the project has access to. - * - budget_id: *Optional[str]* - The id for a budget (tpm/rpm/max budget) for the project. - * ### IF NO BUDGET ID - CREATE ONE WITH THESE PARAMS ### - * - max_budget: *Optional[float]* - Max budget for project - * - tpm_limit: *Optional[int]* - Max tpm limit for project - * - rpm_limit: *Optional[int]* - Max rpm limit for project - * - max_parallel_requests: *Optional[int]* - Max parallel requests for project - * - soft_budget: *Optional[float]* - Get a slack alert when this soft budget is reached. Don't block requests. - * - model_max_budget: *Optional[dict]* - Max budget for a specific model. Example: {"gpt-4": 100.0, "gpt-3.5-turbo": 50.0} - * - model_rpm_limit: *Optional[dict]* - RPM limits per model. Example: {"gpt-4": 1000, "gpt-3.5-turbo": 5000} - * - model_tpm_limit: *Optional[dict]* - TPM limits per model. Example: {"gpt-4": 50000, "gpt-3.5-turbo": 100000} - * - budget_duration: *Optional[str]* - Frequency of reseting project budget - * - metadata: *Optional[dict]* - Metadata for project, store information for project. Example metadata - {"use_case_id": "SNOW-12345", "responsible_ai_id": "RAI-67890"} - * - tags: *Optional[list]* - Tags for the project. Example: ["production", "api"] - * - blocked: *bool* - Flag indicating if the project is blocked or not - will stop all calls from keys with this project_id. - * - object_permission: Optional[LiteLLM_ObjectPermissionBase] - project-specific object permission. Example - {"vector_stores": ["vector_store_1", "vector_store_2"]}. IF null or {} then no object permission. - * - * Example 1: Create new project **without** a budget_id, with model-specific limits - * - * ```bash - * curl --location 'http://0.0.0.0:4000/project/new' \ - * --header 'Authorization: Bearer sk-1234' \ - * --header 'Content-Type: application/json' \ - * --data '{ - * "project_alias": "flight-search-assistant", - * "description": "AI-powered flight search and booking assistant", - * "team_id": "team-123", - * "models": ["gpt-4", "gpt-3.5-turbo"], - * "max_budget": 100, - * "model_rpm_limit": { - * "gpt-4": 1000, - * "gpt-3.5-turbo": 5000 - * }, - * "model_tpm_limit": { - * "gpt-4": 50000, - * "gpt-3.5-turbo": 100000 - * }, - * "metadata": { - * "use_case_id": "SNOW-12345", - * "responsible_ai_id": "RAI-67890" - * } - * }' - * ``` - * - * Example 2: Create new project **with** a budget_id - * - * ```bash - * curl --location 'http://0.0.0.0:4000/project/new' \ - * --header 'Authorization: Bearer sk-1234' \ - * --header 'Content-Type: application/json' \ - * --data '{ - * "project_alias": "hotel-recommendations", - * "description": "Personalized hotel recommendation engine", - * "team_id": "team-123", - * "models": ["claude-3-sonnet"], - * "budget_id": "428eeaa8-f3ac-4e85-a8fb-7dc8d7aa8689", - * "metadata": { - * "use_case_id": "SNOW-54321" - * } - * }' - * ``` - */ - post: operations["new_project_project_new_post"]; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; - "/project/update": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - get?: never; - put?: never; - /** - * Update Project - * @description Update a project - * - * Parameters: - * - project_id: *str* - The project id to update. Required. - * - project_alias: *Optional[str]* - Updated name for the project - * - description: *Optional[str]* - Updated description for the project - * - team_id: *Optional[str]* - Updated team_id for the project - * - metadata: *Optional[dict]* - Updated metadata for project - * - models: *Optional[list]* - Updated list of models for the project - * - blocked: *Optional[bool]* - Updated blocked status - * - max_budget: *Optional[float]* - Updated max budget - * - tpm_limit: *Optional[int]* - Updated tpm limit - * - rpm_limit: *Optional[int]* - Updated rpm limit - * - model_rpm_limit: *Optional[dict]* - Updated RPM limits per model - * - model_tpm_limit: *Optional[dict]* - Updated TPM limits per model - * - budget_duration: *Optional[str]* - Updated budget duration - * - tags: *Optional[list]* - Updated list of tags for the project - * - object_permission: Optional[LiteLLM_ObjectPermissionBase] - Updated object permission - * - * Example: - * ```bash - * curl --location 'http://0.0.0.0:4000/project/update' \ - * --header 'Authorization: Bearer sk-1234' \ - * --header 'Content-Type: application/json' \ - * --data '{ - * "project_id": "project-123", - * "description": "Updated flight search system with enhanced capabilities", - * "max_budget": 200, - * "model_rpm_limit": { - * "gpt-4": 2000, - * "gpt-3.5-turbo": 10000 - * }, - * "metadata": { - * "use_case_id": "SNOW-12345", - * "status": "active" - * } - * }' - * ``` - */ - post: operations["update_project_project_update_post"]; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; "/prompts": { parameters: { query?: never; @@ -11240,27 +10908,6 @@ export interface paths { patch?: never; trace?: never; }; - "/robots.txt": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * Get Robots - * @description Block all web crawlers from indexing the proxy server endpoints - * This is useful for ensuring that the API endpoints aren't indexed by search engines - */ - get: operations["get_robots_robots_txt_get"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; "/router/fields": { parameters: { query?: never; @@ -14300,26 +13947,6 @@ export interface paths { patch?: never; trace?: never; }; - "/user/available_users": { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - /** - * Available Enterprise Users - * @description For keys with `max_users` set, return the list of users that are allowed to use the key. - */ - get: operations["available_enterprise_users_user_available_users_get"]; - put?: never; - post?: never; - delete?: never; - options?: never; - head?: never; - patch?: never; - trace?: never; - }; "/user/bulk_update": { parameters: { query?: never; @@ -20600,37 +20227,6 @@ export interface components { */ unnamed_teams_count: number; }; - /** - * AuditLogResponse - * @description Response model for a single audit log entry - */ - AuditLogResponse: { - /** Action */ - action: string; - /** Before Value */ - before_value?: { - [key: string]: unknown; - } | null; - /** Changed By */ - changed_by: string; - /** Changed By Api Key */ - changed_by_api_key: string; - /** Id */ - id: string; - /** Object Id */ - object_id: string; - /** Table Name */ - table_name: string; - /** - * Updated At - * Format: date-time - */ - updated_at: string; - /** Updated Values */ - updated_values?: { - [key: string]: unknown; - } | null; - }; /** BaseLitellmParams */ "BaseLitellmParams-Input": { /** @@ -23120,14 +22716,6 @@ export interface components { /** Organization Ids */ organization_ids: string[]; }; - /** - * DeleteProjectRequest - * @description Request model for DELETE /project/delete - */ - DeleteProjectRequest: { - /** Project Ids */ - project_ids: string[]; - }; /** * DeleteSkillResponse * @description Response from deleting a skill @@ -23242,27 +22830,6 @@ export interface components { /** Write Capacity Units */ write_capacity_units?: number | null; }; - /** - * EmailEvent - * @enum {string} - */ - EmailEvent: "Virtual Key Created" | "New User Invitation" | "Virtual Key Rotated" | "Soft Budget Crossed" | "Max Budget Alert"; - /** EmailEventSettings */ - EmailEventSettings: { - /** Enabled */ - enabled: boolean; - event: components["schemas"]["EmailEvent"]; - }; - /** EmailEventSettingsResponse */ - EmailEventSettingsResponse: { - /** Settings */ - settings: components["schemas"]["EmailEventSettings"][]; - }; - /** EmailEventSettingsUpdateRequest */ - EmailEventSettingsUpdateRequest: { - /** Settings */ - settings: components["schemas"]["EmailEventSettings"][]; - }; /** EmbeddingRequest */ EmbeddingRequest: { /** @@ -25556,65 +25123,6 @@ export interface components { } & { [key: string]: unknown; }; - /** - * LiteLLM_ProjectTable - * @description Database model representation for project - */ - LiteLLM_ProjectTable: { - /** - * Blocked - * @default false - */ - blocked: boolean; - /** Budget Id */ - budget_id?: string | null; - /** Created At */ - created_at?: string | null; - /** Created By */ - created_by?: string | null; - /** Description */ - description?: string | null; - litellm_budget_table?: components["schemas"]["LiteLLM_BudgetTable"] | null; - /** Metadata */ - metadata?: { - [key: string]: unknown; - } | null; - /** Model Rpm Limit */ - model_rpm_limit?: { - [key: string]: unknown; - } | null; - /** Model Spend */ - model_spend?: { - [key: string]: unknown; - } | null; - /** Model Tpm Limit */ - model_tpm_limit?: { - [key: string]: unknown; - } | null; - /** - * Models - * @default [] - */ - models: string[]; - object_permission?: components["schemas"]["LiteLLM_ObjectPermissionTable"] | null; - /** Object Permission Id */ - object_permission_id?: string | null; - /** Project Alias */ - project_alias?: string | null; - /** Project Id */ - project_id: string; - /** - * Spend - * @default 0 - */ - spend: number; - /** Team Id */ - team_id?: string | null; - /** Updated At */ - updated_at?: string | null; - /** Updated By */ - updated_by?: string | null; - }; /** LiteLLM_ProxyModelTable */ LiteLLM_ProxyModelTable: { /** @@ -27625,134 +27133,6 @@ export interface components { /** Users */ users?: components["schemas"]["LiteLLM_UserTable"][] | null; }; - /** - * NewProjectRequest - * @description Request model for POST /project/new - */ - NewProjectRequest: { - /** Allowed Models */ - allowed_models?: string[] | null; - /** - * Blocked - * @default false - */ - blocked: boolean; - /** Budget Duration */ - budget_duration?: string | null; - /** Budget Id */ - budget_id?: string | null; - /** Description */ - description?: string | null; - /** Guardrails */ - guardrails?: string[] | null; - /** Max Budget */ - max_budget?: number | null; - /** Max Parallel Requests */ - max_parallel_requests?: number | null; - /** Metadata */ - metadata?: { - [key: string]: unknown; - } | null; - /** Model Max Budget */ - model_max_budget?: { - [key: string]: unknown; - } | null; - /** Model Rpm Limit */ - model_rpm_limit?: { - [key: string]: unknown; - } | null; - /** Model Tpm Limit */ - model_tpm_limit?: { - [key: string]: unknown; - } | null; - /** - * Models - * @default [] - */ - models: string[]; - object_permission?: components["schemas"]["LiteLLM_ObjectPermissionBase"] | null; - /** Policies */ - policies?: string[] | null; - /** Project Alias */ - project_alias?: string | null; - /** Project Id */ - project_id?: string | null; - /** Rpm Limit */ - rpm_limit?: number | null; - /** Soft Budget */ - soft_budget?: number | null; - /** Tags */ - tags?: string[] | null; - /** Team Id */ - team_id: string; - /** Tpm Limit */ - tpm_limit?: number | null; - }; - /** - * NewProjectResponse - * @description Response model for POST /project/new - */ - NewProjectResponse: { - /** - * Blocked - * @default false - */ - blocked: boolean; - /** Budget Id */ - budget_id?: string | null; - /** - * Created At - * Format: date-time - */ - created_at: string; - /** Created By */ - created_by?: string | null; - /** Description */ - description?: string | null; - litellm_budget_table?: components["schemas"]["LiteLLM_BudgetTable"] | null; - /** Metadata */ - metadata?: { - [key: string]: unknown; - } | null; - /** Model Rpm Limit */ - model_rpm_limit?: { - [key: string]: unknown; - } | null; - /** Model Spend */ - model_spend?: { - [key: string]: unknown; - } | null; - /** Model Tpm Limit */ - model_tpm_limit?: { - [key: string]: unknown; - } | null; - /** - * Models - * @default [] - */ - models: string[]; - object_permission?: components["schemas"]["LiteLLM_ObjectPermissionTable"] | null; - /** Object Permission Id */ - object_permission_id?: string | null; - /** Project Alias */ - project_alias?: string | null; - /** Project Id */ - project_id: string; - /** - * Spend - * @default 0 - */ - spend: number; - /** Team Id */ - team_id?: string | null; - /** - * Updated At - * Format: date-time - */ - updated_at: string; - /** Updated By */ - updated_by?: string | null; - }; /** NewTeamRequest */ NewTeamRequest: { /** Access Group Ids */ @@ -28262,34 +27642,6 @@ export interface components { /** Organizations */ organizations: string[]; }; - /** - * PaginatedAuditLogResponse - * @description Response model for paginated audit logs - */ - PaginatedAuditLogResponse: { - /** Audit Logs */ - audit_logs: components["schemas"]["AuditLogResponse"][]; - /** - * Page - * @description Current page number - */ - page: number; - /** - * Page Size - * @description Number of items per page - */ - page_size: number; - /** - * Total - * @description Total number of audit logs matching the filters - */ - total: number; - /** - * Total Pages - * @description Total number of pages - */ - total_pages: number; - }; /** PassThroughEndpointResponse */ PassThroughEndpointResponse: { /** Endpoints */ @@ -31800,63 +31152,6 @@ export interface components { /** Model Names */ model_names?: string[] | null; }; - /** - * UpdateProjectRequest - * @description Request model for POST /project/update - */ - UpdateProjectRequest: { - /** Allowed Models */ - allowed_models?: string[] | null; - /** Blocked */ - blocked?: boolean | null; - /** Budget Duration */ - budget_duration?: string | null; - /** Budget Id */ - budget_id?: string | null; - /** Description */ - description?: string | null; - /** Guardrails */ - guardrails?: string[] | null; - /** Max Budget */ - max_budget?: number | null; - /** Max Parallel Requests */ - max_parallel_requests?: number | null; - /** Metadata */ - metadata?: { - [key: string]: unknown; - } | null; - /** Model Max Budget */ - model_max_budget?: { - [key: string]: unknown; - } | null; - /** Model Rpm Limit */ - model_rpm_limit?: { - [key: string]: unknown; - } | null; - /** Model Tpm Limit */ - model_tpm_limit?: { - [key: string]: unknown; - } | null; - /** Models */ - models?: string[] | null; - object_permission?: components["schemas"]["LiteLLM_ObjectPermissionBase"] | null; - /** Policies */ - policies?: string[] | null; - /** Project Alias */ - project_alias?: string | null; - /** Project Id */ - project_id: string; - /** Rpm Limit */ - rpm_limit?: number | null; - /** Soft Budget */ - soft_budget?: number | null; - /** Tags */ - tags?: string[] | null; - /** Team Id */ - team_id?: string | null; - /** Tpm Limit */ - tpm_limit?: number | null; - }; /** * UpdatePublicModelGroupsRequest * @description Request model for updating public model groups @@ -34300,105 +33595,6 @@ export interface operations { }; }; }; - get_audit_logs_audit_get: { - parameters: { - query?: { - page?: number; - page_size?: number; - /** @description Filter by user or system that performed the action */ - changed_by?: string | null; - /** @description Filter by API key hash that performed the action */ - changed_by_api_key?: string | null; - /** @description Filter by action type (create, update, delete) */ - action?: string | null; - /** @description Filter by table name that was modified */ - table_name?: string | null; - /** @description Filter by ID of the object that was modified */ - object_id?: string | null; - /** @description Filter logs after this date */ - start_date?: string | null; - /** @description Filter logs before this date */ - end_date?: string | null; - /** @description Filter by team_id present in before_value or updated_values JSON (PostgreSQL only) */ - object_team_id?: string | null; - /** @description Filter by token (key hash) present in before_value or updated_values JSON (PostgreSQL only) */ - object_key_hash?: string | null; - /** @description Column to sort by (e.g. 'updated_at', 'action', 'table_name') */ - sort_by?: string | null; - /** @description Sort order ('asc' or 'desc') */ - sort_order?: string; - }; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["PaginatedAuditLogResponse"]; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - get_audit_log_by_id_audit__id__get: { - parameters: { - query?: never; - header?: never; - path: { - id: string; - }; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["AuditLogResponse"]; - }; - }; - /** @description Audit log not found */ - 404: { - headers: { - [name: string]: unknown; - }; - content?: never; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - /** @description Database connection error */ - 500: { - headers: { - [name: string]: unknown; - }; - content?: never; - }; - }; - }; azure_proxy_route_azure__endpoint__get: { parameters: { query?: never; @@ -37924,79 +37120,6 @@ export interface operations { }; }; }; - get_email_event_settings_email_event_settings_get: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["EmailEventSettingsResponse"]; - }; - }; - }; - }; - update_event_settings_email_event_settings_patch: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody: { - content: { - "application/json": components["schemas"]["EmailEventSettingsUpdateRequest"]; - }; - }; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - reset_event_settings_email_event_settings_reset_post: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - }; - }; embeddings_embeddings_post: { parameters: { query?: never; @@ -46159,156 +45282,6 @@ export interface operations { }; }; }; - delete_project_project_delete_delete: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody: { - content: { - "application/json": components["schemas"]["DeleteProjectRequest"]; - }; - }; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["LiteLLM_ProjectTable"][]; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - project_info_project_info_get: { - parameters: { - query: { - project_id: string; - }; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["LiteLLM_ProjectTable"]; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - list_projects_project_list_get: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["LiteLLM_ProjectTable"][]; - }; - }; - }; - }; - new_project_project_new_post: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody: { - content: { - "application/json": components["schemas"]["NewProjectRequest"]; - }; - }; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["NewProjectResponse"]; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; - update_project_project_update_post: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody: { - content: { - "application/json": components["schemas"]["UpdateProjectRequest"]; - }; - }; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["LiteLLM_ProjectTable"]; - }; - }; - /** @description Validation Error */ - 422: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": components["schemas"]["HTTPValidationError"]; - }; - }; - }; - }; create_prompt_prompts_post: { parameters: { query?: never; @@ -47221,26 +46194,6 @@ export interface operations { }; }; }; - get_robots_robots_txt_get: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - }; - }; get_router_fields_router_fields_get: { parameters: { query?: never; @@ -50847,26 +49800,6 @@ export interface operations { }; }; }; - available_enterprise_users_user_available_users_get: { - parameters: { - query?: never; - header?: never; - path?: never; - cookie?: never; - }; - requestBody?: never; - responses: { - /** @description Successful Response */ - 200: { - headers: { - [name: string]: unknown; - }; - content: { - "application/json": unknown; - }; - }; - }; - }; bulk_user_update_user_bulk_update_post: { parameters: { query?: never; From 87137710c5fa3f8b6ec3b55e2d55fdb570d154fa Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 3 Jul 2026 19:44:41 +0000 Subject: [PATCH 08/54] fix: restore schema.d.ts from base branch Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ui/litellm-dashboard/src/lib/http/schema.d.ts | 1127 +++++++++++++++++ 1 file changed, 1127 insertions(+) diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 80bc26a810d..bfd1efc99f3 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -706,6 +706,60 @@ export interface paths { patch?: never; trace?: never; }; + "/audit": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Get Audit Logs + * @description Get all audit logs with filtering and pagination. + * + * Returns a paginated response of audit logs matching the specified filters. + * + * Note: object_team_id and object_key_hash use Prisma JSON path filtering, + * which requires PostgreSQL. + */ + get: operations["get_audit_logs_audit_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/audit/{id}": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Get Audit Log By Id + * @description Get detailed information about a specific audit log entry by its ID. + * + * Args: + * id (str): The unique identifier of the audit log entry + * + * Returns: + * AuditLogResponse: Detailed information about the audit log entry + * + * Raises: + * HTTPException: If the audit log is not found or if there's a database connection error + */ + get: operations["get_audit_log_by_id_audit__id__get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/azure/{endpoint}": { parameters: { query?: never; @@ -3116,6 +3170,50 @@ export interface paths { patch?: never; trace?: never; }; + "/email/event_settings": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Get Email Event Settings + * @description Get all email event settings + */ + get: operations["get_email_event_settings_email_event_settings_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + /** + * Update Event Settings + * @description Update the settings for email events + */ + patch: operations["update_event_settings_email_event_settings_patch"]; + trace?: never; + }; + "/email/event_settings/reset": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** + * Reset Event Settings + * @description Reset all email event settings to default (new user invitations on, virtual key creation off) + */ + post: operations["reset_event_settings_email_event_settings_reset_post"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/embeddings": { parameters: { query?: never; @@ -6406,6 +6504,7 @@ export interface paths { * - disable_global_guardrails: Optional[bool] - Whether to disable global guardrails for the key. * - permissions: Optional[dict] - key-specific permissions. Currently just used for turning off pii masking (if connected). Example - {"pii": false} * - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}}. IF null or {} then no model specific budget. + * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - model_rpm_limit: Optional[dict] - key-specific model rpm limit. Example - {"text-davinci-002": 1000, "gpt-3.5-turbo": 1000}. IF null or {} then no model specific rpm limit. * - model_tpm_limit: Optional[dict] - key-specific model tpm limit. Example - {"text-davinci-002": 1000, "gpt-3.5-turbo": 1000}. IF null or {} then no model specific tpm limit. * - mcp_rpm_limit: Optional[dict] - key-specific per-MCP-server rpm limit, keyed by MCP server name (alias if set, else the configured name). Example - {"github": 100, "slack": 200}. IF null or {} then no MCP-specific rpm limit. @@ -6612,6 +6711,7 @@ export interface paths { * - spend: Optional[float] - Amount spent by key * - max_budget: Optional[float] - Max budget for key * - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}} + * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - budget_duration: Optional[str] - Budget reset period ("30d", "1h", etc.) * - soft_budget: Optional[float] - Soft budget limit (warning vs. hard stop). Will trigger a slack alert when this soft budget is reached. * - max_parallel_requests: Optional[int] - Rate limit for parallel requests @@ -6687,6 +6787,7 @@ export interface paths { * - guardrails: Optional[List[str]] - List of active guardrails for the key * - permissions: Optional[dict] - key-specific permissions. Currently just used for turning off pii masking (if connected). Example - {"pii": false} * - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}}}. IF null or {} then no model specific budget. + * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - model_rpm_limit: Optional[dict] - key-specific model rpm limit. Example - {"text-davinci-002": 1000, "gpt-3.5-turbo": 1000}. IF null or {} then no model specific rpm limit. * - model_tpm_limit: Optional[dict] - key-specific model tpm limit. Example - {"text-davinci-002": 1000, "gpt-3.5-turbo": 1000}. IF null or {} then no model specific tpm limit. * - mcp_rpm_limit: Optional[dict] - key-specific per-MCP-server rpm limit, keyed by MCP server name (alias if set, else the configured name). Example - {"github": 100, "slack": 200}. IF null or {} then no MCP-specific rpm limit. @@ -6785,6 +6886,7 @@ export interface paths { * - spend: Optional[float] - Amount spent by key * - max_budget: Optional[float] - Max budget for key * - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}} + * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - budget_duration: Optional[str] - Budget reset period ("30d", "1h", etc.) * - soft_budget: Optional[float] - [TODO] Soft budget limit (warning vs. hard stop). Will trigger a slack alert when this soft budget is reached. * - max_parallel_requests: Optional[int] - Rate limit for parallel requests @@ -6866,6 +6968,7 @@ export interface paths { * - spend: Optional[float] - Amount spent by key * - max_budget: Optional[float] - Max budget for key * - model_max_budget: Optional[Dict[str, BudgetConfig]] - Model-specific budgets {"gpt-4": {"budget_limit": 0.0005, "time_period": "30d"}} + * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - budget_duration: Optional[str] - Budget reset period ("30d", "1h", etc.) * - soft_budget: Optional[float] - Soft budget limit (warning vs. hard stop). Will trigger a slack alert when this soft budget is reached. * - max_parallel_requests: Optional[int] - Rate limit for parallel requests @@ -9866,6 +9969,240 @@ export interface paths { patch?: never; trace?: never; }; + "/project/delete": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + post?: never; + /** + * Delete Project + * @description Delete projects + * + * Parameters: + * - project_ids: *List[str]* - List of project ids to delete + * + * Example: + * ```bash + * curl --location --request DELETE 'http://0.0.0.0:4000/project/delete' \ + * --header 'Authorization: Bearer sk-1234' \ + * --header 'Content-Type: application/json' \ + * --data '{ + * "project_ids": ["project-123", "project-456"] + * }' + * ``` + */ + delete: operations["delete_project_project_delete_delete"]; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/project/info": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Project Info + * @description Get information about a specific project + * + * Parameters: + * - project_id: *str* - The project id to fetch info for + * + * Example: + * ```bash + * curl --location 'http://0.0.0.0:4000/project/info?project_id=project-123' \ + * --header 'Authorization: Bearer sk-1234' + * ``` + */ + get: operations["project_info_project_info_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/project/list": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * List Projects + * @description List all projects that the user has access to + * + * Example: + * ```bash + * curl --location 'http://0.0.0.0:4000/project/list' \ + * --header 'Authorization: Bearer sk-1234' + * ``` + */ + get: operations["list_projects_project_list_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/project/new": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** + * New Project + * @description Create a new project. Projects sit between teams and keys in the hierarchy. + * + * Only admins or team admins can create projects. + * + * # Parameters + * + * - project_alias: *Optional[str]* - The name of the project. + * - description: *Optional[str]* - Description of the project's purpose and use case. + * - team_id: *str* - The team id that this project belongs to. Required. + * - models: *List* - The models the project has access to. + * - budget_id: *Optional[str]* - The id for a budget (tpm/rpm/max budget) for the project. + * ### IF NO BUDGET ID - CREATE ONE WITH THESE PARAMS ### + * - max_budget: *Optional[float]* - Max budget for project + * - tpm_limit: *Optional[int]* - Max tpm limit for project + * - rpm_limit: *Optional[int]* - Max rpm limit for project + * - max_parallel_requests: *Optional[int]* - Max parallel requests for project + * - soft_budget: *Optional[float]* - Get a slack alert when this soft budget is reached. Don't block requests. + * - model_max_budget: *Optional[dict]* - Max budget for a specific model. Example: {"gpt-4": 100.0, "gpt-3.5-turbo": 50.0} + * - model_rpm_limit: *Optional[dict]* - RPM limits per model. Example: {"gpt-4": 1000, "gpt-3.5-turbo": 5000} + * - model_tpm_limit: *Optional[dict]* - TPM limits per model. Example: {"gpt-4": 50000, "gpt-3.5-turbo": 100000} + * - budget_duration: *Optional[str]* - Frequency of reseting project budget + * - metadata: *Optional[dict]* - Metadata for project, store information for project. Example metadata - {"use_case_id": "SNOW-12345", "responsible_ai_id": "RAI-67890"} + * - tags: *Optional[list]* - Tags for the project. Example: ["production", "api"] + * - blocked: *bool* - Flag indicating if the project is blocked or not - will stop all calls from keys with this project_id. + * - object_permission: Optional[LiteLLM_ObjectPermissionBase] - project-specific object permission. Example - {"vector_stores": ["vector_store_1", "vector_store_2"]}. IF null or {} then no object permission. + * + * Example 1: Create new project **without** a budget_id, with model-specific limits + * + * ```bash + * curl --location 'http://0.0.0.0:4000/project/new' \ + * --header 'Authorization: Bearer sk-1234' \ + * --header 'Content-Type: application/json' \ + * --data '{ + * "project_alias": "flight-search-assistant", + * "description": "AI-powered flight search and booking assistant", + * "team_id": "team-123", + * "models": ["gpt-4", "gpt-3.5-turbo"], + * "max_budget": 100, + * "model_rpm_limit": { + * "gpt-4": 1000, + * "gpt-3.5-turbo": 5000 + * }, + * "model_tpm_limit": { + * "gpt-4": 50000, + * "gpt-3.5-turbo": 100000 + * }, + * "metadata": { + * "use_case_id": "SNOW-12345", + * "responsible_ai_id": "RAI-67890" + * } + * }' + * ``` + * + * Example 2: Create new project **with** a budget_id + * + * ```bash + * curl --location 'http://0.0.0.0:4000/project/new' \ + * --header 'Authorization: Bearer sk-1234' \ + * --header 'Content-Type: application/json' \ + * --data '{ + * "project_alias": "hotel-recommendations", + * "description": "Personalized hotel recommendation engine", + * "team_id": "team-123", + * "models": ["claude-3-sonnet"], + * "budget_id": "428eeaa8-f3ac-4e85-a8fb-7dc8d7aa8689", + * "metadata": { + * "use_case_id": "SNOW-54321" + * } + * }' + * ``` + */ + post: operations["new_project_project_new_post"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/project/update": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** + * Update Project + * @description Update a project + * + * Parameters: + * - project_id: *str* - The project id to update. Required. + * - project_alias: *Optional[str]* - Updated name for the project + * - description: *Optional[str]* - Updated description for the project + * - team_id: *Optional[str]* - Updated team_id for the project + * - metadata: *Optional[dict]* - Updated metadata for project + * - models: *Optional[list]* - Updated list of models for the project + * - blocked: *Optional[bool]* - Updated blocked status + * - max_budget: *Optional[float]* - Updated max budget + * - tpm_limit: *Optional[int]* - Updated tpm limit + * - rpm_limit: *Optional[int]* - Updated rpm limit + * - model_rpm_limit: *Optional[dict]* - Updated RPM limits per model + * - model_tpm_limit: *Optional[dict]* - Updated TPM limits per model + * - budget_duration: *Optional[str]* - Updated budget duration + * - tags: *Optional[list]* - Updated list of tags for the project + * - object_permission: Optional[LiteLLM_ObjectPermissionBase] - Updated object permission + * + * Example: + * ```bash + * curl --location 'http://0.0.0.0:4000/project/update' \ + * --header 'Authorization: Bearer sk-1234' \ + * --header 'Content-Type: application/json' \ + * --data '{ + * "project_id": "project-123", + * "description": "Updated flight search system with enhanced capabilities", + * "max_budget": 200, + * "model_rpm_limit": { + * "gpt-4": 2000, + * "gpt-3.5-turbo": 10000 + * }, + * "metadata": { + * "use_case_id": "SNOW-12345", + * "status": "active" + * } + * }' + * ``` + */ + post: operations["update_project_project_update_post"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/prompts": { parameters: { query?: never; @@ -10908,6 +11245,27 @@ export interface paths { patch?: never; trace?: never; }; + "/robots.txt": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Get Robots + * @description Block all web crawlers from indexing the proxy server endpoints + * This is useful for ensuring that the API endpoints aren't indexed by search engines + */ + get: operations["get_robots_robots_txt_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/router/fields": { parameters: { query?: never; @@ -13947,6 +14305,26 @@ export interface paths { patch?: never; trace?: never; }; + "/user/available_users": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Available Enterprise Users + * @description For keys with `max_users` set, return the list of users that are allowed to use the key. + */ + get: operations["available_enterprise_users_user_available_users_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/user/bulk_update": { parameters: { query?: never; @@ -14240,6 +14618,7 @@ export interface paths { * - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x. * - soft_budget: Optional[float] - Get alerts when user crosses given budget, doesn't block requests. * - model_max_budget: Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys) + * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - model_rpm_limit: Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys) * - mcp_rpm_limit: Optional[dict] - Per-MCP-server rpm limit, keyed by MCP server name {"github": 100, "slack": 200}. Enforced for keys and teams only; values set on a user are stored but not enforced per user. * - model_tpm_limit: Optional[float] - Model-specific tpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys) @@ -14320,6 +14699,7 @@ export interface paths { * - max_parallel_requests: Optional[int] - Rate limit a user based on the number of parallel requests. Raises 429 error, if user's parallel requests > x. * - soft_budget: Optional[float] - Get alerts when user crosses given budget, doesn't block requests. * - model_max_budget: Optional[dict] - Model-specific max budget for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-budgets-to-keys) + * - budget_fallbacks: Optional[Dict[str, List[str]]] - Per-model fallback chain tried in order when that model's own `model_max_budget` is exceeded, e.g. {"gpt-4o": ["gpt-4o-mini"]}. * - model_rpm_limit: Optional[float] - Model-specific rpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys) * - mcp_rpm_limit: Optional[dict] - Per-MCP-server rpm limit, keyed by MCP server name {"github": 100, "slack": 200}. Enforced for keys and teams only; values set on a user are stored but not enforced per user. * - model_tpm_limit: Optional[float] - Model-specific tpm limit for user. [Docs](https://docs.litellm.ai/docs/proxy/users#add-model-specific-limits-to-keys) @@ -20227,6 +20607,37 @@ export interface components { */ unnamed_teams_count: number; }; + /** + * AuditLogResponse + * @description Response model for a single audit log entry + */ + AuditLogResponse: { + /** Action */ + action: string; + /** Before Value */ + before_value?: { + [key: string]: unknown; + } | null; + /** Changed By */ + changed_by: string; + /** Changed By Api Key */ + changed_by_api_key: string; + /** Id */ + id: string; + /** Object Id */ + object_id: string; + /** Table Name */ + table_name: string; + /** + * Updated At + * Format: date-time + */ + updated_at: string; + /** Updated Values */ + updated_values?: { + [key: string]: unknown; + } | null; + }; /** BaseLitellmParams */ "BaseLitellmParams-Input": { /** @@ -22716,6 +23127,14 @@ export interface components { /** Organization Ids */ organization_ids: string[]; }; + /** + * DeleteProjectRequest + * @description Request model for DELETE /project/delete + */ + DeleteProjectRequest: { + /** Project Ids */ + project_ids: string[]; + }; /** * DeleteSkillResponse * @description Response from deleting a skill @@ -22830,6 +23249,27 @@ export interface components { /** Write Capacity Units */ write_capacity_units?: number | null; }; + /** + * EmailEvent + * @enum {string} + */ + EmailEvent: "Virtual Key Created" | "New User Invitation" | "Virtual Key Rotated" | "Soft Budget Crossed" | "Max Budget Alert"; + /** EmailEventSettings */ + EmailEventSettings: { + /** Enabled */ + enabled: boolean; + event: components["schemas"]["EmailEvent"]; + }; + /** EmailEventSettingsResponse */ + EmailEventSettingsResponse: { + /** Settings */ + settings: components["schemas"]["EmailEventSettings"][]; + }; + /** EmailEventSettingsUpdateRequest */ + EmailEventSettingsUpdateRequest: { + /** Settings */ + settings: components["schemas"]["EmailEventSettings"][]; + }; /** EmbeddingRequest */ EmbeddingRequest: { /** @@ -23146,6 +23586,10 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** Budget Fallbacks */ + budget_fallbacks?: { + [key: string]: string[]; + } | null; /** Budget Id */ budget_id?: string | null; /** Budget Limits */ @@ -23286,6 +23730,10 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** Budget Fallbacks */ + budget_fallbacks?: { + [key: string]: string[]; + } | null; /** Budget Id */ budget_id?: string | null; /** Budget Limits */ @@ -24269,6 +24717,13 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** + * Budget Fallbacks + * @default {} + */ + budget_fallbacks: { + [key: string]: string[]; + }; /** Budget Id */ budget_id?: string | null; /** Budget Limits */ @@ -25123,6 +25578,65 @@ export interface components { } & { [key: string]: unknown; }; + /** + * LiteLLM_ProjectTable + * @description Database model representation for project + */ + LiteLLM_ProjectTable: { + /** + * Blocked + * @default false + */ + blocked: boolean; + /** Budget Id */ + budget_id?: string | null; + /** Created At */ + created_at?: string | null; + /** Created By */ + created_by?: string | null; + /** Description */ + description?: string | null; + litellm_budget_table?: components["schemas"]["LiteLLM_BudgetTable"] | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model Rpm Limit */ + model_rpm_limit?: { + [key: string]: unknown; + } | null; + /** Model Spend */ + model_spend?: { + [key: string]: unknown; + } | null; + /** Model Tpm Limit */ + model_tpm_limit?: { + [key: string]: unknown; + } | null; + /** + * Models + * @default [] + */ + models: string[]; + object_permission?: components["schemas"]["LiteLLM_ObjectPermissionTable"] | null; + /** Object Permission Id */ + object_permission_id?: string | null; + /** Project Alias */ + project_alias?: string | null; + /** Project Id */ + project_id: string; + /** + * Spend + * @default 0 + */ + spend: number; + /** Team Id */ + team_id?: string | null; + /** Updated At */ + updated_at?: string | null; + /** Updated By */ + updated_by?: string | null; + }; /** LiteLLM_ProxyModelTable */ LiteLLM_ProxyModelTable: { /** @@ -25595,6 +26109,13 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** + * Budget Fallbacks + * @default {} + */ + budget_fallbacks: { + [key: string]: string[]; + }; /** Budget Id */ budget_id?: string | null; /** Budget Limits */ @@ -27133,6 +27654,134 @@ export interface components { /** Users */ users?: components["schemas"]["LiteLLM_UserTable"][] | null; }; + /** + * NewProjectRequest + * @description Request model for POST /project/new + */ + NewProjectRequest: { + /** Allowed Models */ + allowed_models?: string[] | null; + /** + * Blocked + * @default false + */ + blocked: boolean; + /** Budget Duration */ + budget_duration?: string | null; + /** Budget Id */ + budget_id?: string | null; + /** Description */ + description?: string | null; + /** Guardrails */ + guardrails?: string[] | null; + /** Max Budget */ + max_budget?: number | null; + /** Max Parallel Requests */ + max_parallel_requests?: number | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model Max Budget */ + model_max_budget?: { + [key: string]: unknown; + } | null; + /** Model Rpm Limit */ + model_rpm_limit?: { + [key: string]: unknown; + } | null; + /** Model Tpm Limit */ + model_tpm_limit?: { + [key: string]: unknown; + } | null; + /** + * Models + * @default [] + */ + models: string[]; + object_permission?: components["schemas"]["LiteLLM_ObjectPermissionBase"] | null; + /** Policies */ + policies?: string[] | null; + /** Project Alias */ + project_alias?: string | null; + /** Project Id */ + project_id?: string | null; + /** Rpm Limit */ + rpm_limit?: number | null; + /** Soft Budget */ + soft_budget?: number | null; + /** Tags */ + tags?: string[] | null; + /** Team Id */ + team_id: string; + /** Tpm Limit */ + tpm_limit?: number | null; + }; + /** + * NewProjectResponse + * @description Response model for POST /project/new + */ + NewProjectResponse: { + /** + * Blocked + * @default false + */ + blocked: boolean; + /** Budget Id */ + budget_id?: string | null; + /** + * Created At + * Format: date-time + */ + created_at: string; + /** Created By */ + created_by?: string | null; + /** Description */ + description?: string | null; + litellm_budget_table?: components["schemas"]["LiteLLM_BudgetTable"] | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model Rpm Limit */ + model_rpm_limit?: { + [key: string]: unknown; + } | null; + /** Model Spend */ + model_spend?: { + [key: string]: unknown; + } | null; + /** Model Tpm Limit */ + model_tpm_limit?: { + [key: string]: unknown; + } | null; + /** + * Models + * @default [] + */ + models: string[]; + object_permission?: components["schemas"]["LiteLLM_ObjectPermissionTable"] | null; + /** Object Permission Id */ + object_permission_id?: string | null; + /** Project Alias */ + project_alias?: string | null; + /** Project Id */ + project_id: string; + /** + * Spend + * @default 0 + */ + spend: number; + /** Team Id */ + team_id?: string | null; + /** + * Updated At + * Format: date-time + */ + updated_at: string; + /** Updated By */ + updated_by?: string | null; + }; /** NewTeamRequest */ NewTeamRequest: { /** Access Group Ids */ @@ -27275,6 +27924,10 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** Budget Fallbacks */ + budget_fallbacks?: { + [key: string]: string[]; + } | null; /** Budget Limits */ budget_limits?: components["schemas"]["BudgetLimitEntry"][] | null; /** @@ -27409,6 +28062,10 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** Budget Fallbacks */ + budget_fallbacks?: { + [key: string]: string[]; + } | null; /** Budget Id */ budget_id?: string | null; /** Budget Limits */ @@ -27642,6 +28299,34 @@ export interface components { /** Organizations */ organizations: string[]; }; + /** + * PaginatedAuditLogResponse + * @description Response model for paginated audit logs + */ + PaginatedAuditLogResponse: { + /** Audit Logs */ + audit_logs: components["schemas"]["AuditLogResponse"][]; + /** + * Page + * @description Current page number + */ + page: number; + /** + * Page Size + * @description Number of items per page + */ + page_size: number; + /** + * Total + * @description Total number of audit logs matching the filters + */ + total: number; + /** + * Total Pages + * @description Total number of pages + */ + total_pages: number; + }; /** PassThroughEndpointResponse */ PassThroughEndpointResponse: { /** Endpoints */ @@ -28999,6 +29684,10 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** Budget Fallbacks */ + budget_fallbacks?: { + [key: string]: string[]; + } | null; /** Budget Id */ budget_id?: string | null; /** Budget Limits */ @@ -30958,6 +31647,10 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** Budget Fallbacks */ + budget_fallbacks?: { + [key: string]: string[]; + } | null; /** Budget Id */ budget_id?: string | null; /** Budget Limits */ @@ -31152,6 +31845,63 @@ export interface components { /** Model Names */ model_names?: string[] | null; }; + /** + * UpdateProjectRequest + * @description Request model for POST /project/update + */ + UpdateProjectRequest: { + /** Allowed Models */ + allowed_models?: string[] | null; + /** Blocked */ + blocked?: boolean | null; + /** Budget Duration */ + budget_duration?: string | null; + /** Budget Id */ + budget_id?: string | null; + /** Description */ + description?: string | null; + /** Guardrails */ + guardrails?: string[] | null; + /** Max Budget */ + max_budget?: number | null; + /** Max Parallel Requests */ + max_parallel_requests?: number | null; + /** Metadata */ + metadata?: { + [key: string]: unknown; + } | null; + /** Model Max Budget */ + model_max_budget?: { + [key: string]: unknown; + } | null; + /** Model Rpm Limit */ + model_rpm_limit?: { + [key: string]: unknown; + } | null; + /** Model Tpm Limit */ + model_tpm_limit?: { + [key: string]: unknown; + } | null; + /** Models */ + models?: string[] | null; + object_permission?: components["schemas"]["LiteLLM_ObjectPermissionBase"] | null; + /** Policies */ + policies?: string[] | null; + /** Project Alias */ + project_alias?: string | null; + /** Project Id */ + project_id: string; + /** Rpm Limit */ + rpm_limit?: number | null; + /** Soft Budget */ + soft_budget?: number | null; + /** Tags */ + tags?: string[] | null; + /** Team Id */ + team_id?: string | null; + /** Tpm Limit */ + tpm_limit?: number | null; + }; /** * UpdatePublicModelGroupsRequest * @description Request model for updating public model groups @@ -31364,6 +32114,10 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** Budget Fallbacks */ + budget_fallbacks?: { + [key: string]: string[]; + } | null; /** Budget Limits */ budget_limits?: components["schemas"]["BudgetLimitEntry"][] | null; /** @@ -31462,6 +32216,10 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** Budget Fallbacks */ + budget_fallbacks?: { + [key: string]: string[]; + } | null; /** Budget Limits */ budget_limits?: components["schemas"]["BudgetLimitEntry"][] | null; /** @@ -31689,6 +32447,13 @@ export interface components { blocked?: boolean | null; /** Budget Duration */ budget_duration?: string | null; + /** + * Budget Fallbacks + * @default {} + */ + budget_fallbacks: { + [key: string]: string[]; + }; /** Budget Id */ budget_id?: string | null; /** Budget Limits */ @@ -33595,6 +34360,105 @@ export interface operations { }; }; }; + get_audit_logs_audit_get: { + parameters: { + query?: { + page?: number; + page_size?: number; + /** @description Filter by user or system that performed the action */ + changed_by?: string | null; + /** @description Filter by API key hash that performed the action */ + changed_by_api_key?: string | null; + /** @description Filter by action type (create, update, delete) */ + action?: string | null; + /** @description Filter by table name that was modified */ + table_name?: string | null; + /** @description Filter by ID of the object that was modified */ + object_id?: string | null; + /** @description Filter logs after this date */ + start_date?: string | null; + /** @description Filter logs before this date */ + end_date?: string | null; + /** @description Filter by team_id present in before_value or updated_values JSON (PostgreSQL only) */ + object_team_id?: string | null; + /** @description Filter by token (key hash) present in before_value or updated_values JSON (PostgreSQL only) */ + object_key_hash?: string | null; + /** @description Column to sort by (e.g. 'updated_at', 'action', 'table_name') */ + sort_by?: string | null; + /** @description Sort order ('asc' or 'desc') */ + sort_order?: string; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["PaginatedAuditLogResponse"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + get_audit_log_by_id_audit__id__get: { + parameters: { + query?: never; + header?: never; + path: { + id: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["AuditLogResponse"]; + }; + }; + /** @description Audit log not found */ + 404: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + /** @description Database connection error */ + 500: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + }; + }; azure_proxy_route_azure__endpoint__get: { parameters: { query?: never; @@ -37120,6 +37984,79 @@ export interface operations { }; }; }; + get_email_event_settings_email_event_settings_get: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["EmailEventSettingsResponse"]; + }; + }; + }; + }; + update_event_settings_email_event_settings_patch: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["EmailEventSettingsUpdateRequest"]; + }; + }; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": unknown; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + reset_event_settings_email_event_settings_reset_post: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": unknown; + }; + }; + }; + }; embeddings_embeddings_post: { parameters: { query?: never; @@ -45282,6 +46219,156 @@ export interface operations { }; }; }; + delete_project_project_delete_delete: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["DeleteProjectRequest"]; + }; + }; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["LiteLLM_ProjectTable"][]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + project_info_project_info_get: { + parameters: { + query: { + project_id: string; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["LiteLLM_ProjectTable"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + list_projects_project_list_get: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["LiteLLM_ProjectTable"][]; + }; + }; + }; + }; + new_project_project_new_post: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["NewProjectRequest"]; + }; + }; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["NewProjectResponse"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + update_project_project_update_post: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["UpdateProjectRequest"]; + }; + }; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["LiteLLM_ProjectTable"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; create_prompt_prompts_post: { parameters: { query?: never; @@ -46194,6 +47281,26 @@ export interface operations { }; }; }; + get_robots_robots_txt_get: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": unknown; + }; + }; + }; + }; get_router_fields_router_fields_get: { parameters: { query?: never; @@ -49800,6 +50907,26 @@ export interface operations { }; }; }; + available_enterprise_users_user_available_users_get: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": unknown; + }; + }; + }; + }; bulk_user_update_user_bulk_update_post: { parameters: { query?: never; From ed51c96d3f97a4ceedc403e29fe86847dd0d3d8b Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 3 Jul 2026 19:50:36 +0000 Subject: [PATCH 09/54] feat(ui): add budget fallbacks configuration to key create/edit forms Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../BudgetFallbacksEditor.test.tsx | 82 ++++++++++ .../BudgetFallbacksEditor.tsx | 153 ++++++++++++++++++ .../components/key_team_helpers/key_list.tsx | 1 + .../organisms/create_key_button.tsx | 26 +++ .../components/templates/key_edit_view.tsx | 23 +++ .../components/templates/key_info_view.tsx | 15 ++ 6 files changed, 300 insertions(+) create mode 100644 ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.test.tsx create mode 100644 ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.tsx diff --git a/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.test.tsx b/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.test.tsx new file mode 100644 index 00000000000..194f717a92f --- /dev/null +++ b/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.test.tsx @@ -0,0 +1,82 @@ +import { render, screen } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; +import { describe, expect, it, vi } from "vitest"; +import { BudgetFallbacksEditor } from "./BudgetFallbacksEditor"; + +const MODELS = ["gpt-4", "gpt-3.5-turbo", "claude-3", "claude-haiku"]; + +describe("BudgetFallbacksEditor", () => { + it("renders empty state with add button", () => { + const onChange = vi.fn(); + render(); + expect(screen.getByText("Add Budget Fallback")).toBeTruthy(); + expect(screen.getByText(/reroute to fallback models/)).toBeTruthy(); + }); + + it("renders existing entries from value prop", () => { + const onChange = vi.fn(); + render( + , + ); + expect(screen.getByText("IF BUDGET EXCEEDED, TRY")).toBeTruthy(); + expect(screen.getByText("Primary Model")).toBeTruthy(); + expect(screen.getByText("Fallback Models")).toBeTruthy(); + }); + + it("adds a new empty entry when clicking add button", async () => { + const user = userEvent.setup(); + const onChange = vi.fn(); + render(); + + await user.click(screen.getByText("Add Budget Fallback")); + expect(screen.getByText("Primary Model")).toBeTruthy(); + expect(onChange).toHaveBeenCalledWith({}); + }); + + it("removes an entry and emits updated dict", async () => { + const user = userEvent.setup(); + const onChange = vi.fn(); + const { container } = render( + , + ); + + const removeButtons = container.querySelectorAll(".relative > button[type='button']"); + expect(removeButtons.length).toBe(2); + + await user.click(removeButtons[0]); + expect(onChange).toHaveBeenLastCalledWith({ "claude-3": ["claude-haiku"] }); + }); + + it("renders multiple entries for multiple fallback groups", () => { + const onChange = vi.fn(); + render( + , + ); + const labels = screen.getAllByText("Primary Model"); + expect(labels.length).toBe(2); + }); + + it("shows ordering hint when multiple fallback models are configured", () => { + const onChange = vi.fn(); + render( + , + ); + expect(screen.getByText(/first model still within its own budget/)).toBeTruthy(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.tsx b/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.tsx new file mode 100644 index 00000000000..e58d289df8a --- /dev/null +++ b/ui/litellm-dashboard/src/components/key_team_helpers/BudgetFallbacksEditor.tsx @@ -0,0 +1,153 @@ +import { Button, Select, Tooltip } from "antd"; +import { ArrowDown, Plus, X } from "lucide-react"; +import React, { useState } from "react"; + +interface FallbackEntry { + id: string; + primaryModel: string | null; + fallbackModels: string[]; +} + +interface BudgetFallbacksEditorProps { + value: Record; + onChange: (v: Record) => void; + availableModels: string[]; +} + +const entriesToDict = (entries: readonly FallbackEntry[]): Record => + Object.fromEntries( + entries + .filter( + (e): e is FallbackEntry & { primaryModel: string } => e.primaryModel !== null && e.fallbackModels.length > 0, + ) + .map((e) => [e.primaryModel, e.fallbackModels]), + ); + +const dictToEntries = (dict: Record): FallbackEntry[] => { + const keys = Object.keys(dict); + if (keys.length === 0) return []; + return keys.map((model, i) => ({ + id: String(i + 1), + primaryModel: model, + fallbackModels: dict[model], + })); +}; + +export function BudgetFallbacksEditor({ value, onChange, availableModels }: BudgetFallbacksEditorProps) { + const [entries, setEntries] = useState(() => dictToEntries(value)); + + const emitChange = (updated: FallbackEntry[]) => { + setEntries(updated); + onChange(entriesToDict(updated)); + }; + + const addEntry = () => { + emitChange([...entries, { id: Date.now().toString(), primaryModel: null, fallbackModels: [] }]); + }; + + const removeEntry = (id: string) => { + emitChange(entries.filter((e) => e.id !== id)); + }; + + const updateEntry = (id: string, patch: Partial) => { + emitChange(entries.map((e) => (e.id === id ? { ...e, ...patch } : e))); + }; + + const usedPrimaryModels = new Set(entries.map((e) => e.primaryModel).filter(Boolean)); + + if (entries.length === 0) { + return ( +
+
+ When a model exceeds its per-model budget, requests automatically reroute to fallback models +
+ +
+ ); + } + + return ( +
+
+ When a model exceeds its per-model budget, requests automatically reroute to fallback models +
+ {entries.map((entry) => { + const availablePrimaryOptions = availableModels.filter( + (m) => m === entry.primaryModel || !usedPrimaryModels.has(m), + ); + const availableFallbackOptions = availableModels.filter((m) => m !== entry.primaryModel); + + return ( +
+ + +
+ + updateEntry(entry.id, { fallbackModels: values })} + disabled={!entry.primaryModel} + showSearch + filterOption={(input, option) => (option?.label ?? "").toLowerCase().includes(input.toLowerCase())} + options={availableFallbackOptions.map((m) => ({ label: m, value: m }))} + getPopupContainer={(trigger) => trigger.parentElement || document.body} + maxTagCount="responsive" + maxTagPlaceholder={(omittedValues) => ( + v).join(", ")} + > + +{omittedValues.length} more + + )} + /> + {entry.fallbackModels.length > 1 && ( +
+ Tried in order; first model still within its own budget is used +
+ )} +
+
+ ); + })} + +
+ ); +} diff --git a/ui/litellm-dashboard/src/components/key_team_helpers/key_list.tsx b/ui/litellm-dashboard/src/components/key_team_helpers/key_list.tsx index 60568da48ed..99fe23792d6 100644 --- a/ui/litellm-dashboard/src/components/key_team_helpers/key_list.tsx +++ b/ui/litellm-dashboard/src/components/key_team_helpers/key_list.tsx @@ -97,6 +97,7 @@ export interface KeyResponse { agent_access_groups?: string[]; }; access_group_ids?: string[]; + budget_fallbacks?: Record; budget_limits?: Array<{ budget_duration: string; max_budget: number; reset_at?: string }>; auto_rotate?: boolean; rotation_interval?: string; diff --git a/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx b/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx index 2a448d63300..443cbd34e32 100644 --- a/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx +++ b/ui/litellm-dashboard/src/components/organisms/create_key_button.tsx @@ -28,6 +28,7 @@ import TeamDropdown from "../common_components/team_dropdown"; import OrganizationDropdown from "../common_components/OrganizationDropdown"; import ProjectDropdown from "../common_components/ProjectDropdown"; import { CreateUserButton } from "../CreateUserButton"; +import { BudgetFallbacksEditor } from "../key_team_helpers/BudgetFallbacksEditor"; import { BudgetWindowEntry, BudgetWindowsEditor } from "../key_team_helpers/BudgetWindowsEditor"; import { getModelDisplayName } from "../key_team_helpers/fetch_available_models_team_key"; import { Team } from "../key_team_helpers/key_list"; @@ -202,6 +203,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp const [rotationInterval, setRotationInterval] = useState("30d"); const [routerSettings, setRouterSettings] = useState(null); const [budgetLimits, setBudgetLimits] = useState([]); + const [budgetFallbacks, setBudgetFallbacks] = useState>({}); const [routerSettingsKey, setRouterSettingsKey] = useState(0); const [agentsList, setAgentsList] = useState<{ agent_id: string; agent_name: string }[]>([]); const [selectedAgentId, setSelectedAgentId] = useState(null); @@ -220,6 +222,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp setSelectedOrganizationId(null); setSelectedProjectId(null); setBudgetLimits([]); + setBudgetFallbacks({}); }; const handleCancel = () => { @@ -239,6 +242,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp setSelectedOrganizationId(null); setSelectedProjectId(null); setBudgetLimits([]); + setBudgetFallbacks({}); }; useEffect(() => { @@ -536,6 +540,10 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp formValues.budget_limits = validWindows; } + if (Object.keys(budgetFallbacks).length > 0) { + formValues.budget_fallbacks = budgetFallbacks; + } + let response; if (keyOwner === "service_account") { response = await keyCreateServiceAccountCall(accessToken, formValues); @@ -558,6 +566,7 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp NotificationsManager.success("Virtual Key Created"); form.resetFields(); setBudgetLimits([]); + setBudgetFallbacks({}); localStorage.removeItem("userData" + userID); } catch (error) { console.log("error in create key:", error); @@ -1076,6 +1085,23 @@ const CreateKey: React.FC = ({ team, teams, data, addKey, autoOp > + + Budget Fallbacks{" "} + + + + + } + > + + ( Array.isArray(keyData.budget_limits) ? keyData.budget_limits : [], ); + const [budgetFallbacks, setBudgetFallbacks] = useState>( + keyData.budget_fallbacks && typeof keyData.budget_fallbacks === "object" ? keyData.budget_fallbacks : {}, + ); const { data: organizations, isLoading: isOrganizationsLoading } = useOrganizations(); const { data: projects } = useProjects(); const { data: uiSettingsData } = useUISettings(); @@ -304,6 +308,8 @@ export function KeyEditView({ values.budget_limits = []; } + values.budget_fallbacks = budgetFallbacks; + await onSubmit(values); } finally { setIsKeySaving(false); @@ -472,6 +478,23 @@ export function KeyEditView({ + + Budget Fallbacks{" "} + + + + + } + > + + + diff --git a/ui/litellm-dashboard/src/components/templates/key_info_view.tsx b/ui/litellm-dashboard/src/components/templates/key_info_view.tsx index 76d6f7ece22..197f87b6dfe 100644 --- a/ui/litellm-dashboard/src/components/templates/key_info_view.tsx +++ b/ui/litellm-dashboard/src/components/templates/key_info_view.tsx @@ -748,6 +748,21 @@ export default function KeyInfoView({ + {currentKeyData.budget_fallbacks && Object.keys(currentKeyData.budget_fallbacks).length > 0 && ( +
+ Budget Fallbacks +
+ {Object.entries(currentKeyData.budget_fallbacks).map(([model, fallbacks]) => ( +
+ {model} + -> + {fallbacks.join(", ")} +
+ ))} +
+
+ )} +
Tags
From 680f15f1e3f94a44479f99803d729a1a78c7eb3f Mon Sep 17 00:00:00 2001 From: yucheng-berriai Date: Wed, 1 Jul 2026 23:33:52 -0700 Subject: [PATCH 10/54] fix(proxy): stop leaking master_key and database_url in startup DEBUG logs Three startup log statements in litellm/proxy/proxy_server.py dumped secret-bearing values in cleartext when the last-line-of-defense regex scrubber was bypassed (LITELLM_DISABLE_REDACT_SECRETS=true, older versions that predated the SecretRedactionFilter, or any downstream handler that snapshots log records before the module filter runs) ProxyConfig._load_alerting_settings logged the whole general_settings dict under a label that only referred to the alerting callbacks; a copy-paste bug that happened to leak master_key, database_url, and every other secret sitting in general_settings. Now logs only the alerting callback list ProxyConfig.load_config logged the resolved DB URL after secret-manager resolution. The line's stated purpose was to confirm the retrieval ran, which does not need the value. Now logs a value-less breadcrumb proxy_startup_event logged the raw WORKER_CONFIG blob, which docker/K8s deployments hand the proxy as a JSON string containing master_key, database_url, and provider API keys. Now routes through _redact_worker_config_for_logging, which combines the segment-matching SensitiveDataMasker (catches master_key, api_key, *_token) with an explicit pass over _EXTRA_SECRET_GENERAL_SETTINGS_FIELDS (catches database_url and other credential-URL fields the segment masker misses) Regression tests disable the module-level SecretRedactionFilter so assertions see the raw record; without the fix they would trip on the secret substring, so a future refactor cannot silently reconstruct the leaky string --- litellm/proxy/proxy_server.py | 82 ++++++++++---- .../proxy/proxy_server/test_lifecycle.py | 102 ++++++++++++++++++ .../proxy/proxy_server/test_proxy_config.py | 64 +++++++++++ 3 files changed, 230 insertions(+), 18 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index a29acf3b06f..eda7a1b507a 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -63,6 +63,7 @@ from litellm.litellm_core_utils.litellm_logging import ( _init_custom_logger_compatible_class, ) from litellm.litellm_core_utils.safe_json_dumps import safe_dumps +from litellm.litellm_core_utils.safe_json_loads import safe_json_loads from litellm.proxy._types import ( UI_TEAM_ID, CallbackDelete, @@ -637,6 +638,65 @@ premium_user_data: Optional["EnterpriseLicenseData"] = _license_check.airgapped_ global_max_parallel_request_retries_env: Optional[str] = os.getenv("LITELLM_GLOBAL_MAX_PARALLEL_REQUEST_RETRIES") proxy_state = ProxyState() SENSITIVE_DATA_MASKER = SensitiveDataMasker() + + +# Secret-bearing general_settings fields the segment masker does not match by +# name: database_url and database_extra_connection_params embed DB credentials, +# pass_through_endpoints carry upstream Authorization headers, and +# alert_to_webhook_url is itself a webhook secret +_EXTRA_SECRET_GENERAL_SETTINGS_FIELDS = frozenset( + { + "database_url", + "database_extra_connection_params", + "pass_through_endpoints", + "alert_to_webhook_url", + } +) + + +_EXTRA_SECRET_MASK = "REDACTED" + + +def _redact_config_dict_for_logging(data: dict[str, object]) -> dict[str, object]: + """Mask secret-bearing entries in a proxy config-shaped dict. + + Combines the segment-matching masker (catches `master_key`, `api_key`, + `*_token`, etc.) with an explicit whole-value replacement for the + URL/webhook fields that embed credentials but do not contain any + sensitive-pattern segment (e.g. `database_url` splits into + `['database', 'url']`, so segment matching misses it). The whole-value + replacement covers non-string shapes (`alert_to_webhook_url` is a dict, + `pass_through_endpoints` is a list), which the string-only + `mask_sensitive_keys` helper would otherwise pass through unchanged. + """ + segment_masked = SENSITIVE_DATA_MASKER.mask_dict(data) + return { + key: _EXTRA_SECRET_MASK if key in _EXTRA_SECRET_GENERAL_SETTINGS_FIELDS and value is not None else value + for key, value in segment_masked.items() + } + + +def _redact_worker_config_for_logging(worker_config: str | dict[str, object] | None) -> str | dict[str, object] | None: + """Mask sensitive fields in the worker config before it enters a log record. + + `worker_config` reaches `proxy_startup_event` as either the JSON blob + persisted by `save_worker_config` (a string) or the dict passed directly + to `initialize`. Both shapes can carry `master_key`, `database_url`, + provider API keys, etc.; passing the raw value to `verbose_proxy_logger` + leaks them whenever the last-line-of-defense regex filter is bypassed + (`LITELLM_DISABLE_REDACT_SECRETS=true`, an older log sink, a downstream + handler that captures records pre-filter). Redact at the source. + """ + if worker_config is None: + return None + if isinstance(worker_config, dict): + return _redact_config_dict_for_logging(worker_config) + parsed = safe_json_loads(worker_config, default=None) + if isinstance(parsed, dict): + return safe_dumps(_redact_config_dict_for_logging(parsed)) + return worker_config + + if global_max_parallel_request_retries_env is None: global_max_parallel_request_retries: int = 3 else: @@ -833,7 +893,7 @@ async def proxy_startup_event(app: FastAPI): ### LOAD CONFIG ### worker_config: Optional[Union[str, dict]] = get_secret("WORKER_CONFIG") # type: ignore env_config_yaml: Optional[str] = get_secret_str("CONFIG_FILE_PATH") - verbose_proxy_logger.debug("worker_config: %s", worker_config) + verbose_proxy_logger.debug("worker_config: %s", _redact_worker_config_for_logging(worker_config)) # check if it's a valid file path if env_config_yaml is not None: if os.path.isfile(env_config_yaml) and proxy_config.is_yaml(config_file_path=env_config_yaml): @@ -4336,9 +4396,9 @@ class ProxyConfig: ### CONNECT TO DATABASE ### database_url = general_settings.get("database_url", None) if database_url and database_url.startswith("os.environ/"): - verbose_proxy_logger.debug("GOING INTO LITELLM.GET_SECRET!") + verbose_proxy_logger.debug("Resolving database_url via secret manager") database_url = get_secret(database_url) - verbose_proxy_logger.debug("RETRIEVED DB URL: %s", database_url) + verbose_proxy_logger.debug("Resolved database_url from secret manager") ### MASTER KEY ### master_key = general_settings.get("master_key", get_secret("LITELLM_MASTER_KEY", None)) @@ -4738,7 +4798,7 @@ class ProxyConfig: """ _alerting_callbacks = general_settings.get("alerting", None) - verbose_proxy_logger.debug(f"_alerting_callbacks: {general_settings}") + verbose_proxy_logger.debug("_alerting_callbacks: %s", _alerting_callbacks) if _alerting_callbacks is None: return @@ -14218,20 +14278,6 @@ async def update_config_general_settings( return response -# Secret-bearing general_settings fields the segment masker does not match by -# name: database_url and database_extra_connection_params embed DB credentials, -# pass_through_endpoints carry upstream Authorization headers, and -# alert_to_webhook_url is itself a webhook secret -_EXTRA_SECRET_GENERAL_SETTINGS_FIELDS = frozenset( - { - "database_url", - "database_extra_connection_params", - "pass_through_endpoints", - "alert_to_webhook_url", - } -) - - def _is_secret_general_setting_field(field_name: str) -> bool: return field_name in _EXTRA_SECRET_GENERAL_SETTINGS_FIELDS or SENSITIVE_DATA_MASKER.is_sensitive_key(field_name) diff --git a/tests/test_litellm/proxy/proxy_server/test_lifecycle.py b/tests/test_litellm/proxy/proxy_server/test_lifecycle.py index 9343dcbc29f..842b372ba83 100644 --- a/tests/test_litellm/proxy/proxy_server/test_lifecycle.py +++ b/tests/test_litellm/proxy/proxy_server/test_lifecycle.py @@ -212,6 +212,108 @@ def test_save_worker_config_invalid_no_kwargs_yields_empty(monkeypatch): assert os.environ["WORKER_CONFIG"] == "{}" +# --------------------------------------------------------------------------- +# _redact_worker_config_for_logging (LIT-4152) +# --------------------------------------------------------------------------- + + +_LIT4152_SECRETS = ( + "sk-lit4152-regression-master-key-abcdef1234567890", + "leak_password_9090", + "sk-lit4152-provider-api-key-abcdef", + "postgresql://leak_user:leak_password_9090@leak-host.internal:5432/leak_db", +) + + +def _lit4152_worker_config_dict(): + return { + "model": "openai/gpt-4o-mini", + "config": "/tmp/c.yaml", + "master_key": _LIT4152_SECRETS[0], + "database_url": _LIT4152_SECRETS[3], + "api_key": _LIT4152_SECRETS[2], + "telemetry": True, + } + + +def test__redact_worker_config_for_logging_dict_masks_all_secret_shapes(): + """LIT-4152 regression: dict-form worker_config must not embed any raw + secret. Covers the segment-matched fields (`master_key`, `api_key`) and the + URL-with-credentials field (`database_url`), which the segment masker + misses because neither segment matches its sensitive-pattern set. + """ + from litellm.proxy.proxy_server import _redact_worker_config_for_logging + + redacted = _redact_worker_config_for_logging(_lit4152_worker_config_dict()) + rendered = repr(redacted) + for secret in _LIT4152_SECRETS: + assert secret not in rendered, f"leak: {secret} in {rendered!r}" + assert isinstance(redacted, dict) + assert redacted["model"] == "openai/gpt-4o-mini" + assert redacted["telemetry"] is True + + +def test__redact_worker_config_for_logging_json_string_round_trips_masked(): + """Docker/K8s deployments hand the proxy a JSON string via ``WORKER_CONFIG``. + Confirm the string path also masks and that the returned value re-parses + into a dict with the sensitive fields masked. + """ + from litellm.proxy.proxy_server import _redact_worker_config_for_logging + + payload = json.dumps(_lit4152_worker_config_dict()) + redacted = _redact_worker_config_for_logging(payload) + assert isinstance(redacted, str) + for secret in _LIT4152_SECRETS: + assert secret not in redacted, f"leak: {secret} in {redacted!r}" + parsed = json.loads(redacted) + assert parsed["model"] == "openai/gpt-4o-mini" + + +def test__redact_worker_config_for_logging_passthrough_for_none_and_non_json_string(): + """Non-dict, non-JSON-parseable string is passed through verbatim (nothing + to mask) and ``None`` returns ``None``. + """ + from litellm.proxy.proxy_server import _redact_worker_config_for_logging + + assert _redact_worker_config_for_logging(None) is None + assert _redact_worker_config_for_logging("/tmp/some_config.yaml") == "/tmp/some_config.yaml" + + +def test__redact_config_dict_for_logging_masks_non_string_url_webhook_values(): + """The URL/webhook fields the segment masker cannot catch by key name + (``alert_to_webhook_url``, ``pass_through_endpoints``, + ``database_extra_connection_params``) can hold non-string shapes: + ``alert_to_webhook_url`` is typed as ``Optional[Dict]`` and can nest + secret query params under keys the segment masker also misses. Confirm + the whole value is replaced regardless of shape so a nested webhook or + Bearer token under a non-segment-matched key does not slip through. + """ + from litellm.proxy.proxy_server import _redact_config_dict_for_logging + + nested_webhook_secret = "https://hooks.slack.com/services/T0/B0/nested-webhook-secret-xyz" + data = { + "master_key": "sk-should-be-masked", + "alert_to_webhook_url": {"budget_alerts": nested_webhook_secret}, + "pass_through_endpoints": [ + { + "path": "/upstream", + "target": "https://api.provider.com", + "headers": {"Authorization": "Bearer nested-token-should-be-gone"}, + } + ], + "database_extra_connection_params": {"password": "extra-db-password-abc"}, + } + redacted = _redact_config_dict_for_logging(data) + rendered = repr(redacted) + for secret in ( + "sk-should-be-masked", + nested_webhook_secret, + "nested-token-should-be-gone", + "extra-db-password-abc", + ): + assert secret not in rendered, f"leak: {secret} in {rendered!r}" + + # --------------------------------------------------------------------------- # initialize # --------------------------------------------------------------------------- diff --git a/tests/test_litellm/proxy/proxy_server/test_proxy_config.py b/tests/test_litellm/proxy/proxy_server/test_proxy_config.py index 196ff045208..78ec2813c61 100644 --- a/tests/test_litellm/proxy/proxy_server/test_proxy_config.py +++ b/tests/test_litellm/proxy/proxy_server/test_proxy_config.py @@ -784,6 +784,70 @@ def test_ProxyConfig__load_alerting_settings_invalid_alerting_raises(): pc._load_alerting_settings({"alerting": 12345}) +def test_ProxyConfig__load_alerting_settings_does_not_log_general_settings_dict(monkeypatch): + """Regression for LIT-4152. + + ``_load_alerting_settings`` used to log ``general_settings`` verbatim in a + line labelled ``_alerting_callbacks:``, leaking ``master_key``, + ``database_url``, and any other secret sitting in ``general_settings`` in + cleartext at DEBUG. The fix logs only the alerting callback list. + + The regression check runs with the last-line-of-defense regex scrubber + (``SecretRedactionFilter``) DISABLED, since defense in depth is the point. + The caller must not construct the leaky string, so consumers of the log + stream that bypass the module filter (versions before it existed, + ``LITELLM_DISABLE_REDACT_SECRETS=true`` operators, downstream handlers + that snapshot the record pre-filter) still do not see the secret. Uses a + dedicated handler rather than caplog because caplog is unreliable under + pytest-xdist. + """ + import logging + + import litellm._logging as _logging_module + from litellm._logging import verbose_proxy_logger + + monkeypatch.setattr(_logging_module, "_ENABLE_SECRET_REDACTION", False) + + class LogRecordHandler(logging.Handler): + def __init__(self) -> None: + super().__init__() + self.records: list[logging.LogRecord] = [] + + def emit(self, record: logging.LogRecord) -> None: + self.records.append(record) + + master_key_secret = "sk-lit4152-regression-master-key-abcdef1234567890" + db_url_secret = "postgresql://leak_user:leak_password_9090@leak-host.internal:5432/leak_db" + settings = { + "alerting": ["slack"], + "alerting_threshold": 300, + "master_key": master_key_secret, + "database_url": db_url_secret, + } + + handler = LogRecordHandler() + handler.setLevel(logging.DEBUG) + original_level = verbose_proxy_logger.level + verbose_proxy_logger.setLevel(logging.DEBUG) + verbose_proxy_logger.addHandler(handler) + try: + try: + ProxyConfig()._load_alerting_settings(settings) + except Exception: + pass # downstream init may fail without full env; the debug log fires first + rendered = " ".join(record.getMessage() for record in handler.records) + finally: + verbose_proxy_logger.removeHandler(handler) + verbose_proxy_logger.setLevel(original_level) + + assert master_key_secret not in rendered, f"master_key leaked in logs: {rendered!r}" + assert db_url_secret not in rendered, f"database_url leaked in logs: {rendered!r}" + assert "leak_password_9090" not in rendered + assert any("['slack']" in r.getMessage() for r in handler.records), ( + f"expected the alerting callback list to appear in a debug record; got {[r.getMessage() for r in handler.records]!r}" + ) + + # --------------------------------------------------------------------------- # ProxyConfig.initialize_secret_manager # --------------------------------------------------------------------------- From 64d6a1518238c71dabe1fd9b181a21e82ffeee8e Mon Sep 17 00:00:00 2001 From: yucheng-berriai Date: Thu, 2 Jul 2026 11:24:58 -0700 Subject: [PATCH 11/54] fix(proxy): redact secrets on the db-config and litellm_settings log paths too Reuse the existing recursive `_redact_secret_values_in_obj` for the worker config log instead of a hand-rolled top-level pass, so a credential nested under general_settings is masked at any depth and depth overrun fails closed. Route the `_update_config_from_db` param_value log (the store_model_in_db path) and the litellm_settings apply-loop log through the same redactors, so master_key, database_url, and secret-named settings such as api_key stop leaking at DEBUG when the module regex scrubber is bypassed. A plain setting like num_retries still logs its real value. Regression tests disable _ENABLE_SECRET_REDACTION and cover the nested worker config shape, the db-config path, and the litellm_settings loop in both directions. --- litellm/proxy/proxy_server.py | 42 ++--- .../proxy/proxy_server/test_lifecycle.py | 43 ++++- .../proxy/proxy_server/test_proxy_config.py | 150 ++++++++++++++++++ 3 files changed, 204 insertions(+), 31 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index eda7a1b507a..e7ccfc883c3 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -654,29 +654,7 @@ _EXTRA_SECRET_GENERAL_SETTINGS_FIELDS = frozenset( ) -_EXTRA_SECRET_MASK = "REDACTED" - - -def _redact_config_dict_for_logging(data: dict[str, object]) -> dict[str, object]: - """Mask secret-bearing entries in a proxy config-shaped dict. - - Combines the segment-matching masker (catches `master_key`, `api_key`, - `*_token`, etc.) with an explicit whole-value replacement for the - URL/webhook fields that embed credentials but do not contain any - sensitive-pattern segment (e.g. `database_url` splits into - `['database', 'url']`, so segment matching misses it). The whole-value - replacement covers non-string shapes (`alert_to_webhook_url` is a dict, - `pass_through_endpoints` is a list), which the string-only - `mask_sensitive_keys` helper would otherwise pass through unchanged. - """ - segment_masked = SENSITIVE_DATA_MASKER.mask_dict(data) - return { - key: _EXTRA_SECRET_MASK if key in _EXTRA_SECRET_GENERAL_SETTINGS_FIELDS and value is not None else value - for key, value in segment_masked.items() - } - - -def _redact_worker_config_for_logging(worker_config: str | dict[str, object] | None) -> str | dict[str, object] | None: +def _redact_worker_config_for_logging(worker_config: str | dict[str, JsonValue] | None) -> JsonValue: """Mask sensitive fields in the worker config before it enters a log record. `worker_config` reaches `proxy_startup_event` as either the JSON blob @@ -690,10 +668,10 @@ def _redact_worker_config_for_logging(worker_config: str | dict[str, object] | N if worker_config is None: return None if isinstance(worker_config, dict): - return _redact_config_dict_for_logging(worker_config) + return _redact_secret_values_in_obj(worker_config) parsed = safe_json_loads(worker_config, default=None) if isinstance(parsed, dict): - return safe_dumps(_redact_config_dict_for_logging(parsed)) + return safe_dumps(_redact_secret_values_in_obj(parsed)) return worker_config @@ -4333,7 +4311,9 @@ class ProxyConfig: raise Exception( f"team_id missing from default_team_settings at index={idx}\npassed in value={type(team_setting)}" ) - verbose_proxy_logger.debug(f"{blue_color_code} setting litellm.{key}={value}{reset_color_code}") + verbose_proxy_logger.debug( + f"{blue_color_code} setting litellm.{key}={_redact_general_setting_value(key, value, is_full_admin=False)}{reset_color_code}" + ) setattr(litellm, key, value) elif key == "upperbound_key_generate_params": if value is not None and isinstance(value, dict): @@ -4348,7 +4328,9 @@ class ProxyConfig: litellm._turn_on_json() verbose_proxy_logger.debug(f"{blue_color_code} Enabled JSON logging via config{reset_color_code}") else: - verbose_proxy_logger.debug(f"{blue_color_code} setting litellm.{key}={value}{reset_color_code}") + verbose_proxy_logger.debug( + f"{blue_color_code} setting litellm.{key}={_redact_general_setting_value(key, value, is_full_admin=False)}{reset_color_code}" + ) setattr(litellm, key, value) if key == "request_timeout": litellm.request_timeout_explicitly_set = True @@ -5706,7 +5688,11 @@ class ProxyConfig: param_name = getattr(response, "param_name", None) param_value = getattr(response, "param_value", None) - verbose_proxy_logger.debug(f"param_name={param_name}, param_value={param_value}") + verbose_proxy_logger.debug( + "param_name=%s, param_value=%s", + param_name, + _redact_secret_values_in_obj(param_value) if isinstance(param_value, (dict, list)) else param_value, + ) if param_name is not None and param_value is not None: config = self._update_config_fields( diff --git a/tests/test_litellm/proxy/proxy_server/test_lifecycle.py b/tests/test_litellm/proxy/proxy_server/test_lifecycle.py index 842b372ba83..a3f5049ef1d 100644 --- a/tests/test_litellm/proxy/proxy_server/test_lifecycle.py +++ b/tests/test_litellm/proxy/proxy_server/test_lifecycle.py @@ -279,7 +279,7 @@ def test__redact_worker_config_for_logging_passthrough_for_none_and_non_json_str assert _redact_worker_config_for_logging("/tmp/some_config.yaml") == "/tmp/some_config.yaml" -def test__redact_config_dict_for_logging_masks_non_string_url_webhook_values(): +def test__redact_worker_config_for_logging_masks_non_string_url_webhook_values(): """The URL/webhook fields the segment masker cannot catch by key name (``alert_to_webhook_url``, ``pass_through_endpoints``, ``database_extra_connection_params``) can hold non-string shapes: @@ -288,7 +288,7 @@ def test__redact_config_dict_for_logging_masks_non_string_url_webhook_values(): the whole value is replaced regardless of shape so a nested webhook or Bearer token under a non-segment-matched key does not slip through. """ - from litellm.proxy.proxy_server import _redact_config_dict_for_logging + from litellm.proxy.proxy_server import _redact_worker_config_for_logging nested_webhook_secret = "https://hooks.slack.com/services/T0/B0/nested-webhook-secret-xyz" data = { @@ -303,7 +303,7 @@ def test__redact_config_dict_for_logging_masks_non_string_url_webhook_values(): ], "database_extra_connection_params": {"password": "extra-db-password-abc"}, } - redacted = _redact_config_dict_for_logging(data) + redacted = _redact_worker_config_for_logging(data) rendered = repr(redacted) for secret in ( "sk-should-be-masked", @@ -314,6 +314,43 @@ def test__redact_config_dict_for_logging_masks_non_string_url_webhook_values(): assert secret not in rendered, f"leak: {secret} in {rendered!r}" +def test__redact_worker_config_for_logging_masks_nested_secret_fields(): + """LIT-4152 nested regression: the URL/webhook credential fields the segment + masker cannot catch by name (``database_url``, + ``database_extra_connection_params``, ``pass_through_endpoints``, + ``alert_to_webhook_url``) must be redacted at any depth, not just the top + level. A worker_config that nests ``general_settings`` under a parent key + must not leak a nested ``database_url`` or webhook secret; the earlier + top-level-only redaction would have passed these through raw. + """ + from litellm.proxy.proxy_server import _redact_worker_config_for_logging + + nested_db_url = "postgresql://nested_user:nested_pw_4152@nested-host:5432/db" + nested_webhook = "https://hooks.slack.com/services/T0/B0/nested-4152-webhook" + nested_extra_pw = "nested-extra-conn-pw-4152" + nested_bearer = "Bearer nested-passthrough-token-4152" + data = { + "config": { + "general_settings": { + "database_url": nested_db_url, + "database_extra_connection_params": {"password": nested_extra_pw}, + "alert_to_webhook_url": {"budget_alerts": nested_webhook}, + "pass_through_endpoints": [ + {"path": "/up", "headers": {"Authorization": nested_bearer}} + ], + } + } + } + redacted = _redact_worker_config_for_logging(data) + rendered = repr(redacted) + for secret in (nested_db_url, nested_webhook, nested_extra_pw, nested_bearer): + assert secret not in rendered, f"nested leak: {secret} in {rendered!r}" + + inner = redacted["config"]["general_settings"] + assert inner["database_url"] == "REDACTED" + assert inner["pass_through_endpoints"] == "REDACTED" + + # --------------------------------------------------------------------------- # initialize # --------------------------------------------------------------------------- diff --git a/tests/test_litellm/proxy/proxy_server/test_proxy_config.py b/tests/test_litellm/proxy/proxy_server/test_proxy_config.py index 78ec2813c61..08b825f41d4 100644 --- a/tests/test_litellm/proxy/proxy_server/test_proxy_config.py +++ b/tests/test_litellm/proxy/proxy_server/test_proxy_config.py @@ -1612,3 +1612,153 @@ def test_ProxyConfig__update_config_fields_invalid_param_raises(): with pytest.raises(Exception): # Missing required arg. pc._update_config_fields(current_config={}, param_name="general_settings") # type: ignore[call-arg] + + +# --------------------------------------------------------------------------- +# ProxyConfig._update_config_from_db +# --------------------------------------------------------------------------- + + +@pytest.mark.asyncio +async def test_ProxyConfig__update_config_from_db_does_not_log_general_settings_secrets( + monkeypatch, +): + """Regression for LIT-4152 on the store_model_in_db path. + + ``_update_config_from_db`` logged each DB ``param_value`` verbatim at DEBUG; + for ``general_settings`` that value is the whole dict, leaking ``master_key`` + and ``database_url`` the same way the startup config load did. The value now + routes through the recursive redactor. Asserted with the module regex + scrubber (``_ENABLE_SECRET_REDACTION``) disabled so the caller itself must + not build the leaky string. The merge into the returned config must still + carry the raw values, proving only the log record is redacted. + """ + import logging + + import litellm._logging as _logging_module + from litellm._logging import verbose_proxy_logger + + monkeypatch.setattr(_logging_module, "_ENABLE_SECRET_REDACTION", False) + + master_key_secret = "sk-lit4152-db-path-master-key-abcdef1234567890" + db_url_secret = "postgresql://leak_user:leak_password_9090@leak-host.internal:5432/leak_db" + nested_webhook_secret = "https://hooks.slack.com/services/T0/B0/db-path-webhook-secret" + + responses = { + "general_settings": SimpleNamespace( + param_name="general_settings", + param_value={ + "master_key": master_key_secret, + "database_url": db_url_secret, + "alert_to_webhook_url": {"budget_alerts": nested_webhook_secret}, + }, + ), + "router_settings": None, + "litellm_settings": None, + "environment_variables": None, + } + + async def _fake_get_config_param(prisma_client, key): + return responses[key] + + monkeypatch.setattr( + "litellm.proxy.proxy_server.get_config_param", _fake_get_config_param + ) + + class LogRecordHandler(logging.Handler): + def __init__(self) -> None: + super().__init__() + self.records: list[logging.LogRecord] = [] + + def emit(self, record: logging.LogRecord) -> None: + self.records.append(record) + + handler = LogRecordHandler() + handler.setLevel(logging.DEBUG) + original_level = verbose_proxy_logger.level + verbose_proxy_logger.setLevel(logging.DEBUG) + verbose_proxy_logger.addHandler(handler) + try: + merged = await ProxyConfig()._update_config_from_db( + prisma_client=MagicMock(), + config={"general_settings": {}}, + store_model_in_db=True, + ) + rendered = " ".join(record.getMessage() for record in handler.records) + finally: + verbose_proxy_logger.removeHandler(handler) + verbose_proxy_logger.setLevel(original_level) + + for secret in ( + master_key_secret, + db_url_secret, + nested_webhook_secret, + "leak_password_9090", + ): + assert secret not in rendered, f"leak: {secret} in {rendered!r}" + assert merged["general_settings"]["master_key"] == master_key_secret + assert merged["general_settings"]["database_url"] == db_url_secret + + +@pytest.mark.asyncio +async def test_ProxyConfig_load_config_redacts_secret_litellm_setting_keeps_plain( + tmp_path, monkeypatch +): + """Regression for LIT-4152 on the ``litellm_settings`` apply loop. + + ``load_config`` logged ``setting litellm.=`` verbatim at DEBUG, + so a secret-bearing setting such as ``api_key`` leaked in cleartext. The + value now routes through ``_redact_general_setting_value``. Crucially the + redaction must be surgical: a secret-named key is masked, but a plain + operational setting like ``num_retries`` must still log its real value, so + the debug line keeps its signal. Asserted with the module regex scrubber + (``_ENABLE_SECRET_REDACTION``) disabled. + """ + import logging + + import litellm._logging as _logging_module + from litellm._logging import verbose_proxy_logger + + monkeypatch.setattr(_logging_module, "_ENABLE_SECRET_REDACTION", False) + + api_key_secret = "sk-lit4152-litellm-settings-secret-abcdef1234567890" + f = tmp_path / "c.yaml" + f.write_text( + "model_list: []\n" + "general_settings: {}\n" + "litellm_settings:\n" + f" api_key: {api_key_secret}\n" + " num_retries: 7\n" + ) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None) + monkeypatch.setattr("litellm.proxy.proxy_server.store_model_in_db", False) + monkeypatch.delenv("LITELLM_CONFIG_BUCKET_NAME", raising=False) + + class LogRecordHandler(logging.Handler): + def __init__(self) -> None: + super().__init__() + self.records: list[logging.LogRecord] = [] + + def emit(self, record: logging.LogRecord) -> None: + self.records.append(record) + + handler = LogRecordHandler() + handler.setLevel(logging.DEBUG) + original_level = verbose_proxy_logger.level + original_api_key = getattr(litellm, "api_key", None) + original_num_retries = getattr(litellm, "num_retries", None) + verbose_proxy_logger.setLevel(logging.DEBUG) + verbose_proxy_logger.addHandler(handler) + try: + await ProxyConfig().load_config(router=None, config_file_path=str(f)) + rendered = " ".join(record.getMessage() for record in handler.records) + finally: + verbose_proxy_logger.removeHandler(handler) + verbose_proxy_logger.setLevel(original_level) + litellm.api_key = original_api_key + litellm.num_retries = original_num_retries + + assert api_key_secret not in rendered, f"api_key leaked in logs: {rendered!r}" + assert "num_retries=7" in rendered, ( + f"non-secret num_retries value was over-redacted; expected it visible in {rendered!r}" + ) From 89d3f2a7b8213b490e95b43c1130ccad5069b32d Mon Sep 17 00:00:00 2001 From: yucheng-berriai Date: Thu, 2 Jul 2026 19:01:43 +0000 Subject: [PATCH 12/54] fix: redact db environment variable debug logs --- litellm/proxy/proxy_server.py | 10 +++++++++- .../proxy/proxy_server/test_proxy_config.py | 19 +++++++++++++++---- 2 files changed, 24 insertions(+), 5 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index e7ccfc883c3..0b09f661ea5 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5691,7 +5691,7 @@ class ProxyConfig: verbose_proxy_logger.debug( "param_name=%s, param_value=%s", param_name, - _redact_secret_values_in_obj(param_value) if isinstance(param_value, (dict, list)) else param_value, + _redact_config_param_value_for_logging(param_name, param_value), ) if param_name is not None and param_value is not None: @@ -14293,6 +14293,14 @@ def _redact_secret_values_in_obj(value: JsonValue, depth: int = 0) -> JsonValue: return value +def _redact_config_param_value_for_logging(param_name: Optional[str], param_value: JsonValue) -> JsonValue: + if param_name == "environment_variables" and isinstance(param_value, dict): + return {key: "REDACTED" for key in param_value} + if isinstance(param_value, (dict, list)): + return _redact_secret_values_in_obj(param_value) + return param_value + + def _redact_general_setting_value(field_name: str, value: JsonValue, is_full_admin: bool) -> JsonValue: if is_full_admin: return value diff --git a/tests/test_litellm/proxy/proxy_server/test_proxy_config.py b/tests/test_litellm/proxy/proxy_server/test_proxy_config.py index 08b825f41d4..e7c036dd039 100644 --- a/tests/test_litellm/proxy/proxy_server/test_proxy_config.py +++ b/tests/test_litellm/proxy/proxy_server/test_proxy_config.py @@ -1639,9 +1639,16 @@ async def test_ProxyConfig__update_config_from_db_does_not_log_general_settings_ from litellm._logging import verbose_proxy_logger monkeypatch.setattr(_logging_module, "_ENABLE_SECRET_REDACTION", False) + monkeypatch.delenv("DATABASE_URL", raising=False) + + def _fake_decrypt_value_helper(value, key, **_kwargs): + return value + + monkeypatch.setattr("litellm.proxy.proxy_server.decrypt_value_helper", _fake_decrypt_value_helper) master_key_secret = "sk-lit4152-db-path-master-key-abcdef1234567890" db_url_secret = "postgresql://leak_user:leak_password_9090@leak-host.internal:5432/leak_db" + env_db_url_secret = "postgresql://env_leak_user:env_leak_password_9090@env-leak-host.internal:5432/env_leak_db" nested_webhook_secret = "https://hooks.slack.com/services/T0/B0/db-path-webhook-secret" responses = { @@ -1655,15 +1662,16 @@ async def test_ProxyConfig__update_config_from_db_does_not_log_general_settings_ ), "router_settings": None, "litellm_settings": None, - "environment_variables": None, + "environment_variables": SimpleNamespace( + param_name="environment_variables", + param_value={"DATABASE_URL": env_db_url_secret}, + ), } async def _fake_get_config_param(prisma_client, key): return responses[key] - monkeypatch.setattr( - "litellm.proxy.proxy_server.get_config_param", _fake_get_config_param - ) + monkeypatch.setattr("litellm.proxy.proxy_server.get_config_param", _fake_get_config_param) class LogRecordHandler(logging.Handler): def __init__(self) -> None: @@ -1692,12 +1700,15 @@ async def test_ProxyConfig__update_config_from_db_does_not_log_general_settings_ for secret in ( master_key_secret, db_url_secret, + env_db_url_secret, nested_webhook_secret, "leak_password_9090", + "env_leak_password_9090", ): assert secret not in rendered, f"leak: {secret} in {rendered!r}" assert merged["general_settings"]["master_key"] == master_key_secret assert merged["general_settings"]["database_url"] == db_url_secret + assert merged["environment_variables"]["DATABASE_URL"] == env_db_url_secret @pytest.mark.asyncio From bd6ae9effa5b607ff1e2097de0fa4f292f4ca88e Mon Sep 17 00:00:00 2001 From: yucheng-berriai Date: Thu, 2 Jul 2026 13:32:24 -0700 Subject: [PATCH 13/54] fix(proxy): stop the decrypt-failure debug log from leaking the raw value decrypt_value_helper logged `Unable to decrypt value={value}` at DEBUG, which printed the raw secret whenever decryption failed (for example after a salt or master key change). This is the same environment_variables config path the db-config redaction covers, so a DATABASE_URL connection string could still leak here when the module regex scrubber is bypassed. Drop the value; the key already identifies the failing pair. Regression forces a decrypt failure with the redaction filter disabled and asserts the raw value never reaches a log record while the key stays visible. --- .../common_utils/encrypt_decrypt_utils.py | 2 +- .../test_encrypt_decrypt_utils.py | 63 +++++++++++++++---- 2 files changed, 52 insertions(+), 13 deletions(-) diff --git a/litellm/proxy/common_utils/encrypt_decrypt_utils.py b/litellm/proxy/common_utils/encrypt_decrypt_utils.py index 8599b3ace7f..9a0b4f8b982 100644 --- a/litellm/proxy/common_utils/encrypt_decrypt_utils.py +++ b/litellm/proxy/common_utils/encrypt_decrypt_utils.py @@ -150,7 +150,7 @@ def decrypt_value_helper( verbose_proxy_logger.debug(error_message) return value if return_original_value else None - verbose_proxy_logger.debug(f"Unable to decrypt value={value} for key: {key}, returning None") + verbose_proxy_logger.debug(f"Unable to decrypt value for key: {key}, returning None") if return_original_value: return value else: diff --git a/tests/test_litellm/proxy/common_utils/test_encrypt_decrypt_utils.py b/tests/test_litellm/proxy/common_utils/test_encrypt_decrypt_utils.py index bee39e01dd6..08cf1e45812 100644 --- a/tests/test_litellm/proxy/common_utils/test_encrypt_decrypt_utils.py +++ b/tests/test_litellm/proxy/common_utils/test_encrypt_decrypt_utils.py @@ -18,9 +18,7 @@ from litellm.proxy.common_utils.encrypt_decrypt_utils import ( def _use_aes(monkeypatch): """Flip the write-time algorithm to AES-256-GCM for the duration of a test.""" - monkeypatch.setattr( - proxy_server, "general_settings", {"encryption_algorithm": "aes-256-gcm"} - ) + monkeypatch.setattr(proxy_server, "general_settings", {"encryption_algorithm": "aes-256-gcm"}) @pytest.fixture(autouse=True) @@ -96,12 +94,7 @@ def test_aes_decrypt_failure_returns_original_when_requested(monkeypatch): _use_aes(monkeypatch) garbled = _V2_GCM_PREFIX + "###" - assert ( - decrypt_value_helper( - garbled, key="t", exception_type="debug", return_original_value=True - ) - == garbled - ) + assert decrypt_value_helper(garbled, key="t", exception_type="debug", return_original_value=True) == garbled def test_empty_string_round_trips_under_aes(monkeypatch): @@ -139,10 +132,56 @@ def test_callback_prefix_composes_with_v2(monkeypatch): def test_unknown_algorithm_falls_back_to_legacy(monkeypatch): """An unrecognized encryption_algorithm value does not produce v2 writes.""" - monkeypatch.setattr( - proxy_server, "general_settings", {"encryption_algorithm": "rot13"} - ) + monkeypatch.setattr(proxy_server, "general_settings", {"encryption_algorithm": "rot13"}) ct = encrypt_value_helper("secret") assert not ct.startswith(_V2_GCM_PREFIX) assert decrypt_value_helper(ct, key="t") == "secret" + + +def test_decrypt_failure_debug_log_omits_raw_value(monkeypatch): + """Regression for LIT-4152: the decrypt-failure debug breadcrumb must not + embed the raw value. + + A DB ``environment_variables`` secret (e.g. a ``DATABASE_URL`` connection + string) reaches this path when it cannot be decrypted, for example after a + salt or master key change, and previously printed in cleartext when the + module regex scrubber was bypassed. The failing key still names the pair so + the breadcrumb keeps its debugging value. Uses a dedicated handler rather + than caplog because caplog is unreliable under pytest-xdist. + """ + import logging + + import litellm._logging as _logging_module + from litellm._logging import verbose_proxy_logger + + monkeypatch.setattr(_logging_module, "_ENABLE_SECRET_REDACTION", False) + + secret = "postgresql://leak_user:leak_pw_decrypt@leak-host:5432/leak_db" + + class LogRecordHandler(logging.Handler): + def __init__(self) -> None: + super().__init__() + self.records: list[logging.LogRecord] = [] + + def emit(self, record: logging.LogRecord) -> None: + self.records.append(record) + + handler = LogRecordHandler() + handler.setLevel(logging.DEBUG) + original_level = verbose_proxy_logger.level + verbose_proxy_logger.setLevel(logging.DEBUG) + verbose_proxy_logger.addHandler(handler) + try: + result = decrypt_value_helper(secret, key="DATABASE_URL", return_original_value=True) + rendered = " ".join(record.getMessage() for record in handler.records) + finally: + verbose_proxy_logger.removeHandler(handler) + verbose_proxy_logger.setLevel(original_level) + + assert secret not in rendered, f"raw value leaked in decrypt-failure log: {rendered!r}" + assert "leak_pw_decrypt" not in rendered + assert any("DATABASE_URL" in record.getMessage() for record in handler.records), ( + "the failing key should still be named in the breadcrumb" + ) + assert result == secret From 11314f4baea3665d01625d64a77e46a1f851d6c0 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 3 Jul 2026 20:15:33 +0000 Subject: [PATCH 14/54] feat(ui): migrate chat UI from antd to shadcn/ui Replace all Ant Design components (Table, Modal, Popover, Tooltip, Skeleton, Select, Spin, Popconfirm, Switch) with shadcn/ui primitives and Lucide React icons across all chat components: - ChatPage: sidebar, model selector, input bar, comparison mode - ConversationList: search dialog, delete confirmation, scroll area - ChatMessages: message bubbles, tool cards, copy button - MCPAppsPanel: list/detail views, OAuth2 flow, tabs - MCPConnectPicker: server toggle switches - MCPCredentialsTab: credentials table with delete - KeysPanel: API key management with rotation dialog (enterprise) - UsagePanel: spend/request stats with sparkline charts Add design.md as the design specification guiding the migration. Install 15 shadcn/ui components (dialog, popover, tooltip, table, etc.). All existing functionality preserved; no backend changes. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ui/litellm-dashboard/eslint-metrics.json | 2 +- .../src/app/chat/page.test.tsx | 1 + ui/litellm-dashboard/src/app/chat/page.tsx | 3 +- .../src/components/chat/ChatMessages.tsx | 331 ++---- .../src/components/chat/ChatPage.tsx | 985 ++++++------------ .../src/components/chat/ConversationList.tsx | 374 +++---- .../src/components/chat/KeysPanel.tsx | 357 +++++++ .../src/components/chat/MCPAppsPanel.tsx | 397 ++----- .../src/components/chat/MCPConnectPicker.tsx | 96 +- .../src/components/chat/MCPCredentialsTab.tsx | 134 ++- .../src/components/chat/UsagePanel.tsx | 192 ++++ .../src/components/chat/design.md | 360 +++++++ .../src/components/ui/alert-dialog.tsx | 163 +++ .../src/components/ui/badge.tsx | 40 + .../src/components/ui/collapsible.tsx | 17 + .../src/components/ui/dialog.tsx | 138 +++ .../src/components/ui/input.tsx | 21 + .../src/components/ui/label.tsx | 21 + .../src/components/ui/popover.tsx | 54 + .../src/components/ui/scroll-area.tsx | 48 + .../src/components/ui/select.tsx | 162 +++ .../src/components/ui/separator.tsx | 28 + .../src/components/ui/skeleton.tsx | 7 + .../src/components/ui/switch.tsx | 35 + .../src/components/ui/table.tsx | 78 ++ .../src/components/ui/tabs.tsx | 71 ++ .../src/components/ui/tooltip.tsx | 44 + 27 files changed, 2557 insertions(+), 1602 deletions(-) create mode 100644 ui/litellm-dashboard/src/components/chat/KeysPanel.tsx create mode 100644 ui/litellm-dashboard/src/components/chat/UsagePanel.tsx create mode 100644 ui/litellm-dashboard/src/components/chat/design.md create mode 100644 ui/litellm-dashboard/src/components/ui/alert-dialog.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/badge.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/collapsible.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/dialog.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/input.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/label.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/popover.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/scroll-area.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/select.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/separator.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/skeleton.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/switch.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/table.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/tabs.tsx create mode 100644 ui/litellm-dashboard/src/components/ui/tooltip.tsx diff --git a/ui/litellm-dashboard/eslint-metrics.json b/ui/litellm-dashboard/eslint-metrics.json index cf754a1bb75..23774254b7d 100644 --- a/ui/litellm-dashboard/eslint-metrics.json +++ b/ui/litellm-dashboard/eslint-metrics.json @@ -1,5 +1,5 @@ { "@typescript-eslint/no-explicit-any": 1991, - "complexity": 128, + "complexity": 129, "max-depth": 59 } diff --git a/ui/litellm-dashboard/src/app/chat/page.test.tsx b/ui/litellm-dashboard/src/app/chat/page.test.tsx index 658142d47a6..af85b6b8c7f 100644 --- a/ui/litellm-dashboard/src/app/chat/page.test.tsx +++ b/ui/litellm-dashboard/src/app/chat/page.test.tsx @@ -17,6 +17,7 @@ const { mockUseAuthorized, mockUseUISettings, mockUseUIConfig, mockReplace, stat userRole: state.userRole, userId: "user-1", userEmail: "user@example.com", + premiumUser: false, })), mockUseUISettings: vi.fn(() => ({ data: { values: { enable_chat_ui: state.enableChatUI } }, diff --git a/ui/litellm-dashboard/src/app/chat/page.tsx b/ui/litellm-dashboard/src/app/chat/page.tsx index f1053bec78d..21f34ba9f6e 100644 --- a/ui/litellm-dashboard/src/app/chat/page.tsx +++ b/ui/litellm-dashboard/src/app/chat/page.tsx @@ -9,7 +9,7 @@ import ChatPage from "@/components/chat/ChatPage"; // ChatPage uses useSearchParams() which requires a Suspense boundary for static export. const ChatPageContent = () => { - const { accessToken, userRole, userId, userEmail } = useAuthorized(); + const { accessToken, userRole, userId, userEmail, premiumUser } = useAuthorized(); const { data: uiSettings, isLoading: isUISettingsLoading } = useUISettings(); const { data: uiConfig } = useUIConfig(); const router = useRouter(); @@ -33,6 +33,7 @@ const ChatPageContent = () => { userRole={userRole ?? ""} userId={userId ?? ""} userEmail={userEmail ?? ""} + premiumUser={premiumUser ?? false} /> ); }; diff --git a/ui/litellm-dashboard/src/components/chat/ChatMessages.tsx b/ui/litellm-dashboard/src/components/chat/ChatMessages.tsx index f29adcfb6a5..c92b9494e2c 100644 --- a/ui/litellm-dashboard/src/components/chat/ChatMessages.tsx +++ b/ui/litellm-dashboard/src/components/chat/ChatMessages.tsx @@ -1,7 +1,8 @@ "use client"; -import { ToolOutlined, CopyOutlined, CheckOutlined, EditOutlined } from "@ant-design/icons"; -import { Collapse, Tooltip } from "antd"; +import { Wrench, Copy, Check, Pencil } from "lucide-react"; +import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip"; +import { Collapsible, CollapsibleContent, CollapsibleTrigger } from "@/components/ui/collapsible"; import React, { useEffect, useRef, useState } from "react"; import ReactMarkdown from "react-markdown"; import remarkGfm from "remark-gfm"; @@ -11,9 +12,6 @@ import ReasoningContent from "@/components/chat_ui/ReasoningContent"; import MCPEventsDisplay from "@/components/chat_ui/MCPEventsDisplay"; import { ChatMessage } from "./types"; -const { Panel } = Collapse; - -// Keys whose values must be redacted in tool args display const REDACTED_KEY_PATTERNS = /token|key|secret|password|auth/i; function redactSensitiveValues(obj: Record): Record { @@ -43,8 +41,6 @@ function formatTimestamp(ts: number): string { return `${hh}:${mm}`; } -// Shared markdown code renderer matching ReasoningContent style. -// react-markdown v9 removed the `inline` prop; detect fenced blocks via language className. function MarkdownCodeRenderer({ node, className, @@ -63,14 +59,12 @@ function MarkdownCodeRenderer({ {String(children).replace(/\n$/, "")} ) : ( - + {children} ); } -// ------- Sub-components ------- - interface UserBubbleProps { message: ChatMessage; onEdit?: (messageId: string, newContent: string) => void; @@ -90,7 +84,6 @@ function UserBubble({ message, onEdit, isStreaming }: UserBubbleProps) { } }, [editing]); - // Auto-resize textarea useEffect(() => { const ta = textareaRef.current; if (!ta) return; @@ -119,76 +112,34 @@ function UserBubble({ message, onEdit, isStreaming }: UserBubbleProps) { if (editing) { return ( -
-
+
+