From b87c2f66a6287451911538db19ee9dda0ca70c31 Mon Sep 17 00:00:00 2001 From: user <70670632+stuxf@users.noreply.github.com> Date: Sun, 3 May 2026 10:08:28 +0000 Subject: [PATCH] fix(vector_store): resolve embedding config at request time, never persist creds The vector store create/update path previously called ``_resolve_embedding_config`` against the admin-configured router/DB model and persisted the resolved ``litellm_embedding_config`` dict (``api_key`` / ``api_base`` / ``api_version``) into the ``litellm_managedvectorstorestable.litellm_params`` column. Because the resolver expanded ``os.environ/...`` references via ``get_secret``, the DB row carried cleartext provider credentials, and the ``/vector_store/{new,info,update,list}`` responses returned them to any authenticated caller who could supply a known admin model name. Move the auto-resolve out of ``create_vector_store_in_db`` and out of the update path. Persist only the user-supplied ``litellm_embedding_model`` reference. Resolve at request-handling time inside ``_update_request_data_with_litellm_managed_vector_store_registry`` so the resolved config lives in the per-request ``data`` dict and is garbage-collected after the response. Legacy rows that were created by an earlier proxy version and already carry a resolved ``litellm_embedding_config`` skip the re-resolution and pass through unchanged so embedding calls keep working. The ``new_vector_store`` response now also runs the existing ``_redact_sensitive_litellm_params`` masker (already used by ``info``, ``update``, and ``list``), defending against caller-supplied cleartext on the create path and against legacy rows whose persisted credentials are still in the database. Existing tests that asserted the old write-time-resolve behaviour are updated to assert the new persistence shape (no embedding config stored, just the model reference). Two new tests cover the use-time path: one asserting fresh resolution happens when a row carries only the model reference, the other asserting legacy rows with persisted config skip re-resolution and continue to work. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../proxy/vector_store_endpoints/endpoints.py | 22 +++ .../management_endpoints.py | 57 ++++--- .../test_vector_store_endpoints.py | 146 +++++++++++++++--- 3 files changed, 172 insertions(+), 53 deletions(-) diff --git a/litellm/proxy/vector_store_endpoints/endpoints.py b/litellm/proxy/vector_store_endpoints/endpoints.py index 86e316e7f40..0f9753a303a 100644 --- a/litellm/proxy/vector_store_endpoints/endpoints.py +++ b/litellm/proxy/vector_store_endpoints/endpoints.py @@ -56,6 +56,28 @@ async def _update_request_data_with_litellm_managed_vector_store_registry( if "litellm_params" in vector_store_to_run: litellm_params = vector_store_to_run.get("litellm_params", {}) or {} + # Resolve ``litellm_embedding_config`` here, at request-handling + # time, instead of at row-creation time. The resolved + # ``api_key`` / ``api_base`` / ``api_version`` lives only in + # this per-request ``data`` dict and is never persisted. + # Legacy rows that already carry a resolved (cleartext) + # ``litellm_embedding_config`` skip the lookup and pass through + # unchanged so the embed call keeps working. + embedding_model = litellm_params.get("litellm_embedding_model") + if embedding_model and not litellm_params.get("litellm_embedding_config"): + from litellm.proxy.proxy_server import prisma_client + from litellm.proxy.vector_store_endpoints.management_endpoints import ( + _resolve_embedding_config, + ) + + resolved_config = await _resolve_embedding_config( + embedding_model=embedding_model, prisma_client=prisma_client + ) + if resolved_config: + litellm_params = { + **litellm_params, + "litellm_embedding_config": resolved_config, + } data.update(litellm_params) return data diff --git a/litellm/proxy/vector_store_endpoints/management_endpoints.py b/litellm/proxy/vector_store_endpoints/management_endpoints.py index 99a2085bfcd..bff43cacaaf 100644 --- a/litellm/proxy/vector_store_endpoints/management_endpoints.py +++ b/litellm/proxy/vector_store_endpoints/management_endpoints.py @@ -432,20 +432,17 @@ async def create_vector_store_in_db( if user_id is not None: data_to_create["user_id"] = user_id - # Handle litellm_params - always provide at least an empty dict + # Handle litellm_params - always provide at least an empty dict. + # The earlier behaviour resolved ``litellm_embedding_config`` from the + # admin-configured router/DB model and persisted the cleartext result + # (``api_key``, ``api_base``, ``api_version``) into this row. That + # exposed every env-stored embedding-model credential on the + # ``/vector_store/{new,info,update,list}`` responses. Keep the user's + # raw ``litellm_embedding_model`` reference; resolution now happens in + # ``_update_request_data_with_litellm_managed_vector_store_registry`` + # at request-handling time so the cleartext config exists only in + # per-request memory and never reaches the database. if litellm_params: - # Auto-resolve embedding config if embedding model is provided but config is not - embedding_model = litellm_params.get("litellm_embedding_model") - if embedding_model and not litellm_params.get("litellm_embedding_config"): - resolved_config = await _resolve_embedding_config( - embedding_model=embedding_model, prisma_client=prisma_client - ) - if resolved_config: - litellm_params["litellm_embedding_config"] = resolved_config - verbose_proxy_logger.info( - f"Auto-resolved embedding config for model {embedding_model}" - ) - litellm_params_dict = GenericLiteLLMParams(**litellm_params).model_dump( exclude_none=True ) @@ -531,10 +528,19 @@ async def new_vector_store( user_id=user_api_key_dict.user_id, ) + # Apply the same litellm_params redaction the list / info / update + # endpoints already use, so a caller-supplied credential or a + # cleartext value persisted by an earlier proxy version doesn't + # come back in the response. + response_vs = LiteLLM_ManagedVectorStore(**new_vector_store) + response_vs["litellm_params"] = _redact_sensitive_litellm_params( + new_vector_store.get("litellm_params") + ) + return { "status": "success", "message": f"Vector store {vector_store.get('vector_store_id')} created successfully", - "vector_store": new_vector_store, + "vector_store": response_vs, } except Exception as e: verbose_proxy_logger.exception(f"Error creating vector store: {str(e)}") @@ -865,24 +871,15 @@ async def update_vector_store( update_data["vector_store_metadata"] ) - # Handle litellm_params if provided + # Handle litellm_params if provided. As with the create path, the + # embedding-config auto-resolve previously persisted cleartext + # credentials into the row; resolution now happens at request- + # handling time in + # ``_update_request_data_with_litellm_managed_vector_store_registry`` + # so this row only ever stores the user-supplied + # ``litellm_embedding_model`` reference. if "litellm_params" in update_data: _input_litellm_params: dict = update_data.get("litellm_params", {}) or {} - - # Auto-resolve embedding config if embedding model is provided but config is not - embedding_model = _input_litellm_params.get("litellm_embedding_model") - if embedding_model and not _input_litellm_params.get( - "litellm_embedding_config" - ): - resolved_config = await _resolve_embedding_config( - embedding_model=embedding_model, prisma_client=prisma_client - ) - if resolved_config: - _input_litellm_params["litellm_embedding_config"] = resolved_config - verbose_proxy_logger.info( - f"Auto-resolved embedding config for model {embedding_model}" - ) - litellm_params_dict = GenericLiteLLMParams( **_input_litellm_params ).model_dump(exclude_none=True) diff --git a/tests/test_litellm/proxy/vector_store_endpoints/test_vector_store_endpoints.py b/tests/test_litellm/proxy/vector_store_endpoints/test_vector_store_endpoints.py index e67a04c749a..7f5557f41ed 100644 --- a/tests/test_litellm/proxy/vector_store_endpoints/test_vector_store_endpoints.py +++ b/tests/test_litellm/proxy/vector_store_endpoints/test_vector_store_endpoints.py @@ -170,6 +170,95 @@ async def test_update_request_data_with_litellm_managed_vector_store_registry(): assert result == original_data +@pytest.mark.asyncio +async def test_update_request_data_resolves_embedding_config_at_use_time(): + """When the persisted vector store row carries only a + ``litellm_embedding_model`` reference (the new behaviour after + moving the auto-resolve out of write time), the request-handling + layer must resolve the embedding config so the downstream embed + call still has ``api_key`` / ``api_base`` / ``api_version``. The + resolved config lives in this per-request data dict only — never + persisted.""" + mock_vector_store: LiteLLM_ManagedVectorStore = { + "vector_store_id": "test_store", + "custom_llm_provider": "azure_ai", + "litellm_params": { + "litellm_embedding_model": "azure/text-embedding-3-large", + # Note: no litellm_embedding_config persisted + }, + } + + mock_registry = MagicMock() + mock_registry.get_litellm_managed_vector_store_from_registry.return_value = ( + mock_vector_store + ) + + resolved = { + "api_key": "use-time-resolved-key", + "api_base": "https://my-azure.example", + "api_version": "2024-09-01", + } + + with ( + patch.object(litellm, "vector_store_registry", mock_registry), + patch( + "litellm.proxy.vector_store_endpoints.management_endpoints._resolve_embedding_config", + new=AsyncMock(return_value=resolved), + ), + ): + result = await _update_request_data_with_litellm_managed_vector_store_registry( + data={}, vector_store_id="test_store" + ) + + assert result["litellm_embedding_model"] == "azure/text-embedding-3-large" + assert result["litellm_embedding_config"] == resolved + + +@pytest.mark.asyncio +async def test_update_request_data_passes_through_legacy_embedding_config(): + """A vector store row created by an older proxy version may already + carry a fully-resolved ``litellm_embedding_config`` in its persisted + ``litellm_params`` (the very leak this PR closes). Those legacy rows + must still work — the use-time resolver skips re-resolution when + the config is already present so the embed call keeps succeeding.""" + legacy_config = { + "api_key": "legacy-cleartext-key", + "api_base": "https://legacy-azure.example", + "api_version": "2024-01-01", + } + mock_vector_store: LiteLLM_ManagedVectorStore = { + "vector_store_id": "legacy_store", + "custom_llm_provider": "azure_ai", + "litellm_params": { + "litellm_embedding_model": "azure/text-embedding-3-large", + "litellm_embedding_config": legacy_config, + }, + } + + mock_registry = MagicMock() + mock_registry.get_litellm_managed_vector_store_from_registry.return_value = ( + mock_vector_store + ) + + resolve_mock = AsyncMock( + return_value={"api_key": "should-not-be-used", "api_base": "wrong"} + ) + + with ( + patch.object(litellm, "vector_store_registry", mock_registry), + patch( + "litellm.proxy.vector_store_endpoints.management_endpoints._resolve_embedding_config", + new=resolve_mock, + ), + ): + result = await _update_request_data_with_litellm_managed_vector_store_registry( + data={}, vector_store_id="legacy_store" + ) + + assert result["litellm_embedding_config"] == legacy_config + resolve_mock.assert_not_awaited() + + class TestCheckVectorStorePermission: """Test suite for check_vector_store_permission function.""" @@ -1417,20 +1506,31 @@ async def test_new_vector_store_auto_resolves_embedding_config(): ) assert result["status"] == "success" - # Verify that embedding config was resolved and included in the create call + # Auto-resolve no longer happens at create time — the persisted row + # carries only the model reference, never the resolved cleartext + # credential. Resolution now happens at request-handling time inside + # ``_update_request_data_with_litellm_managed_vector_store_registry``, + # where the resolved config lives in per-request memory and is never + # written to the database. litellm_params_json = captured_create_data.get("litellm_params") assert litellm_params_json is not None litellm_params_dict = json.loads(litellm_params_json) - assert "litellm_embedding_config" in litellm_params_dict - assert ( - litellm_params_dict["litellm_embedding_config"]["api_key"] == "resolved-api-key" - ) - assert ( - litellm_params_dict["litellm_embedding_config"]["api_base"] - == "https://api.openai.com" - ) - assert ( - litellm_params_dict["litellm_embedding_config"]["api_version"] == "2024-01-01" + assert "litellm_embedding_config" not in litellm_params_dict + assert litellm_params_dict["litellm_embedding_model"] == "text-embedding-ada-002" + + # The response must also not echo a cleartext credential — even on + # the create response, where redaction guards against caller-supplied + # cleartext or pre-existing rows that were created by an earlier + # proxy version. + response_vs = result["vector_store"] + response_params = response_vs.get("litellm_params") + # The redact helper preserves the persisted shape (string or dict); + # serialise to text either way and assert the cleartext credential + # never appears. + assert "resolved-api-key" not in ( + response_params + if isinstance(response_params, str) + else json.dumps(response_params or {}) ) @@ -1687,21 +1787,21 @@ async def test_new_vector_store_auto_resolves_from_router(): ) assert result["status"] == "success" - # Verify that embedding config was resolved from router and included in the create call + # Resolution against the router happens at request-handling time now, + # not at row creation. The persisted ``litellm_params`` carries only + # the model reference, never the cleartext credential. litellm_params_json = captured_create_data.get("litellm_params") assert litellm_params_json is not None litellm_params_dict = json.loads(litellm_params_json) - assert "litellm_embedding_config" in litellm_params_dict - assert ( - litellm_params_dict["litellm_embedding_config"]["api_key"] - == "router-resolved-api-key" - ) - assert ( - litellm_params_dict["litellm_embedding_config"]["api_base"] - == "https://router-resolved-base.com" - ) - assert ( - litellm_params_dict["litellm_embedding_config"]["api_version"] == "2024-03-01" + assert "litellm_embedding_config" not in litellm_params_dict + assert litellm_params_dict["litellm_embedding_model"] == "config-embedding-model" + + response_vs = result["vector_store"] + response_params = response_vs.get("litellm_params") + assert "router-resolved-api-key" not in ( + response_params + if isinstance(response_params, str) + else json.dumps(response_params or {}) )