From ba61f70f0823cc19f7840db1e341c41e8e31cee6 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Mon, 16 Mar 2026 19:59:13 +0530 Subject: [PATCH] docs(encrypted_content_affinity): add version requirement 1.82.1+, reduce verbosity Made-with: Cursor --- .../index.md | 2 +- docs/my-website/docs/proxy/config_settings.md | 2 +- docs/my-website/docs/proxy/load_balancing.md | 2 +- docs/my-website/docs/response_api.md | 105 ++++-------------- .../exception_mapping_utils.py | 4 +- 5 files changed, 24 insertions(+), 91 deletions(-) diff --git a/docs/my-website/blog/responses_api_encrypted_content_incident/index.md b/docs/my-website/blog/responses_api_encrypted_content_incident/index.md index 19b55898caa..0f151526924 100644 --- a/docs/my-website/blog/responses_api_encrypted_content_incident/index.md +++ b/docs/my-website/blog/responses_api_encrypted_content_incident/index.md @@ -22,7 +22,7 @@ hide_table_of_contents: false **Date:** Feb 24, 2026 **Duration:** Ongoing (until fix deployed) **Severity:** High (for users load balancing Responses API across different API keys) -**Status:** Resolved +**Status:** Resolved (fix available in **LiteLLM 1.82.1+**) ## Summary diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md index a0e404e3a18..9f94fa789ca 100644 --- a/docs/my-website/docs/proxy/config_settings.md +++ b/docs/my-website/docs/proxy/config_settings.md @@ -361,7 +361,7 @@ router_settings: | redis_url | str | URL for Redis server. **Known performance issue with Redis URL.** | | cache_responses | boolean | Flag to enable caching LLM Responses, if cache set under `router_settings`. If true, caches responses. Defaults to False. | | router_general_settings | RouterGeneralSettings | [SDK-Only] Router general settings - contains optimizations like 'async_only_mode'. [Docs](../routing.md#router-general-settings) | -| optional_pre_call_checks | List[str] | List of pre-call checks to add to the router. Supported: `router_budget_limiting`, `prompt_caching`, `responses_api_deployment_check`, `encrypted_content_affinity`, `deployment_affinity`, `session_affinity`, `forward_client_headers_by_model_group` | +| optional_pre_call_checks | List[str] | List of pre-call checks to add to the router. Supported: `router_budget_limiting`, `prompt_caching`, `responses_api_deployment_check`, `encrypted_content_affinity` (requires 1.82.1+), `deployment_affinity`, `session_affinity`, `forward_client_headers_by_model_group` | | deployment_affinity_ttl_seconds | int | TTL (seconds) for user-key → deployment affinity mapping when `deployment_affinity` is enabled (configured at Router init / proxy startup). Defaults to `3600` (1 hour). | | ignore_invalid_deployments | boolean | If true, ignores invalid deployments. Default for proxy is True - to prevent invalid models from blocking other models from being loaded. | | search_tools | List[SearchToolTypedDict] | List of search tool configurations for Search API integration. Each tool specifies a search_tool_name and litellm_params with search_provider, api_key, api_base, etc. [Further Docs](../search.md) | diff --git a/docs/my-website/docs/proxy/load_balancing.md b/docs/my-website/docs/proxy/load_balancing.md index 5bf39d179f6..96b4b2f18a5 100644 --- a/docs/my-website/docs/proxy/load_balancing.md +++ b/docs/my-website/docs/proxy/load_balancing.md @@ -352,7 +352,7 @@ If `order=1` deployment is unavailable (e.g., rate-limited), the router falls ba When load balancing OpenAI's Responses API across deployments with **different API keys** (e.g., different Azure regions or organizations), encrypted content items (like `rs_...` reasoning items) can only be decrypted by the originating API key. -**Solution:** Use the `encrypted_content_affinity` pre-call check to automatically route follow-up requests containing encrypted items to the correct deployment: +**Solution:** Use the `encrypted_content_affinity` pre-call check to automatically route follow-up requests containing encrypted items to the correct deployment. **Requires LiteLLM 1.82.1+**. ```yaml model_list: diff --git a/docs/my-website/docs/response_api.md b/docs/my-website/docs/response_api.md index fb55ae9f9d0..459a17f4c0b 100644 --- a/docs/my-website/docs/response_api.md +++ b/docs/my-website/docs/response_api.md @@ -1160,12 +1160,12 @@ follow_up = await router.aresponses( To enable session continuity for Responses API in your LiteLLM proxy, set `optional_pre_call_checks` in your proxy config.yaml. - `responses_api_deployment_check`: high priority routing when `previous_response_id` is provided -- `encrypted_content_affinity`: **[Recommended]** content-aware routing for encrypted items (e.g., `rs_...` reasoning items) +- `encrypted_content_affinity`: **[Recommended]** Routes encrypted items to originating deployment. Requires 1.82.1+. - `session_affinity`: sticky sessions based on session id (takes priority over `deployment_affinity`) - `deployment_affinity`: sticky sessions based on user key (applies even without `previous_response_id`) :::tip Recommended: Use `encrypted_content_affinity` -For Responses API with load balancing across deployments with **different API keys**, use `encrypted_content_affinity` instead of `deployment_affinity`. It only pins requests that contain encrypted content, avoiding quota reduction while preventing `invalid_encrypted_content` errors. +For multi-region Responses API with different API keys, use `encrypted_content_affinity` instead of `deployment_affinity` — pins only requests with encrypted content, avoids quota reduction. Requires 1.82.1+. ::: Notes: @@ -1230,46 +1230,13 @@ follow_up = client.responses.create( ## Encrypted Content Affinity (Multi-Region Load Balancing) -When load balancing Responses API across deployments with **different API keys** (e.g., different Azure regions or OpenAI organizations), encrypted content items (like `rs_...` reasoning items) can only be decrypted by the API key that created them. +:::info Version requirement +Requires **LiteLLM 1.82.1+**. Upgrade: `pip install litellm>=1.82.1` +::: -### The Problem +When load balancing Responses API across deployments with **different API keys**, encrypted items (e.g. `rs_...` reasoning items) can only be decrypted by the API key that created them. If a follow-up request gets routed to a different deployment, you'll see `invalid_encrypted_content`. -```json -{ - "error": { - "message": "The encrypted content for item rs_0d09d6e56879e76500699d6feee41c8197bd268aae76141f87 could not be verified. Reason: Encrypted content organization_id did not match the target organization.", - "type": "invalid_request_error", - "code": "invalid_encrypted_content" - } -} -``` - -This error occurs when: -1. Initial request goes to Deployment A (API Key 1) → produces encrypted item `rs_xyz` -2. Follow-up request with `rs_xyz` in input gets load balanced to Deployment B (API Key 2) -3. Deployment B cannot decrypt content created by Deployment A → **request fails** - -### The Solution: `encrypted_content_affinity` - -The `encrypted_content_affinity` pre-call check routes follow-up requests containing encrypted items to the originating deployment **only when necessary** - -**Key Benefits:** -- ✅ **No quota reduction**: Unlike `deployment_affinity`, only pins requests that contain encrypted items -- ✅ **Bypasses rate limits**: When encrypted content requires a specific deployment, RPM/TPM limits are bypassed (the request would fail on any other deployment anyway) -- ✅ **No `previous_response_id` required**: Works by encoding `model_id` directly into item IDs -- ✅ **No cache required**: `model_id` is decoded on-the-fly — no Redis dependency, no TTL to manage -- ✅ **Globally safe**: Can be enabled for all models; non-Responses-API calls (chat, embeddings) are unaffected - -### How It Works - -1. **Encoding Phase** (on response): - - For each output item that contains `encrypted_content`, LiteLLM rewrites the item ID to embed the originating `model_id`: `rs_xyz` → `encitem_{base64("litellm:model_id:{model_id};item_id:rs_xyz")}` - - The original item ID is restored before forwarding the request to the upstream provider - -2. **Routing Phase** (before request): - - Scans request `input` for `encitem_` prefixed IDs - - If found → decodes `model_id`, pins to originating deployment, bypasses rate limits - - If no encoded items → normal load balancing +The `encrypted_content_affinity` pre-call check routes follow-up requests containing encrypted items to the originating deployment **only when necessary** — no quota reduction, no Redis, no `previous_response_id` required. ### Configuration @@ -1281,87 +1248,53 @@ from litellm import Router router = Router( model_list=[ - { - "model_name": "gpt-5.1-codex", - "litellm_params": { - "model": "openai/gpt-5.1-codex", - "api_key": "org-1-api-key", # Different API key - }, - "model_info": {"id": "deployment-us-east"}, - }, - { - "model_name": "gpt-5.1-codex", - "litellm_params": { - "model": "openai/gpt-5.1-codex", - "api_key": "org-2-api-key", # Different API key - }, - "model_info": {"id": "deployment-eu-west"}, - }, + {"model_name": "gpt-5.1-codex", "litellm_params": {"model": "openai/gpt-5.1-codex", "api_key": "org-1-key"}, "model_info": {"id": "deployment-us-east"}}, + {"model_name": "gpt-5.1-codex", "litellm_params": {"model": "openai/gpt-5.1-codex", "api_key": "org-2-key"}, "model_info": {"id": "deployment-eu-west"}}, ], optional_pre_call_checks=["encrypted_content_affinity"], ) -# Initial request - routes to any deployment -response1 = await router.aresponses( - model="gpt-5.1-codex", - input="Explain quantum computing", -) - -# Follow-up with encrypted items - automatically routes to same deployment -response2 = await router.aresponses( - model="gpt-5.1-codex", - input=response1.output, # Contains encrypted items from response1 -) +response1 = await router.aresponses(model="gpt-5.1-codex", input="Explain quantum computing") +response2 = await router.aresponses(model="gpt-5.1-codex", input=response1.output) # Auto-routes to same deployment ``` -```yaml showLineNumbers title="config.yaml" +```yaml model_list: - model_name: gpt-5.1-codex litellm_params: model: azure/gpt-5.1-codex api_base: https://eastus.openai.azure.com/ api_key: os.environ/AZURE_API_KEY_EASTUS - rpm: 600 - tpm: 100000 model_info: id: "gpt-5.1-codex-eastus" - - model_name: gpt-5.1-codex litellm_params: model: azure/gpt-5.1-codex api_base: https://westeurope.openai.azure.com/ api_key: os.environ/AZURE_API_KEY_WESTEUROPE - rpm: 600 - tpm: 100000 model_info: id: "gpt-5.1-codex-westeurope" router_settings: - routing_strategy: usage-based-routing-v2 enable_pre_call_checks: true optional_pre_call_checks: - encrypted_content_affinity ``` -**Start proxy:** -```bash -litellm --config config.yaml -``` - -### When to Use Each Affinity Type +### Affinity Comparison -| Affinity Type | Use Case | Scope | Quota Impact | -|---------------|----------|-------|--------------| -| **`encrypted_content_affinity`** | **[Recommended]** Multi-region Responses API with different API keys | Only requests with tracked encrypted items | ✅ None (surgical pinning) | -| `responses_api_deployment_check` | When `previous_response_id` is available | Requests with `previous_response_id` | ✅ None | -| `session_affinity` | Session-based applications | All requests with same `session_id` | ⚠️ Reduces quota by # of sessions | -| `deployment_affinity` | Simple sticky sessions | All requests from same API key | ❌ Reduces quota by # of users | +| Affinity | Use Case | Quota Impact | +|----------|----------|--------------| +| **`encrypted_content_affinity`** | Multi-region Responses API, different API keys | ✅ None | +| `responses_api_deployment_check` | When `previous_response_id` available | ✅ None | +| `session_affinity` | Session-based apps | ⚠️ Reduces by # sessions | +| `deployment_affinity` | Sticky sessions by API key | ❌ Reduces by # users | ## Calling non-Responses API endpoints (`/responses` to `/chat/completions` Bridge) diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index bc54786420a..7ae9bb25f11 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -455,7 +455,7 @@ def exception_type( # type: ignore # noqa: PLR0915 f"{exception_provider} - {message}\n\n" " This error occurs when load balancing Responses API across deployments with different API keys.\n" " Encrypted content is tied to the organization that created it and cannot be decrypted by other organizations.\n\n" - " Solution: Enable 'encrypted_content_affinity' to route follow-up requests to the correct deployment:\n\n" + " Solution: Enable 'encrypted_content_affinity' to route follow-up requests to the correct deployment (requires LiteLLM 1.82.1+):\n\n" " router_settings:\n" " enable_pre_call_checks: true\n" " optional_pre_call_checks:\n" @@ -2170,7 +2170,7 @@ def exception_type( # type: ignore # noqa: PLR0915 f"AzureException - {message}\n\n" "This error occurs when load balancing Responses API across deployments with different API keys.\n" " Encrypted content is tied to the organization that created it and cannot be decrypted by other organizations.\n\n" - " Solution: Enable 'encrypted_content_affinity' to route follow-up requests to the correct deployment:\n\n" + " Solution: Enable 'encrypted_content_affinity' to route follow-up requests to the correct deployment (requires LiteLLM 1.82.1+):\n\n" " router_settings:\n" " enable_pre_call_checks: true\n" " optional_pre_call_checks:\n"