From 570c29a2583fc37feb6d106d44147e5c05a9aaa7 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Sat, 14 Mar 2026 17:20:50 -0700 Subject: [PATCH] fix: forward HTTP method and all provider headers through sidecar - Forward original HTTP method (GET/PUT/PATCH/DELETE/HEAD) via X-LiteLLM-Method header instead of hardcoding POST. The Rust sidecar now dispatches to the correct reqwest method. - Forward all provider headers (Azure, Bedrock, Cohere, Vertex, OpenAI org, extra_headers) via X-LiteLLM-Fwd-* prefix instead of a 3-header whitelist. Only hop-by-hop and already-handled headers are skipped. - Return a shallow copy from UserAPIKeyLabelValues.get_label_dict() to prevent callers from mutating the internal cache. Co-Authored-By: Claude Opus 4.6 --- litellm-sidecar/src/main.rs | 23 +++++++++++++++---- .../llms/custom_httpx/sidecar_transport.py | 22 ++++++++++++++---- litellm/types/integrations/prometheus.py | 2 +- 3 files changed, 36 insertions(+), 11 deletions(-) diff --git a/litellm-sidecar/src/main.rs b/litellm-sidecar/src/main.rs index bd1287c6091..560b83a1a79 100644 --- a/litellm-sidecar/src/main.rs +++ b/litellm-sidecar/src/main.rs @@ -117,6 +117,13 @@ async fn handle_request( .map(|v| v == "true") .unwrap_or(false); + let method = req + .headers() + .get("x-litellm-method") + .and_then(|v| v.to_str().ok()) + .unwrap_or("POST") + .to_uppercase(); + let content_type = req .headers() .get("content-type") @@ -157,11 +164,17 @@ async fn handle_request( let client = sidecar.get_or_create_client(&host); let full_url = format!("{}{}", provider_url.trim_end_matches('/'), request_path); - let mut req_builder = client - .post(&full_url) - .header("content-type", &content_type) - .timeout(std::time::Duration::from_secs(timeout_secs)) - .body(body_bytes.to_vec()); + let mut req_builder = match method.as_str() { + "GET" => client.get(&full_url), + "PUT" => client.put(&full_url), + "PATCH" => client.patch(&full_url), + "DELETE" => client.delete(&full_url), + "HEAD" => client.head(&full_url), + _ => client.post(&full_url), + } + .header("content-type", &content_type) + .timeout(std::time::Duration::from_secs(timeout_secs)) + .body(body_bytes.to_vec()); if !api_key.is_empty() { req_builder = req_builder.header("authorization", format!("Bearer {}", api_key)); diff --git a/litellm/llms/custom_httpx/sidecar_transport.py b/litellm/llms/custom_httpx/sidecar_transport.py index bc0250a39db..71006b4d415 100644 --- a/litellm/llms/custom_httpx/sidecar_transport.py +++ b/litellm/llms/custom_httpx/sidecar_transport.py @@ -103,20 +103,32 @@ class LiteLLMSidecarTransport(httpx.AsyncBaseTransport): else: timeout_secs = 300 + # Forward the original HTTP method so the sidecar uses the correct verb + method = request.method.upper() + headers = { "X-LiteLLM-Provider-URL": provider_base, "X-LiteLLM-API-Key": api_key, "X-LiteLLM-Timeout": str(timeout_secs), "X-LiteLLM-Stream": "true" if is_stream else "false", "X-LiteLLM-Path": path, + "X-LiteLLM-Method": method, "Content-Type": request.headers.get("content-type", "application/json"), } - # Copy provider-specific headers the sidecar should forward - for key in ("x-api-key", "anthropic-version", "x-goog-api-key"): - val = request.headers.get(key) - if val: - headers[f"X-LiteLLM-Fwd-{key}"] = val + # Forward all non-host, non-internal headers to the sidecar + _skip_headers = { + "host", + "content-length", + "transfer-encoding", + "connection", + "content-type", # already set above + "authorization", # sent via X-LiteLLM-API-Key + } + for key, val in request.headers.items(): + lower_key = key.lower() + if lower_key not in _skip_headers: + headers[f"X-LiteLLM-Fwd-{lower_key}"] = val session = await self._get_session() resp = await session.post( diff --git a/litellm/types/integrations/prometheus.py b/litellm/types/integrations/prometheus.py index ead8cde00ea..f3ef82747ce 100644 --- a/litellm/types/integrations/prometheus.py +++ b/litellm/types/integrations/prometheus.py @@ -735,7 +735,7 @@ class UserAPIKeyLabelValues(BaseModel): """Return cached model_dump() dict to avoid re-serializing on every prometheus_label_factory call.""" if self._cached_dump is None: self._cached_dump = self.model_dump() - return self._cached_dump + return dict(self._cached_dump) class PrometheusMetricsConfig(BaseModel):