mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix: forward HTTP method and all provider headers through sidecar
- Forward original HTTP method (GET/PUT/PATCH/DELETE/HEAD) via X-LiteLLM-Method header instead of hardcoding POST. The Rust sidecar now dispatches to the correct reqwest method. - Forward all provider headers (Azure, Bedrock, Cohere, Vertex, OpenAI org, extra_headers) via X-LiteLLM-Fwd-* prefix instead of a 3-header whitelist. Only hop-by-hop and already-handled headers are skipped. - Return a shallow copy from UserAPIKeyLabelValues.get_label_dict() to prevent callers from mutating the internal cache. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
parent
748ea4ece9
commit
570c29a258
3 changed files with 36 additions and 11 deletions
|
|
@ -117,6 +117,13 @@ async fn handle_request(
|
|||
.map(|v| v == "true")
|
||||
.unwrap_or(false);
|
||||
|
||||
let method = req
|
||||
.headers()
|
||||
.get("x-litellm-method")
|
||||
.and_then(|v| v.to_str().ok())
|
||||
.unwrap_or("POST")
|
||||
.to_uppercase();
|
||||
|
||||
let content_type = req
|
||||
.headers()
|
||||
.get("content-type")
|
||||
|
|
@ -157,11 +164,17 @@ async fn handle_request(
|
|||
let client = sidecar.get_or_create_client(&host);
|
||||
let full_url = format!("{}{}", provider_url.trim_end_matches('/'), request_path);
|
||||
|
||||
let mut req_builder = client
|
||||
.post(&full_url)
|
||||
.header("content-type", &content_type)
|
||||
.timeout(std::time::Duration::from_secs(timeout_secs))
|
||||
.body(body_bytes.to_vec());
|
||||
let mut req_builder = match method.as_str() {
|
||||
"GET" => client.get(&full_url),
|
||||
"PUT" => client.put(&full_url),
|
||||
"PATCH" => client.patch(&full_url),
|
||||
"DELETE" => client.delete(&full_url),
|
||||
"HEAD" => client.head(&full_url),
|
||||
_ => client.post(&full_url),
|
||||
}
|
||||
.header("content-type", &content_type)
|
||||
.timeout(std::time::Duration::from_secs(timeout_secs))
|
||||
.body(body_bytes.to_vec());
|
||||
|
||||
if !api_key.is_empty() {
|
||||
req_builder = req_builder.header("authorization", format!("Bearer {}", api_key));
|
||||
|
|
|
|||
|
|
@ -103,20 +103,32 @@ class LiteLLMSidecarTransport(httpx.AsyncBaseTransport):
|
|||
else:
|
||||
timeout_secs = 300
|
||||
|
||||
# Forward the original HTTP method so the sidecar uses the correct verb
|
||||
method = request.method.upper()
|
||||
|
||||
headers = {
|
||||
"X-LiteLLM-Provider-URL": provider_base,
|
||||
"X-LiteLLM-API-Key": api_key,
|
||||
"X-LiteLLM-Timeout": str(timeout_secs),
|
||||
"X-LiteLLM-Stream": "true" if is_stream else "false",
|
||||
"X-LiteLLM-Path": path,
|
||||
"X-LiteLLM-Method": method,
|
||||
"Content-Type": request.headers.get("content-type", "application/json"),
|
||||
}
|
||||
|
||||
# Copy provider-specific headers the sidecar should forward
|
||||
for key in ("x-api-key", "anthropic-version", "x-goog-api-key"):
|
||||
val = request.headers.get(key)
|
||||
if val:
|
||||
headers[f"X-LiteLLM-Fwd-{key}"] = val
|
||||
# Forward all non-host, non-internal headers to the sidecar
|
||||
_skip_headers = {
|
||||
"host",
|
||||
"content-length",
|
||||
"transfer-encoding",
|
||||
"connection",
|
||||
"content-type", # already set above
|
||||
"authorization", # sent via X-LiteLLM-API-Key
|
||||
}
|
||||
for key, val in request.headers.items():
|
||||
lower_key = key.lower()
|
||||
if lower_key not in _skip_headers:
|
||||
headers[f"X-LiteLLM-Fwd-{lower_key}"] = val
|
||||
|
||||
session = await self._get_session()
|
||||
resp = await session.post(
|
||||
|
|
|
|||
|
|
@ -735,7 +735,7 @@ class UserAPIKeyLabelValues(BaseModel):
|
|||
"""Return cached model_dump() dict to avoid re-serializing on every prometheus_label_factory call."""
|
||||
if self._cached_dump is None:
|
||||
self._cached_dump = self.model_dump()
|
||||
return self._cached_dump
|
||||
return dict(self._cached_dump)
|
||||
|
||||
|
||||
class PrometheusMetricsConfig(BaseModel):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue