From 849bcc91c05493135790d589613dcab5d296085e Mon Sep 17 00:00:00 2001 From: rainbowgits <164521089+rainbowgits@users.noreply.github.com> Date: Fri, 21 Aug 2026 06:03:21 +0300 Subject: [PATCH] fix(proxy): forward anthropic 429 rate-limit headers on /v1/messages --- .../proxy/anthropic_endpoints/endpoints.py | 44 ++++- .../anthropic_endpoints/test_endpoints.py | 166 ++++++++++++++++++ 2 files changed, 208 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/anthropic_endpoints/endpoints.py b/litellm/proxy/anthropic_endpoints/endpoints.py index f742965ade2..3ada7744bba 100644 --- a/litellm/proxy/anthropic_endpoints/endpoints.py +++ b/litellm/proxy/anthropic_endpoints/endpoints.py @@ -2,6 +2,7 @@ Unified /v1/messages endpoint - (Anthropic Spec) """ +from collections.abc import Mapping from typing import Final from fastapi import APIRouter, Depends, HTTPException, Request, Response @@ -29,6 +30,43 @@ from litellm.types.utils import TokenCountResponse router: Final = APIRouter() +# Upstream response headers forwarded unprefixed on the /v1/messages error +# path. Restricted to Anthropic-native rate-limit headers so a hostile upstream +# cannot spoof proxy-owned ``x-litellm-*`` metadata or set browser/framing +# headers on the proxy's own response. +_FORWARDABLE_UPSTREAM_HEADERS: Final = frozenset({"retry-after", "request-id"}) +_FORWARDABLE_UPSTREAM_PREFIX: Final = "anthropic-" + + +def _extract_upstream_anthropic_headers(e: Exception) -> Mapping[str, str]: + """Return the upstream provider's Anthropic-native rate-limit headers from a + mapped LiteLLM exception, so they survive the proxy's error wrapping. + + The ``/v1/messages`` endpoint speaks the Anthropic protocol, so these are + forwarded unprefixed — unlike the OpenAI-format path, which namespaces + upstream headers under ``llm_provider-``. Keys are lowercased and limited to + an allowlist (``retry-after``, ``request-id``, ``anthropic-*``) so a hostile + upstream cannot forge case-variant ``x-litellm-*`` headers or inject + browser/framing headers. Source order: ``litellm_response_headers`` (set by + the exception mapper), then ``e.headers``, then ``e.response.headers``. + """ + raw = getattr(e, "litellm_response_headers", None) or getattr(e, "headers", None) + if not raw: + _response = getattr(e, "response", None) + raw = getattr(_response, "headers", None) if _response is not None else None + if not raw: + return {} + try: + items = raw.items() + except AttributeError: + return {} + return { + key: str(v) + for k, v in items + if (key := str(k).lower()) in _FORWARDABLE_UPSTREAM_HEADERS or key.startswith(_FORWARDABLE_UPSTREAM_PREFIX) + } + + def _strip_total_tokens_from_anthropic_response(response: Any) -> None: """Remove the OpenAI-flavored `usage.total_tokens` field that LiteLLM injects into Anthropic /v1/messages responses. @@ -201,7 +239,9 @@ async def anthropic_response( model_info: Final = litellm_metadata.get("model_info", {}) or {} model_id: Final = model_info.get("id", "") or "" - # Get headers + # Preserve upstream rate-limit headers so clients can fail fast on 429 (#37754). + upstream_headers: Final = _extract_upstream_anthropic_headers(e) + headers: Final = ProxyBaseLLMRequestProcessing.get_custom_headers( user_api_key_dict=user_api_key_dict, call_id=data.get("litellm_call_id", ""), @@ -220,7 +260,7 @@ async def anthropic_response( type=getattr(e, "type", "None"), param=getattr(e, "param", "None"), code=getattr(e, "status_code", 500), - headers=headers, + headers={**upstream_headers, **headers}, # proxy-owned headers win ) diff --git a/tests/test_litellm/proxy/anthropic_endpoints/test_endpoints.py b/tests/test_litellm/proxy/anthropic_endpoints/test_endpoints.py index 9a90daeccb7..230421cf6f8 100644 --- a/tests/test_litellm/proxy/anthropic_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/anthropic_endpoints/test_endpoints.py @@ -307,3 +307,169 @@ class TestStripTotalTokensFeatureFlag(unittest.TestCase): import litellm assert litellm.strip_anthropic_total_tokens is False + + +class TestUpstreamRateLimitHeaderPassthrough: + """Issue #37754: on a 429 from an Anthropic-compatible upstream (e.g. GLM), + the `/v1/messages` endpoint must forward the upstream's Anthropic-native + rate-limit headers (`retry-after`, `anthropic-ratelimit-unified-status`) + to the client. Without them, Claude Code treats the 429 as a transient + throttle and retries forever. + """ + + def _build_upstream_ratelimit_exception(self): + """Produce the exception the proxy sees when an Anthropic-compatible + upstream returns 429, through the REAL litellm exception mapping so + `litellm_response_headers` is populated exactly as in production.""" + import httpx + + import litellm + from litellm.llms.anthropic.common_utils import AnthropicError + + upstream_headers = httpx.Headers( + { + "anthropic-ratelimit-unified-status": "rejected", + "retry-after": "287441", + "request-id": "20260821091810b67f7ae6443d450c", + # A vendor header that MUST be stripped before the proxy + # forwards it as its own response header. + "set-cookie": "acw_tc=abc; path=/; HttpOnly", + } + ) + raw = AnthropicError( + status_code=429, + message=( + '{"type":"error","error":{"type":"rate_limit_error",' + '"code":"1310","message":"quota exhausted"}}' + ), + headers=upstream_headers, + ) + try: + litellm.exception_type( + model="glm-5.2", + original_exception=raw, + custom_llm_provider="anthropic", + ) + except Exception as mapped: + return mapped + raise AssertionError("exception_type did not raise") + + @pytest.mark.asyncio + async def test_429_forwards_anthropic_ratelimit_headers(self): + import litellm.proxy.anthropic_endpoints.endpoints as ep + import litellm.proxy.proxy_server as proxy_server + from litellm.proxy._types import ProxyException, UserAPIKeyAuth + + mapped = self._build_upstream_ratelimit_exception() + + with ( + patch.object(ep, "_read_request_body", new=AsyncMock(return_value={"model": "glm"})), + patch.object( + ep.ProxyBaseLLMRequestProcessing, + "base_process_llm_request", + new=AsyncMock(side_effect=mapped), + ), + patch.object(proxy_server, "proxy_logging_obj") as mock_logging, + ): + mock_logging.post_call_failure_hook = AsyncMock() + with pytest.raises(ProxyException) as exc_info: + await ep.anthropic_response( + fastapi_response=MagicMock(), + request=MagicMock(), + user_api_key_dict=UserAPIKeyAuth(), + ) + + raised = exc_info.value + headers = {k.lower(): v for k, v in (raised.headers or {}).items()} + assert raised.code == "429" + # The two headers the issue is about — forwarded verbatim (unprefixed). + assert headers.get("anthropic-ratelimit-unified-status") == "rejected" + assert headers.get("retry-after") == "287441" + # Unsafe vendor headers must not leak onto the proxy's own response. + assert "set-cookie" not in headers + # LiteLLM's own headers are still present. + assert "x-litellm-version" in headers + mock_logging.post_call_failure_hook.assert_awaited_once() + + @pytest.mark.asyncio + async def test_upstream_cannot_spoof_litellm_headers(self): + """A hostile upstream must not be able to forge proxy-owned x-litellm-* + metadata — including via case-variant header names, which HTTP treats as + the same header. Only allowlisted Anthropic-native headers pass.""" + import litellm.proxy.anthropic_endpoints.endpoints as ep + import litellm.proxy.proxy_server as proxy_server + from litellm.proxy._types import ProxyException, UserAPIKeyAuth + + mapped = self._build_upstream_ratelimit_exception() + mapped.litellm_response_headers = { + "retry-after": "287441", + "x-litellm-version": "spoofed", # exact-case spoof + "X-LiteLLM-Model-ID": "forged-model", # case-variant spoof + "X-LiteLLM-Response-Cost": "999.99", + } + + with ( + patch.object(ep, "_read_request_body", new=AsyncMock(return_value={"model": "glm"})), + patch.object( + ep.ProxyBaseLLMRequestProcessing, + "base_process_llm_request", + new=AsyncMock(side_effect=mapped), + ), + patch.object(proxy_server, "proxy_logging_obj") as mock_logging, + ): + mock_logging.post_call_failure_hook = AsyncMock() + with pytest.raises(ProxyException) as exc_info: + await ep.anthropic_response( + fastapi_response=MagicMock(), + request=MagicMock(), + user_api_key_dict=UserAPIKeyAuth(), + ) + + raised_headers = exc_info.value.headers or {} + headers = {k.lower(): v for k, v in raised_headers.items()} + # The legit rate-limit header still passes. + assert headers.get("retry-after") == "287441" + # No forged value reaches the client, under any casing. + assert all(v not in ("spoofed", "forged-model", "999.99") for v in raised_headers.values()) + # No case-variant duplicate x-litellm-* key survived. + assert [k for k in raised_headers if k.lower() == "x-litellm-model-id"] in ([], ["x-litellm-model-id"]) + + def test_extract_upstream_headers_empty_when_none(self): + """No upstream headers anywhere -> empty dict (no crash).""" + from litellm.proxy.anthropic_endpoints.endpoints import ( + _extract_upstream_anthropic_headers, + ) + + e = Exception("boom") + assert _extract_upstream_anthropic_headers(e) == {} + + def test_extract_upstream_headers_non_mapping_source(self): + """A header source that isn't a mapping (no .items()) is tolerated -> {}.""" + from litellm.proxy.anthropic_endpoints.endpoints import ( + _extract_upstream_anthropic_headers, + ) + + e = Exception("boom") + e.litellm_response_headers = ["retry-after", "10"] # list, not a mapping + assert _extract_upstream_anthropic_headers(e) == {} + + def test_extract_upstream_headers_strips_unsafe(self): + from litellm.proxy.anthropic_endpoints.endpoints import ( + _extract_upstream_anthropic_headers, + ) + + e = Exception("boom") + e.litellm_response_headers = { + "retry-after": "10", + "Anthropic-RateLimit-Unified-Status": "rejected", # case-normalized to lowercase + "set-cookie": "x=1", + "content-length": "123", + "access-control-allow-origin": "*", + "X-LiteLLM-Model-ID": "forged", # proxy-owned namespace, must be dropped + "x-process-time": "0.1", # not allowlisted + } + out = {k: v for k, v in _extract_upstream_anthropic_headers(e).items()} + assert out == { + "retry-after": "10", + "anthropic-ratelimit-unified-status": "rejected", + }