feat(anthropic): auto-inject context-1m beta header for [1m] model suffix

A model name ending in [1m] (case-insensitive) now adds the
context-1m-2025-08-07 anthropic-beta value and is sent upstream without the
suffix, on both the chat completions and the /v1/messages paths

Ported onto the current default branch: the pass-through module moved from
experimental_pass_through to pass_through and the tests moved to tests/unit

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
hx 2026-09-27 18:05:36 +08:00
parent f4308bc124
commit d5f6274c1e
4 changed files with 236 additions and 5 deletions

View file

@ -99,10 +99,12 @@ from litellm.utils import (
from ..common_utils import (
AnthropicError,
AnthropicModelInfo,
context_1m_requested,
eager_input_streaming_flag,
process_anthropic_headers,
requires_native_compaction_beta,
strip_advisor_blocks_from_messages,
strip_context_1m_suffix,
)
if TYPE_CHECKING:
@ -1779,6 +1781,17 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
)
return tools
def _maybe_add_context_1m_beta(
self,
headers: dict,
*,
model: str,
optional_params: dict,
litellm_params: Mapping[str, object] | None,
) -> None:
if context_1m_requested(model=model, optional_params=optional_params, litellm_params=litellm_params):
self._ensure_beta_header(headers, "context-1m-2025-08-07")
def _ensure_beta_header(self, headers: dict[str, str], beta_value: str) -> None:
"""
Ensure a beta header value is present in the anthropic-beta header.
@ -1837,7 +1850,12 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
)
def update_headers_with_optional_anthropic_beta(
self, headers: dict, optional_params: dict, messages: Sequence[object] = ()
self,
headers: dict,
optional_params: dict,
messages: Sequence[object] = (),
litellm_params: Mapping[str, object] | None = None,
model: str = "",
) -> dict:
"""Update headers with optional anthropic beta."""
@ -1850,6 +1868,10 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
if requires_native_compaction_beta(self._resolved_provider, optional_params, messages):
self._ensure_beta_header(headers, ANTHROPIC_BETA_HEADER_VALUES.COMPACT_2026_09_04.value)
self._maybe_add_context_1m_beta(
headers, model=model, optional_params=optional_params, litellm_params=litellm_params
)
_tools: Final = optional_params.get("tools", [])
for tool in _tools:
if tool.get("type", None) and tool.get("type").startswith(ANTHROPIC_HOSTED_TOOLS.WEB_FETCH.value):
@ -2004,7 +2026,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
) # don't use verbose_logger.exception, if exception is raised
self.update_headers_with_optional_anthropic_beta(
headers=headers, optional_params=optional_params, messages=anthropic_messages
headers=headers,
optional_params=optional_params,
messages=anthropic_messages,
litellm_params=litellm_params,
model=model,
)
## Auto-strip advisor blocks from history if advisor tool is absent.
@ -2072,7 +2098,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
)
data: Final = {
"model": model,
"model": strip_context_1m_suffix(model),
"messages": anthropic_messages,
**optional_params,
}

View file

@ -71,6 +71,46 @@ _BEDROCK_VERSION_SUFFIX_RE: Final = re.compile(r"-v\d+(?::\d+)?$")
_INFERENCE_PROFILE_MINOR_RE: Final = re.compile(r":\d+$")
_DATED_RELEASE_SUFFIX_RE: Final = re.compile(r"-\d{8}$")
_DOTTED_VERSION_RE: Final = re.compile(r"(\d)\.(\d)")
_CONTEXT_1M_SUFFIX: Final = re.compile(r"\[1m\]$", flags=re.IGNORECASE)
def model_has_context_1m_suffix(model: object) -> bool:
return isinstance(model, str) and _CONTEXT_1M_SUFFIX.search(model) is not None
def strip_context_1m_suffix(model: str) -> str:
return _CONTEXT_1M_SUFFIX.sub("", model)
def _original_model_from(params: object) -> object:
if not isinstance(params, Mapping):
return None
return params.get("_original_model")
def context_1m_requested(
*,
model: object = "",
optional_params: object = None,
litellm_params: object = None,
) -> bool:
return any(
model_has_context_1m_suffix(candidate)
for candidate in (
model,
_original_model_from(optional_params),
_original_model_from(litellm_params),
)
)
_CONTEXT_1M_BETA: Final = "context-1m-2025-08-07"
def context_1m_beta_values(supported: bool) -> frozenset[str]:
return frozenset((_CONTEXT_1M_BETA,)) if supported else frozenset()
_CLAUDE_CODE_BILLING_HEADER_PREFIX: Final = "x-anthropic-billing-header:"
_CLAUDE_CODE_OBJECT_MAPPING_ADAPTER: Final = TypeAdapter(dict[object, object])
_CLAUDE_CODE_OBJECT_LIST_ADAPTER: Final = TypeAdapter(list[object])
@ -915,8 +955,9 @@ class AnthropicModelInfo(BaseLLMModelInfo):
container_with_skills_used: bool = False,
api_base: str | None = None,
use_bearer_for_custom_base: bool = False,
context_1m_supported: bool = False,
) -> dict:
betas: Final = set()
betas: Final = set(context_1m_beta_values(context_1m_supported))
# Anthropic no longer requires the prompt-caching beta header
# Prompt caching now works automatically when cache_control is used in messages
# Reference: https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching
@ -1025,8 +1066,14 @@ class AnthropicModelInfo(BaseLLMModelInfo):
user_anthropic_beta_headers: Final = self._get_user_anthropic_beta_headers(
anthropic_beta_header=headers.get("anthropic-beta")
)
context_1m_supported: Final = context_1m_requested(
model=model,
optional_params=optional_params,
litellm_params=litellm_params,
)
anthropic_headers: Final = self.get_anthropic_headers(
computer_tool_used=computer_tool_used,
context_1m_supported=context_1m_supported,
prompt_caching_set=prompt_caching_set,
pdf_used=pdf_used,
api_key=api_key,

View file

@ -23,9 +23,12 @@ from litellm.types.router import GenericLiteLLMParams
from ...common_utils import (
AnthropicError,
AnthropicModelInfo,
context_1m_beta_values,
context_1m_requested,
optionally_handle_anthropic_oauth,
requires_native_compaction_beta,
strip_advisor_blocks_from_messages,
strip_context_1m_suffix,
strip_encrypted_reasoning_blocks_from_anthropic_messages,
)
from .mid_conversation_system import (
@ -277,6 +280,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
headers=headers,
optional_params=optional_params,
messages=messages,
model=model,
litellm_params=litellm_params,
)
return headers, api_base
@ -562,7 +567,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
anthropic_messages_request: Final[AnthropicMessagesRequest] = AnthropicMessagesRequest(
messages=strip_encrypted_reasoning_blocks_from_anthropic_messages(messages),
max_tokens=max_tokens,
model=model,
model=strip_context_1m_suffix(model),
**anthropic_messages_optional_request_params,
)
return dict(anthropic_messages_request)
@ -611,6 +616,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
optional_params: dict,
custom_llm_provider: str = "anthropic",
messages: Sequence[object] = (),
model: str = "",
litellm_params: object = None,
) -> dict:
"""
Auto-inject anthropic-beta headers based on features used.
@ -621,6 +628,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
- output_format: adds 'structured-outputs-2025-11-13'
- speed: adds 'fast-mode-2026-02-01'
- a message carrying output_config: adds 'per-turn-control-2026-07-01'
- [1m] suffix: adds 'context-1m-2025-08-07'
Args:
headers: Request headers dict
@ -642,6 +650,12 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
if requires_native_compaction_beta(custom_llm_provider, optional_params, messages):
beta_values.add(ANTHROPIC_BETA_HEADER_VALUES.COMPACT_2026_09_04.value)
beta_values.update(
context_1m_beta_values(
context_1m_requested(model=model, optional_params=optional_params, litellm_params=litellm_params)
)
)
# Check for context management
context_management_param: Final = optional_params.get("context_management")
if context_management_param is not None:

View file

@ -0,0 +1,144 @@
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.common_utils import (
context_1m_requested,
strip_context_1m_suffix,
)
from litellm.llms.anthropic.pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
class TestAnthropicContext1mBetaHeader:
def test_strip_context_1m_suffix_is_pure(self):
assert strip_context_1m_suffix("claude-sonnet-4-6[1m]") == "claude-sonnet-4-6"
assert strip_context_1m_suffix("claude-sonnet-4-6[1M]") == "claude-sonnet-4-6"
assert strip_context_1m_suffix("claude-sonnet-4-6") == "claude-sonnet-4-6"
assert strip_context_1m_suffix("gpt-4o[1m]") == "gpt-4o"
def test_context_1m_requested_from_model_or_stashed_original(self):
assert context_1m_requested(model="claude-sonnet-4-6[1m]")
assert context_1m_requested(
model="claude-sonnet-4-6",
litellm_params={"_original_model": "claude-sonnet-4-6[1M]"},
)
assert not context_1m_requested(model="claude-sonnet-4-6")
assert not context_1m_requested(model="gpt-4o")
def test_get_anthropic_headers_with_context_1m(self):
headers = AnthropicConfig().get_anthropic_headers(api_key="sk-test", context_1m_supported=True)
assert "context-1m-2025-08-07" in headers.get("anthropic-beta", "")
def test_get_anthropic_headers_without_context_1m(self):
headers = AnthropicConfig().get_anthropic_headers(api_key="sk-test", context_1m_supported=False)
assert "context-1m-2025-08-07" not in headers.get("anthropic-beta", "")
def test_validate_environment_adds_context_1m_for_suffixed_model(self):
headers = AnthropicConfig().validate_environment(
headers={},
model="claude-sonnet-4-6[1m]",
messages=[{"role": "user", "content": "hi"}],
optional_params={},
litellm_params={},
api_key="sk-test",
)
assert "context-1m-2025-08-07" in headers.get("anthropic-beta", "")
def test_validate_environment_no_context_1m_without_suffix(self):
headers = AnthropicConfig().validate_environment(
headers={},
model="claude-sonnet-4-6",
messages=[{"role": "user", "content": "hi"}],
optional_params={},
litellm_params={},
api_key="sk-test",
)
assert "context-1m-2025-08-07" not in headers.get("anthropic-beta", "")
def test_validate_environment_uses_original_model_from_litellm_params(self):
headers = AnthropicConfig().validate_environment(
headers={},
model="claude-sonnet-4-6",
messages=[{"role": "user", "content": "hi"}],
optional_params={},
litellm_params={"_original_model": "claude-sonnet-4-6[1M]"},
api_key="sk-test",
)
assert "context-1m-2025-08-07" in headers.get("anthropic-beta", "")
def test_update_headers_adds_context_1m_with_original_model(self):
headers = AnthropicConfig().update_headers_with_optional_anthropic_beta(
headers={},
optional_params={"_original_model": "claude-sonnet-4-6[1m]"},
)
assert "context-1m-2025-08-07" in headers.get("anthropic-beta", "")
def test_update_headers_no_context_1m_without_original_model(self):
headers = AnthropicConfig().update_headers_with_optional_anthropic_beta(
headers={},
optional_params={},
)
assert "context-1m-2025-08-07" not in headers.get("anthropic-beta", "")
def test_pass_through_adds_context_1m(self):
headers = AnthropicMessagesConfig._update_headers_with_anthropic_beta(
headers={},
optional_params={"_original_model": "claude-sonnet-4-6[1m]"},
)
assert "context-1m-2025-08-07" in headers.get("anthropic-beta", "")
def test_pass_through_no_context_1m_without_suffix(self):
headers = AnthropicMessagesConfig._update_headers_with_anthropic_beta(
headers={},
optional_params={},
)
assert "context-1m-2025-08-07" not in headers.get("anthropic-beta", "")
def test_pass_through_adds_context_1m_from_model_without_original_model(self):
headers, _ = AnthropicMessagesConfig().validate_anthropic_messages_environment(
headers={},
model="claude-sonnet-4-6[1m]",
messages=[{"role": "user", "content": "hi"}],
optional_params={},
litellm_params={},
api_key="sk-test",
)
assert "context-1m-2025-08-07" in headers.get("anthropic-beta", "")
def test_pass_through_transform_strips_1m_from_upstream_model(self):
result = AnthropicMessagesConfig().transform_anthropic_messages_request(
model="claude-sonnet-4-6[1m]",
messages=[{"role": "user", "content": "hi"}],
anthropic_messages_optional_request_params={"max_tokens": 16},
litellm_params={},
headers={},
)
assert result["model"] == "claude-sonnet-4-6"
def test_chat_transform_request_strips_1m_and_injects_header(self):
headers = {}
result = AnthropicConfig().transform_request(
model="claude-sonnet-4-6[1m]",
messages=[{"role": "user", "content": "hi"}],
optional_params={"max_tokens": 16},
litellm_params={},
headers=headers,
)
assert result["model"] == "claude-sonnet-4-6"
assert "context-1m-2025-08-07" in headers.get("anthropic-beta", "")
def test_chat_transform_request_uppercase_suffix_strips_and_injects_header(self):
headers = {}
result = AnthropicConfig().transform_request(
model="claude-sonnet-4-6[1M]",
messages=[{"role": "user", "content": "hi"}],
optional_params={"max_tokens": 16},
litellm_params={},
headers=headers,
)
assert result["model"] == "claude-sonnet-4-6"
assert "context-1m-2025-08-07" in headers.get("anthropic-beta", "")
def test_proxy_does_not_globally_strip_non_anthropic_suffix(self):
from litellm.proxy import route_llm_request
assert not hasattr(route_llm_request, "stash_and_strip_context_1m_model_suffix")