mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
fix(compact_20260112): attribute summary subcall spend to parent key/team
The compact_20260112 polyfill summary subrequest propagated metadata via the Anthropic-shape `metadata` parameter, which only carries `user_id`. The proxy auth fields used for spend attribution (`user_api_key`, `user_api_key_team_id`, `litellm_call_id`, ...) live in `data["litellm_metadata"]`. As a result, summary subcalls landed on the router with an empty propagated metadata and the resulting tokens were not attributed to the caller's key/team budget. Rename the polyfill chain's spend-propagation parameter to `litellm_metadata` and pull it from `kwargs["litellm_metadata"]` in both the async and sync handlers, so the post-call hooks see the parent key/team and bill the summary tokens accordingly. Add an `_extract_proxy_litellm_metadata` helper and refactor `_extract_user_api_key_auth` to use it.
This commit is contained in:
parent
8fb1852444
commit
54d1bfc0af
4 changed files with 111 additions and 32 deletions
|
|
@ -52,21 +52,30 @@ def _messages_have_compaction_block(messages: List[Dict]) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def _extract_user_api_key_auth(kwargs: Dict[str, Any]) -> Any:
|
||||
"""Pull the parent request's ``UserAPIKeyAuth`` out of ``litellm_metadata``.
|
||||
def _extract_proxy_litellm_metadata(kwargs: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
"""Return ``kwargs["litellm_metadata"]`` when it's a dict; ``None`` otherwise.
|
||||
|
||||
The proxy attaches the full auth object under
|
||||
``data["litellm_metadata"]["user_api_key_auth"]`` (see
|
||||
``LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata``).
|
||||
The context-management polyfill uses it to gate the summary subrequest on
|
||||
the parent key/team's model allowlist; without it, the summary call would
|
||||
bypass the proxy auth checks. Returns ``None`` for SDK callers that bypass
|
||||
the proxy entirely.
|
||||
The proxy attaches its auth/spend-attribution fields (``user_api_key``,
|
||||
``user_api_key_team_id``, ``litellm_call_id``, the full ``UserAPIKeyAuth``
|
||||
object under ``user_api_key_auth``, ...) to ``data["litellm_metadata"]``
|
||||
for ``/v1/messages`` (see
|
||||
``LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata`` and
|
||||
``LITELLM_METADATA_ROUTES``). The Anthropic-shape ``metadata`` arg only
|
||||
carries ``user_id`` and must not be conflated. Returns ``None`` for SDK
|
||||
callers that bypass the proxy entirely.
|
||||
"""
|
||||
litellm_metadata = kwargs.get("litellm_metadata")
|
||||
if not isinstance(litellm_metadata, dict):
|
||||
return None
|
||||
return litellm_metadata.get("user_api_key_auth")
|
||||
return litellm_metadata
|
||||
|
||||
|
||||
def _extract_user_api_key_auth(kwargs: Dict[str, Any]) -> Any:
|
||||
"""Pull the parent request's ``UserAPIKeyAuth`` out of ``litellm_metadata``."""
|
||||
proxy_metadata = _extract_proxy_litellm_metadata(kwargs)
|
||||
if proxy_metadata is None:
|
||||
return None
|
||||
return proxy_metadata.get("user_api_key_auth")
|
||||
|
||||
|
||||
async def _prepare_context_managed_request(
|
||||
|
|
@ -76,7 +85,7 @@ async def _prepare_context_managed_request(
|
|||
tools: Optional[List[Dict]],
|
||||
system: Optional[Any],
|
||||
context_management_spec: Any,
|
||||
metadata: Optional[Dict],
|
||||
litellm_metadata: Optional[Dict],
|
||||
drop_params: Optional[bool],
|
||||
llm_router: Any,
|
||||
user_api_key_auth: Any = None,
|
||||
|
|
@ -117,7 +126,7 @@ async def _prepare_context_managed_request(
|
|||
tools=tools,
|
||||
system=working_system,
|
||||
context_management_spec=context_management_spec,
|
||||
metadata=metadata,
|
||||
litellm_metadata=litellm_metadata,
|
||||
drop_params=drop_params,
|
||||
llm_router=llm_router,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
|
|
@ -237,7 +246,7 @@ async def _run_polyfill_if_enabled(
|
|||
tools: Optional[List[Dict]],
|
||||
system: Optional[Any],
|
||||
context_management_spec: Any,
|
||||
metadata: Optional[Dict],
|
||||
litellm_metadata: Optional[Dict],
|
||||
drop_params: Optional[bool],
|
||||
llm_router: Any,
|
||||
user_api_key_auth: Any = None,
|
||||
|
|
@ -265,7 +274,7 @@ async def _run_polyfill_if_enabled(
|
|||
tools=tools,
|
||||
system=system,
|
||||
context_management_spec=context_management_spec,
|
||||
metadata=metadata,
|
||||
litellm_metadata=litellm_metadata,
|
||||
llm_router=llm_router,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
|
|
@ -578,7 +587,12 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
except Exception:
|
||||
pass
|
||||
|
||||
user_api_key_auth = _extract_user_api_key_auth(kwargs)
|
||||
proxy_litellm_metadata = _extract_proxy_litellm_metadata(kwargs)
|
||||
user_api_key_auth = (
|
||||
proxy_litellm_metadata.get("user_api_key_auth")
|
||||
if proxy_litellm_metadata is not None
|
||||
else None
|
||||
)
|
||||
|
||||
polyfill_result = await _prepare_context_managed_request(
|
||||
model=model,
|
||||
|
|
@ -586,7 +600,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
tools=tools,
|
||||
system=system,
|
||||
context_management_spec=context_management,
|
||||
metadata=metadata,
|
||||
litellm_metadata=proxy_litellm_metadata,
|
||||
drop_params=drop_params,
|
||||
llm_router=litellm_router,
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
|
|
@ -722,6 +736,12 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
if context_management is None and not _messages_have_compaction_block(messages):
|
||||
polyfill_result: Optional[PolyfillResult] = None
|
||||
else:
|
||||
proxy_litellm_metadata = _extract_proxy_litellm_metadata(kwargs)
|
||||
user_api_key_auth = (
|
||||
proxy_litellm_metadata.get("user_api_key_auth")
|
||||
if proxy_litellm_metadata is not None
|
||||
else None
|
||||
)
|
||||
polyfill_result = run_async_function(
|
||||
_prepare_context_managed_request,
|
||||
model=model,
|
||||
|
|
@ -729,10 +749,10 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
tools=tools,
|
||||
system=system,
|
||||
context_management_spec=context_management,
|
||||
metadata=metadata,
|
||||
litellm_metadata=proxy_litellm_metadata,
|
||||
drop_params=drop_params,
|
||||
llm_router=litellm_router,
|
||||
user_api_key_auth=_extract_user_api_key_auth(kwargs),
|
||||
user_api_key_auth=user_api_key_auth,
|
||||
)
|
||||
|
||||
effective_messages = (
|
||||
|
|
|
|||
|
|
@ -60,7 +60,7 @@ async def apply_context_management(
|
|||
tools: Optional[List[Dict[str, Any]]],
|
||||
system: Any,
|
||||
context_management_spec: Union[Dict[str, Any], List[Dict[str, Any]], None],
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
litellm_metadata: Optional[Dict[str, Any]] = None,
|
||||
llm_router: Any = None,
|
||||
user_api_key_auth: Any = None,
|
||||
) -> PolyfillResult:
|
||||
|
|
@ -101,7 +101,7 @@ async def apply_context_management(
|
|||
# Only async editors accept these — passing them to sync v0 editors
|
||||
# would break their signature.
|
||||
if inspect.iscoroutinefunction(editor):
|
||||
kwargs["metadata"] = metadata
|
||||
kwargs["litellm_metadata"] = litellm_metadata
|
||||
kwargs["llm_router"] = llm_router
|
||||
kwargs["user_api_key_auth"] = user_api_key_auth
|
||||
raw_result = await cast(Callable[..., Awaitable[Any]], editor)(**kwargs)
|
||||
|
|
|
|||
|
|
@ -269,13 +269,23 @@ def _build_summary_prompt(
|
|||
return prompt
|
||||
|
||||
|
||||
def _propagate_metadata(parent_metadata: Optional[Dict[str, Any]]) -> Dict[str, Any]:
|
||||
if not parent_metadata:
|
||||
def _propagate_metadata(
|
||||
parent_litellm_metadata: Optional[Dict[str, Any]],
|
||||
) -> Dict[str, Any]:
|
||||
"""Extract the parent request's auth/spend-attribution fields for the summary subcall.
|
||||
|
||||
The proxy attaches ``user_api_key``, ``user_api_key_team_id`` etc. to
|
||||
``data["litellm_metadata"]`` (see
|
||||
``LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata``).
|
||||
Without these on the summary subrequest, the router's post-call hooks
|
||||
cannot attribute summary tokens to the caller's key/team budget.
|
||||
"""
|
||||
if not parent_litellm_metadata:
|
||||
return {}
|
||||
propagated: Dict[str, Any] = {}
|
||||
for key in _PROPAGATED_METADATA_KEYS:
|
||||
if key in parent_metadata:
|
||||
propagated[key] = parent_metadata[key]
|
||||
if key in parent_litellm_metadata:
|
||||
propagated[key] = parent_litellm_metadata[key]
|
||||
return propagated
|
||||
|
||||
|
||||
|
|
@ -615,7 +625,7 @@ async def apply_compact_20260112( # noqa: PLR0915
|
|||
tools: Optional[List[Dict[str, Any]]],
|
||||
system: Optional[Union[str, List[Dict[str, Any]]]],
|
||||
edit_spec: Dict[str, Any],
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
litellm_metadata: Optional[Dict[str, Any]] = None,
|
||||
llm_router: Any = None,
|
||||
user_api_key_auth: Any = None,
|
||||
) -> PolyfillResult:
|
||||
|
|
@ -738,7 +748,7 @@ async def apply_compact_20260112( # noqa: PLR0915
|
|||
summary_messages = _build_summary_messages(
|
||||
effective_messages, prompt, system=augmented_system
|
||||
)
|
||||
propagated_metadata = _propagate_metadata(metadata)
|
||||
propagated_metadata = _propagate_metadata(litellm_metadata)
|
||||
|
||||
try:
|
||||
response = await _call_summary_model(
|
||||
|
|
|
|||
|
|
@ -537,11 +537,11 @@ async def test_full_summary_path_uses_router_when_available():
|
|||
assert result.compaction_block["content"] == "Router summary"
|
||||
|
||||
|
||||
async def test_metadata_propagated_to_summary_call():
|
||||
"""Auth metadata from the parent request is forwarded to the summary call."""
|
||||
async def test_litellm_metadata_propagated_to_summary_call():
|
||||
"""Auth fields from the proxy ``litellm_metadata`` are forwarded to the summary call."""
|
||||
messages = _simple_messages()
|
||||
mock_response = _make_mock_response("<summary>Summary</summary>")
|
||||
parent_metadata = {
|
||||
parent_litellm_metadata = {
|
||||
"user_api_key": "sk-test",
|
||||
"user_api_key_team_id": "team-123",
|
||||
"user_api_key_user_id": "user-456",
|
||||
|
|
@ -567,7 +567,7 @@ async def test_metadata_propagated_to_summary_call():
|
|||
tools=None,
|
||||
system=None,
|
||||
edit_spec=_EDIT_SPEC_DEFAULT,
|
||||
metadata=parent_metadata,
|
||||
litellm_metadata=parent_litellm_metadata,
|
||||
)
|
||||
|
||||
call_kwargs = mock_call.call_args.kwargs
|
||||
|
|
@ -1255,7 +1255,7 @@ async def test_run_polyfill_skipped_when_drop_params_true():
|
|||
tools=None,
|
||||
system=None,
|
||||
context_management_spec={"edits": [{"type": "compact_20260112"}]},
|
||||
metadata={},
|
||||
litellm_metadata={},
|
||||
drop_params=True,
|
||||
llm_router=None,
|
||||
)
|
||||
|
|
@ -1274,13 +1274,62 @@ async def test_run_polyfill_skipped_when_spec_empty():
|
|||
tools=None,
|
||||
system=None,
|
||||
context_management_spec=None,
|
||||
metadata={},
|
||||
litellm_metadata={},
|
||||
drop_params=False,
|
||||
llm_router=None,
|
||||
)
|
||||
assert result is None
|
||||
|
||||
|
||||
async def test_prepare_context_managed_request_forwards_proxy_litellm_metadata():
|
||||
"""The handler must hand the polyfill the proxy ``litellm_metadata`` (which
|
||||
carries ``user_api_key`` / ``user_api_key_team_id`` / ...), not the
|
||||
Anthropic-shape ``metadata`` arg (which only carries ``user_id``). Otherwise
|
||||
the summary subcall lands on the router with no parent attribution, and
|
||||
those tokens go unbilled to the caller's key/team."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.handler import (
|
||||
_prepare_context_managed_request,
|
||||
)
|
||||
|
||||
captured_summary_metadata: Dict[str, Any] = {}
|
||||
|
||||
class _RouterStub:
|
||||
async def acompletion(self, **kwargs):
|
||||
captured_summary_metadata.update(kwargs.get("metadata", {}))
|
||||
return _make_mock_response("<summary>s</summary>")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"litellm.llms.anthropic.experimental_pass_through.context_management.editors.compact._read_summary_model_setting",
|
||||
return_value="claude-haiku-4-5",
|
||||
),
|
||||
patch("litellm.token_counter", return_value=200_000),
|
||||
):
|
||||
result = await _prepare_context_managed_request(
|
||||
model=MODEL,
|
||||
messages=_simple_messages(),
|
||||
tools=None,
|
||||
system=None,
|
||||
context_management_spec={"edits": [_EDIT_SPEC_DEFAULT]},
|
||||
litellm_metadata={
|
||||
"user_api_key": "sk-parent",
|
||||
"user_api_key_team_id": "team-abc",
|
||||
"user_api_key_user_id": "user-xyz",
|
||||
"litellm_call_id": "call-1",
|
||||
},
|
||||
drop_params=False,
|
||||
llm_router=_RouterStub(),
|
||||
)
|
||||
|
||||
assert result is not None
|
||||
assert captured_summary_metadata.get("user_api_key") == "sk-parent"
|
||||
assert captured_summary_metadata.get("user_api_key_team_id") == "team-abc"
|
||||
assert captured_summary_metadata.get("user_api_key_user_id") == "user-xyz"
|
||||
assert captured_summary_metadata.get("litellm_call_id") == "call-1"
|
||||
# Anthropic-shape ``metadata.user_id`` must not leak in as a propagated field.
|
||||
assert "user_id" not in captured_summary_metadata
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Endpoint error format: AnthropicContextManagementError → Anthropic 400 body
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue