From fd6e5be46bebe10e16fe5b7d8b041e968bfed294 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Thu, 28 May 2026 06:58:09 +0000 Subject: [PATCH] fix: bug fixes from PR review - streaming_iterator: don't set sent_content_block_finish during compaction block lifecycle; that flag tracks the regular text/tool_use/thinking block state machine, conflating the two leaks bad state to introspection paths. - compact._call_summary_model: send propagated proxy auth/spend-attribution fields as 'litellm_metadata' instead of 'metadata' so the router's post-call hooks attribute summary tokens to the caller's key/team budget. Co-authored-by: Yassin Kortam --- .../adapters/streaming_iterator.py | 10 ++++++---- .../context_management/editors/compact.py | 7 ++++++- .../context_management/test_compact.py | 2 +- 3 files changed, 13 insertions(+), 6 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py index e6fd6290042..4f3a8b8ffb7 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py @@ -245,12 +245,14 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): "type": "content_block_stop", "index": compaction_index, } - # Advance state atomically with returning the terminal event so - # outside observers never see ``sent_content_block_finish=True`` - # before the client has received ``content_block_stop``. + # Don't touch ``sent_content_block_finish`` here: that flag is the + # state machine for the regular text/tool_use/thinking block and is + # independent of the synthetic compaction block lifecycle. Conflating + # them would let outside observers (subclass overrides, introspection + # hooks, exception paths) see ``sent_content_block_finish=True`` + # without any regular content block ever having started. self._increment_content_block_index() self.sent_compaction_block = True - self.sent_content_block_finish = True return stop_event def _create_initial_usage_delta(self) -> UsageDelta: diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index e942633aedc..4f152ce5d70 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -536,11 +536,16 @@ async def _call_summary_model( # accepted by providers that don't strictly require it (OpenAI etc.). # Setting a sensible default here means the feature works regardless of # which model an admin configures as ``context_management_summary_model``. + # The propagated proxy auth/spend-attribution fields (``user_api_key`` etc.) + # must travel as ``litellm_metadata`` — that is the parameter the proxy's + # post-call spend hooks read for budget attribution. The provider-level + # ``metadata`` kwarg corresponds to the upstream API request body and would + # not flow into spend tracking. call_kwargs: Dict[str, Any] = { "model": summary_model, "messages": summary_messages, "max_tokens": COMPACT_SUMMARY_MAX_TOKENS, - "metadata": metadata, + "litellm_metadata": metadata, } if llm_router is not None and hasattr(llm_router, "acompletion"): return await llm_router.acompletion(**call_kwargs) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py index ef44c54d379..1c438e61963 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py @@ -1295,7 +1295,7 @@ async def test_prepare_context_managed_request_forwards_proxy_litellm_metadata() class _RouterStub: async def acompletion(self, **kwargs): - captured_summary_metadata.update(kwargs.get("metadata", {})) + captured_summary_metadata.update(kwargs.get("litellm_metadata", {})) return _make_mock_response("s") with (