diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py index e6fd6290042..4f3a8b8ffb7 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/streaming_iterator.py @@ -245,12 +245,14 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper): "type": "content_block_stop", "index": compaction_index, } - # Advance state atomically with returning the terminal event so - # outside observers never see ``sent_content_block_finish=True`` - # before the client has received ``content_block_stop``. + # Don't touch ``sent_content_block_finish`` here: that flag is the + # state machine for the regular text/tool_use/thinking block and is + # independent of the synthetic compaction block lifecycle. Conflating + # them would let outside observers (subclass overrides, introspection + # hooks, exception paths) see ``sent_content_block_finish=True`` + # without any regular content block ever having started. self._increment_content_block_index() self.sent_compaction_block = True - self.sent_content_block_finish = True return stop_event def _create_initial_usage_delta(self) -> UsageDelta: diff --git a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py index e942633aedc..4f152ce5d70 100644 --- a/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py +++ b/litellm/llms/anthropic/experimental_pass_through/context_management/editors/compact.py @@ -536,11 +536,16 @@ async def _call_summary_model( # accepted by providers that don't strictly require it (OpenAI etc.). # Setting a sensible default here means the feature works regardless of # which model an admin configures as ``context_management_summary_model``. + # The propagated proxy auth/spend-attribution fields (``user_api_key`` etc.) + # must travel as ``litellm_metadata`` — that is the parameter the proxy's + # post-call spend hooks read for budget attribution. The provider-level + # ``metadata`` kwarg corresponds to the upstream API request body and would + # not flow into spend tracking. call_kwargs: Dict[str, Any] = { "model": summary_model, "messages": summary_messages, "max_tokens": COMPACT_SUMMARY_MAX_TOKENS, - "metadata": metadata, + "litellm_metadata": metadata, } if llm_router is not None and hasattr(llm_router, "acompletion"): return await llm_router.acompletion(**call_kwargs) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py index ef44c54d379..1c438e61963 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/context_management/test_compact.py @@ -1295,7 +1295,7 @@ async def test_prepare_context_managed_request_forwards_proxy_litellm_metadata() class _RouterStub: async def acompletion(self, **kwargs): - captured_summary_metadata.update(kwargs.get("metadata", {})) + captured_summary_metadata.update(kwargs.get("litellm_metadata", {})) return _make_mock_response("s") with (