fix: bug fixes from PR review

- streaming_iterator: don't set sent_content_block_finish during compaction
  block lifecycle; that flag tracks the regular text/tool_use/thinking block
  state machine, conflating the two leaks bad state to introspection paths.
- compact._call_summary_model: send propagated proxy auth/spend-attribution
  fields as 'litellm_metadata' instead of 'metadata' so the router's post-call
  hooks attribute summary tokens to the caller's key/team budget.

Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
Cursor Agent 2026-05-28 06:58:09 +00:00
parent ad2e1ba916
commit fd6e5be46b
No known key found for this signature in database
3 changed files with 13 additions and 6 deletions

View file

@ -245,12 +245,14 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
"type": "content_block_stop",
"index": compaction_index,
}
# Advance state atomically with returning the terminal event so
# outside observers never see ``sent_content_block_finish=True``
# before the client has received ``content_block_stop``.
# Don't touch ``sent_content_block_finish`` here: that flag is the
# state machine for the regular text/tool_use/thinking block and is
# independent of the synthetic compaction block lifecycle. Conflating
# them would let outside observers (subclass overrides, introspection
# hooks, exception paths) see ``sent_content_block_finish=True``
# without any regular content block ever having started.
self._increment_content_block_index()
self.sent_compaction_block = True
self.sent_content_block_finish = True
return stop_event
def _create_initial_usage_delta(self) -> UsageDelta:

View file

@ -536,11 +536,16 @@ async def _call_summary_model(
# accepted by providers that don't strictly require it (OpenAI etc.).
# Setting a sensible default here means the feature works regardless of
# which model an admin configures as ``context_management_summary_model``.
# The propagated proxy auth/spend-attribution fields (``user_api_key`` etc.)
# must travel as ``litellm_metadata`` — that is the parameter the proxy's
# post-call spend hooks read for budget attribution. The provider-level
# ``metadata`` kwarg corresponds to the upstream API request body and would
# not flow into spend tracking.
call_kwargs: Dict[str, Any] = {
"model": summary_model,
"messages": summary_messages,
"max_tokens": COMPACT_SUMMARY_MAX_TOKENS,
"metadata": metadata,
"litellm_metadata": metadata,
}
if llm_router is not None and hasattr(llm_router, "acompletion"):
return await llm_router.acompletion(**call_kwargs)

View file

@ -1295,7 +1295,7 @@ async def test_prepare_context_managed_request_forwards_proxy_litellm_metadata()
class _RouterStub:
async def acompletion(self, **kwargs):
captured_summary_metadata.update(kwargs.get("metadata", {}))
captured_summary_metadata.update(kwargs.get("litellm_metadata", {}))
return _make_mock_response("<summary>s</summary>")
with (