mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
fix: bug fixes from PR review
- streaming_iterator: don't set sent_content_block_finish during compaction block lifecycle; that flag tracks the regular text/tool_use/thinking block state machine, conflating the two leaks bad state to introspection paths. - compact._call_summary_model: send propagated proxy auth/spend-attribution fields as 'litellm_metadata' instead of 'metadata' so the router's post-call hooks attribute summary tokens to the caller's key/team budget. Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
parent
ad2e1ba916
commit
fd6e5be46b
3 changed files with 13 additions and 6 deletions
|
|
@ -245,12 +245,14 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
"type": "content_block_stop",
|
||||
"index": compaction_index,
|
||||
}
|
||||
# Advance state atomically with returning the terminal event so
|
||||
# outside observers never see ``sent_content_block_finish=True``
|
||||
# before the client has received ``content_block_stop``.
|
||||
# Don't touch ``sent_content_block_finish`` here: that flag is the
|
||||
# state machine for the regular text/tool_use/thinking block and is
|
||||
# independent of the synthetic compaction block lifecycle. Conflating
|
||||
# them would let outside observers (subclass overrides, introspection
|
||||
# hooks, exception paths) see ``sent_content_block_finish=True``
|
||||
# without any regular content block ever having started.
|
||||
self._increment_content_block_index()
|
||||
self.sent_compaction_block = True
|
||||
self.sent_content_block_finish = True
|
||||
return stop_event
|
||||
|
||||
def _create_initial_usage_delta(self) -> UsageDelta:
|
||||
|
|
|
|||
|
|
@ -536,11 +536,16 @@ async def _call_summary_model(
|
|||
# accepted by providers that don't strictly require it (OpenAI etc.).
|
||||
# Setting a sensible default here means the feature works regardless of
|
||||
# which model an admin configures as ``context_management_summary_model``.
|
||||
# The propagated proxy auth/spend-attribution fields (``user_api_key`` etc.)
|
||||
# must travel as ``litellm_metadata`` — that is the parameter the proxy's
|
||||
# post-call spend hooks read for budget attribution. The provider-level
|
||||
# ``metadata`` kwarg corresponds to the upstream API request body and would
|
||||
# not flow into spend tracking.
|
||||
call_kwargs: Dict[str, Any] = {
|
||||
"model": summary_model,
|
||||
"messages": summary_messages,
|
||||
"max_tokens": COMPACT_SUMMARY_MAX_TOKENS,
|
||||
"metadata": metadata,
|
||||
"litellm_metadata": metadata,
|
||||
}
|
||||
if llm_router is not None and hasattr(llm_router, "acompletion"):
|
||||
return await llm_router.acompletion(**call_kwargs)
|
||||
|
|
|
|||
|
|
@ -1295,7 +1295,7 @@ async def test_prepare_context_managed_request_forwards_proxy_litellm_metadata()
|
|||
|
||||
class _RouterStub:
|
||||
async def acompletion(self, **kwargs):
|
||||
captured_summary_metadata.update(kwargs.get("metadata", {}))
|
||||
captured_summary_metadata.update(kwargs.get("litellm_metadata", {}))
|
||||
return _make_mock_response("<summary>s</summary>")
|
||||
|
||||
with (
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue