mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(router): preserve explicit nonstreaming Fusion calls
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
f38bbad108
commit
6496a4cb53
5 changed files with 15 additions and 9 deletions
8
.github/workflows/test-unit.yml
vendored
8
.github/workflows/test-unit.yml
vendored
|
|
@ -101,6 +101,14 @@ jobs:
|
|||
timeout-minutes: 20
|
||||
job-timeout-minutes: 60
|
||||
|
||||
- shard: fusion
|
||||
artifact-name: fusion
|
||||
test-path: "tests/test_litellm/test_fusion_router.py"
|
||||
workers: 2
|
||||
reruns: 0
|
||||
timeout-minutes: 20
|
||||
job-timeout-minutes: 60
|
||||
|
||||
- shard: misc
|
||||
artifact-name: misc
|
||||
test-path: >-
|
||||
|
|
|
|||
|
|
@ -51,7 +51,7 @@ The outer model must support function calling. Panel and analyst models only nee
|
|||
- Initial outer, panel, analyst, continuation, and search calls are marked separately in spend logs. They inherit the caller identity and remain part of one logical Fusion request.
|
||||
- Client-visible `usage` describes the outer response returned to that client. Hidden panel, analyst, and search usage remains in its separately tagged spend-log rows; budget reconciliation includes the cost of every hidden call rather than merging heterogeneous model tokens into one public token count.
|
||||
- Admission control reserves the worst-case model-call cost. Hidden calls accumulate against that shared reservation, and the direct initial response or final continuation reconciles it once. This keeps concurrent requests from spending the same remaining budget while Fusion is still running.
|
||||
- A deliberately timed-out panel or analyst call does not charge the worst-case cost of the entire Fusion request. Completed calls retain their known spend; the missing-callback fallback remains fail-closed for calls that completed without reporting cost.
|
||||
- If a cancelled or timed-out call has unknown provider cost, admission accounting retains the full reservation. The same conservative fallback applies when a completed call never reports cost. Spend logs still record actual reported costs; cancellation does not prove the provider stopped billing.
|
||||
- Chat-completion streaming is buffered until LiteLLM knows whether the private tool was invoked. A direct response is replayed as a normal stream; a Fusion invocation suppresses the private tool-call stream and exposes only the final outer-model stream.
|
||||
- A request-level `tool_choice: required` is considered satisfied when Fusion runs. The continuation changes it to `auto` when client tools exist, or removes it when they do not, so the outer model can finish instead of being forced into a second tool call.
|
||||
- A client tool named `litellm_fusion` is rejected because that name is reserved for the private server tool.
|
||||
|
|
|
|||
|
|
@ -836,10 +836,8 @@ class FusionReplayStream(CustomStreamWrapper):
|
|||
custom_llm_provider=source.custom_llm_provider,
|
||||
stream_options=source.stream_options,
|
||||
)
|
||||
self._hidden_params = { # mutable-ok: stream consumers attach response metadata
|
||||
**source._hidden_params,
|
||||
"fusion": dict(fusion_metadata),
|
||||
}
|
||||
self._hidden_params.update(source._hidden_params)
|
||||
self._hidden_params["fusion"] = dict(fusion_metadata)
|
||||
self._iterator = iter(chunks)
|
||||
|
||||
def __next__(self) -> ModelResponseStream:
|
||||
|
|
|
|||
|
|
@ -2472,8 +2472,7 @@ class Router:
|
|||
messages=cast( # cast-ok: the public completion message shape is a valid acompletion subset
|
||||
list[AllMessageValues], messages
|
||||
),
|
||||
stream=False,
|
||||
**kwargs,
|
||||
**(kwargs | {"stream": False}),
|
||||
)
|
||||
kwargs["model"] = model
|
||||
kwargs["messages"] = messages
|
||||
|
|
|
|||
|
|
@ -1332,9 +1332,10 @@ async def test_router_responses_and_anthropic_adapters_stream_direct_outer_respo
|
|||
assert all(isinstance(event, bytes) for event in anthropic_events)
|
||||
|
||||
|
||||
def test_sync_router_and_responses_support_nonstreaming_fusion() -> None:
|
||||
@pytest.mark.parametrize("request_kwargs", [{}, {"stream": False}, {"stream": None}])
|
||||
def test_sync_router_and_responses_support_nonstreaming_fusion(request_kwargs: dict[str, bool | None]) -> None:
|
||||
router = Router(model_list=_router_model_list())
|
||||
response = router.completion(model="fusion/test", messages=[{"role": "user", "content": "Answer"}])
|
||||
response = router.completion(model="fusion/test", messages=[{"role": "user", "content": "Answer"}], **request_kwargs)
|
||||
assert isinstance(response, ModelResponse)
|
||||
assert response.choices[0].message.content == "Final"
|
||||
responses_result = router._fusion_aware_responses(model="fusion/test", input="Answer")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue