diff --git a/.github/workflows/test-unit.yml b/.github/workflows/test-unit.yml index 49e6d7040d4..33f69779394 100644 --- a/.github/workflows/test-unit.yml +++ b/.github/workflows/test-unit.yml @@ -101,6 +101,14 @@ jobs: timeout-minutes: 20 job-timeout-minutes: 60 + - shard: fusion + artifact-name: fusion + test-path: "tests/test_litellm/test_fusion_router.py" + workers: 2 + reruns: 0 + timeout-minutes: 20 + job-timeout-minutes: 60 + - shard: misc artifact-name: misc test-path: >- diff --git a/cookbook/fusion_models.md b/cookbook/fusion_models.md index 9a7bffe004d..ee39cbde049 100644 --- a/cookbook/fusion_models.md +++ b/cookbook/fusion_models.md @@ -51,7 +51,7 @@ The outer model must support function calling. Panel and analyst models only nee - Initial outer, panel, analyst, continuation, and search calls are marked separately in spend logs. They inherit the caller identity and remain part of one logical Fusion request. - Client-visible `usage` describes the outer response returned to that client. Hidden panel, analyst, and search usage remains in its separately tagged spend-log rows; budget reconciliation includes the cost of every hidden call rather than merging heterogeneous model tokens into one public token count. - Admission control reserves the worst-case model-call cost. Hidden calls accumulate against that shared reservation, and the direct initial response or final continuation reconciles it once. This keeps concurrent requests from spending the same remaining budget while Fusion is still running. -- A deliberately timed-out panel or analyst call does not charge the worst-case cost of the entire Fusion request. Completed calls retain their known spend; the missing-callback fallback remains fail-closed for calls that completed without reporting cost. +- If a cancelled or timed-out call has unknown provider cost, admission accounting retains the full reservation. The same conservative fallback applies when a completed call never reports cost. Spend logs still record actual reported costs; cancellation does not prove the provider stopped billing. - Chat-completion streaming is buffered until LiteLLM knows whether the private tool was invoked. A direct response is replayed as a normal stream; a Fusion invocation suppresses the private tool-call stream and exposes only the final outer-model stream. - A request-level `tool_choice: required` is considered satisfied when Fusion runs. The continuation changes it to `auto` when client tools exist, or removes it when they do not, so the outer model can finish instead of being forced into a second tool call. - A client tool named `litellm_fusion` is rejected because that name is reserved for the private server tool. diff --git a/litellm/fusion_router.py b/litellm/fusion_router.py index 14b03dc5b74..90da7ee5cdd 100644 --- a/litellm/fusion_router.py +++ b/litellm/fusion_router.py @@ -836,10 +836,8 @@ class FusionReplayStream(CustomStreamWrapper): custom_llm_provider=source.custom_llm_provider, stream_options=source.stream_options, ) - self._hidden_params = { # mutable-ok: stream consumers attach response metadata - **source._hidden_params, - "fusion": dict(fusion_metadata), - } + self._hidden_params.update(source._hidden_params) + self._hidden_params["fusion"] = dict(fusion_metadata) self._iterator = iter(chunks) def __next__(self) -> ModelResponseStream: diff --git a/litellm/router.py b/litellm/router.py index 56a03943f6a..c7a982cc67d 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -2472,8 +2472,7 @@ class Router: messages=cast( # cast-ok: the public completion message shape is a valid acompletion subset list[AllMessageValues], messages ), - stream=False, - **kwargs, + **(kwargs | {"stream": False}), ) kwargs["model"] = model kwargs["messages"] = messages diff --git a/tests/test_litellm/test_fusion_router.py b/tests/test_litellm/test_fusion_router.py index bccc4427ea6..22cbcccae13 100644 --- a/tests/test_litellm/test_fusion_router.py +++ b/tests/test_litellm/test_fusion_router.py @@ -1332,9 +1332,10 @@ async def test_router_responses_and_anthropic_adapters_stream_direct_outer_respo assert all(isinstance(event, bytes) for event in anthropic_events) -def test_sync_router_and_responses_support_nonstreaming_fusion() -> None: +@pytest.mark.parametrize("request_kwargs", [{}, {"stream": False}, {"stream": None}]) +def test_sync_router_and_responses_support_nonstreaming_fusion(request_kwargs: dict[str, bool | None]) -> None: router = Router(model_list=_router_model_list()) - response = router.completion(model="fusion/test", messages=[{"role": "user", "content": "Answer"}]) + response = router.completion(model="fusion/test", messages=[{"role": "user", "content": "Answer"}], **request_kwargs) assert isinstance(response, ModelResponse) assert response.choices[0].message.content == "Final" responses_result = router._fusion_aware_responses(model="fusion/test", input="Answer")