fix(router): preserve explicit nonstreaming Fusion calls

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Moe Khalil 2026-09-22 03:51:21 +00:00
parent f38bbad108
commit 6496a4cb53
5 changed files with 15 additions and 9 deletions

View file

@ -101,6 +101,14 @@ jobs:
timeout-minutes: 20
job-timeout-minutes: 60
- shard: fusion
artifact-name: fusion
test-path: "tests/test_litellm/test_fusion_router.py"
workers: 2
reruns: 0
timeout-minutes: 20
job-timeout-minutes: 60
- shard: misc
artifact-name: misc
test-path: >-

View file

@ -51,7 +51,7 @@ The outer model must support function calling. Panel and analyst models only nee
- Initial outer, panel, analyst, continuation, and search calls are marked separately in spend logs. They inherit the caller identity and remain part of one logical Fusion request.
- Client-visible `usage` describes the outer response returned to that client. Hidden panel, analyst, and search usage remains in its separately tagged spend-log rows; budget reconciliation includes the cost of every hidden call rather than merging heterogeneous model tokens into one public token count.
- Admission control reserves the worst-case model-call cost. Hidden calls accumulate against that shared reservation, and the direct initial response or final continuation reconciles it once. This keeps concurrent requests from spending the same remaining budget while Fusion is still running.
- A deliberately timed-out panel or analyst call does not charge the worst-case cost of the entire Fusion request. Completed calls retain their known spend; the missing-callback fallback remains fail-closed for calls that completed without reporting cost.
- If a cancelled or timed-out call has unknown provider cost, admission accounting retains the full reservation. The same conservative fallback applies when a completed call never reports cost. Spend logs still record actual reported costs; cancellation does not prove the provider stopped billing.
- Chat-completion streaming is buffered until LiteLLM knows whether the private tool was invoked. A direct response is replayed as a normal stream; a Fusion invocation suppresses the private tool-call stream and exposes only the final outer-model stream.
- A request-level `tool_choice: required` is considered satisfied when Fusion runs. The continuation changes it to `auto` when client tools exist, or removes it when they do not, so the outer model can finish instead of being forced into a second tool call.
- A client tool named `litellm_fusion` is rejected because that name is reserved for the private server tool.

View file

@ -836,10 +836,8 @@ class FusionReplayStream(CustomStreamWrapper):
custom_llm_provider=source.custom_llm_provider,
stream_options=source.stream_options,
)
self._hidden_params = { # mutable-ok: stream consumers attach response metadata
**source._hidden_params,
"fusion": dict(fusion_metadata),
}
self._hidden_params.update(source._hidden_params)
self._hidden_params["fusion"] = dict(fusion_metadata)
self._iterator = iter(chunks)
def __next__(self) -> ModelResponseStream:

View file

@ -2472,8 +2472,7 @@ class Router:
messages=cast( # cast-ok: the public completion message shape is a valid acompletion subset
list[AllMessageValues], messages
),
stream=False,
**kwargs,
**(kwargs | {"stream": False}),
)
kwargs["model"] = model
kwargs["messages"] = messages

View file

@ -1332,9 +1332,10 @@ async def test_router_responses_and_anthropic_adapters_stream_direct_outer_respo
assert all(isinstance(event, bytes) for event in anthropic_events)
def test_sync_router_and_responses_support_nonstreaming_fusion() -> None:
@pytest.mark.parametrize("request_kwargs", [{}, {"stream": False}, {"stream": None}])
def test_sync_router_and_responses_support_nonstreaming_fusion(request_kwargs: dict[str, bool | None]) -> None:
router = Router(model_list=_router_model_list())
response = router.completion(model="fusion/test", messages=[{"role": "user", "content": "Answer"}])
response = router.completion(model="fusion/test", messages=[{"role": "user", "content": "Answer"}], **request_kwargs)
assert isinstance(response, ModelResponse)
assert response.choices[0].message.content == "Final"
responses_result = router._fusion_aware_responses(model="fusion/test", input="Answer")