diff --git a/litellm/integrations/advisor_interception/handler.py b/litellm/integrations/advisor_interception/handler.py index 0eb81ab14f3..1fee42eb8b3 100644 --- a/litellm/integrations/advisor_interception/handler.py +++ b/litellm/integrations/advisor_interception/handler.py @@ -238,6 +238,11 @@ class AdvisorInterceptionLogger(CustomLogger): response=current_response, response_cost=total_response_cost ) return current_response + if len(advisor_calls) != len(raw_tool_calls): + verbose_logger.debug( + "AdvisorInterception: Mixed tool calls detected inside advisor loop, stopping interception" + ) + return current_response assistant_content = self._extract_message_content(current_response) assistant_message: Dict[str, Any] = { @@ -410,6 +415,7 @@ class AdvisorInterceptionLogger(CustomLogger): raw_tool_calls: List[Dict] = [] for tool_call in tool_calls: + raw_tool_calls.append(tool_call) function = tool_call.get("function", {}) function_name = function.get("name") if function_name not in {"advisor", LITELLM_ADVISOR_TOOL_NAME}: @@ -432,7 +438,6 @@ class AdvisorInterceptionLogger(CustomLogger): "question": parsed_args.get("question"), } ) - raw_tool_calls.append(tool_call) # Some providers (e.g. Gemini in certain modes) can return legacy function_call. if not advisor_calls and function_call is not None: @@ -633,6 +638,7 @@ class AdvisorInterceptionLogger(CustomLogger): "safety_identifier", "service_tier", "stream", + "litellm_call_id", } return { k: v diff --git a/tests/test_litellm/integrations/advisor_interception/test_advisor_interception_handler.py b/tests/test_litellm/integrations/advisor_interception/test_advisor_interception_handler.py index 1a197b71498..740c0b2cff7 100644 --- a/tests/test_litellm/integrations/advisor_interception/test_advisor_interception_handler.py +++ b/tests/test_litellm/integrations/advisor_interception/test_advisor_interception_handler.py @@ -323,3 +323,74 @@ async def test_post_call_hook_cleans_up_config_when_should_run_is_false(): assert result is None assert "cleanup-call-2" not in logger._advisor_config_by_call_id + + +@pytest.mark.asyncio +async def test_should_run_chat_completion_agentic_loop_skips_mixed_tool_calls(): + logger = AdvisorInterceptionLogger(enabled_providers=["openai"]) + logger._advisor_config_by_call_id["mixed-call-1"] = { + "advisor_model": "claude-opus-4-6", + "max_uses": 3, + } + mock_response = ModelResponse( + id="test-mixed", + choices=[ + Choices( + finish_reason="tool_calls", + index=0, + message=Message( + role="assistant", + content=None, + tool_calls=[ + ChatCompletionMessageToolCall( + id="call_advisor", + type="function", + function=Function( + name=LITELLM_ADVISOR_TOOL_NAME, + arguments='{"question":"Need advisor guidance"}', + ), + ), + ChatCompletionMessageToolCall( + id="call_weather", + type="function", + function=Function( + name="get_weather", + arguments='{"city":"San Francisco"}', + ), + ), + ], + ), + ) + ], + model="gpt-4o-mini", + object="chat.completion", + created=123, + ) + + should_run, tools_dict = await logger.async_should_run_chat_completion_agentic_loop( + response=mock_response, + model="gpt-4o-mini", + messages=[{"role": "user", "content": "Help"}], + tools=[get_litellm_advisor_tool_openai()], + stream=False, + custom_llm_provider="openai", + kwargs={"litellm_call_id": "mixed-call-1"}, + ) + + assert should_run is False + assert tools_dict == {} + assert "mixed-call-1" not in logger._advisor_config_by_call_id + + +def test_prepare_followup_kwargs_removes_litellm_call_id(): + kwargs = { + "litellm_call_id": "original-call-id", + "metadata": {"k": "v"}, + "user_defined_key": "should_remain", + } + + filtered_kwargs = AdvisorInterceptionLogger._prepare_followup_kwargs(kwargs) + + assert "litellm_call_id" not in filtered_kwargs + assert "metadata" not in filtered_kwargs + assert filtered_kwargs["user_defined_key"] == "should_remain"