From a1ea01c6f17294ec07a6cedf238e651c2e09a709 Mon Sep 17 00:00:00 2001 From: Vrajesh Sulakhe Date: Fri, 2 Oct 2026 20:52:35 +0530 Subject: [PATCH 1/5] fix(agentic): preserve provider prefixes for organization models --- .../chat_completion_agentic_loop.py | 2 +- litellm/llms/custom_httpx/llm_http_handler.py | 2 +- .../test_agentic_loop_prefix_44069.py | 82 +++++++++++++++++++ 3 files changed, 84 insertions(+), 2 deletions(-) create mode 100644 tests/test_litellm/test_agentic_loop_prefix_44069.py diff --git a/litellm/litellm_core_utils/chat_completion_agentic_loop.py b/litellm/litellm_core_utils/chat_completion_agentic_loop.py index 9d7c9864e62..2e842df4718 100644 --- a/litellm/litellm_core_utils/chat_completion_agentic_loop.py +++ b/litellm/litellm_core_utils/chat_completion_agentic_loop.py @@ -171,7 +171,7 @@ async def _execute_chat_completion_agentic_plan( raise ValueError("Agentic loop plan missing patched messages") full_model_name = patch.model or model - if "/" not in full_model_name: + if custom_llm_provider and not full_model_name.startswith(f"{custom_llm_provider}/"): full_model_name = f"{custom_llm_provider}/{full_model_name}" optional_params_for_followup: Final = {**optional_params, **patch.optional_params} diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index dd97db45a88..bfd60bb0f19 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -5494,7 +5494,7 @@ class BaseLLMHTTPHandler: raise ValueError("Agentic loop plan missing patched messages") full_model_name = patch.model or model - if "/" not in full_model_name: + if custom_llm_provider and not full_model_name.startswith(f"{custom_llm_provider}/"): full_model_name = f"{custom_llm_provider}/{full_model_name}" optional_params_for_followup: Final = dict(optional_params) diff --git a/tests/test_litellm/test_agentic_loop_prefix_44069.py b/tests/test_litellm/test_agentic_loop_prefix_44069.py new file mode 100644 index 00000000000..c3c5c7ddc7b --- /dev/null +++ b/tests/test_litellm/test_agentic_loop_prefix_44069.py @@ -0,0 +1,82 @@ +from typing import Final, Literal +from unittest.mock import AsyncMock, patch + +import pytest + +import litellm +from litellm.integrations.custom_logger import CustomLogger +from litellm.litellm_core_utils.chat_completion_agentic_loop import ( + _execute_chat_completion_agentic_plan, +) +from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler +from litellm.types.integrations.custom_logger import ( + AgenticLoopPlan, + AgenticLoopRequestPatch, +) +from litellm.types.utils import ModelResponse + + +@pytest.mark.asyncio +@pytest.mark.parametrize("execution_path", ("http", "sdk")) +@pytest.mark.parametrize("use_patch_model", (False, True), ids=("original-model", "patch-model")) +@pytest.mark.parametrize( + ("model_name", "custom_llm_provider", "expected_model"), + ( + ("zai-org/GLM-5.3-Flash", "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ("hosted_vllm/zai-org/GLM-5.3-Flash", "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ("hosted_vllm/zai-org/GLM-5.3-Flash", "", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ), + ids=("organization-model", "already-prefixed", "no-provider"), +) +async def test_agentic_followup_preserves_provider_prefix( + execution_path: Literal["http", "sdk"], + use_patch_model: bool, + model_name: str, + custom_llm_provider: str, + expected_model: str, +) -> None: + model: Final = "original-model" if use_patch_model else model_name + plan: Final = AgenticLoopPlan( + run_agentic_loop=True, + request_patch=AgenticLoopRequestPatch( + model=model_name if use_patch_model else None, + messages=[{"role": "user", "content": "Continue"}], + ), + ) + followup: Final = AsyncMock(wraps=litellm.acompletion) + + with patch("litellm.acompletion", new=followup): + response: Final = ( + await BaseLLMHTTPHandler()._execute_chat_completion_agentic_plan( + plan=plan, + model=model, + messages=[], + optional_params={"mock_response": "Follow-up complete"}, + kwargs={"custom_llm_provider": custom_llm_provider}, + custom_llm_provider=custom_llm_provider, + depth=0, + max_loops=3, + fingerprints=[], + fingerprint="followup", + ) + if execution_path == "http" + else await _execute_chat_completion_agentic_plan( + plan=plan, + callback=CustomLogger(), + model=model, + optional_params={"mock_response": "Follow-up complete"}, + kwargs={"custom_llm_provider": custom_llm_provider}, + logging_obj=None, + custom_llm_provider=custom_llm_provider, + depth=0, + max_loops=3, + fingerprints=[], + fingerprint="followup", + ) + ) + + assert isinstance(response, ModelResponse) + assert response.model == expected_model.removeprefix("hosted_vllm/") + followup.assert_awaited_once() + assert followup.await_args is not None + assert followup.await_args.kwargs["model"] == expected_model From 3011431d41c850552ceb9f3edf3b9672ea22fc0c Mon Sep 17 00:00:00 2001 From: Vrajesh Sulakhe Date: Sat, 3 Oct 2026 06:32:22 +0530 Subject: [PATCH 2/5] fix(agentic): preserve cross-provider follow-up overrides --- litellm/litellm_core_utils/chat_completion_agentic_loop.py | 7 ++++++- litellm/llms/custom_httpx/llm_http_handler.py | 7 ++++++- tests/test_litellm/test_agentic_loop_prefix_44069.py | 7 ++++--- 3 files changed, 16 insertions(+), 5 deletions(-) diff --git a/litellm/litellm_core_utils/chat_completion_agentic_loop.py b/litellm/litellm_core_utils/chat_completion_agentic_loop.py index 2e842df4718..b9d03e0cc8a 100644 --- a/litellm/litellm_core_utils/chat_completion_agentic_loop.py +++ b/litellm/litellm_core_utils/chat_completion_agentic_loop.py @@ -171,7 +171,12 @@ async def _execute_chat_completion_agentic_plan( raise ValueError("Agentic loop plan missing patched messages") full_model_name = patch.model or model - if custom_llm_provider and not full_model_name.startswith(f"{custom_llm_provider}/"): + known_providers: Final = getattr(litellm, "provider_list", []) + has_provider_prefix: Final = ( + full_model_name.startswith(f"{custom_llm_provider}/") + or ("/" in full_model_name and full_model_name.split("/", 1)[0] in known_providers) + ) + if custom_llm_provider and not has_provider_prefix: full_model_name = f"{custom_llm_provider}/{full_model_name}" optional_params_for_followup: Final = {**optional_params, **patch.optional_params} diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index bfd60bb0f19..9377bdab6c3 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -5494,7 +5494,12 @@ class BaseLLMHTTPHandler: raise ValueError("Agentic loop plan missing patched messages") full_model_name = patch.model or model - if custom_llm_provider and not full_model_name.startswith(f"{custom_llm_provider}/"): + known_providers: Final = getattr(litellm, "provider_list", []) + has_provider_prefix: Final = ( + full_model_name.startswith(f"{custom_llm_provider}/") + or ("/" in full_model_name and full_model_name.split("/", 1)[0] in known_providers) + ) + if custom_llm_provider and not has_provider_prefix: full_model_name = f"{custom_llm_provider}/{full_model_name}" optional_params_for_followup: Final = dict(optional_params) diff --git a/tests/test_litellm/test_agentic_loop_prefix_44069.py b/tests/test_litellm/test_agentic_loop_prefix_44069.py index c3c5c7ddc7b..74c78f082cb 100644 --- a/tests/test_litellm/test_agentic_loop_prefix_44069.py +++ b/tests/test_litellm/test_agentic_loop_prefix_44069.py @@ -25,8 +25,9 @@ from litellm.types.utils import ModelResponse ("zai-org/GLM-5.3-Flash", "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), ("hosted_vllm/zai-org/GLM-5.3-Flash", "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), ("hosted_vllm/zai-org/GLM-5.3-Flash", "", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ("openai/gpt-4o", "hosted_vllm", "openai/gpt-4o"), ), - ids=("organization-model", "already-prefixed", "no-provider"), + ids=("organization-model", "already-prefixed", "no-provider", "cross-provider"), ) async def test_agentic_followup_preserves_provider_prefix( execution_path: Literal["http", "sdk"], @@ -75,8 +76,8 @@ async def test_agentic_followup_preserves_provider_prefix( ) ) - assert isinstance(response, ModelResponse) - assert response.model == expected_model.removeprefix("hosted_vllm/") followup.assert_awaited_once() assert followup.await_args is not None assert followup.await_args.kwargs["model"] == expected_model + assert isinstance(response, ModelResponse) + assert response.model == expected_model.split("/", 1)[1] From dba346d816d535e9e119cda51fb7c0e6b982db1c Mon Sep 17 00:00:00 2001 From: Vrajesh Sulakhe Date: Sun, 4 Oct 2026 07:22:12 +0530 Subject: [PATCH 3/5] fix(agentic): keep provider prefix for org-named models on follow-up calls --- .../agentic_followup_kwargs.py | 17 +++++++++ .../chat_completion_agentic_loop.py | 17 ++++----- litellm/llms/custom_httpx/llm_http_handler.py | 17 ++++----- .../test_agentic_loop_prefix_44069.py | 35 ++++++++++++------- 4 files changed, 57 insertions(+), 29 deletions(-) diff --git a/litellm/litellm_core_utils/agentic_followup_kwargs.py b/litellm/litellm_core_utils/agentic_followup_kwargs.py index d9ffa9a9582..9fc96f36b36 100644 --- a/litellm/litellm_core_utils/agentic_followup_kwargs.py +++ b/litellm/litellm_core_utils/agentic_followup_kwargs.py @@ -4,6 +4,23 @@ from types import MappingProxyType from typing import Final +def resolve_agentic_followup_model( + *, + request_model: str, + patch_model: str | None, + custom_llm_provider: str, + known_providers: Collection[str], +) -> str: + """The request model is already provider-stripped, so a leading "openai/" there is an org name. + Only a callback-supplied model may carry its own provider prefix and switch providers""" + model: Final = patch_model or request_model + if not custom_llm_provider or model.startswith(f"{custom_llm_provider}/"): + return model + if patch_model and "/" in patch_model and patch_model.split("/", 1)[0] in known_providers: + return patch_model + return f"{custom_llm_provider}/{model}" + + def build_agentic_followup_kwargs( *, request_kwargs: Mapping[str, object], diff --git a/litellm/litellm_core_utils/chat_completion_agentic_loop.py b/litellm/litellm_core_utils/chat_completion_agentic_loop.py index b9d03e0cc8a..052663da1a3 100644 --- a/litellm/litellm_core_utils/chat_completion_agentic_loop.py +++ b/litellm/litellm_core_utils/chat_completion_agentic_loop.py @@ -8,7 +8,10 @@ from typing import Final, cast from litellm._logging import verbose_logger from litellm.integrations.custom_logger import CustomLogger -from litellm.litellm_core_utils.agentic_followup_kwargs import build_agentic_followup_kwargs +from litellm.litellm_core_utils.agentic_followup_kwargs import ( + build_agentic_followup_kwargs, + resolve_agentic_followup_model, +) from litellm.litellm_core_utils.agentic_loop_settings import ( DEFAULT_MAX_AGENTIC_LOOPS, validated_max_agentic_loops, @@ -170,14 +173,12 @@ async def _execute_chat_completion_agentic_plan( if patch.messages is None: raise ValueError("Agentic loop plan missing patched messages") - full_model_name = patch.model or model - known_providers: Final = getattr(litellm, "provider_list", []) - has_provider_prefix: Final = ( - full_model_name.startswith(f"{custom_llm_provider}/") - or ("/" in full_model_name and full_model_name.split("/", 1)[0] in known_providers) + full_model_name: Final = resolve_agentic_followup_model( + request_model=model, + patch_model=patch.model, + custom_llm_provider=custom_llm_provider, + known_providers=litellm.provider_list, ) - if custom_llm_provider and not has_provider_prefix: - full_model_name = f"{custom_llm_provider}/{full_model_name}" optional_params_for_followup: Final = {**optional_params, **patch.optional_params} if patch.tools is not None: diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 9309640d027..5ea9e1915fa 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -37,7 +37,10 @@ from litellm._logging import _redact_string, verbose_logger from litellm.anthropic_beta_headers_manager import update_headers_with_filtered_beta from litellm.constants import MAX_FILE_LIST_LIMIT, REALTIME_WEBSOCKET_MAX_MESSAGE_SIZE_BYTES from litellm.files.types import FileContentStreamingResult -from litellm.litellm_core_utils.agentic_followup_kwargs import build_agentic_followup_kwargs +from litellm.litellm_core_utils.agentic_followup_kwargs import ( + build_agentic_followup_kwargs, + resolve_agentic_followup_model, +) from litellm.litellm_core_utils.agentic_loop_settings import ( DEFAULT_MAX_AGENTIC_LOOPS, validated_max_agentic_loops, @@ -5685,14 +5688,12 @@ class BaseLLMHTTPHandler: if patch.messages is None: raise ValueError("Agentic loop plan missing patched messages") - full_model_name = patch.model or model - known_providers: Final = getattr(litellm, "provider_list", []) - has_provider_prefix: Final = ( - full_model_name.startswith(f"{custom_llm_provider}/") - or ("/" in full_model_name and full_model_name.split("/", 1)[0] in known_providers) + full_model_name: Final = resolve_agentic_followup_model( + request_model=model, + patch_model=patch.model, + custom_llm_provider=custom_llm_provider, + known_providers=litellm.provider_list, ) - if custom_llm_provider and not has_provider_prefix: - full_model_name = f"{custom_llm_provider}/{full_model_name}" optional_params_for_followup: Final = dict(optional_params) optional_params_for_followup.update(patch.optional_params) diff --git a/tests/test_litellm/test_agentic_loop_prefix_44069.py b/tests/test_litellm/test_agentic_loop_prefix_44069.py index 74c78f082cb..708b6561a83 100644 --- a/tests/test_litellm/test_agentic_loop_prefix_44069.py +++ b/tests/test_litellm/test_agentic_loop_prefix_44069.py @@ -18,29 +18,38 @@ from litellm.types.utils import ModelResponse @pytest.mark.asyncio @pytest.mark.parametrize("execution_path", ("http", "sdk")) -@pytest.mark.parametrize("use_patch_model", (False, True), ids=("original-model", "patch-model")) @pytest.mark.parametrize( - ("model_name", "custom_llm_provider", "expected_model"), + ("request_model", "patch_model", "custom_llm_provider", "expected_model"), ( - ("zai-org/GLM-5.3-Flash", "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), - ("hosted_vllm/zai-org/GLM-5.3-Flash", "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), - ("hosted_vllm/zai-org/GLM-5.3-Flash", "", "hosted_vllm/zai-org/GLM-5.3-Flash"), - ("openai/gpt-4o", "hosted_vllm", "openai/gpt-4o"), + ("zai-org/GLM-5.3-Flash", None, "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ("hosted_vllm/zai-org/GLM-5.3-Flash", None, "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ("hosted_vllm/zai-org/GLM-5.3-Flash", None, "", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ("openai/whisper-large-v3", None, "hosted_vllm", "hosted_vllm/openai/whisper-large-v3"), + ("original-model", "zai-org/GLM-5.3-Flash", "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ("original-model", "hosted_vllm/zai-org/GLM-5.3-Flash", "hosted_vllm", "hosted_vllm/zai-org/GLM-5.3-Flash"), + ("original-model", "openai/gpt-4o", "hosted_vllm", "openai/gpt-4o"), + ), + ids=( + "organization-model", + "already-prefixed", + "no-provider", + "organization-named-after-provider", + "patch-organization-model", + "patch-already-prefixed", + "patch-cross-provider", ), - ids=("organization-model", "already-prefixed", "no-provider", "cross-provider"), ) async def test_agentic_followup_preserves_provider_prefix( execution_path: Literal["http", "sdk"], - use_patch_model: bool, - model_name: str, + request_model: str, + patch_model: str | None, custom_llm_provider: str, expected_model: str, ) -> None: - model: Final = "original-model" if use_patch_model else model_name plan: Final = AgenticLoopPlan( run_agentic_loop=True, request_patch=AgenticLoopRequestPatch( - model=model_name if use_patch_model else None, + model=patch_model, messages=[{"role": "user", "content": "Continue"}], ), ) @@ -50,7 +59,7 @@ async def test_agentic_followup_preserves_provider_prefix( response: Final = ( await BaseLLMHTTPHandler()._execute_chat_completion_agentic_plan( plan=plan, - model=model, + model=request_model, messages=[], optional_params={"mock_response": "Follow-up complete"}, kwargs={"custom_llm_provider": custom_llm_provider}, @@ -64,7 +73,7 @@ async def test_agentic_followup_preserves_provider_prefix( else await _execute_chat_completion_agentic_plan( plan=plan, callback=CustomLogger(), - model=model, + model=request_model, optional_params={"mock_response": "Follow-up complete"}, kwargs={"custom_llm_provider": custom_llm_provider}, logging_obj=None, From 746b18f0a894e874b2dc9d9aef3be60897e9428a Mon Sep 17 00:00:00 2001 From: Vrajesh Sulakhe Date: Sun, 4 Oct 2026 08:33:50 +0530 Subject: [PATCH 4/5] test(agentic): relocate regression test to mirror core utils module path --- .../test_chat_completion_agentic_loop_44069.py} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename tests/test_litellm/{test_agentic_loop_prefix_44069.py => litellm_core_utils/test_chat_completion_agentic_loop_44069.py} (100%) diff --git a/tests/test_litellm/test_agentic_loop_prefix_44069.py b/tests/test_litellm/litellm_core_utils/test_chat_completion_agentic_loop_44069.py similarity index 100% rename from tests/test_litellm/test_agentic_loop_prefix_44069.py rename to tests/test_litellm/litellm_core_utils/test_chat_completion_agentic_loop_44069.py From 58d0dc37d6ea704509740d25f1d089181811b517 Mon Sep 17 00:00:00 2001 From: Vrajesh Sulakhe Date: Sun, 4 Oct 2026 09:53:35 +0530 Subject: [PATCH 5/5] Revert "test(agentic): relocate regression test to mirror core utils module path" This reverts commit 746b18f0a894e874b2dc9d9aef3be60897e9428a. --- ...on_agentic_loop_44069.py => test_agentic_loop_prefix_44069.py} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename tests/test_litellm/{litellm_core_utils/test_chat_completion_agentic_loop_44069.py => test_agentic_loop_prefix_44069.py} (100%) diff --git a/tests/test_litellm/litellm_core_utils/test_chat_completion_agentic_loop_44069.py b/tests/test_litellm/test_agentic_loop_prefix_44069.py similarity index 100% rename from tests/test_litellm/litellm_core_utils/test_chat_completion_agentic_loop_44069.py rename to tests/test_litellm/test_agentic_loop_prefix_44069.py