From e9424bf3acb43708175187ce4174b5a510bf5d1b Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 28 Feb 2026 08:22:32 +0000 Subject: [PATCH] fix(test): pre-populate WatsonX IAM token cache to prevent parallel test interference The watsonx prompt transformation test was failing in parallel execution because litellm.module_level_client.post mock was being interfered with by other tests. Pre-populating the IAM token cache avoids the HTTP call entirely. Co-authored-by: Ishaan Jaff --- tests/test_litellm/llms/watsonx/test_watsonx.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/tests/test_litellm/llms/watsonx/test_watsonx.py b/tests/test_litellm/llms/watsonx/test_watsonx.py index 4360a154cfa..17c006f5548 100644 --- a/tests/test_litellm/llms/watsonx/test_watsonx.py +++ b/tests/test_litellm/llms/watsonx/test_watsonx.py @@ -209,7 +209,7 @@ def test_watsonx_completion_regular_model_includes_model_id( @pytest.mark.asyncio @pytest.mark.xdist_group("watsonx_heavy") -async def test_watsonx_gpt_oss_prompt_transformation(monkeypatch): +async def test_watsonx_gpt_oss_prompt_transformation(monkeypatch): # noqa: PLR0915 """ Test that gpt-oss-120b model transforms messages to proper format instead of simple concatenation. @@ -310,6 +310,11 @@ async def test_watsonx_gpt_oss_prompt_transformation(monkeypatch): } mock_token_get_response.raise_for_status = Mock() + # Pre-populate the WatsonX IAM token cache to avoid any HTTP calls for token generation. + # This prevents parallel test interference with litellm.module_level_client. + from litellm.llms.watsonx.common_utils import iam_token_cache + iam_token_cache.set_cache(key="test_api_key", value="mock_access_token", ttl=3600) + with patch.object(client, "post", side_effect=mock_post_func) as mock_post, patch.object( litellm.module_level_client, "post", return_value=mock_token_get_response ), patch(