fix(test): pre-populate WatsonX IAM token cache to prevent parallel test interference

The watsonx prompt transformation test was failing in parallel execution because
litellm.module_level_client.post mock was being interfered with by other tests.
Pre-populating the IAM token cache avoids the HTTP call entirely.

Co-authored-by: Ishaan Jaff <ishaan-jaff@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-02-28 08:22:32 +00:00
parent 9b8a7f1f5b
commit e9424bf3ac

View file

@ -209,7 +209,7 @@ def test_watsonx_completion_regular_model_includes_model_id(
@pytest.mark.asyncio
@pytest.mark.xdist_group("watsonx_heavy")
async def test_watsonx_gpt_oss_prompt_transformation(monkeypatch):
async def test_watsonx_gpt_oss_prompt_transformation(monkeypatch): # noqa: PLR0915
"""
Test that gpt-oss-120b model transforms messages to proper format instead of simple concatenation.
@ -310,6 +310,11 @@ async def test_watsonx_gpt_oss_prompt_transformation(monkeypatch):
}
mock_token_get_response.raise_for_status = Mock()
# Pre-populate the WatsonX IAM token cache to avoid any HTTP calls for token generation.
# This prevents parallel test interference with litellm.module_level_client.
from litellm.llms.watsonx.common_utils import iam_token_cache
iam_token_cache.set_cache(key="test_api_key", value="mock_access_token", ttl=3600)
with patch.object(client, "post", side_effect=mock_post_func) as mock_post, patch.object(
litellm.module_level_client, "post", return_value=mock_token_get_response
), patch(