Fix test isolation for test_watsonx_gpt_oss_prompt_transformation

Set cached tokenizer config directly and mock both sync and async
tokenizer functions to avoid race conditions when running with
parallel test execution (-n 16).

The issue was that parallel tests could populate the
litellm.known_tokenizer_config cache between clearing it and
when the code checked it. This caused the sync code path to be
used instead of the async path, bypassing the mocked async functions.

Fix:
1. Set cache directly instead of clearing it
2. Also mock sync versions _get_tokenizer_config and _get_chat_template_file

This ensures the test is deterministic regardless of test execution order.
This commit is contained in:
Shin 2026-02-05 06:29:05 +00:00
parent 1017c3a0e4
commit e7f4665091

View file

@ -283,10 +283,19 @@ async def test_watsonx_gpt_oss_prompt_transformation(monkeypatch):
# Return failure to use tokenizer_config instead
return {"status": "failure"}
# Clear any cached tokenizer config for this model to ensure fresh fetch
# Set cached tokenizer config directly to avoid race conditions with parallel tests.
# When running with pytest-xdist (-n 16), another test might populate the cache between
# clearing it and the actual usage. By setting the cache directly, we ensure the correct
# template is always used regardless of test execution order.
hf_model = "openai/gpt-oss-120b"
if hf_model in litellm.known_tokenizer_config:
del litellm.known_tokenizer_config[hf_model]
litellm.known_tokenizer_config[hf_model] = mock_tokenizer_config
# Also create sync mock functions in case the fallback sync path is used
def mock_get_tokenizer_config(hf_model_name: str):
return mock_tokenizer_config
def mock_get_chat_template_file(hf_model_name: str):
return {"status": "failure"}
with patch.object(client, "post") as mock_post, patch.object(
litellm.module_level_client, "post", return_value=mock_token_response
@ -296,6 +305,12 @@ async def test_watsonx_gpt_oss_prompt_transformation(monkeypatch):
), patch(
"litellm.litellm_core_utils.prompt_templates.huggingface_template_handler._aget_chat_template_file",
side_effect=mock_aget_chat_template_file,
), patch(
"litellm.litellm_core_utils.prompt_templates.huggingface_template_handler._get_tokenizer_config",
side_effect=mock_get_tokenizer_config,
), patch(
"litellm.litellm_core_utils.prompt_templates.huggingface_template_handler._get_chat_template_file",
side_effect=mock_get_chat_template_file,
):
# Set the mock to return the completion response
mock_post.return_value = mock_completion_response