diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py index 2164a3b82e3..eef206ca667 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_chunk_builder_utils.py @@ -328,6 +328,42 @@ def test_stream_chunk_builder_litellm_usage_chunks(): assert usage.total_tokens == 77 +def test_get_model_from_chunks_azure_model_router(): + """ + Test that _get_model_from_chunks finds the actual model from Azure Model Router chunks. + + Azure Model Router returns the request model (e.g., 'azure-model-router') in the first chunk, + but subsequent chunks contain the actual model (e.g., 'gpt-4.1-nano-2025-04-14'). + This is important for accurate cost calculation. + """ + # First chunk has request model, subsequent chunks have actual model + chunks = [ + {"model": "azure-model-router", "id": "chatcmpl-123", "choices": []}, + {"model": "gpt-4.1-nano-2025-04-14", "id": "chatcmpl-123", "choices": []}, + {"model": "gpt-4.1-nano-2025-04-14", "id": "chatcmpl-123", "choices": []}, + ] + + result = ChunkProcessor._get_model_from_chunks( + chunks=chunks, first_chunk_model="azure-model-router" + ) + + # Should return the actual model, not the request model + assert result == "gpt-4.1-nano-2025-04-14" + + # Test when all chunks have the same model (non-router case) + chunks_same_model = [ + {"model": "gpt-4", "id": "chatcmpl-456", "choices": []}, + {"model": "gpt-4", "id": "chatcmpl-456", "choices": []}, + ] + + result_same = ChunkProcessor._get_model_from_chunks( + chunks=chunks_same_model, first_chunk_model="gpt-4" + ) + + # Should return the first chunk's model when all are the same + assert result_same == "gpt-4" + + def test_stream_chunk_builder_anthropic_web_search(): # Prepare two mocked streaming chunks with usage split across them chunk1 = ModelResponseStream(