Merge pull request #41623 from BerriAI/litellm_lit_8010_mock_response_provider_custom_pricing

fix(mock_completion): keep the resolved provider so router custom pricing resolves for azure_ai deployments
This commit is contained in:
kerry-berri 2026-09-17 11:49:33 -07:00 • committed by GitHub
commit ab03850666
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 63 additions and 5 deletions

View file

@ -999,12 +999,15 @@ def mock_completion(
),
)
try:
_, custom_llm_provider, _, _ = litellm.utils.get_llm_provider(model=model)
if custom_llm_provider is not None:
model_response._hidden_params["custom_llm_provider"] = custom_llm_provider
except Exception:
# dont let setting a hidden param block a mock_respose
pass
else:
try:
_, inferred_provider, _, _ = litellm.utils.get_llm_provider(model=model)
model_response._hidden_params["custom_llm_provider"] = inferred_provider
except Exception:
# dont let setting a hidden param block a mock_respose
pass
if logging is not None:
logging.post_call(

View file

@ -2432,6 +2432,61 @@ def test_mock_completion_usage_falls_back_to_default_without_admission_count():
assert response.usage.prompt_tokens == litellm_main.DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT
_AZURE_AI_CUSTOM_PRICED_DEPLOYMENT: Final = {
"model_name": "azure-ai-custom-priced",
"litellm_params": {
"model": "azure_ai/gpt-5.6",
"api_key": "mock",
"api_base": "https://example.services.ai.azure.com",
"mock_response": "ok",
"input_cost_per_token": 3e-6,
"output_cost_per_token": 7e-6,
"cache_read_input_token_cost": 1e-7,
"cache_creation_input_token_cost": 5e-7,
},
"model_info": {"id": "azure-ai-custom-priced-deployment-id"},
}
def _expected_custom_price(response: litellm.ModelResponse) -> float:
params: Final = _AZURE_AI_CUSTOM_PRICED_DEPLOYMENT["litellm_params"]
return (
response.usage.prompt_tokens * params["input_cost_per_token"]
+ response.usage.completion_tokens * params["output_cost_per_token"]
)
@pytest.mark.asyncio
@pytest.mark.parametrize("use_async", (False, True))
async def test_mock_completion_prices_azure_ai_router_deployment_with_custom_pricing(use_async: bool):
router: Final = litellm.Router(model_list=[_AZURE_AI_CUSTOM_PRICED_DEPLOYMENT])
messages: Final = [{"role": "user", "content": "hello"}]
response: Final = (
await router.acompletion(model="azure-ai-custom-priced", messages=messages)
if use_async
else router.completion(model="azure-ai-custom-priced", messages=messages)
)
assert response._hidden_params["response_cost"] == pytest.approx(_expected_custom_price(response))
assert response._hidden_params["custom_llm_provider"] == "azure_ai"
@pytest.mark.parametrize(
("model", "expected_provider"),
(("anthropic/claude-sonnet-5", "anthropic"), ("no-such-provider-model", None)),
)
def test_mock_completion_infers_provider_when_called_directly_without_one(model: str, expected_provider: str | None):
response: Final = litellm.mock_completion(
model=model,
messages=[{"role": "user", "content": "hello"}],
mock_response="ok",
)
assert response.choices[0].message.content == "ok"
assert response._hidden_params.get("custom_llm_provider") == expected_provider
_ADMISSION_INPUT_TOKENS: Final = 51234