mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
Merge pull request #41623 from BerriAI/litellm_lit_8010_mock_response_provider_custom_pricing
fix(mock_completion): keep the resolved provider so router custom pricing resolves for azure_ai deployments
This commit is contained in:
commit
ab03850666
2 changed files with 63 additions and 5 deletions
|
|
@ -999,12 +999,15 @@ def mock_completion(
|
|||
),
|
||||
)
|
||||
|
||||
try:
|
||||
_, custom_llm_provider, _, _ = litellm.utils.get_llm_provider(model=model)
|
||||
if custom_llm_provider is not None:
|
||||
model_response._hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
except Exception:
|
||||
# dont let setting a hidden param block a mock_respose
|
||||
pass
|
||||
else:
|
||||
try:
|
||||
_, inferred_provider, _, _ = litellm.utils.get_llm_provider(model=model)
|
||||
model_response._hidden_params["custom_llm_provider"] = inferred_provider
|
||||
except Exception:
|
||||
# dont let setting a hidden param block a mock_respose
|
||||
pass
|
||||
|
||||
if logging is not None:
|
||||
logging.post_call(
|
||||
|
|
|
|||
|
|
@ -2432,6 +2432,61 @@ def test_mock_completion_usage_falls_back_to_default_without_admission_count():
|
|||
assert response.usage.prompt_tokens == litellm_main.DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT
|
||||
|
||||
|
||||
_AZURE_AI_CUSTOM_PRICED_DEPLOYMENT: Final = {
|
||||
"model_name": "azure-ai-custom-priced",
|
||||
"litellm_params": {
|
||||
"model": "azure_ai/gpt-5.6",
|
||||
"api_key": "mock",
|
||||
"api_base": "https://example.services.ai.azure.com",
|
||||
"mock_response": "ok",
|
||||
"input_cost_per_token": 3e-6,
|
||||
"output_cost_per_token": 7e-6,
|
||||
"cache_read_input_token_cost": 1e-7,
|
||||
"cache_creation_input_token_cost": 5e-7,
|
||||
},
|
||||
"model_info": {"id": "azure-ai-custom-priced-deployment-id"},
|
||||
}
|
||||
|
||||
|
||||
def _expected_custom_price(response: litellm.ModelResponse) -> float:
|
||||
params: Final = _AZURE_AI_CUSTOM_PRICED_DEPLOYMENT["litellm_params"]
|
||||
return (
|
||||
response.usage.prompt_tokens * params["input_cost_per_token"]
|
||||
+ response.usage.completion_tokens * params["output_cost_per_token"]
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("use_async", (False, True))
|
||||
async def test_mock_completion_prices_azure_ai_router_deployment_with_custom_pricing(use_async: bool):
|
||||
router: Final = litellm.Router(model_list=[_AZURE_AI_CUSTOM_PRICED_DEPLOYMENT])
|
||||
messages: Final = [{"role": "user", "content": "hello"}]
|
||||
|
||||
response: Final = (
|
||||
await router.acompletion(model="azure-ai-custom-priced", messages=messages)
|
||||
if use_async
|
||||
else router.completion(model="azure-ai-custom-priced", messages=messages)
|
||||
)
|
||||
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(_expected_custom_price(response))
|
||||
assert response._hidden_params["custom_llm_provider"] == "azure_ai"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "expected_provider"),
|
||||
(("anthropic/claude-sonnet-5", "anthropic"), ("no-such-provider-model", None)),
|
||||
)
|
||||
def test_mock_completion_infers_provider_when_called_directly_without_one(model: str, expected_provider: str | None):
|
||||
response: Final = litellm.mock_completion(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
mock_response="ok",
|
||||
)
|
||||
|
||||
assert response.choices[0].message.content == "ok"
|
||||
assert response._hidden_params.get("custom_llm_provider") == expected_provider
|
||||
|
||||
|
||||
_ADMISSION_INPUT_TOKENS: Final = 51234
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue