fix(cost_calc): update custom_llm_provider when base_model has different provider prefix (#22906)

* fix(cost_calc): update custom_llm_provider when base_model has different provider

When base_model carries a provider prefix that differs from the
deployment provider (e.g. base_model='gemini/gemini-2.0-flash' on an
anthropic/ deployment), the custom_llm_provider was not updated,
causing cost_per_token to build an invalid model key and return 0.

After _select_model_name_for_cost_calc resolves the model name from
base_model, extract the provider prefix and update custom_llm_provider
so the downstream cost lookup uses the correct provider.

Fixes #22257

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>

* fix(cost_calc): gate base_model provider override on custom_pricing and diff check

- Skip custom_llm_provider override when custom_pricing=True (base_model unused)
- Only override when extracted provider differs from current custom_llm_provider
- Add direct completion_cost unit test for cross-provider base_model
- Add same-provider no-regression test (e.g. openai/gpt-4o on openai deployment)

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>

* fix: guard hidden_params override when provider already overridden by base_model

- Add _provider_overridden flag to prevent hidden_params from undoing base_model fix
- Add direct unit test verifying hidden_params doesn't override extracted provider

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>

* fix: gate reasoning_auto_summary injection on supports_reasoning()

When reasoning_auto_summary is globally enabled, the reasoning param
was injected unconditionally for all models including non-reasoning
ones (e.g. gpt-4o-mini), causing OpenAI API errors. Now gated on
supports_reasoning(model, custom_llm_provider) check.

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>

* fix: revert unrelated transformation.py changes causing TypeError

The transformation.py changes pass model= and custom_llm_provider=
kwargs to _map_optional_params_to_responses_api_request() which only
accepts (self, optional_params, responses_api_request) — causing a
TypeError at runtime for every Responses API request.

Reverted to upstream version; cost_calculator.py fix is unaffected.

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>

---------

Co-authored-by: giulio-leone <6887247+giulio-leone@users.noreply.github.com>
Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
Giulio Leone 2026-03-07 03:01:28 +01:00 • committed by GitHub
parent a4fe06134f
commit b622694c90
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 91 additions and 3 deletions

View file

@ -1093,6 +1093,19 @@ def completion_cost( # noqa: PLR0915
router_model_id=router_model_id,
)
# When base_model overrides model and carries its own provider prefix
# (e.g. base_model="gemini/gemini-2.0-flash" on an anthropic deployment),
# align custom_llm_provider so cost_per_token builds the correct key.
# Skip when custom_pricing is True (base_model is ignored in that path).
_provider_overridden = False
if base_model is not None and selected_model is not None and not custom_pricing:
_parts = selected_model.split("/", 1)
if len(_parts) > 1 and _parts[0] in LlmProvidersSet:
extracted = _parts[0]
if extracted != custom_llm_provider:
custom_llm_provider = extracted
_provider_overridden = True
potential_model_names = [
selected_model,
_get_response_model(completion_response),
@ -1174,9 +1187,10 @@ def completion_cost( # noqa: PLR0915
hidden_params = getattr(completion_response, "_hidden_params", None)
if hidden_params is not None:
custom_llm_provider = hidden_params.get(
"custom_llm_provider", custom_llm_provider or None
)
if not _provider_overridden:
custom_llm_provider = hidden_params.get(
"custom_llm_provider", custom_llm_provider or None
)
region_name = hidden_params.get("region_name", region_name)
# For Gemini/Vertex AI responses, trafficType is stored in

View file

@ -2945,3 +2945,77 @@ def test_batch_cost_calculator():
cost = completion_cost(**args)
assert cost > 0
def test_cost_calculator_base_model_cross_provider():
"""
When base_model has a different provider prefix than the deployment,
custom_llm_provider should be updated so cost_per_token builds the
correct model key. Regression test for #22257.
"""
resp = litellm.completion(
model="anthropic/my-custom-deployment",
messages=[{"role": "user", "content": "Hello"}],
base_model="gemini/gemini-2.0-flash",
mock_response="Hi there!",
)
assert resp._hidden_params["response_cost"] > 0
def test_cost_calculator_base_model_cross_provider_direct():
"""
Direct completion_cost unit test for cross-provider base_model override.
Verifies that completion_cost correctly routes to the base_model provider.
"""
from litellm import ModelResponse, Usage
resp = ModelResponse(
id="chatcmpl-test",
model="gemini/gemini-2.0-flash",
usage=Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30),
)
cost = completion_cost(
model="anthropic/my-custom-deployment",
completion_response=resp,
base_model="gemini/gemini-2.0-flash",
custom_llm_provider="anthropic",
)
assert cost > 0
def test_cost_calculator_base_model_cross_provider_hidden_params_guard():
"""
Verify that hidden_params.custom_llm_provider does not undo the
base_model provider override when the response carries a stale provider.
"""
from litellm import ModelResponse, Usage
resp = ModelResponse(
id="chatcmpl-guard",
model="gemini/gemini-2.0-flash",
usage=Usage(prompt_tokens=10, completion_tokens=20, total_tokens=30),
)
# Simulate hidden_params carrying the original (wrong) provider
resp._hidden_params = {"custom_llm_provider": "anthropic"}
cost = completion_cost(
model="anthropic/my-custom-deployment",
completion_response=resp,
base_model="gemini/gemini-2.0-flash",
custom_llm_provider="anthropic",
)
assert cost > 0
def test_cost_calculator_base_model_same_provider_no_regression():
"""
When base_model has the same provider prefix as the deployment,
custom_llm_provider should remain unchanged (no-regression).
"""
resp = litellm.completion(
model="openai/my-custom-deployment",
messages=[{"role": "user", "content": "Hello"}],
base_model="openai/gpt-4o",
mock_response="Hi there!",
)
assert resp._hidden_params["response_cost"] > 0