feat(cost-calc): infer data_residency from response hidden_params

completion_cost(completion_response=resp) is a user-facing entry point
that reruns the cost calculation from the response object alone — it
does not have access to the logging object's litellm_params. Without
this, users calling completion_cost() after a completion would see the
base cost, while the internal cost calculation (via _response_cost_calculator
in litellm_logging.py) would apply the uplift and store it on
_hidden_params["response_cost"]. Symmetry: also read api_base from
hidden_params and infer data_residency there.

https://claude.ai/code/session_012ebH44s7ohYxjoix5CXzTW
This commit is contained in:
mateo-berri 2026-08-17 16:48:43 +00:00
parent 7697518fae
commit 6c313e9b5e
No known key found for this signature in database
2 changed files with 52 additions and 0 deletions

View file

@ -15,6 +15,7 @@ from litellm.constants import (
DEFAULT_MAX_LRU_CACHE_SIZE,
DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND,
)
from litellm.litellm_core_utils.data_residency import infer_openai_data_residency
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
StandardBuiltInToolCostTracking,
)
@ -1302,6 +1303,15 @@ def completion_cost( # noqa: PLR0915
)
region_name = hidden_params.get("region_name", region_name)
# For OpenAI, infer data_residency from api_base if it
# wasn't passed in explicitly. Keeps external
# completion_cost(response) in sync with the internal
# cost calculation (which reads it from litellm_params).
if data_residency is None and custom_llm_provider == "openai":
data_residency = infer_openai_data_residency(
hidden_params.get("api_base")
)
# For Gemini/Vertex AI responses, trafficType is stored in
# provider_specific_fields. Map it to the service_tier used
# by the cost key lookup (_priority / _flex suffixes) so that

View file

@ -1530,3 +1530,45 @@ def test_data_residency_composes_with_service_tier(_local_model_cost_map):
assert priority_base_total > 0
assert priority_eu_total == pytest.approx(priority_base_total * 1.10, rel=1e-9)
def test_completion_cost_infers_data_residency_from_hidden_params(
_local_model_cost_map,
):
"""External completion_cost(response) infers data_residency from
the api_base recorded on hidden_params, so it matches the internal cost
calculation from the completion flow."""
from litellm import ModelResponse, completion_cost
from litellm.types.utils import Choices, Message
from litellm.types.utils import Usage as U
def _resp(api_base):
r = ModelResponse(
id="test",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(content="ok", role="assistant"),
)
],
model="gpt-5",
object="chat.completion",
created=0,
usage=U(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
)
r._hidden_params = {
"custom_llm_provider": "openai",
"api_base": api_base,
}
return r
global_cost = completion_cost(
completion_response=_resp("https://api.openai.com/v1")
)
eu_cost = completion_cost(completion_response=_resp("https://eu.api.openai.com/v1"))
us_cost = completion_cost(completion_response=_resp("https://us.api.openai.com/v1"))
assert global_cost > 0
assert eu_cost == pytest.approx(global_cost * 1.10, rel=1e-9)
assert us_cost == pytest.approx(global_cost * 1.10, rel=1e-9)