diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 2e7cbf4d039..45d1ec131c2 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -15,6 +15,7 @@ from litellm.constants import ( DEFAULT_MAX_LRU_CACHE_SIZE, DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND, ) +from litellm.litellm_core_utils.data_residency import infer_openai_data_residency from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import ( StandardBuiltInToolCostTracking, ) @@ -1302,6 +1303,15 @@ def completion_cost( # noqa: PLR0915 ) region_name = hidden_params.get("region_name", region_name) + # For OpenAI, infer data_residency from api_base if it + # wasn't passed in explicitly. Keeps external + # completion_cost(response) in sync with the internal + # cost calculation (which reads it from litellm_params). + if data_residency is None and custom_llm_provider == "openai": + data_residency = infer_openai_data_residency( + hidden_params.get("api_base") + ) + # For Gemini/Vertex AI responses, trafficType is stored in # provider_specific_fields. Map it to the service_tier used # by the cost key lookup (_priority / _flex suffixes) so that diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index dc6607e7009..667baf65ed3 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -1530,3 +1530,45 @@ def test_data_residency_composes_with_service_tier(_local_model_cost_map): assert priority_base_total > 0 assert priority_eu_total == pytest.approx(priority_base_total * 1.10, rel=1e-9) + + +def test_completion_cost_infers_data_residency_from_hidden_params( + _local_model_cost_map, +): + """External completion_cost(response) infers data_residency from + the api_base recorded on hidden_params, so it matches the internal cost + calculation from the completion flow.""" + from litellm import ModelResponse, completion_cost + from litellm.types.utils import Choices, Message + from litellm.types.utils import Usage as U + + def _resp(api_base): + r = ModelResponse( + id="test", + choices=[ + Choices( + finish_reason="stop", + index=0, + message=Message(content="ok", role="assistant"), + ) + ], + model="gpt-5", + object="chat.completion", + created=0, + usage=U(prompt_tokens=1000, completion_tokens=500, total_tokens=1500), + ) + r._hidden_params = { + "custom_llm_provider": "openai", + "api_base": api_base, + } + return r + + global_cost = completion_cost( + completion_response=_resp("https://api.openai.com/v1") + ) + eu_cost = completion_cost(completion_response=_resp("https://eu.api.openai.com/v1")) + us_cost = completion_cost(completion_response=_resp("https://us.api.openai.com/v1")) + + assert global_cost > 0 + assert eu_cost == pytest.approx(global_cost * 1.10, rel=1e-9) + assert us_cost == pytest.approx(global_cost * 1.10, rel=1e-9)