fix(logging): stop billing and logging response reads as LLM calls

Retrieving, deleting or cancelling a stored response, and vector store management calls, run through the same logging lifecycle as inference. A retrieved response replays the usage of the call that created it, so every read priced it again and wrote a second spend log row for the same tokens. Non-inference calls now cost 0, report no usage, log no placeholder chat message, and get a litellm.responses_management operation name instead of reading as chat.

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Devin AI 2026-08-14 05:27:24 +00:00
parent 0a25756e78
commit 6285192141
8 changed files with 200 additions and 2 deletions

View file

@ -1736,6 +1736,33 @@ BROWSER_SECURITY_HEADERS: Final[frozenset[str]] = frozenset(
UNSAFE_PROXY_RESPONSE_HEADERS: Final[frozenset[str]] = HTTP_FRAMING_HEADERS | BROWSER_SECURITY_HEADERS
# Read/management routes that run through the same request processing and logging
# lifecycle as inference, but never send a prompt to a model. Their responses can still
# carry a usage object (a retrieved response replays the usage of the call that created
# it), so pricing them bills the same tokens twice.
NON_INFERENCE_CALL_TYPES: Final[frozenset[str]] = frozenset(
{
"get_responses",
"aget_responses",
"delete_responses",
"adelete_responses",
"cancel_responses",
"acancel_responses",
"list_input_items",
"alist_input_items",
"vector_store_create",
"avector_store_create",
"vector_store_retrieve",
"avector_store_retrieve",
"vector_store_list",
"avector_store_list",
"vector_store_update",
"avector_store_update",
"vector_store_delete",
"avector_store_delete",
}
)
# PTU reservation rollup writes rows to LiteLLM_DailyTeamSpend with this
# sentinel api_key so PTU flat cost stays distinguishable from real per-request
# spend under the table's composite unique constraint.

View file

@ -30,6 +30,7 @@ class GenAIOperation(str, Enum):
EXECUTE_TOOL = "execute_tool" # MCP tool-call spans
LITELLM_VECTOR_STORE_MANAGEMENT = "litellm.vector_store_management"
LITELLM_VECTOR_STORE_FILE_MANAGEMENT = "litellm.vector_store_file_management"
LITELLM_RESPONSES_MANAGEMENT = "litellm.responses_management" # fetch/delete/cancel a stored response
class GenAIProvider(str, Enum):
@ -348,6 +349,14 @@ _OPERATION_BY_CALL_TYPE: Final[dict[str, GenAIOperation]] = {
"aembedding": GenAIOperation.EMBEDDINGS,
"responses": GenAIOperation.CHAT,
"aresponses": GenAIOperation.CHAT,
"get_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
"aget_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
"delete_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
"adelete_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
"cancel_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
"acancel_responses": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
"list_input_items": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
"alist_input_items": GenAIOperation.LITELLM_RESPONSES_MANAGEMENT,
"call_mcp_tool": GenAIOperation.EXECUTE_TOOL,
"vector_store_search": GenAIOperation.RETRIEVAL,
"avector_store_search": GenAIOperation.RETRIEVAL,

View file

@ -41,6 +41,7 @@ from litellm.caching.caching_handler import LLMCachingHandler
from litellm.constants import (
DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT,
DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT,
NON_INFERENCE_CALL_TYPES,
SENTRY_DENYLIST,
SENTRY_PII_DENYLIST,
)
@ -1435,6 +1436,9 @@ class Logging(LiteLLMLoggingBaseClass):
if cache_hit is True:
return 0.0
if self.call_type in NON_INFERENCE_CALL_TYPES:
return 0.0
transformed_result: Final = self._generate_content_result_as_model_response(result)
if transformed_result is not None:
result = transformed_result
@ -5487,7 +5491,7 @@ def get_standard_logging_object_payload(
cache_hit: Final = kwargs.get("cache_hit", False)
# Extract usage as a plain dict, avoiding Pydantic round-trip
raw_usage_dict: Final = StandardLoggingPayloadSetup.get_usage_as_dict(
response_obj=response_obj,
response_obj=None if call_type in NON_INFERENCE_CALL_TYPES else response_obj,
combined_usage_object=cast(Usage | None, kwargs.get("combined_usage_object")),
)
usage_dict: Final = (

View file

@ -14,6 +14,7 @@ from litellm._logging import verbose_proxy_logger
from litellm.constants import (
LITELLM_TRUNCATED_PAYLOAD_FIELD,
LITELLM_TRUNCATION_DB_SAFEGUARD_NOTE,
NON_INFERENCE_CALL_TYPES,
REDACTED_BY_LITELM_STRING,
)
from litellm.constants import (
@ -258,7 +259,7 @@ def get_logging_payload(kwargs, response_obj, start_time, end_time) -> SpendLogs
usage: dict = {}
if call_type in ["ocr", "aocr"]:
usage = _extract_usage_for_ocr_call(response_obj, response_obj_dict)
else:
elif call_type not in NON_INFERENCE_CALL_TYPES:
# Use response_obj_dict instead of response_obj to avoid calling .get() on Pydantic models
_usage: Final = response_obj_dict.get("usage", None) or {}
if isinstance(_usage, litellm.Usage):

View file

@ -72,6 +72,7 @@ from litellm.constants import (
MAX_RETRY_DELAY,
MAX_TOKEN_TRIMMING_ATTEMPTS,
MINIMUM_PROMPT_CACHE_TOKEN_COUNT_OVERRIDE,
NON_INFERENCE_CALL_TYPES,
OPENAI_EMBEDDING_PARAMS,
TOOL_CHOICE_OBJECT_TOKEN_COUNT,
)
@ -1070,6 +1071,8 @@ def function_setup(
except Exception as e:
verbose_logger.debug("Error extracting messages from Google contents: %s", e)
messages = "default-message-value"
elif call_type in NON_INFERENCE_CALL_TYPES:
messages = ()
else:
messages = "default-message-value"
stream = False

View file

@ -264,6 +264,27 @@ def test_vector_store_file_management_is_not_chat(call_type):
assert resolve_operation(call_type).value == "litellm.vector_store_file_management"
@pytest.mark.parametrize(
"call_type",
[
f"{prefix}{operation}"
for operation in ("get_responses", "delete_responses", "cancel_responses", "list_input_items")
for prefix in ("", "a")
],
)
def test_responses_management_is_not_chat(call_type):
"""Fetching, deleting or cancelling a stored response runs no inference, so it must not
read as a chat completion: the retrieved object replays the original call's tokens and
would inflate the chat series on every read. Regression test for LIT-5602."""
assert resolve_operation(call_type) is GenAIOperation.LITELLM_RESPONSES_MANAGEMENT
assert resolve_operation(call_type).value == "litellm.responses_management"
def test_creating_a_response_is_still_chat():
"""Guards the test above: ``/v1/responses`` itself is a chat completion."""
assert resolve_operation("aresponses") is GenAIOperation.CHAT
def test_vendor_operation_values_are_namespaced():
"""A vendor value must stay under the ``litellm.`` prefix: an unprefixed invented
name could collide with a value the convention adds later, silently changing what

View file

@ -4539,3 +4539,93 @@ async def test_restore_correlation_context_works_across_asyncio_task_boundary():
finally:
trace_id_var.set("")
session_id_var.set("")
class TestNonInferenceCallTypesAreNotBilled:
"""A retrieved response replays the usage of the call that created it, so pricing a read
of it double bills the same tokens. Regression tests for LIT-5602."""
RETRIEVED_RESPONSE_USAGE = {"input_tokens": 4000, "output_tokens": 2000, "total_tokens": 6000}
def _logging_obj(self, call_type: str):
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
obj = LiteLLMLoggingObj(
model="gpt-4o",
messages=[],
stream=False,
call_type=call_type,
start_time=time.time(),
litellm_call_id=f"lit5602-{call_type}",
function_id="fn-lit5602",
)
obj.update_environment_variables(
model="gpt-4o",
user="",
optional_params={},
litellm_params={"api_base": "", "custom_llm_provider": "openai"},
)
return obj
def _retrieved_response(self):
from litellm.types.llms.openai import ResponsesAPIResponse
return ResponsesAPIResponse(
id="resp_lit5602",
created_at=1234567890,
model="gpt-4o",
output=[],
usage=self.RETRIEVED_RESPONSE_USAGE,
)
def test_creating_a_response_is_still_priced(self):
"""Guards the tests below: the same response object must cost money on the create path."""
cost = self._logging_obj("aresponses")._response_cost_calculator(result=self._retrieved_response())
assert cost is not None and cost > 0
@pytest.mark.parametrize(
"call_type",
["aget_responses", "adelete_responses", "acancel_responses", "alist_input_items", "avector_store_delete"],
)
def test_read_and_management_calls_cost_nothing(self, call_type):
cost = self._logging_obj(call_type)._response_cost_calculator(result=self._retrieved_response())
assert cost == 0.0
def test_retrieved_usage_is_not_re_reported_in_standard_logging_payload(self):
from litellm.litellm_core_utils.litellm_logging import (
get_standard_logging_object_payload,
)
from datetime import datetime
logging_obj = self._logging_obj("aget_responses")
now = datetime.now()
payload = get_standard_logging_object_payload(
kwargs={
"litellm_call_id": "lit5602-payload",
"model": "gpt-4o",
"call_type": "aget_responses",
"litellm_params": {},
},
init_response_obj=self._retrieved_response(),
start_time=now,
end_time=now,
logging_obj=logging_obj,
status="success",
)
assert payload is not None
assert payload["prompt_tokens"] == 0
assert payload["completion_tokens"] == 0
assert payload["total_tokens"] == 0
assert payload["response_cost"] == 0.0
def test_read_calls_do_not_log_a_placeholder_chat_message(self):
logging_obj, _ = litellm.utils.function_setup(
original_function="aget_responses",
rules_obj=litellm.utils.Rules(),
start_time=time.time(),
**{"litellm_call_id": "lit5602-setup", "response_id": "resp_lit5602"},
)
assert logging_obj.model_call_details["messages"] == ()

View file

@ -2959,3 +2959,46 @@ def test_user_traffic_carries_no_internal_call_origin():
)
metadata = json.loads(payload["metadata"])
assert metadata["internal_call_origin"] is None
def _spend_log_for_call_type(call_type: str) -> dict:
from litellm.types.llms.openai import ResponsesAPIResponse
return cast(
dict,
get_logging_payload(
kwargs={
"model": "gpt-4o",
"call_type": call_type,
"response_cost": 0.0,
"litellm_params": {"metadata": {"user_api_key": "test-key"}},
},
response_obj=ResponsesAPIResponse(
id="resp_lit5602",
created_at=1234567890,
model="gpt-4o",
output=[],
usage={"input_tokens": 4000, "output_tokens": 2000, "total_tokens": 6000},
),
start_time=datetime.datetime.now(timezone.utc),
end_time=datetime.datetime.now(timezone.utc),
),
)
def test_spend_log_for_response_retrieval_does_not_replay_the_created_responses_tokens():
"""A retrieved response carries the usage of the call that created it, so counting it again
bills the same tokens twice. Regression test for LIT-5602."""
payload = _spend_log_for_call_type("aget_responses")
assert payload["prompt_tokens"] == 0
assert payload["completion_tokens"] == 0
assert payload["total_tokens"] == 0
assert payload["spend"] == 0.0
def test_spend_log_for_response_creation_still_counts_tokens():
"""Guards the test above: the same response object must still be counted on the create path."""
payload = _spend_log_for_call_type("aresponses")
assert payload["total_tokens"] == 6000