mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
test(spend-tracking): add regression coverage for uncovered spend paths
Spend tracking is already heavily tested, but a few high-value paths had no direct coverage. Add focused, mutation-surviving regression tests: - _get_status_for_spend_log: failure-status tracking had zero tests - cache-hit cost zeroing in _PROXY_track_cost_callback (anti double-charge) - cache-hit unique request_id suffix in get_logging_payload (SpendLogs duplicate-key collision fix) - failure-status and zero-spend wiring through get_logging_payload - embedding completion_cost (no dedicated embedding cost test existed) Each test fails if its guard is mutated (verified by hand-mutation). Also add a spend tracking test runbook documenting the offline suite plus an operator-run live e2e check against real provider APIs and SpendLogs rows.
This commit is contained in:
parent
3448bf79f8
commit
e6db497fb8
5 changed files with 285 additions and 1 deletions
|
|
@ -254,6 +254,55 @@ async def test_track_cost_callback_releases_budget_reservation_when_spend_tracki
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_track_cost_callback_zeroes_response_cost_on_cache_hit():
|
||||
logger = _ProxyDBLogger()
|
||||
kwargs = {
|
||||
"model": "gpt-4",
|
||||
"call_type": "acompletion",
|
||||
"cache_hit": True,
|
||||
"litellm_params": {
|
||||
"metadata": {
|
||||
"user_api_key": "hashed-key",
|
||||
"user_api_key_user_id": "test_user",
|
||||
"user_api_key_team_id": "test_team",
|
||||
}
|
||||
},
|
||||
"standard_logging_object": {
|
||||
"response_cost": 0.25,
|
||||
"request_tags": None,
|
||||
},
|
||||
"stream": False,
|
||||
}
|
||||
|
||||
mock_proxy_logging = MagicMock()
|
||||
mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock()
|
||||
mock_proxy_logging.slack_alerting_instance.customer_spend_alert = AsyncMock()
|
||||
|
||||
with (
|
||||
patch("litellm.proxy.proxy_server.proxy_logging_obj", mock_proxy_logging),
|
||||
patch(
|
||||
"litellm.proxy.proxy_server.increment_spend_counters",
|
||||
new_callable=AsyncMock,
|
||||
),
|
||||
patch("litellm.proxy.proxy_server.update_cache", new_callable=AsyncMock),
|
||||
):
|
||||
await logger._PROXY_track_cost_callback(
|
||||
kwargs=kwargs,
|
||||
completion_response=None,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
mock_proxy_logging.db_spend_update_writer.update_database.assert_awaited_once()
|
||||
recorded_cost = (
|
||||
mock_proxy_logging.db_spend_update_writer.update_database.await_args.kwargs[
|
||||
"response_cost"
|
||||
]
|
||||
)
|
||||
assert recorded_cost == 0.0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_track_cost_callback_releases_budget_reservation_when_response_cost_missing():
|
||||
logger = _ProxyDBLogger()
|
||||
|
|
|
|||
|
|
@ -0,0 +1,102 @@
|
|||
# Spend Tracking Test Runbook
|
||||
|
||||
This runbook documents the regression coverage that guards litellm's spend/cost
|
||||
tracking against silent breakage before a release. It has two halves: a set of
|
||||
deterministic offline unit tests (Part A) that run in CI, and an operator-driven
|
||||
live end-to-end check (Part B) that hits real provider APIs and asserts real
|
||||
SpendLogs rows. Spend tracking is already heavily tested across the codebase
|
||||
(cost calculators, SpendLogsPayload construction, the proxy cost callback, the DB
|
||||
spend-update writer, daily-spend queues); the tests below fill the specific
|
||||
high-value gaps that previously had no direct coverage.
|
||||
|
||||
## Part A: offline regression tests (CI)
|
||||
|
||||
Run them with:
|
||||
|
||||
```bash
|
||||
uv run --no-sync python -m pytest \
|
||||
tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py \
|
||||
tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py \
|
||||
tests/test_litellm/test_cost_calculator.py::test_embedding_completion_cost_uses_input_cost_per_token \
|
||||
-q
|
||||
```
|
||||
|
||||
Each test is written to fail if the specific guard it covers is mutated; the
|
||||
"What it guards" column names the source line that, when broken, turns the test
|
||||
red. Verified by hand-mutation (revert each guard and watch the row go red).
|
||||
|
||||
| Test name | What it guards | Status |
|
||||
| --- | --- | --- |
|
||||
| `TestGetStatusForSpendLog::test_missing_status_key_defaults_to_success` | `_get_status_for_spend_log` default branch (spend_tracking_utils.py) returns success when no status set | pass |
|
||||
| `TestGetStatusForSpendLog::test_explicit_success_returns_success` | same helper, explicit success preserved | pass |
|
||||
| `TestGetStatusForSpendLog::test_failure_returns_failure` | the `== "failure"` guard so failed requests are logged as failures | pass |
|
||||
| `TestGetStatusForSpendLog::test_non_failure_value_returns_success` | kills the "any non-None status -> failure" mutant | pass |
|
||||
| `test_get_logging_payload_cache_hit_appends_unique_suffix_to_request_id` | cache-hit `_cache_hit{time}` suffix on request_id; without it SpendLogs hits duplicate-key collisions | pass |
|
||||
| `test_get_logging_payload_failure_status_and_zero_spend` | status wiring at the `get_logging_payload` call site plus `spend` sourced from `response_cost` | pass |
|
||||
| `test_get_logging_payload_default_status_success` | default status path through `get_logging_payload` | pass |
|
||||
| `test_track_cost_callback_zeroes_response_cost_on_cache_hit` | cache-hit cost zeroing in `_PROXY_track_cost_callback`; the anti double-charge guard | pass |
|
||||
| `test_embedding_completion_cost_uses_input_cost_per_token` | embedding cost = `prompt_tokens * input_cost_per_token`; previously no dedicated embedding cost test | pass |
|
||||
|
||||
All 9 pass on `claude/spend-tracking-tests-4gbezl`.
|
||||
|
||||
## Part B: live end-to-end check (operator-run, real spend logs)
|
||||
|
||||
This proves spend tracking against real provider responses and a real database,
|
||||
which is the closest mirror of what a customer sees. It needs a Postgres
|
||||
`DATABASE_URL`, an `OPENAI_API_KEY` in `.env`, and outbound access to
|
||||
api.openai.com. It uses `gpt-5.4-nano` (current cheap small model as of 2026-06)
|
||||
and `text-embedding-3-small`, with local caching enabled so a repeated identical
|
||||
chat request produces a cache hit.
|
||||
|
||||
Config: `tests/test_litellm/proxy/spend_tracking/e2e_spend_config.yaml`
|
||||
|
||||
1. Point at a database and start the proxy (spend logs require a DB):
|
||||
|
||||
```bash
|
||||
export DATABASE_URL='postgresql://user:pass@localhost:5432/litellm'
|
||||
python litellm/proxy/proxy_cli.py \
|
||||
--config tests/test_litellm/proxy/spend_tracking/e2e_spend_config.yaml \
|
||||
--detailed_debug 2>&1 | tee litellm.log
|
||||
```
|
||||
|
||||
2. First chat call (real cost expected):
|
||||
|
||||
```bash
|
||||
curl -s http://localhost:4000/v1/chat/completions \
|
||||
-H 'Authorization: Bearer sk-1234' -H 'Content-Type: application/json' \
|
||||
-d '{"model":"gpt-5.4-nano","messages":[{"role":"user","content":"say hello in one word"}]}' | jq '{id, usage}'
|
||||
```
|
||||
|
||||
3. Identical chat call again to trigger a cache hit (cost should be recorded as 0):
|
||||
|
||||
```bash
|
||||
curl -s http://localhost:4000/v1/chat/completions \
|
||||
-H 'Authorization: Bearer sk-1234' -H 'Content-Type: application/json' \
|
||||
-d '{"model":"gpt-5.4-nano","messages":[{"role":"user","content":"say hello in one word"}]}' | jq '{id, usage}'
|
||||
```
|
||||
|
||||
4. Embedding call (real cost expected):
|
||||
|
||||
```bash
|
||||
curl -s http://localhost:4000/v1/embeddings \
|
||||
-H 'Authorization: Bearer sk-1234' -H 'Content-Type: application/json' \
|
||||
-d '{"model":"text-embedding-3-small","input":"hello world"}' | jq '{model, usage}'
|
||||
```
|
||||
|
||||
5. Wait for the spend-log flush (the proxy batches writes roughly once a minute),
|
||||
then read the logs back and assert the three rows:
|
||||
|
||||
```bash
|
||||
sleep 65
|
||||
curl -s 'http://localhost:4000/spend/logs' -H 'Authorization: Bearer sk-1234' \
|
||||
| jq '[.[] | {request_id, call_type, model, spend, cache_hit}]'
|
||||
```
|
||||
|
||||
Expected: the chat row has `spend > 0`, the embedding row has `spend > 0`, and the
|
||||
cache-hit row has `spend == 0` with a `request_id` containing `_cache_hit`.
|
||||
|
||||
Status: not run in this sandbox. The container has no `.env`/`OPENAI_API_KEY`, no
|
||||
`DATABASE_URL`, and egress to api.openai.com is blocked by the environment's
|
||||
network policy (a billed call is not possible here). Run the steps above on a host
|
||||
with credentials and network access to fill this status; the offline suite in
|
||||
Part A is what gates CI
|
||||
|
|
@ -0,0 +1,17 @@
|
|||
model_list:
|
||||
- model_name: gpt-5.4-nano
|
||||
litellm_params:
|
||||
model: openai/gpt-5.4-nano
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: text-embedding-3-small
|
||||
litellm_params:
|
||||
model: openai/text-embedding-3-small
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
cache: true
|
||||
cache_params:
|
||||
type: local
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234
|
||||
|
|
@ -28,6 +28,7 @@ from litellm.proxy.spend_tracking.spend_tracking_utils import (
|
|||
_get_request_duration_ms,
|
||||
_get_response_for_spend_logs_payload,
|
||||
_get_spend_logs_metadata,
|
||||
_get_status_for_spend_log,
|
||||
_get_vector_store_request_for_spend_logs_payload,
|
||||
_is_master_key,
|
||||
_redact_prompt_leaks_in_error_string,
|
||||
|
|
@ -2073,3 +2074,88 @@ def test_sanitize_error_information_redacts_pydantic_assignment_form(
|
|||
assert sanitized is not None
|
||||
assert "leaked-via-pydantic-msg" not in sanitized["error_message"]
|
||||
assert REDACTED_BY_LITELM_STRING in sanitized["error_message"]
|
||||
|
||||
|
||||
class TestGetStatusForSpendLog:
|
||||
def test_missing_status_key_defaults_to_success(self):
|
||||
assert _get_status_for_spend_log(metadata={}) == "success"
|
||||
|
||||
def test_explicit_success_returns_success(self):
|
||||
assert _get_status_for_spend_log(metadata={"status": "success"}) == "success"
|
||||
|
||||
def test_failure_returns_failure(self):
|
||||
assert _get_status_for_spend_log(metadata={"status": "failure"}) == "failure"
|
||||
|
||||
def test_non_failure_value_returns_success(self):
|
||||
assert _get_status_for_spend_log(metadata={"status": "error"}) == "success"
|
||||
|
||||
|
||||
@patch("litellm.proxy.proxy_server.master_key", None)
|
||||
@patch("litellm.proxy.proxy_server.general_settings", {})
|
||||
def test_get_logging_payload_cache_hit_appends_unique_suffix_to_request_id():
|
||||
base_response_id = "chatcmpl-deterministic-base-id"
|
||||
response_obj = {
|
||||
"id": base_response_id,
|
||||
"choices": [{"message": {"content": "hi"}}],
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 3, "total_tokens": 8},
|
||||
}
|
||||
kwargs = {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"litellm_params": {"metadata": {"user_api_key": "sk-test-key"}},
|
||||
}
|
||||
start_time = datetime.datetime.now(timezone.utc)
|
||||
end_time = datetime.datetime.now(timezone.utc)
|
||||
|
||||
no_cache_payload = get_logging_payload(
|
||||
kwargs={**kwargs},
|
||||
response_obj=response_obj,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
)
|
||||
assert no_cache_payload["request_id"] == base_response_id
|
||||
|
||||
cache_payload = get_logging_payload(
|
||||
kwargs={**kwargs, "cache_hit": True},
|
||||
response_obj=response_obj,
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
)
|
||||
assert cache_payload["request_id"].startswith(base_response_id)
|
||||
assert "_cache_hit" in cache_payload["request_id"]
|
||||
assert cache_payload["request_id"] != base_response_id
|
||||
|
||||
|
||||
@patch("litellm.proxy.proxy_server.master_key", None)
|
||||
@patch("litellm.proxy.proxy_server.general_settings", {})
|
||||
def test_get_logging_payload_failure_status_and_zero_spend():
|
||||
kwargs = {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"response_cost": 0.0,
|
||||
"litellm_params": {
|
||||
"metadata": {"user_api_key": "sk-test-key", "status": "failure"}
|
||||
},
|
||||
}
|
||||
payload = get_logging_payload(
|
||||
kwargs=kwargs,
|
||||
response_obj={},
|
||||
start_time=datetime.datetime.now(timezone.utc),
|
||||
end_time=datetime.datetime.now(timezone.utc),
|
||||
)
|
||||
assert payload["status"] == "failure"
|
||||
assert payload["spend"] == 0.0
|
||||
|
||||
|
||||
@patch("litellm.proxy.proxy_server.master_key", None)
|
||||
@patch("litellm.proxy.proxy_server.general_settings", {})
|
||||
def test_get_logging_payload_default_status_success():
|
||||
kwargs = {
|
||||
"model": "gpt-3.5-turbo",
|
||||
"litellm_params": {"metadata": {"user_api_key": "sk-test-key"}},
|
||||
}
|
||||
payload = get_logging_payload(
|
||||
kwargs=kwargs,
|
||||
response_obj={},
|
||||
start_time=datetime.datetime.now(timezone.utc),
|
||||
end_time=datetime.datetime.now(timezone.utc),
|
||||
)
|
||||
assert payload["status"] == "success"
|
||||
|
|
|
|||
|
|
@ -310,6 +310,36 @@ def test_transcription_cost_falls_back_to_duration():
|
|||
assert pytest.approx(cost, rel=1e-6) == expected_cost
|
||||
|
||||
|
||||
def test_embedding_completion_cost_uses_input_cost_per_token():
|
||||
from litellm.types.utils import EmbeddingResponse
|
||||
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
model = "text-embedding-3-small"
|
||||
prompt_tokens = 100
|
||||
response = EmbeddingResponse(
|
||||
model=model,
|
||||
data=[{"embedding": [0.1, 0.2, 0.3], "index": 0, "object": "embedding"}],
|
||||
usage=Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=0,
|
||||
total_tokens=prompt_tokens,
|
||||
),
|
||||
)
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model=model,
|
||||
custom_llm_provider="openai",
|
||||
call_type="embedding",
|
||||
)
|
||||
|
||||
expected = prompt_tokens * litellm.model_cost[model]["input_cost_per_token"]
|
||||
assert pytest.approx(cost, rel=1e-6) == expected
|
||||
assert cost > 0
|
||||
|
||||
|
||||
def test_handle_realtime_stream_cost_calculation():
|
||||
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
|
||||
|
||||
|
|
@ -385,7 +415,7 @@ def test_handle_realtime_stream_cost_calculation():
|
|||
)
|
||||
assert cost == 0.0 # No usage, no cost
|
||||
|
||||
|
||||
|
||||
def test_realtime_stream_combines_text_and_audio_token_details():
|
||||
"""Realtime response.done usage with input_token_details / output_token_details."""
|
||||
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue