From e6db497fb8dd56709b634cace715457de1059021 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 8 Jun 2026 18:10:57 +0000 Subject: [PATCH] test(spend-tracking): add regression coverage for uncovered spend paths Spend tracking is already heavily tested, but a few high-value paths had no direct coverage. Add focused, mutation-surviving regression tests: - _get_status_for_spend_log: failure-status tracking had zero tests - cache-hit cost zeroing in _PROXY_track_cost_callback (anti double-charge) - cache-hit unique request_id suffix in get_logging_payload (SpendLogs duplicate-key collision fix) - failure-status and zero-spend wiring through get_logging_payload - embedding completion_cost (no dedicated embedding cost test existed) Each test fails if its guard is mutated (verified by hand-mutation). Also add a spend tracking test runbook documenting the offline suite plus an operator-run live e2e check against real provider APIs and SpendLogs rows. --- .../hooks/test_proxy_track_cost_callback.py | 49 +++++++++ .../SPEND_TRACKING_TEST_RUNBOOK.md | 102 ++++++++++++++++++ .../spend_tracking/e2e_spend_config.yaml | 17 +++ .../test_spend_tracking_utils.py | 86 +++++++++++++++ tests/test_litellm/test_cost_calculator.py | 32 +++++- 5 files changed, 285 insertions(+), 1 deletion(-) create mode 100644 tests/test_litellm/proxy/spend_tracking/SPEND_TRACKING_TEST_RUNBOOK.md create mode 100644 tests/test_litellm/proxy/spend_tracking/e2e_spend_config.yaml diff --git a/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py b/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py index 771e10a54a0..56d26ce9178 100644 --- a/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py +++ b/tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py @@ -254,6 +254,55 @@ async def test_track_cost_callback_releases_budget_reservation_when_spend_tracki ) +@pytest.mark.asyncio +async def test_track_cost_callback_zeroes_response_cost_on_cache_hit(): + logger = _ProxyDBLogger() + kwargs = { + "model": "gpt-4", + "call_type": "acompletion", + "cache_hit": True, + "litellm_params": { + "metadata": { + "user_api_key": "hashed-key", + "user_api_key_user_id": "test_user", + "user_api_key_team_id": "test_team", + } + }, + "standard_logging_object": { + "response_cost": 0.25, + "request_tags": None, + }, + "stream": False, + } + + mock_proxy_logging = MagicMock() + mock_proxy_logging.db_spend_update_writer.update_database = AsyncMock() + mock_proxy_logging.slack_alerting_instance.customer_spend_alert = AsyncMock() + + with ( + patch("litellm.proxy.proxy_server.proxy_logging_obj", mock_proxy_logging), + patch( + "litellm.proxy.proxy_server.increment_spend_counters", + new_callable=AsyncMock, + ), + patch("litellm.proxy.proxy_server.update_cache", new_callable=AsyncMock), + ): + await logger._PROXY_track_cost_callback( + kwargs=kwargs, + completion_response=None, + start_time=datetime.now(), + end_time=datetime.now(), + ) + + mock_proxy_logging.db_spend_update_writer.update_database.assert_awaited_once() + recorded_cost = ( + mock_proxy_logging.db_spend_update_writer.update_database.await_args.kwargs[ + "response_cost" + ] + ) + assert recorded_cost == 0.0 + + @pytest.mark.asyncio async def test_track_cost_callback_releases_budget_reservation_when_response_cost_missing(): logger = _ProxyDBLogger() diff --git a/tests/test_litellm/proxy/spend_tracking/SPEND_TRACKING_TEST_RUNBOOK.md b/tests/test_litellm/proxy/spend_tracking/SPEND_TRACKING_TEST_RUNBOOK.md new file mode 100644 index 00000000000..3d040ef2b02 --- /dev/null +++ b/tests/test_litellm/proxy/spend_tracking/SPEND_TRACKING_TEST_RUNBOOK.md @@ -0,0 +1,102 @@ +# Spend Tracking Test Runbook + +This runbook documents the regression coverage that guards litellm's spend/cost +tracking against silent breakage before a release. It has two halves: a set of +deterministic offline unit tests (Part A) that run in CI, and an operator-driven +live end-to-end check (Part B) that hits real provider APIs and asserts real +SpendLogs rows. Spend tracking is already heavily tested across the codebase +(cost calculators, SpendLogsPayload construction, the proxy cost callback, the DB +spend-update writer, daily-spend queues); the tests below fill the specific +high-value gaps that previously had no direct coverage. + +## Part A: offline regression tests (CI) + +Run them with: + +```bash +uv run --no-sync python -m pytest \ + tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py \ + tests/test_litellm/proxy/hooks/test_proxy_track_cost_callback.py \ + tests/test_litellm/test_cost_calculator.py::test_embedding_completion_cost_uses_input_cost_per_token \ + -q +``` + +Each test is written to fail if the specific guard it covers is mutated; the +"What it guards" column names the source line that, when broken, turns the test +red. Verified by hand-mutation (revert each guard and watch the row go red). + +| Test name | What it guards | Status | +| --- | --- | --- | +| `TestGetStatusForSpendLog::test_missing_status_key_defaults_to_success` | `_get_status_for_spend_log` default branch (spend_tracking_utils.py) returns success when no status set | pass | +| `TestGetStatusForSpendLog::test_explicit_success_returns_success` | same helper, explicit success preserved | pass | +| `TestGetStatusForSpendLog::test_failure_returns_failure` | the `== "failure"` guard so failed requests are logged as failures | pass | +| `TestGetStatusForSpendLog::test_non_failure_value_returns_success` | kills the "any non-None status -> failure" mutant | pass | +| `test_get_logging_payload_cache_hit_appends_unique_suffix_to_request_id` | cache-hit `_cache_hit{time}` suffix on request_id; without it SpendLogs hits duplicate-key collisions | pass | +| `test_get_logging_payload_failure_status_and_zero_spend` | status wiring at the `get_logging_payload` call site plus `spend` sourced from `response_cost` | pass | +| `test_get_logging_payload_default_status_success` | default status path through `get_logging_payload` | pass | +| `test_track_cost_callback_zeroes_response_cost_on_cache_hit` | cache-hit cost zeroing in `_PROXY_track_cost_callback`; the anti double-charge guard | pass | +| `test_embedding_completion_cost_uses_input_cost_per_token` | embedding cost = `prompt_tokens * input_cost_per_token`; previously no dedicated embedding cost test | pass | + +All 9 pass on `claude/spend-tracking-tests-4gbezl`. + +## Part B: live end-to-end check (operator-run, real spend logs) + +This proves spend tracking against real provider responses and a real database, +which is the closest mirror of what a customer sees. It needs a Postgres +`DATABASE_URL`, an `OPENAI_API_KEY` in `.env`, and outbound access to +api.openai.com. It uses `gpt-5.4-nano` (current cheap small model as of 2026-06) +and `text-embedding-3-small`, with local caching enabled so a repeated identical +chat request produces a cache hit. + +Config: `tests/test_litellm/proxy/spend_tracking/e2e_spend_config.yaml` + +1. Point at a database and start the proxy (spend logs require a DB): + +```bash +export DATABASE_URL='postgresql://user:pass@localhost:5432/litellm' +python litellm/proxy/proxy_cli.py \ + --config tests/test_litellm/proxy/spend_tracking/e2e_spend_config.yaml \ + --detailed_debug 2>&1 | tee litellm.log +``` + +2. First chat call (real cost expected): + +```bash +curl -s http://localhost:4000/v1/chat/completions \ + -H 'Authorization: Bearer sk-1234' -H 'Content-Type: application/json' \ + -d '{"model":"gpt-5.4-nano","messages":[{"role":"user","content":"say hello in one word"}]}' | jq '{id, usage}' +``` + +3. Identical chat call again to trigger a cache hit (cost should be recorded as 0): + +```bash +curl -s http://localhost:4000/v1/chat/completions \ + -H 'Authorization: Bearer sk-1234' -H 'Content-Type: application/json' \ + -d '{"model":"gpt-5.4-nano","messages":[{"role":"user","content":"say hello in one word"}]}' | jq '{id, usage}' +``` + +4. Embedding call (real cost expected): + +```bash +curl -s http://localhost:4000/v1/embeddings \ + -H 'Authorization: Bearer sk-1234' -H 'Content-Type: application/json' \ + -d '{"model":"text-embedding-3-small","input":"hello world"}' | jq '{model, usage}' +``` + +5. Wait for the spend-log flush (the proxy batches writes roughly once a minute), + then read the logs back and assert the three rows: + +```bash +sleep 65 +curl -s 'http://localhost:4000/spend/logs' -H 'Authorization: Bearer sk-1234' \ + | jq '[.[] | {request_id, call_type, model, spend, cache_hit}]' +``` + +Expected: the chat row has `spend > 0`, the embedding row has `spend > 0`, and the +cache-hit row has `spend == 0` with a `request_id` containing `_cache_hit`. + +Status: not run in this sandbox. The container has no `.env`/`OPENAI_API_KEY`, no +`DATABASE_URL`, and egress to api.openai.com is blocked by the environment's +network policy (a billed call is not possible here). Run the steps above on a host +with credentials and network access to fill this status; the offline suite in +Part A is what gates CI diff --git a/tests/test_litellm/proxy/spend_tracking/e2e_spend_config.yaml b/tests/test_litellm/proxy/spend_tracking/e2e_spend_config.yaml new file mode 100644 index 00000000000..9da2295a4ab --- /dev/null +++ b/tests/test_litellm/proxy/spend_tracking/e2e_spend_config.yaml @@ -0,0 +1,17 @@ +model_list: + - model_name: gpt-5.4-nano + litellm_params: + model: openai/gpt-5.4-nano + api_key: os.environ/OPENAI_API_KEY + - model_name: text-embedding-3-small + litellm_params: + model: openai/text-embedding-3-small + api_key: os.environ/OPENAI_API_KEY + +litellm_settings: + cache: true + cache_params: + type: local + +general_settings: + master_key: sk-1234 diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py index 0c7511589de..cc9bfda87c8 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_tracking_utils.py @@ -28,6 +28,7 @@ from litellm.proxy.spend_tracking.spend_tracking_utils import ( _get_request_duration_ms, _get_response_for_spend_logs_payload, _get_spend_logs_metadata, + _get_status_for_spend_log, _get_vector_store_request_for_spend_logs_payload, _is_master_key, _redact_prompt_leaks_in_error_string, @@ -2073,3 +2074,88 @@ def test_sanitize_error_information_redacts_pydantic_assignment_form( assert sanitized is not None assert "leaked-via-pydantic-msg" not in sanitized["error_message"] assert REDACTED_BY_LITELM_STRING in sanitized["error_message"] + + +class TestGetStatusForSpendLog: + def test_missing_status_key_defaults_to_success(self): + assert _get_status_for_spend_log(metadata={}) == "success" + + def test_explicit_success_returns_success(self): + assert _get_status_for_spend_log(metadata={"status": "success"}) == "success" + + def test_failure_returns_failure(self): + assert _get_status_for_spend_log(metadata={"status": "failure"}) == "failure" + + def test_non_failure_value_returns_success(self): + assert _get_status_for_spend_log(metadata={"status": "error"}) == "success" + + +@patch("litellm.proxy.proxy_server.master_key", None) +@patch("litellm.proxy.proxy_server.general_settings", {}) +def test_get_logging_payload_cache_hit_appends_unique_suffix_to_request_id(): + base_response_id = "chatcmpl-deterministic-base-id" + response_obj = { + "id": base_response_id, + "choices": [{"message": {"content": "hi"}}], + "usage": {"prompt_tokens": 5, "completion_tokens": 3, "total_tokens": 8}, + } + kwargs = { + "model": "gpt-3.5-turbo", + "litellm_params": {"metadata": {"user_api_key": "sk-test-key"}}, + } + start_time = datetime.datetime.now(timezone.utc) + end_time = datetime.datetime.now(timezone.utc) + + no_cache_payload = get_logging_payload( + kwargs={**kwargs}, + response_obj=response_obj, + start_time=start_time, + end_time=end_time, + ) + assert no_cache_payload["request_id"] == base_response_id + + cache_payload = get_logging_payload( + kwargs={**kwargs, "cache_hit": True}, + response_obj=response_obj, + start_time=start_time, + end_time=end_time, + ) + assert cache_payload["request_id"].startswith(base_response_id) + assert "_cache_hit" in cache_payload["request_id"] + assert cache_payload["request_id"] != base_response_id + + +@patch("litellm.proxy.proxy_server.master_key", None) +@patch("litellm.proxy.proxy_server.general_settings", {}) +def test_get_logging_payload_failure_status_and_zero_spend(): + kwargs = { + "model": "gpt-3.5-turbo", + "response_cost": 0.0, + "litellm_params": { + "metadata": {"user_api_key": "sk-test-key", "status": "failure"} + }, + } + payload = get_logging_payload( + kwargs=kwargs, + response_obj={}, + start_time=datetime.datetime.now(timezone.utc), + end_time=datetime.datetime.now(timezone.utc), + ) + assert payload["status"] == "failure" + assert payload["spend"] == 0.0 + + +@patch("litellm.proxy.proxy_server.master_key", None) +@patch("litellm.proxy.proxy_server.general_settings", {}) +def test_get_logging_payload_default_status_success(): + kwargs = { + "model": "gpt-3.5-turbo", + "litellm_params": {"metadata": {"user_api_key": "sk-test-key"}}, + } + payload = get_logging_payload( + kwargs=kwargs, + response_obj={}, + start_time=datetime.datetime.now(timezone.utc), + end_time=datetime.datetime.now(timezone.utc), + ) + assert payload["status"] == "success" diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 82a4a60bf82..d050e77830d 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -310,6 +310,36 @@ def test_transcription_cost_falls_back_to_duration(): assert pytest.approx(cost, rel=1e-6) == expected_cost +def test_embedding_completion_cost_uses_input_cost_per_token(): + from litellm.types.utils import EmbeddingResponse + + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + model = "text-embedding-3-small" + prompt_tokens = 100 + response = EmbeddingResponse( + model=model, + data=[{"embedding": [0.1, 0.2, 0.3], "index": 0, "object": "embedding"}], + usage=Usage( + prompt_tokens=prompt_tokens, + completion_tokens=0, + total_tokens=prompt_tokens, + ), + ) + + cost = completion_cost( + completion_response=response, + model=model, + custom_llm_provider="openai", + call_type="embedding", + ) + + expected = prompt_tokens * litellm.model_cost[model]["input_cost_per_token"] + assert pytest.approx(cost, rel=1e-6) == expected + assert cost > 0 + + def test_handle_realtime_stream_cost_calculation(): from litellm.cost_calculator import RealtimeAPITokenUsageProcessor @@ -385,7 +415,7 @@ def test_handle_realtime_stream_cost_calculation(): ) assert cost == 0.0 # No usage, no cost - + def test_realtime_stream_combines_text_and_audio_token_details(): """Realtime response.done usage with input_token_details / output_token_details.""" from litellm.cost_calculator import RealtimeAPITokenUsageProcessor