mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
* feat(spend): track prompt compression saved tokens in daily spend aggregates Native compression interception now records tokens_before/after/saved into the request litellm_metadata so savings land in the SpendLog metadata JSON under a typed compression_savings key. A single normalizer (extract_compression_saved_tokens) sums that key with Headroom guardrail tokens_saved; the two writers are disjoint and run at different stages, so summing never double-counts. The spend-log redactor now preserves purely numeric compression stats inside guardrail_response so Headroom savings survive the store_prompts_in_spend_logs=false default. compression_saved_tokens is threaded through BaseDailySpendTransaction, queue aggregation, the daily upsert blocks, a new BigInt column on all six daily spend tables, and the daily activity read path (SpendMetrics, DailySpendMetadata, raw-SQL rollups) * fix(spend): normalize legacy guardrail shapes and float token stats in compression savings reader * feat(spend): aggregate compression and prompt caching dollar savings in daily rollups Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(spend): update daily spend aggregation fixtures for savings columns Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(ui): add Cost Optimization dashboard page New left-nav Cost Optimization page under Observability that surfaces money saved by prompt compression and prompt caching. It reads the daily activity rollup (userDailyActivityCall / get_daily_activity) and never scans SpendLogs, so it stays fast at 1M+ rows. Renders a Total saved card, per-driver Compression and Prompt caching cards, a savings-over-time area chart, and a savings-by-driver donut, all aggregated in memory from the per-day metrics.compression_savings_spend and metrics.prompt_caching_savings_spend fields. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Krrish Dholakia <krrishdholakia@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
78 lines
2.5 KiB
Python
78 lines
2.5 KiB
Python
import os
|
|
import sys
|
|
|
|
sys.path.insert(0, os.path.abspath("../../../.."))
|
|
|
|
import pytest
|
|
|
|
import litellm
|
|
from litellm.proxy.spend_tracking.savings import compute_savings_spend
|
|
|
|
|
|
def _anthropic_costs(model: str) -> tuple[float, float]:
|
|
info = litellm.get_model_info(model=model, custom_llm_provider="anthropic")
|
|
input_cost = info["input_cost_per_token"] or 0.0
|
|
cache_read_cost = info.get("cache_read_input_token_cost") or input_cost
|
|
return input_cost, cache_read_cost
|
|
|
|
|
|
def test_compression_savings_priced_at_input_rate():
|
|
input_cost, _ = _anthropic_costs("claude-sonnet-5")
|
|
result = compute_savings_spend(
|
|
model="claude-sonnet-5",
|
|
custom_llm_provider="anthropic",
|
|
compression_saved_tokens=4389,
|
|
cache_read_input_tokens=0,
|
|
)
|
|
assert result.compression == pytest.approx(4389 * input_cost)
|
|
assert result.compression > 0
|
|
assert result.prompt_caching == 0.0
|
|
|
|
|
|
def test_prompt_caching_savings_priced_at_input_minus_cache_read():
|
|
input_cost, cache_read_cost = _anthropic_costs("claude-sonnet-5")
|
|
# A model that supports prompt caching must charge less to read from cache;
|
|
# otherwise this test is asserting nothing.
|
|
assert cache_read_cost < input_cost
|
|
result = compute_savings_spend(
|
|
model="claude-sonnet-5",
|
|
custom_llm_provider="anthropic",
|
|
compression_saved_tokens=0,
|
|
cache_read_input_tokens=8200,
|
|
)
|
|
assert result.prompt_caching == pytest.approx(8200 * (input_cost - cache_read_cost))
|
|
assert result.prompt_caching > 0
|
|
assert result.compression == 0.0
|
|
|
|
|
|
def test_unknown_model_fails_open_to_zero():
|
|
result = compute_savings_spend(
|
|
model="totally-made-up-model-xyz",
|
|
custom_llm_provider="anthropic",
|
|
compression_saved_tokens=1000,
|
|
cache_read_input_tokens=1000,
|
|
)
|
|
assert result.compression == 0.0
|
|
assert result.prompt_caching == 0.0
|
|
|
|
|
|
def test_missing_model_fails_open_to_zero():
|
|
result = compute_savings_spend(
|
|
model=None,
|
|
custom_llm_provider=None,
|
|
compression_saved_tokens=1000,
|
|
cache_read_input_tokens=1000,
|
|
)
|
|
assert result.compression == 0.0
|
|
assert result.prompt_caching == 0.0
|
|
|
|
|
|
def test_negative_token_counts_clamp_to_zero():
|
|
result = compute_savings_spend(
|
|
model="claude-sonnet-5",
|
|
custom_llm_provider="anthropic",
|
|
compression_saved_tokens=-500,
|
|
cache_read_input_tokens=-500,
|
|
)
|
|
assert result.compression == 0.0
|
|
assert result.prompt_caching == 0.0
|