mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
feat(cost_breakdown): show cache read/write cost in Logs UI
Add cache_read_input_cost and cache_write_input_cost to CostBreakdown. Backend (litellm_logging.py): - Add _enrich_cost_breakdown_with_cache_costs() that reads cached_tokens and cache_creation_tokens from usage, computes their costs via _get_token_base_cost, and injects them into the existing cost_breakdown dict before it reaches the spend log. Frontend (CostBreakdownViewer.tsx): - Show 'Cache Read Cost' and 'Cache Write Cost' sub-rows under Input Cost when the values are present and > 0. Types (types/utils.py): - Add cache_read_input_cost and cache_write_input_cost to CostBreakdown. Co-authored-by: Ishaan Jaff <ishaan-jaff@users.noreply.github.com>
This commit is contained in:
parent
6f56852cf1
commit
8fb8e497ec
3 changed files with 92 additions and 1 deletions
|
|
@ -5210,6 +5210,78 @@ def _extract_response_obj_and_hidden_params(
|
|||
return response_obj, hidden_params
|
||||
|
||||
|
||||
def _enrich_cost_breakdown_with_cache_costs(
|
||||
cost_breakdown: Optional[CostBreakdown],
|
||||
init_response_obj: Any,
|
||||
logging_obj: "Logging",
|
||||
) -> Optional[CostBreakdown]:
|
||||
"""
|
||||
Enrich cost_breakdown with cache_read_input_cost / cache_write_input_cost
|
||||
when the usage object contains cached or cache-creation tokens.
|
||||
"""
|
||||
if cost_breakdown is None:
|
||||
return None
|
||||
if "cache_read_input_cost" in cost_breakdown:
|
||||
return cost_breakdown
|
||||
|
||||
try:
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import (
|
||||
_get_token_base_cost,
|
||||
_parse_prompt_tokens_details,
|
||||
calculate_cache_writing_cost,
|
||||
)
|
||||
from litellm.responses.utils import ResponseAPILoggingUtils
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
usage_obj = getattr(init_response_obj, "usage", None)
|
||||
if usage_obj is None:
|
||||
return cost_breakdown
|
||||
|
||||
# Normalise Response-API usage to chat Usage
|
||||
if hasattr(usage_obj, "input_tokens") and not hasattr(usage_obj, "prompt_tokens"):
|
||||
if ResponseAPILoggingUtils._is_response_api_usage(usage_obj):
|
||||
usage_obj = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(usage_obj)
|
||||
else:
|
||||
return cost_breakdown
|
||||
|
||||
if not hasattr(usage_obj, "prompt_tokens") or not usage_obj.prompt_tokens_details:
|
||||
return cost_breakdown
|
||||
|
||||
custom_llm_provider = logging_obj.model_call_details.get("custom_llm_provider")
|
||||
model = logging_obj.model
|
||||
if not custom_llm_provider or not model:
|
||||
return cost_breakdown
|
||||
|
||||
details = _parse_prompt_tokens_details(usage_obj)
|
||||
cache_hit_tokens = details["cache_hit_tokens"]
|
||||
cache_creation_tokens = details["cache_creation_tokens"]
|
||||
if cache_hit_tokens == 0 and cache_creation_tokens == 0:
|
||||
return cost_breakdown
|
||||
|
||||
model_info = get_model_info(model=model, custom_llm_provider=custom_llm_provider)
|
||||
(
|
||||
_prompt_base,
|
||||
_comp_base,
|
||||
cache_creation_rate,
|
||||
cache_creation_rate_1hr,
|
||||
cache_read_rate,
|
||||
) = _get_token_base_cost(model_info=model_info, usage=usage_obj)
|
||||
|
||||
if cache_hit_tokens > 0:
|
||||
cost_breakdown["cache_read_input_cost"] = float(cache_hit_tokens) * cache_read_rate
|
||||
if cache_creation_tokens > 0:
|
||||
cost_breakdown["cache_write_input_cost"] = calculate_cache_writing_cost(
|
||||
cache_creation_tokens=cache_creation_tokens,
|
||||
cache_creation_token_details=details["cache_creation_token_details"],
|
||||
cache_creation_cost_above_1hr=cache_creation_rate_1hr,
|
||||
cache_creation_cost=cache_creation_rate,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return cost_breakdown
|
||||
|
||||
|
||||
def get_standard_logging_object_payload(
|
||||
kwargs: Optional[dict],
|
||||
init_response_obj: Union[Any, BaseModel, dict],
|
||||
|
|
@ -5374,7 +5446,9 @@ def get_standard_logging_object_payload(
|
|||
metadata=clean_metadata,
|
||||
cache_key=clean_hidden_params["cache_key"],
|
||||
response_cost=response_cost,
|
||||
cost_breakdown=logging_obj.cost_breakdown,
|
||||
cost_breakdown=_enrich_cost_breakdown_with_cache_costs(
|
||||
logging_obj.cost_breakdown, init_response_obj, logging_obj
|
||||
),
|
||||
total_tokens=usage_dict.get("total_tokens", 0),
|
||||
prompt_tokens=usage_dict.get("prompt_tokens", 0),
|
||||
completion_tokens=usage_dict.get("completion_tokens", 0),
|
||||
|
|
|
|||
|
|
@ -2736,6 +2736,9 @@ class CostBreakdown(TypedDict, total=False):
|
|||
margin_fixed_amount: float # Fixed margin amount in USD (optional)
|
||||
margin_total_amount: float # Total margin added in USD (optional)
|
||||
|
||||
cache_read_input_cost: float # Cost for cache-read/hit input tokens
|
||||
cache_write_input_cost: float # Cost for cache-creation/write input tokens
|
||||
|
||||
|
||||
class StandardLoggingPayloadStatusFields(TypedDict, total=False):
|
||||
"""Status fields for easy filtering and analytics"""
|
||||
|
|
|
|||
|
|
@ -14,6 +14,8 @@ export interface CostBreakdown {
|
|||
margin_percent?: number;
|
||||
margin_fixed_amount?: number;
|
||||
margin_total_amount?: number;
|
||||
cache_read_input_cost?: number;
|
||||
cache_write_input_cost?: number;
|
||||
}
|
||||
|
||||
interface CostBreakdownViewerProps {
|
||||
|
|
@ -110,6 +112,18 @@ export const CostBreakdownViewer: React.FC<CostBreakdownViewerProps> = ({
|
|||
)}
|
||||
</span>
|
||||
</div>
|
||||
{!isCached && costBreakdown?.cache_read_input_cost !== undefined && costBreakdown.cache_read_input_cost > 0 && (
|
||||
<div className="flex text-xs ml-6 border-l-2 border-gray-200 pl-3">
|
||||
<span className="text-gray-500 w-1/3">Cache Read Cost:</span>
|
||||
<span className="text-gray-700">{formatCost(costBreakdown.cache_read_input_cost)}</span>
|
||||
</div>
|
||||
)}
|
||||
{!isCached && costBreakdown?.cache_write_input_cost !== undefined && costBreakdown.cache_write_input_cost > 0 && (
|
||||
<div className="flex text-xs ml-6 border-l-2 border-gray-200 pl-3">
|
||||
<span className="text-gray-500 w-1/3">Cache Write Cost:</span>
|
||||
<span className="text-gray-700">{formatCost(costBreakdown.cache_write_input_cost)}</span>
|
||||
</div>
|
||||
)}
|
||||
<div className="flex text-sm">
|
||||
<span className="text-gray-600 font-medium w-1/3">Output Cost:</span>
|
||||
<span className="text-gray-900">
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue