mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
fix(usage): preserve unknown historical throughput
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
7455796c04
commit
ed862773d1
17 changed files with 133 additions and 51 deletions
|
|
@ -1,6 +1,6 @@
|
|||
ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0;
|
||||
ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0;
|
||||
ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0;
|
||||
ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0;
|
||||
ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0;
|
||||
ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0;
|
||||
ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT;
|
||||
ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT;
|
||||
ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT;
|
||||
ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT;
|
||||
ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT;
|
||||
ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT;
|
||||
|
|
|
|||
|
|
@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
@@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
|
||||
|
|
@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
@@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
|
||||
|
|
@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
ptu_flat_cost Float @default(0.0)
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
|
@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
|
|||
|
|
@ -4364,9 +4364,15 @@
|
|||
"title": "Output Tokens Per Second"
|
||||
},
|
||||
"timed_completion_tokens": {
|
||||
"default": 0,
|
||||
"title": "Timed Completion Tokens",
|
||||
"type": "integer"
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "Timed Completion Tokens"
|
||||
},
|
||||
"timed_requests": {
|
||||
"default": 0,
|
||||
|
|
|
|||
|
|
@ -5703,7 +5703,7 @@ class BaseDailySpendTransaction(TypedDict):
|
|||
failed_requests: int
|
||||
total_response_time_ms: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place
|
||||
timed_requests: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place
|
||||
timed_completion_tokens: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place
|
||||
timed_completion_tokens: NotRequired[int | None]
|
||||
|
||||
|
||||
class DailyTeamSpendTransaction(BaseDailySpendTransaction):
|
||||
|
|
|
|||
|
|
@ -125,6 +125,20 @@ def _as_float(value: object) -> float:
|
|||
return float(value) if isinstance(value, (int, float)) else 0.0
|
||||
|
||||
|
||||
def _counter_total(column: str, group: Sequence[SpendRow]) -> int | None:
|
||||
values: Final = tuple(row.get(column) for row in group)
|
||||
if column == "timed_completion_tokens" and any(value is None for value in values):
|
||||
return None
|
||||
return sum(_as_int(value) for value in values)
|
||||
|
||||
|
||||
def _counter_value(column: str, transaction: SpendRow) -> int | None:
|
||||
value: Final = transaction.get(column)
|
||||
if column == "timed_completion_tokens" and value is None:
|
||||
return None
|
||||
return _as_int(value)
|
||||
|
||||
|
||||
def conflict_key(table: DailySpendTable, transaction: SpendRow) -> tuple[str, ...]:
|
||||
"""The tuple the database arbitrates the upsert on, normalized free of NULLs."""
|
||||
return tuple(_as_text(transaction.get(column)) for column in (table.entity_id_column, *_KEY_COLUMNS))
|
||||
|
|
@ -135,7 +149,7 @@ def _merge(group: Sequence[SpendRow]) -> SpendRow:
|
|||
return group[0]
|
||||
return {
|
||||
**group[0],
|
||||
**{column: sum(_as_int(row.get(column)) for row in group) for column in _COUNTER_COLUMNS},
|
||||
**{column: _counter_total(column, group) for column in _COUNTER_COLUMNS},
|
||||
**{column: sum(_as_float(row.get(column)) for row in group) for column in _SPEND_COLUMNS},
|
||||
}
|
||||
|
||||
|
|
@ -166,7 +180,7 @@ def _row_params(
|
|||
str(uuid.uuid4()),
|
||||
*key,
|
||||
None if transaction.get("model_group") is None else _as_text(transaction.get("model_group")),
|
||||
*(_as_int(transaction.get(column)) for column in _COUNTER_COLUMNS),
|
||||
*(_counter_value(column, transaction) for column in _COUNTER_COLUMNS),
|
||||
*(_as_float(transaction.get(column)) for column in _SPEND_COLUMNS),
|
||||
*((None if request_id is None else _as_text(request_id),) if table.carries_request_id else ()),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -22,6 +22,10 @@ def _daily_spend_updates(
|
|||
yield from update.items()
|
||||
|
||||
|
||||
def _sum_timed_completion_tokens(left: int | None, right: int | None) -> int | None:
|
||||
return left + right if left is not None and right is not None else None
|
||||
|
||||
|
||||
def _merge_daily_spend_transactions(
|
||||
existing: BaseDailySpendTransaction,
|
||||
payload: BaseDailySpendTransaction,
|
||||
|
|
@ -51,8 +55,10 @@ def _merge_daily_spend_transactions(
|
|||
"total_response_time_ms": (existing.get("total_response_time_ms", 0) or 0)
|
||||
+ (payload.get("total_response_time_ms", 0) or 0),
|
||||
"timed_requests": (existing.get("timed_requests", 0) or 0) + (payload.get("timed_requests", 0) or 0),
|
||||
"timed_completion_tokens": (existing.get("timed_completion_tokens", 0) or 0)
|
||||
+ (payload.get("timed_completion_tokens", 0) or 0),
|
||||
"timed_completion_tokens": _sum_timed_completion_tokens(
|
||||
existing.get("timed_completion_tokens"),
|
||||
payload.get("timed_completion_tokens"),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -162,7 +162,7 @@ class DailySpendRecord(Protocol):
|
|||
def timed_requests(self) -> int: ...
|
||||
|
||||
@property
|
||||
def timed_completion_tokens(self) -> int: ...
|
||||
def timed_completion_tokens(self) -> int | None: ...
|
||||
|
||||
|
||||
class _KeyMetadataDict(TypedDict, total=False):
|
||||
|
|
@ -241,13 +241,13 @@ def update_metrics(existing_metrics: SpendMetrics, record: DailySpendRecord) ->
|
|||
|
||||
|
||||
def _provider_throughput(
|
||||
timed_completion_tokens: int,
|
||||
timed_completion_tokens: int | None,
|
||||
total_response_time_ms: int,
|
||||
timed_requests: int,
|
||||
) -> ProviderThroughputMetrics:
|
||||
output_tokens_per_second: Final = (
|
||||
timed_completion_tokens * 1000 / total_response_time_ms
|
||||
if timed_requests > 0 and total_response_time_ms > 0
|
||||
if timed_completion_tokens is not None and timed_requests > 0 and total_response_time_ms > 0
|
||||
else None
|
||||
)
|
||||
return ProviderThroughputMetrics(
|
||||
|
|
@ -258,18 +258,34 @@ def _provider_throughput(
|
|||
)
|
||||
|
||||
|
||||
def _combined_timed_completion_tokens(
|
||||
existing: ProviderThroughputMetrics | None,
|
||||
current: int | None,
|
||||
) -> int | None:
|
||||
if existing is None:
|
||||
return current
|
||||
if existing.timed_completion_tokens is None or current is None:
|
||||
return None
|
||||
return existing.timed_completion_tokens + current
|
||||
|
||||
|
||||
def _update_provider_throughput(
|
||||
target: MetricWithMetadata,
|
||||
provider: str,
|
||||
record: DailySpendRecord,
|
||||
) -> None:
|
||||
existing: Final = target.provider_breakdown.get(provider, ProviderThroughputMetrics())
|
||||
existing: Final = target.provider_breakdown.get(provider)
|
||||
timed_completion_tokens: Final = _combined_timed_completion_tokens(
|
||||
existing,
|
||||
record.timed_completion_tokens,
|
||||
)
|
||||
target.provider_breakdown = {
|
||||
**target.provider_breakdown,
|
||||
provider: _provider_throughput(
|
||||
timed_completion_tokens=existing.timed_completion_tokens + (record.timed_completion_tokens or 0),
|
||||
total_response_time_ms=existing.total_response_time_ms + (record.total_response_time_ms or 0),
|
||||
timed_requests=existing.timed_requests + (record.timed_requests or 0),
|
||||
timed_completion_tokens=timed_completion_tokens,
|
||||
total_response_time_ms=(existing.total_response_time_ms if existing is not None else 0)
|
||||
+ (record.total_response_time_ms or 0),
|
||||
timed_requests=(existing.timed_requests if existing is not None else 0) + (record.timed_requests or 0),
|
||||
),
|
||||
}
|
||||
|
||||
|
|
@ -844,7 +860,7 @@ def _aggregate_grouping_sets_records_sync(
|
|||
metadata={},
|
||||
provider_breakdown={
|
||||
provider: _provider_throughput(
|
||||
record.timed_completion_tokens or 0,
|
||||
record.timed_completion_tokens,
|
||||
record.total_response_time_ms or 0,
|
||||
record.timed_requests or 0,
|
||||
)
|
||||
|
|
@ -854,7 +870,7 @@ def _aggregate_grouping_sets_records_sync(
|
|||
parent.provider_breakdown = {
|
||||
**parent.provider_breakdown,
|
||||
provider: _provider_throughput(
|
||||
record.timed_completion_tokens or 0,
|
||||
record.timed_completion_tokens,
|
||||
record.total_response_time_ms or 0,
|
||||
record.timed_requests or 0,
|
||||
),
|
||||
|
|
|
|||
|
|
@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
@@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
|
||||
|
|
@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
@@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
|
||||
|
|
@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
ptu_flat_cost Float @default(0.0)
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
|
@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
|
|||
|
|
@ -130,7 +130,10 @@ def _rollup_metric_select(table: DailyActivityTable) -> str:
|
|||
SUM(failed_requests)::bigint AS failed_requests,
|
||||
SUM(total_response_time_ms)::bigint AS total_response_time_ms,
|
||||
SUM(timed_requests)::bigint AS timed_requests,
|
||||
SUM(timed_completion_tokens)::bigint AS timed_completion_tokens"""
|
||||
CASE
|
||||
WHEN COUNT(timed_completion_tokens) = COUNT(*) THEN SUM(timed_completion_tokens)::bigint
|
||||
ELSE NULL::bigint
|
||||
END AS timed_completion_tokens"""
|
||||
|
||||
|
||||
def _validate_api_key_limit(api_key_limit: int) -> None:
|
||||
|
|
|
|||
|
|
@ -57,7 +57,7 @@ class KeyMetricWithMetadata(MetricBase):
|
|||
|
||||
|
||||
class ProviderThroughputMetrics(BaseModel):
|
||||
timed_completion_tokens: int = Field(default=0)
|
||||
timed_completion_tokens: int | None = Field(default=None)
|
||||
total_response_time_ms: int = Field(default=0)
|
||||
timed_requests: int = Field(default=0)
|
||||
output_tokens_per_second: float | None = Field(default=None)
|
||||
|
|
|
|||
|
|
@ -185,7 +185,7 @@ class DailyActivityRow(Protocol):
|
|||
failed_requests: int
|
||||
total_response_time_ms: int
|
||||
timed_requests: int
|
||||
timed_completion_tokens: int
|
||||
timed_completion_tokens: int | None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
|
|
|
|||
|
|
@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
@@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
|
||||
|
|
@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
@@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
|
||||
|
|
@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
ptu_flat_cost Float @default(0.0)
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
|
@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend {
|
|||
failed_requests BigInt @default(0)
|
||||
total_response_time_ms BigInt @default(0)
|
||||
timed_requests BigInt @default(0)
|
||||
timed_completion_tokens BigInt @default(0)
|
||||
timed_completion_tokens BigInt?
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
|
|
|
|||
|
|
@ -575,7 +575,15 @@ async def test_optional_metric_missing_from_an_older_payload_still_aggregates(
|
|||
|
||||
await daily_spend_update_queue.add_update({test_key: dict(base)})
|
||||
await daily_spend_update_queue.add_update(
|
||||
{test_key: {**base, "autorouter_savings_spend": 0.25, "total_response_time_ms": 900, "timed_requests": 1}}
|
||||
{
|
||||
test_key: {
|
||||
**base,
|
||||
"autorouter_savings_spend": 0.25,
|
||||
"total_response_time_ms": 900,
|
||||
"timed_requests": 1,
|
||||
"timed_completion_tokens": 5,
|
||||
}
|
||||
}
|
||||
)
|
||||
await daily_spend_update_queue.aggregate_queue_updates()
|
||||
updates = await daily_spend_update_queue.flush_all_updates_from_in_memory_queue()
|
||||
|
|
@ -583,3 +591,4 @@ async def test_optional_metric_missing_from_an_older_payload_still_aggregates(
|
|||
assert updates[0][test_key]["autorouter_savings_spend"] == pytest.approx(0.25)
|
||||
assert updates[0][test_key]["total_response_time_ms"] == 900
|
||||
assert updates[0][test_key]["timed_requests"] == 1
|
||||
assert updates[0][test_key]["timed_completion_tokens"] is None
|
||||
|
|
|
|||
|
|
@ -72,6 +72,15 @@ def test_null_and_empty_provider_merge_into_one_row(order):
|
|||
assert folded["api_requests"] == 4
|
||||
|
||||
|
||||
def test_unknown_timed_tokens_remain_unknown_when_rows_merge():
|
||||
legacy = tag_txn()
|
||||
del legacy["timed_completion_tokens"]
|
||||
|
||||
merged = merge_by_conflict_key(TAG_TABLE, (legacy, tag_txn(timed_completion_tokens=7)))
|
||||
|
||||
assert merged[0][1]["timed_completion_tokens"] is None
|
||||
|
||||
|
||||
def test_distinct_keys_are_not_merged_and_are_ordered_deterministically():
|
||||
unordered = (tag_txn(tag="z-team"), tag_txn(tag="a-team"), tag_txn(tag="m-team"))
|
||||
|
||||
|
|
|
|||
|
|
@ -1880,6 +1880,28 @@ def test_grouping_sets_dispatcher_returns_zero_for_timed_requests_without_comple
|
|||
assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 0.0
|
||||
|
||||
|
||||
def test_grouping_sets_dispatcher_returns_no_throughput_for_legacy_rows_without_timed_tokens():
|
||||
from litellm.proxy.management_endpoints.common_daily_activity import (
|
||||
_GROUP_DATE_MODEL_PROVIDER,
|
||||
_aggregate_grouping_sets_records_sync,
|
||||
)
|
||||
|
||||
records = [
|
||||
_grouping_row(
|
||||
_GROUP_DATE_MODEL_PROVIDER,
|
||||
model="gpt-4o",
|
||||
completion_tokens=900,
|
||||
timed_completion_tokens=None,
|
||||
total_response_time_ms=3000,
|
||||
timed_requests=1,
|
||||
)
|
||||
]
|
||||
|
||||
day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0]
|
||||
|
||||
assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second is None
|
||||
|
||||
|
||||
def test_grouping_sets_dispatcher_keeps_ptu_flat_cost_out_of_the_provider_breakdown():
|
||||
"""Sentinel rows carry no provider, so their flat cost must not surface under the
|
||||
"unknown" provider - the per-row path skips them for exactly the same reason."""
|
||||
|
|
|
|||
|
|
@ -42,7 +42,7 @@ export interface MetricWithMetadata {
|
|||
}
|
||||
|
||||
export interface ProviderThroughputMetrics {
|
||||
timed_completion_tokens: number;
|
||||
timed_completion_tokens?: number | null;
|
||||
total_response_time_ms: number;
|
||||
timed_requests: number;
|
||||
output_tokens_per_second?: number | null;
|
||||
|
|
|
|||
7
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
7
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -41076,11 +41076,8 @@ export interface components {
|
|||
ProviderThroughputMetrics: {
|
||||
/** Output Tokens Per Second */
|
||||
output_tokens_per_second?: number | null;
|
||||
/**
|
||||
* Timed Completion Tokens
|
||||
* @default 0
|
||||
*/
|
||||
timed_completion_tokens: number;
|
||||
/** Timed Completion Tokens */
|
||||
timed_completion_tokens?: number | null;
|
||||
/**
|
||||
* Timed Requests
|
||||
* @default 0
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue