mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
* test(integration): streamed Bedrock Messages usage cost equals the recorded spend (Pylon #6667) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): echoed cost-map model info is not persisted as deployment overrides (Pylon #6844) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): reset sweep runs on one pod per tick while replicas share the lease (Pylon #6521) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): bedrock post-call guardrail scans streamed Anthropic Messages tool use without 500 (Pylon #6503) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): realtime cached audio tokens bill at the audio cache-read rate (Pylon #6704) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): legacy GET /spend/logs returns at most the 10000 most recent rows and flags truncation (Pylon #6752) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): itemize Responses API cache write tokens as cache creation cost (Pylon #6454) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): migration entrypoint deploys pending migrations before proxy startup (Pylon #6649) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): opted-in team keys stop at the owner's personal budget (Pylon #6641) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): guardrail information stays in the spend log when the caller sends metadata on /v1/messages (Pylon #6614) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): key model allowlist is enforced on Bedrock passthrough routes (Pylon #6419) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): JWT mapped key backfills a null user email from token claims (Pylon #6266) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): scheduled budget reset recovers from a transient DB transport failure (Pylon #6582) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): team key lists models granted through a team access group (Pylon #6044) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): JWT subject without team claim lands in the configured default team (Pylon #5895) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): end-user spend lands for a key without user_id when the auth cache is Redis (Pylon #6021) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): stale-low redis counter still blocks team member over budget (Pylon #5824) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): prompt-carrying spend rows are written in byte-bounded statements (Pylon #6083) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): logs UI session_total_spend sums every round of a multi-round session (Pylon #5928) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): config.yaml guardrails are served by the guardrail usage detail and overview (Pylon #5813) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): plain chat request skips the object permission lookup (Pylon #5965) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): register july accounting regression contracts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): isolate cost map override clear on owned proxy Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): make reset lease claim and db relay refusal deterministic Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): poll pg_stat settle, bound unbanned relay refusals, clear reset lease on teardown Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): bound relay refusals so the budget sweep can reconnect Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
75 lines
3.1 KiB
Python
75 lines
3.1 KiB
Python
import json
|
|
import os
|
|
from pathlib import Path
|
|
from types import MappingProxyType
|
|
from typing import Final
|
|
|
|
import psycopg
|
|
import pytest
|
|
from integration._support.client import Gateway, eventually
|
|
from integration._support.database import read_rows
|
|
from integration._support.process import owned_proxy
|
|
from redis import Redis
|
|
|
|
RESET_LEASE_KEY: Final = "cronjob_lock:reset_budget_job"
|
|
PEER_POD_LEASE: Final = json.dumps("integration-peer-pod-holding-the-reset-lease")
|
|
FAST_RESET_TICK: Final = MappingProxyType(
|
|
{"PROXY_BUDGET_RESCHEDULER_MIN_TIME": "2", "PROXY_BUDGET_RESCHEDULER_MAX_TIME": "2"}
|
|
)
|
|
|
|
|
|
def _team_spend(team: str) -> float:
|
|
rows: Final = read_rows('SELECT spend FROM "LiteLLM_TeamTable" WHERE team_id = %s', (team,))
|
|
assert len(rows) == 1, rows
|
|
spend: Final = rows[0]["spend"]
|
|
assert isinstance(spend, (int, float)), rows
|
|
return float(spend)
|
|
|
|
|
|
def _make_team_budget_due(team: str, spend: float) -> None:
|
|
with psycopg.connect(os.environ["DATABASE_URL"]) as connection:
|
|
connection.execute(
|
|
"UPDATE \"LiteLLM_TeamTable\" SET spend = %s, budget_reset_at = now() - interval '1 day' "
|
|
"WHERE team_id = %s",
|
|
(spend, team),
|
|
)
|
|
|
|
|
|
@pytest.mark.covers("spend.budget_reset.one_pod_sweeps_per_tick")
|
|
def test_reset_sweep_skips_ticks_while_another_pod_holds_the_lease_and_resumes_after_release(
|
|
gateway: Gateway, tmp_path: Path
|
|
) -> None:
|
|
with (
|
|
gateway.scenario() as scenario,
|
|
Redis(host=os.environ["REDIS_HOST"], port=int(os.environ["REDIS_PORT"])) as cache,
|
|
):
|
|
model: Final = scenario.model(input_cost_per_token=0.0, output_cost_per_token=0.0)
|
|
team: Final = scenario.team(max_budget=1.0, budget_duration="30d")
|
|
key: Final = scenario.key(team_id=team)
|
|
_make_team_budget_due(team, spend=0.5)
|
|
assert _team_spend(team) == 0.5
|
|
eventually(
|
|
lambda: cache.set(RESET_LEASE_KEY, PEER_POD_LEASE, ex=120, nx=True),
|
|
lambda claimed: claimed is True,
|
|
seconds=60,
|
|
)
|
|
try:
|
|
with owned_proxy(gateway, tmp_path, FAST_RESET_TICK) as replica:
|
|
response: Final = replica.request(
|
|
"POST",
|
|
"/v1/chat/completions",
|
|
{"model": model, "messages": [{"role": "user", "content": "classification request"}]},
|
|
key=key,
|
|
)
|
|
assert response.status_code == 200, response.text
|
|
held: Final = eventually(
|
|
lambda: _team_spend(team), lambda spend: spend != 0.5, seconds=8, return_last_on_timeout=True
|
|
)
|
|
assert held == 0.5, f"team {team} was swept while another pod held the reset lease: spend={held}"
|
|
assert cache.get(RESET_LEASE_KEY) == PEER_POD_LEASE.encode()
|
|
cache.delete(RESET_LEASE_KEY)
|
|
swept: Final = eventually(lambda: _team_spend(team), lambda spend: spend == 0.0, seconds=15)
|
|
assert swept == 0.0
|
|
eventually(lambda: cache.get(RESET_LEASE_KEY), lambda value: value is None, seconds=15)
|
|
finally:
|
|
cache.delete(RESET_LEASE_KEY)
|