mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-26 01:12:21 +00:00
test(integration): gemini passthrough success releases its budget reservation from the spend counter (Pylon #7295)
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
c65118041a
commit
50c6e18ca7
2 changed files with 105 additions and 0 deletions
|
|
@ -122,6 +122,9 @@
|
|||
"tests/integration/spend/test_cache_and_quota.py::test_different_system_messages_do_not_share_a_cached_response": [
|
||||
"quota_management.response_cache.system_messages_partition_cache_identity"
|
||||
],
|
||||
"tests/integration/spend/test_passthrough_budget_reservation.py::test_repeated_gemini_passthrough_calls_stay_served_while_key_spend_is_below_max_budget": [
|
||||
"spend.budget_reservation.gemini_passthrough_success_releases_reservation_from_spend_counter"
|
||||
],
|
||||
"tests/integration/database/test_transaction_atomicity.py::test_access_group_second_key_constraint_failure_rolls_back_all_writes": [
|
||||
"other.database.access_group.failed_second_write_rolls_back_first"
|
||||
],
|
||||
|
|
|
|||
102
tests/integration/spend/test_passthrough_budget_reservation.py
Normal file
102
tests/integration/spend/test_passthrough_budget_reservation.py
Normal file
|
|
@ -0,0 +1,102 @@
|
|||
import uuid
|
||||
from hashlib import sha256
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from integration._support.client import Gateway, eventually, object_value, string_value
|
||||
from integration._support.database import read_rows
|
||||
from integration._support.upstream import delete_scenario, register_scenario
|
||||
from integration.cost_calculation.cost_tracking_case import JsonResponse
|
||||
from pydantic import JsonValue
|
||||
|
||||
INPUT_COST_PER_TOKEN: Final = 0.000001
|
||||
OUTPUT_COST_PER_TOKEN: Final = 0.001
|
||||
PROMPT_TOKENS: Final = 10
|
||||
CANDIDATE_TOKENS: Final = 5
|
||||
COST_PER_CALL: Final = PROMPT_TOKENS * INPUT_COST_PER_TOKEN + CANDIDATE_TOKENS * OUTPUT_COST_PER_TOKEN
|
||||
MAX_BUDGET: Final = 0.02
|
||||
CALLS_WITHIN_BUDGET: Final = 4
|
||||
|
||||
|
||||
def _key_spend(digest: str) -> float:
|
||||
rows: Final = read_rows('SELECT spend FROM "LiteLLM_VerificationToken" WHERE token=%s', (digest,))
|
||||
assert len(rows) == 1, rows
|
||||
return float(rows[0]["spend"])
|
||||
|
||||
|
||||
def _generate_content_request(model: str) -> dict[str, JsonValue]:
|
||||
return {"contents": [{"role": "user", "parts": [{"text": f"budget {model}"}]}]}
|
||||
|
||||
|
||||
def _generate_content_response(model: str) -> JsonResponse:
|
||||
return JsonResponse(
|
||||
content_type="application/json",
|
||||
body={
|
||||
"candidates": [
|
||||
{
|
||||
"content": {"parts": [{"text": f"scripted answer {model}"}], "role": "model"},
|
||||
"finishReason": "STOP",
|
||||
"index": 0,
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": PROMPT_TOKENS,
|
||||
"candidatesTokenCount": CANDIDATE_TOKENS,
|
||||
"totalTokenCount": PROMPT_TOKENS + CANDIDATE_TOKENS,
|
||||
},
|
||||
"modelVersion": model,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _served_call(gateway: Gateway, model: str, key: str, scenario_id: str, call: int) -> None:
|
||||
digest: Final = sha256(key.encode()).hexdigest()
|
||||
spend_before: Final = _key_spend(digest)
|
||||
assert spend_before == pytest.approx((call - 1) * COST_PER_CALL) and spend_before < MAX_BUDGET
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
f"/gemini/v1beta/models/{model}:generateContent",
|
||||
_generate_content_request(model),
|
||||
headers={"x-goog-api-key": key, "x-pass-x-scripted-scenario": scenario_id},
|
||||
)
|
||||
assert response.status_code == 200, f"call {call} with key spend {spend_before}: {response.text}"
|
||||
assert response.json() == _generate_content_response(model).body, response.text
|
||||
eventually(lambda: _key_spend(digest), lambda spend: spend >= call * COST_PER_CALL - 1e-9, seconds=70)
|
||||
|
||||
|
||||
@pytest.mark.covers("spend.budget_reservation.gemini_passthrough_success_releases_reservation_from_spend_counter")
|
||||
def test_repeated_gemini_passthrough_calls_stay_served_while_key_spend_is_below_max_budget(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
gateway.post(
|
||||
"/config/update",
|
||||
{"environment_variables": {"GEMINI_API_BASE": gateway.upstream_url, "GEMINI_API_KEY": "scripted"}},
|
||||
)
|
||||
model: Final = f"gemini-passthrough-{uuid.uuid4().hex}"
|
||||
created: Final = gateway.post(
|
||||
"/model/new",
|
||||
{
|
||||
"model_name": model,
|
||||
"litellm_params": {
|
||||
"model": "gemini/gemini-2.5-flash",
|
||||
"api_key": "scripted",
|
||||
"api_base": gateway.upstream_url,
|
||||
"input_cost_per_token": INPUT_COST_PER_TOKEN,
|
||||
"output_cost_per_token": OUTPUT_COST_PER_TOKEN,
|
||||
},
|
||||
"model_info": {"id": model, "max_output_tokens": 10},
|
||||
},
|
||||
)
|
||||
scenario.cleanups.callback(scenario.delete_model, string_value(object_value(created["model_info"])["id"]))
|
||||
handle: Final = register_scenario(f"sc-{model}", _generate_content_response(model))
|
||||
scenario.cleanups.callback(delete_scenario, handle)
|
||||
key: Final = scenario.key(models=[model], max_budget=MAX_BUDGET)
|
||||
for call in range(1, CALLS_WITHIN_BUDGET + 1):
|
||||
_served_call(gateway, model, key, handle.scenario_id, call)
|
||||
assert _key_spend(sha256(key.encode()).hexdigest()) == pytest.approx(CALLS_WITHIN_BUDGET * COST_PER_CALL)
|
||||
denied: Final = gateway.request(
|
||||
"POST",
|
||||
f"/gemini/v1beta/models/{model}:generateContent",
|
||||
_generate_content_request(model),
|
||||
headers={"x-goog-api-key": key, "x-pass-x-scripted-scenario": handle.scenario_id},
|
||||
)
|
||||
assert denied.status_code == 422 and denied.json()["error"]["type"] == "budget_exceeded", denied.text
|
||||
Loading…
Add table
Reference in a new issue