mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
* test(integration): streamed Bedrock Messages usage cost equals the recorded spend (Pylon #6667) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): echoed cost-map model info is not persisted as deployment overrides (Pylon #6844) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): reset sweep runs on one pod per tick while replicas share the lease (Pylon #6521) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): bedrock post-call guardrail scans streamed Anthropic Messages tool use without 500 (Pylon #6503) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): realtime cached audio tokens bill at the audio cache-read rate (Pylon #6704) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): legacy GET /spend/logs returns at most the 10000 most recent rows and flags truncation (Pylon #6752) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): itemize Responses API cache write tokens as cache creation cost (Pylon #6454) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): migration entrypoint deploys pending migrations before proxy startup (Pylon #6649) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): opted-in team keys stop at the owner's personal budget (Pylon #6641) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): guardrail information stays in the spend log when the caller sends metadata on /v1/messages (Pylon #6614) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): key model allowlist is enforced on Bedrock passthrough routes (Pylon #6419) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): JWT mapped key backfills a null user email from token claims (Pylon #6266) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): scheduled budget reset recovers from a transient DB transport failure (Pylon #6582) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): team key lists models granted through a team access group (Pylon #6044) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): JWT subject without team claim lands in the configured default team (Pylon #5895) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): end-user spend lands for a key without user_id when the auth cache is Redis (Pylon #6021) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): stale-low redis counter still blocks team member over budget (Pylon #5824) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): prompt-carrying spend rows are written in byte-bounded statements (Pylon #6083) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): logs UI session_total_spend sums every round of a multi-round session (Pylon #5928) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): config.yaml guardrails are served by the guardrail usage detail and overview (Pylon #5813) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): plain chat request skips the object permission lookup (Pylon #5965) Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): register july accounting regression contracts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): isolate cost map override clear on owned proxy Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): make reset lease claim and db relay refusal deterministic Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): poll pg_stat settle, bound unbanned relay refusals, clear reset lease on teardown Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): bound relay refusals so the budget sweep can reconnect Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
176 lines
7.2 KiB
Python
176 lines
7.2 KiB
Python
import base64
|
|
import json
|
|
import uuid
|
|
from pathlib import Path
|
|
from typing import Final
|
|
|
|
import pytest
|
|
from integration._support.client import Gateway, eventually
|
|
from integration._support.database import read_rows
|
|
from integration._support.process import owned_proxy
|
|
from integration._support.upstream import _aws_event_frame
|
|
from integration._support.wire import Reply, Request, wire_server
|
|
from pydantic import JsonValue, TypeAdapter
|
|
|
|
BEDROCK_MODEL: Final = "anthropic.claude-haiku-4-5-20251001-v1:0"
|
|
INPUT_TOKENS: Final = 30
|
|
CACHE_READ_TOKENS: Final = 900
|
|
CACHE_CREATION_TOKENS: Final = 400
|
|
OUTPUT_TOKENS: Final = 57
|
|
INPUT_RATE: Final = 0.001
|
|
OUTPUT_RATE: Final = 0.002
|
|
CACHE_READ_RATE: Final = 0.0001
|
|
CACHE_CREATION_RATE: Final = 0.00125
|
|
STREAMED_USAGE: Final = TypeAdapter(dict[str, float])
|
|
EXPECTED_SPEND: Final = (
|
|
INPUT_TOKENS * INPUT_RATE
|
|
+ CACHE_READ_TOKENS * CACHE_READ_RATE
|
|
+ CACHE_CREATION_TOKENS * CACHE_CREATION_RATE
|
|
+ OUTPUT_TOKENS * OUTPUT_RATE
|
|
)
|
|
|
|
|
|
def _invoke_chunk(payload: dict[str, JsonValue]) -> bytes:
|
|
encoded: Final = base64.b64encode(json.dumps(payload, separators=(",", ":")).encode()).decode()
|
|
return _aws_event_frame("chunk", {"bytes": encoded}, "", "")
|
|
|
|
|
|
def _stream(message_id: str) -> bytes:
|
|
return (
|
|
_invoke_chunk(
|
|
{
|
|
"type": "message_start",
|
|
"message": {
|
|
"id": message_id,
|
|
"type": "message",
|
|
"role": "assistant",
|
|
"model": BEDROCK_MODEL,
|
|
"content": [],
|
|
"stop_reason": None,
|
|
"stop_sequence": None,
|
|
"usage": {
|
|
"input_tokens": INPUT_TOKENS,
|
|
"cache_read_input_tokens": CACHE_READ_TOKENS,
|
|
"cache_creation_input_tokens": CACHE_CREATION_TOKENS,
|
|
"output_tokens": 0,
|
|
},
|
|
},
|
|
}
|
|
)
|
|
+ _invoke_chunk({"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}})
|
|
+ _invoke_chunk(
|
|
{"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "cached answer"}}
|
|
)
|
|
+ _invoke_chunk({"type": "content_block_stop", "index": 0})
|
|
+ _invoke_chunk(
|
|
{
|
|
"type": "message_delta",
|
|
"delta": {"stop_reason": "end_turn"},
|
|
"usage": {
|
|
"input_tokens": INPUT_TOKENS,
|
|
"cache_read_input_tokens": CACHE_READ_TOKENS,
|
|
"cache_creation_input_tokens": CACHE_CREATION_TOKENS,
|
|
"output_tokens": OUTPUT_TOKENS,
|
|
},
|
|
}
|
|
)
|
|
+ _invoke_chunk({"type": "message_stop"})
|
|
)
|
|
|
|
|
|
def _proxy_config(directory: Path, model: str, upstream_url: str) -> Path:
|
|
config: Final = directory / "streamed_usage_cost_config.yaml"
|
|
config.write_text(
|
|
json.dumps(
|
|
{
|
|
"model_list": [
|
|
{
|
|
"model_name": model,
|
|
"litellm_params": {
|
|
"model": f"bedrock/invoke/{BEDROCK_MODEL}",
|
|
"api_base": upstream_url,
|
|
"aws_access_key_id": "AKIASCRIPTEDPROVIDER",
|
|
"aws_secret_access_key": "scripted-secret",
|
|
"aws_region_name": "us-east-1",
|
|
"input_cost_per_token": INPUT_RATE,
|
|
"output_cost_per_token": OUTPUT_RATE,
|
|
"cache_read_input_token_cost": CACHE_READ_RATE,
|
|
"cache_creation_input_token_cost": CACHE_CREATION_RATE,
|
|
},
|
|
}
|
|
],
|
|
"general_settings": {
|
|
"master_key": "os.environ/LITELLM_MASTER_KEY",
|
|
"database_url": "os.environ/DATABASE_URL",
|
|
"store_model_in_db": True,
|
|
"disable_spend_logs": False,
|
|
"proxy_batch_write_at": 1,
|
|
"proxy_batch_polling_interval": 1,
|
|
},
|
|
"litellm_settings": {"include_cost_in_streaming_usage": True},
|
|
"router_settings": {"disable_cooldowns": True},
|
|
}
|
|
)
|
|
)
|
|
return config
|
|
|
|
|
|
def _data_events(body: str) -> tuple[dict[str, JsonValue], ...]:
|
|
return tuple(json.loads(line.removeprefix("data:")) for line in body.splitlines() if line.startswith("data:"))
|
|
|
|
|
|
@pytest.mark.covers("spend.anthropic_messages_stream.streamed_usage_cost_equals_recorded_spend")
|
|
@pytest.mark.timeout(180)
|
|
def test_bedrock_messages_stream_usage_cost_matches_recorded_spend_with_custom_cache_rates(
|
|
gateway: Gateway, tmp_path: Path
|
|
) -> None:
|
|
message_id: Final = f"msg_{uuid.uuid4().hex}"
|
|
model: Final = f"{BEDROCK_MODEL}-{uuid.uuid4().hex}"
|
|
|
|
def respond(request: Request) -> Reply:
|
|
assert request.target == f"/model/{BEDROCK_MODEL}/invoke-with-response-stream", request.target
|
|
assert json.loads(request.body)["messages"] == [{"role": "user", "content": "cached cost control"}], (
|
|
request.body
|
|
)
|
|
return Reply(content_type="application/vnd.amazon.eventstream", chunks=(_stream(message_id),))
|
|
|
|
with wire_server(respond) as wire:
|
|
config: Final = _proxy_config(tmp_path, model, wire.url)
|
|
with owned_proxy(gateway, tmp_path, {}, config=config) as candidate:
|
|
response: Final = candidate.request(
|
|
"POST",
|
|
"/v1/messages",
|
|
{
|
|
"model": model,
|
|
"messages": [{"role": "user", "content": "cached cost control"}],
|
|
"max_tokens": OUTPUT_TOKENS,
|
|
"stream": True,
|
|
},
|
|
)
|
|
assert response.status_code == 200, response.text
|
|
message_delta: Final = next(
|
|
event for event in _data_events(response.text) if event["type"] == "message_delta"
|
|
)
|
|
streamed_usage: Final = STREAMED_USAGE.validate_python(message_delta["usage"])
|
|
rows: Final = eventually(
|
|
lambda: read_rows(
|
|
'SELECT status, prompt_tokens, completion_tokens, spend FROM "LiteLLM_SpendLogs" '
|
|
"WHERE request_id=%s",
|
|
(message_id,),
|
|
),
|
|
lambda values: len(values) == 1,
|
|
seconds=70,
|
|
)
|
|
assert rows[0]["status"] == "success", rows
|
|
assert rows[0]["prompt_tokens"] == INPUT_TOKENS + CACHE_READ_TOKENS + CACHE_CREATION_TOKENS, rows
|
|
assert rows[0]["completion_tokens"] == OUTPUT_TOKENS, rows
|
|
recorded_spend: Final = float(str(rows[0]["spend"]))
|
|
assert recorded_spend == pytest.approx(EXPECTED_SPEND), rows
|
|
assert streamed_usage == {
|
|
"input_tokens": INPUT_TOKENS,
|
|
"cache_read_input_tokens": CACHE_READ_TOKENS,
|
|
"cache_creation_input_tokens": CACHE_CREATION_TOKENS,
|
|
"output_tokens": OUTPUT_TOKENS,
|
|
"cost": pytest.approx(recorded_spend),
|
|
}, (streamed_usage, rows, response.text)
|
|
assert len(wire.drain()) == 1
|