mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
chore: merge upstream main into neuraltrust guardrail PR
This commit is contained in:
commit
c9c268fa90
85 changed files with 1107 additions and 243 deletions
3
.github/workflows/test-e2e-changed.yml
vendored
3
.github/workflows/test-e2e-changed.yml
vendored
|
|
@ -175,6 +175,8 @@ jobs:
|
|||
env:
|
||||
TESTS: ${{ needs.detect.outputs.tests }}
|
||||
E2E_FIXTURE_MODE: live
|
||||
E2E_PROVIDER_EDGE_HOST_REACHABLE: '1'
|
||||
COLUMNS: '400'
|
||||
run: |
|
||||
umask 077
|
||||
read -r -a test_files <<< "${TESTS}"
|
||||
|
|
@ -189,6 +191,7 @@ jobs:
|
|||
uv run --no-sync python .github/e2e-stack/assert_tests_ran.py "${report}" "${test_files[@]}"
|
||||
verified=$?
|
||||
set -e
|
||||
grep -E '^(FAILED|ERROR) ' "${log}" || true
|
||||
grep -E '^=+ .* in [0-9.]+s( \([0-9:]+\))? =+$' "${log}" | tail -n 1
|
||||
echo "::endgroup::"
|
||||
if [ "${status}" = "5" ]; then
|
||||
|
|
|
|||
|
|
@ -7,6 +7,9 @@ metadata:
|
|||
{{- include "litellm.commonLabels" . | nindent 4 }}
|
||||
app.kubernetes.io/component: backend
|
||||
spec:
|
||||
{{- if and (not .Values.backend.hpa.enabled) (not (kindIs "invalid" .Values.backend.replicaCount)) }}
|
||||
replicas: {{ .Values.backend.replicaCount }}
|
||||
{{- end }}
|
||||
{{- with .Values.backend.strategy }}
|
||||
strategy:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
|
|
|
|||
|
|
@ -7,6 +7,9 @@ metadata:
|
|||
{{- include "litellm.commonLabels" . | nindent 4 }}
|
||||
app.kubernetes.io/component: gateway
|
||||
spec:
|
||||
{{- if and (not .Values.gateway.hpa.enabled) (not (kindIs "invalid" .Values.gateway.replicaCount)) }}
|
||||
replicas: {{ .Values.gateway.replicaCount }}
|
||||
{{- end }}
|
||||
{{- with .Values.gateway.strategy }}
|
||||
strategy:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
|
|
|
|||
|
|
@ -7,6 +7,9 @@ metadata:
|
|||
{{- include "litellm.commonLabels" . | nindent 4 }}
|
||||
app.kubernetes.io/component: ui
|
||||
spec:
|
||||
{{- if and (not .Values.ui.hpa.enabled) (not (kindIs "invalid" .Values.ui.replicaCount)) }}
|
||||
replicas: {{ .Values.ui.replicaCount }}
|
||||
{{- end }}
|
||||
{{- with .Values.ui.strategy }}
|
||||
strategy:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
|
|
|
|||
100
helm/litellm/tests/replica_count_tests.yaml
Normal file
100
helm/litellm/tests/replica_count_tests.yaml
Normal file
|
|
@ -0,0 +1,100 @@
|
|||
suite: test fixed replica count when HPA is disabled
|
||||
templates:
|
||||
- gateway/deployment.yaml
|
||||
- gateway/configmap.yaml
|
||||
- backend/deployment.yaml
|
||||
- ui/deployment.yaml
|
||||
values:
|
||||
- ./values/required.yaml
|
||||
tests:
|
||||
- it: gateway renders replicaCount into spec.replicas when its HPA is disabled
|
||||
template: gateway/deployment.yaml
|
||||
set:
|
||||
gateway.hpa.enabled: false
|
||||
gateway.replicaCount: 3
|
||||
asserts:
|
||||
- isKind:
|
||||
of: Deployment
|
||||
- equal:
|
||||
path: spec.replicas
|
||||
value: 3
|
||||
|
||||
- it: backend renders replicaCount into spec.replicas when its HPA is disabled
|
||||
template: backend/deployment.yaml
|
||||
set:
|
||||
backend.hpa.enabled: false
|
||||
backend.replicaCount: 2
|
||||
asserts:
|
||||
- equal:
|
||||
path: spec.replicas
|
||||
value: 2
|
||||
|
||||
- it: ui renders replicaCount into spec.replicas when its HPA is disabled
|
||||
template: ui/deployment.yaml
|
||||
set:
|
||||
ui.hpa.enabled: false
|
||||
ui.replicaCount: 2
|
||||
asserts:
|
||||
- equal:
|
||||
path: spec.replicas
|
||||
value: 2
|
||||
|
||||
- it: replicaCount 0 scales the gateway to zero instead of being treated as unset
|
||||
template: gateway/deployment.yaml
|
||||
set:
|
||||
gateway.hpa.enabled: false
|
||||
gateway.replicaCount: 0
|
||||
asserts:
|
||||
- equal:
|
||||
path: spec.replicas
|
||||
value: 0
|
||||
|
||||
- it: a component with HPA disabled but no replicaCount set keeps omitting spec.replicas, so upgrades do not reset a hand-scaled Deployment
|
||||
set:
|
||||
gateway.hpa.enabled: false
|
||||
backend.hpa.enabled: false
|
||||
ui.hpa.enabled: false
|
||||
asserts:
|
||||
- notExists:
|
||||
path: spec.replicas
|
||||
template: gateway/deployment.yaml
|
||||
- notExists:
|
||||
path: spec.replicas
|
||||
template: backend/deployment.yaml
|
||||
- notExists:
|
||||
path: spec.replicas
|
||||
template: ui/deployment.yaml
|
||||
|
||||
- it: every component omits spec.replicas when its HPA is enabled, so the autoscaler owns the count
|
||||
set:
|
||||
gateway.hpa.enabled: true
|
||||
gateway.replicaCount: 3
|
||||
backend.hpa.enabled: true
|
||||
backend.replicaCount: 3
|
||||
ui.hpa.enabled: true
|
||||
ui.replicaCount: 3
|
||||
asserts:
|
||||
- notExists:
|
||||
path: spec.replicas
|
||||
template: gateway/deployment.yaml
|
||||
- notExists:
|
||||
path: spec.replicas
|
||||
template: backend/deployment.yaml
|
||||
- notExists:
|
||||
path: spec.replicas
|
||||
template: ui/deployment.yaml
|
||||
|
||||
- it: a component with HPA disabled renders replicas while a sibling with HPA enabled does not
|
||||
set:
|
||||
gateway.hpa.enabled: false
|
||||
gateway.replicaCount: 4
|
||||
backend.hpa.enabled: true
|
||||
backend.replicaCount: 4
|
||||
asserts:
|
||||
- equal:
|
||||
path: spec.replicas
|
||||
value: 4
|
||||
template: gateway/deployment.yaml
|
||||
- notExists:
|
||||
path: spec.replicas
|
||||
template: backend/deployment.yaml
|
||||
|
|
@ -397,6 +397,11 @@ gateway:
|
|||
# failureThreshold: 30
|
||||
# periodSeconds: 10
|
||||
startupProbe: {}
|
||||
# Optional fixed pod count, rendered into the Deployment's spec.replicas only
|
||||
# when hpa.enabled is false. Unset by default so an existing Deployment keeps
|
||||
# its current count; with the HPA on, the autoscaler owns the count, e.g.:
|
||||
# replicaCount: 3
|
||||
replicaCount:
|
||||
hpa:
|
||||
enabled: true
|
||||
minReplicas: 1
|
||||
|
|
@ -524,6 +529,8 @@ backend:
|
|||
strategy: {}
|
||||
# Optional startupProbe; same shape as gateway.startupProbe. Empty by default.
|
||||
startupProbe: {}
|
||||
# Same semantics as gateway.replicaCount.
|
||||
replicaCount:
|
||||
hpa:
|
||||
enabled: true
|
||||
minReplicas: 1
|
||||
|
|
@ -590,6 +597,8 @@ ui:
|
|||
strategy: {}
|
||||
# Optional startupProbe; same shape as gateway.startupProbe. Empty by default.
|
||||
startupProbe: {}
|
||||
# Same semantics as gateway.replicaCount.
|
||||
replicaCount:
|
||||
hpa:
|
||||
enabled: false
|
||||
minReplicas: 1
|
||||
|
|
|
|||
|
|
@ -400,6 +400,7 @@ default_redis_batch_cache_expiry: Optional[float] = None
|
|||
model_alias_map: Dict[str, str] = {}
|
||||
model_group_settings: Optional["ModelGroupSettings"] = None
|
||||
max_budget: float = 0.0 # set the max budget across all providers
|
||||
budget_exceeded_status_code: int = 422 # set to 429 to restore the pre-422 budget_exceeded response code
|
||||
budget_duration: Optional[str] = (
|
||||
None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
|
||||
)
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ import logging
|
|||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections.abc import Sequence
|
||||
from datetime import datetime
|
||||
from logging import Formatter
|
||||
from typing import Any, Final, TextIO
|
||||
|
|
@ -186,7 +187,8 @@ class SecretRedactionFilter(logging.Filter):
|
|||
record.stack_info = _redact_string(record.stack_info) # rebind-ok: a Filter scrubs records in place
|
||||
|
||||
# Redact extra fields passed via logger.debug("msg", extra={...})
|
||||
for key, value in list(record.__dict__.items()):
|
||||
record_items: Final[Sequence[tuple[str, object]]] = list(record.__dict__.items())
|
||||
for key, value in record_items:
|
||||
if key in _STANDARD_RECORD_ATTRS:
|
||||
continue
|
||||
if isinstance(value, str):
|
||||
|
|
@ -507,7 +509,7 @@ handler.addFilter(_secret_filter)
|
|||
handler.addFilter(_correlation_filter)
|
||||
|
||||
|
||||
def _try_parse_json_message(message: str) -> dict[str, Any] | None:
|
||||
def _try_parse_json_message(message: str) -> dict[str, object] | None:
|
||||
"""
|
||||
Try to parse a log message as JSON. Returns parsed dict if valid, else None.
|
||||
Handles messages that are entirely valid JSON (e.g. json.dumps output).
|
||||
|
|
@ -585,7 +587,7 @@ class JsonFormatter(Formatter):
|
|||
|
||||
def format(self, record):
|
||||
message_str: Final = record.getMessage()
|
||||
json_record: Final[dict[str, Any]] = {
|
||||
json_record: Final[dict[str, object]] = {
|
||||
"message": message_str,
|
||||
"level": record.levelname,
|
||||
"timestamp": self.formatTime(record),
|
||||
|
|
|
|||
|
|
@ -98,7 +98,7 @@ class BedrockAgentCoreA2AHandler:
|
|||
request_id=request_id,
|
||||
params=params,
|
||||
litellm_params=litellm_params,
|
||||
method="message/send",
|
||||
method="message/stream",
|
||||
stream=True,
|
||||
agent_extra_headers=agent_extra_headers,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ A2A Streaming Iterator with token tracking and logging support.
|
|||
import asyncio
|
||||
from collections.abc import AsyncIterator
|
||||
from datetime import datetime
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Final
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
|
|
@ -15,7 +15,7 @@ from litellm.litellm_core_utils.asyncify import asyncify
|
|||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from a2a.types import SendStreamingMessageRequest, SendStreamingMessageResponse
|
||||
from a2a.compat.v0_3.types import SendStreamingMessageRequest, SendStreamingMessageResponse
|
||||
|
||||
|
||||
class A2AStreamingIterator:
|
||||
|
|
@ -39,9 +39,9 @@ class A2AStreamingIterator:
|
|||
self.start_time = datetime.now()
|
||||
|
||||
# Collect chunks for token counting
|
||||
self.chunks: list[Any] = []
|
||||
self.chunks: list[SendStreamingMessageResponse] = []
|
||||
self.collected_text_parts: list[str] = []
|
||||
self.final_chunk: Any | None = None
|
||||
self.final_chunk: SendStreamingMessageResponse | None = None
|
||||
|
||||
def __aiter__(self):
|
||||
return self
|
||||
|
|
@ -69,7 +69,7 @@ class A2AStreamingIterator:
|
|||
await self._handle_stream_complete()
|
||||
raise
|
||||
|
||||
def _collect_text_from_chunk(self, chunk: Any) -> None:
|
||||
def _collect_text_from_chunk(self, chunk: "SendStreamingMessageResponse") -> None:
|
||||
"""Extract text from a streaming chunk and add to collected parts."""
|
||||
try:
|
||||
chunk_dict: Final = chunk.model_dump(mode="json", exclude_none=True) if hasattr(chunk, "model_dump") else {}
|
||||
|
|
@ -79,7 +79,7 @@ class A2AStreamingIterator:
|
|||
except Exception:
|
||||
verbose_logger.debug("Failed to extract text from A2A streaming chunk")
|
||||
|
||||
def _is_completed_chunk(self, chunk: Any) -> bool:
|
||||
def _is_completed_chunk(self, chunk: "SendStreamingMessageResponse") -> bool:
|
||||
"""Check if chunk indicates stream completion."""
|
||||
try:
|
||||
chunk_dict: Final = chunk.model_dump(mode="json", exclude_none=True) if hasattr(chunk, "model_dump") else {}
|
||||
|
|
|
|||
|
|
@ -16,6 +16,7 @@ from typing import Any, Final
|
|||
import httpx
|
||||
import openai
|
||||
|
||||
import litellm
|
||||
from litellm.types.utils import LiteLLMCommonStrings
|
||||
from litellm.types.vector_stores import VectorStoreSearchFailure
|
||||
|
||||
|
|
@ -1002,7 +1003,7 @@ class BudgetExceededError(Exception):
|
|||
):
|
||||
self.current_cost = current_cost
|
||||
self.max_budget = max_budget
|
||||
self.status_code = 429
|
||||
self.status_code = litellm.budget_exceeded_status_code
|
||||
self.llm_provider = llm_provider or ""
|
||||
self.entity_type = entity_type
|
||||
self.entity_id = entity_id
|
||||
|
|
|
|||
|
|
@ -15,6 +15,8 @@ from .destinations import FocusTimeWindow
|
|||
if TYPE_CHECKING:
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
|
||||
from litellm.proxy.db.db_transaction_queue.pod_lock_manager import PodLockManager
|
||||
|
||||
from .export_engine import FocusExportEngine
|
||||
else:
|
||||
AsyncIOScheduler = Any
|
||||
|
|
@ -111,7 +113,7 @@ class FocusLogger(CustomLogger):
|
|||
"""Entry point for scheduler jobs to run export cycle with locking."""
|
||||
from litellm.proxy.proxy_server import proxy_logging_obj
|
||||
|
||||
pod_lock_manager = None
|
||||
pod_lock_manager: PodLockManager | None = None
|
||||
if proxy_logging_obj is not None:
|
||||
writer: Final = getattr(proxy_logging_obj, "db_spend_update_writer", None)
|
||||
if writer is not None:
|
||||
|
|
|
|||
|
|
@ -58,7 +58,7 @@ class GenericPromptManager(CustomPromptManagement):
|
|||
api_key: str | None = None,
|
||||
timeout: int = 30,
|
||||
prompt_id: str | None = None,
|
||||
additional_provider_specific_query_params: dict[str, Any] | None = None,
|
||||
additional_provider_specific_query_params: Mapping[str, object] | None = None,
|
||||
**kwargs,
|
||||
):
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ from __future__ import annotations
|
|||
|
||||
import os
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, Protocol
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
|
|
@ -35,6 +35,17 @@ else:
|
|||
AsyncIOScheduler = Any
|
||||
|
||||
|
||||
class _PodLockManager(Protocol):
|
||||
"""The subset of PodLockManager this logger drives to serialize the export across pods."""
|
||||
|
||||
@property
|
||||
def redis_cache(self) -> object: ...
|
||||
|
||||
async def acquire_lock(self, cronjob_id: str) -> bool | None: ...
|
||||
|
||||
async def release_lock(self, cronjob_id: str) -> None: ...
|
||||
|
||||
|
||||
def _parse_metrics_marker(
|
||||
marker: object | None,
|
||||
) -> datetime | None:
|
||||
|
|
@ -226,9 +237,9 @@ class MavvrikFocusLogger(FocusLogger):
|
|||
"""Scheduler entry point — uses Mavvrik-specific pod-lock key."""
|
||||
from litellm.proxy.proxy_server import proxy_logging_obj # noqa: PLC0415
|
||||
|
||||
pod_lock_manager = None
|
||||
pod_lock_manager: _PodLockManager | None = None
|
||||
if proxy_logging_obj is not None:
|
||||
writer: Final = getattr(proxy_logging_obj, "db_spend_update_writer", None)
|
||||
writer: Final[object] = getattr(proxy_logging_obj, "db_spend_update_writer", None)
|
||||
if writer is not None:
|
||||
pod_lock_manager = getattr(writer, "pod_lock_manager", None)
|
||||
|
||||
|
|
|
|||
|
|
@ -9,9 +9,12 @@ this preset registers a custom exporter (``kind="agentops"``) that mints the JWT
|
|||
worker thread, off any event loop — and caches it for the process lifetime.
|
||||
"""
|
||||
|
||||
from collections.abc import Sequence
|
||||
from typing import Any, Final
|
||||
|
||||
import httpx
|
||||
from opentelemetry.sdk.trace import ReadableSpan
|
||||
from opentelemetry.sdk.trace.export import SpanExporter, SpanExportResult
|
||||
from pydantic import Field
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
|
|
@ -71,7 +74,7 @@ def agentops_preset(
|
|||
)
|
||||
|
||||
|
||||
def _build_agentops_exporter(spec: ExporterSpec) -> Any:
|
||||
def _build_agentops_exporter(spec: ExporterSpec) -> SpanExporter:
|
||||
"""Factory for the ``agentops`` exporter kind: a lazy-auth OTLP/HTTP exporter."""
|
||||
from opentelemetry.exporter.otlp.proto.http.trace_exporter import (
|
||||
OTLPSpanExporter,
|
||||
|
|
@ -106,7 +109,7 @@ def _build_agentops_exporter(spec: ExporterSpec) -> Any:
|
|||
except Exception as e:
|
||||
verbose_logger.debug("AgentOps JWT fetch failed: %s", e)
|
||||
|
||||
def export(self, spans: Any) -> Any:
|
||||
def export(self, spans: Sequence[ReadableSpan]) -> SpanExportResult:
|
||||
self._ensure_authenticated()
|
||||
return super().export(spans)
|
||||
|
||||
|
|
|
|||
|
|
@ -8,13 +8,16 @@ identity unconditionally.
|
|||
"""
|
||||
|
||||
from collections.abc import Callable, Iterator
|
||||
from contextlib import contextmanager
|
||||
from contextlib import AbstractContextManager, contextmanager
|
||||
from functools import cache
|
||||
from typing import Any, Final
|
||||
from typing import TYPE_CHECKING, Final
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from opentelemetry.trace import Span
|
||||
|
||||
|
||||
@cache
|
||||
def _otel_runtime() -> "tuple[Callable[[str], Any], Callable[..., None]] | None":
|
||||
def _otel_runtime() -> "tuple[Callable[[str], AbstractContextManager[Span | None]], Callable[..., None]] | None":
|
||||
"""Resolve the SDK-backed hooks once and cache the outcome, absence included.
|
||||
|
||||
CPython never caches a failed import, so without this memoization every call
|
||||
|
|
@ -29,7 +32,7 @@ def _otel_runtime() -> "tuple[Callable[[str], Any], Callable[..., None]] | None"
|
|||
|
||||
|
||||
@contextmanager
|
||||
def phase_span(name: str) -> "Iterator[Any]":
|
||||
def phase_span(name: str) -> "Iterator[Span | None]":
|
||||
"""Run a request phase inside a live active span so its DB/service calls nest.
|
||||
|
||||
Yields ``None`` (a plain no-op) when the OTel SDK is unavailable or V2 is not
|
||||
|
|
@ -43,7 +46,7 @@ def phase_span(name: str) -> "Iterator[Any]":
|
|||
yield span
|
||||
|
||||
|
||||
def seed_request_identity(user_api_key_dict: Any, model: Any = None) -> None:
|
||||
def seed_request_identity(user_api_key_dict: object, model: object = None) -> None:
|
||||
"""Seed request-identity Baggage at the auth boundary (no-op without V2)."""
|
||||
runtime: Final = _otel_runtime()
|
||||
if runtime is None:
|
||||
|
|
|
|||
|
|
@ -3,7 +3,13 @@ from __future__ import annotations
|
|||
import time
|
||||
from collections import OrderedDict
|
||||
from threading import RLock
|
||||
from typing import Any, Final
|
||||
from typing import Final, Protocol
|
||||
|
||||
|
||||
class _RemovableMetric(Protocol):
|
||||
"""The one prometheus-client metric method this tracker calls."""
|
||||
|
||||
def remove(self, *labelvalues: object) -> None: ...
|
||||
|
||||
|
||||
class BoundedPrometheusSeriesTracker:
|
||||
|
|
@ -21,7 +27,7 @@ class BoundedPrometheusSeriesTracker:
|
|||
|
||||
def track_series(
|
||||
self,
|
||||
metric: Any,
|
||||
metric: _RemovableMetric,
|
||||
metric_name: str,
|
||||
label_values: tuple[str | None, ...],
|
||||
max_series: int | None,
|
||||
|
|
@ -60,7 +66,7 @@ class BoundedPrometheusSeriesTracker:
|
|||
break
|
||||
del series[tracked_label_values]
|
||||
|
||||
def remove_series(self, metric: object, label_values: tuple[str | None, ...]) -> bool:
|
||||
def remove_series(self, metric: _RemovableMetric, label_values: tuple[str | None, ...]) -> bool:
|
||||
"""Drop one child series, True when it is gone (removed or never existed)."""
|
||||
return self._remove_metric_child(metric, label_values)
|
||||
|
||||
|
|
@ -82,7 +88,7 @@ class BoundedPrometheusSeriesTracker:
|
|||
|
||||
def _remove_metric_series(
|
||||
self,
|
||||
metric: Any,
|
||||
metric: _RemovableMetric,
|
||||
series: OrderedDict[tuple[str | None, ...], float],
|
||||
label_values: tuple[str | None, ...],
|
||||
) -> None:
|
||||
|
|
@ -90,7 +96,7 @@ class BoundedPrometheusSeriesTracker:
|
|||
series.pop(label_values, None)
|
||||
|
||||
@staticmethod
|
||||
def _remove_metric_child(metric: Any, label_values: tuple[str | None, ...]) -> bool:
|
||||
def _remove_metric_child(metric: _RemovableMetric, label_values: tuple[str | None, ...]) -> bool:
|
||||
"""
|
||||
Remove the Prometheus child for ``label_values`` and report whether the
|
||||
tracker should commit the matching state change.
|
||||
|
|
|
|||
|
|
@ -406,7 +406,7 @@ class VectorStorePreCallHook(CustomLogger):
|
|||
request_data: dict,
|
||||
response_chunk: Any,
|
||||
call_type: CallTypes | None,
|
||||
) -> Any | None:
|
||||
) -> object | None:
|
||||
"""
|
||||
Add search results to the final streaming chunk.
|
||||
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ imported_openAIResponse = True
|
|||
try:
|
||||
import io
|
||||
import logging
|
||||
from collections.abc import Mapping
|
||||
from typing import Any, Literal, Protocol, TypeVar
|
||||
|
||||
from wandb.sdk.data_types import trace_tree
|
||||
|
|
@ -43,7 +44,7 @@ try:
|
|||
|
||||
@staticmethod
|
||||
def results_to_trace_tree(
|
||||
request: dict[str, Any],
|
||||
request: Mapping[str, object],
|
||||
response: OpenAIResponse,
|
||||
results: list[trace_tree.Result],
|
||||
time_elapsed: float,
|
||||
|
|
@ -73,7 +74,7 @@ try:
|
|||
|
||||
def _resolve_edit(
|
||||
self,
|
||||
request: dict[str, Any],
|
||||
request: Mapping[str, object],
|
||||
response: OpenAIResponse,
|
||||
time_elapsed: float,
|
||||
) -> trace_tree.WBTraceTree:
|
||||
|
|
@ -91,7 +92,7 @@ try:
|
|||
|
||||
def _resolve_completion(
|
||||
self,
|
||||
request: dict[str, Any],
|
||||
request: Mapping[str, object],
|
||||
response: OpenAIResponse,
|
||||
time_elapsed: float,
|
||||
) -> trace_tree.WBTraceTree:
|
||||
|
|
@ -134,7 +135,7 @@ try:
|
|||
|
||||
def _request_response_result_to_trace(
|
||||
self,
|
||||
request: dict[str, Any],
|
||||
request: Mapping[str, object],
|
||||
response: OpenAIResponse,
|
||||
request_str: str,
|
||||
choices: list[str],
|
||||
|
|
|
|||
|
|
@ -50,7 +50,7 @@ _INTERACTIONS_MODALITY_FIELDS: Final[Mapping[str, str]] = MappingProxyType(
|
|||
)
|
||||
|
||||
|
||||
def _modality_field(entry: Mapping[str, Any]) -> str | None:
|
||||
def _modality_field(entry: Mapping[str, object]) -> str | None:
|
||||
return _INTERACTIONS_MODALITY_FIELDS.get(str(entry.get("modality", "")).lower())
|
||||
|
||||
|
||||
|
|
@ -58,7 +58,7 @@ def _token_count(value: object) -> int:
|
|||
return value if isinstance(value, int) else 0
|
||||
|
||||
|
||||
def _modality_token_sums(entries: Sequence[Mapping[str, Any]]) -> Mapping[str, int]:
|
||||
def _modality_token_sums(entries: Sequence[Mapping[str, object]]) -> Mapping[str, int]:
|
||||
fields: Final = frozenset(field for entry in entries if (field := _modality_field(entry)) is not None)
|
||||
return MappingProxyType(
|
||||
{
|
||||
|
|
@ -68,7 +68,7 @@ def _modality_token_sums(entries: Sequence[Mapping[str, Any]]) -> Mapping[str, i
|
|||
)
|
||||
|
||||
|
||||
def _google_search_query_count(usage_object: Mapping[str, Any]) -> int:
|
||||
def _google_search_query_count(usage_object: Mapping[str, object]) -> int:
|
||||
entries: Final = usage_object.get("grounding_tool_count")
|
||||
if not isinstance(entries, Sequence):
|
||||
return 0
|
||||
|
|
|
|||
|
|
@ -85,7 +85,7 @@ def safe_json_structure(
|
|||
|
||||
|
||||
def safe_dumps(
|
||||
data: Any,
|
||||
data: object,
|
||||
max_depth: int = DEFAULT_MAX_RECURSE_DEPTH,
|
||||
value_transform: Callable[[str | None, str], str] | None = None,
|
||||
) -> str:
|
||||
|
|
|
|||
|
|
@ -35,7 +35,7 @@ if TYPE_CHECKING:
|
|||
from litellm.router import Router
|
||||
|
||||
# Anthropic-only keys already mapped by the translator; strip on extra_kwargs re-merge.
|
||||
ANTHROPIC_ONLY_REQUEST_KEYS: Final[frozenset[str]] = frozenset({"output_config"})
|
||||
ANTHROPIC_ONLY_REQUEST_KEYS: Final[frozenset[str]] = frozenset({"output_config", "safeguards"})
|
||||
|
||||
_AnthropicMessages: TypeAlias = "list[dict[str, object]]"
|
||||
_AnthropicSystem: TypeAlias = "str | list[dict[str, object]] | None"
|
||||
|
|
|
|||
|
|
@ -79,10 +79,14 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
"speed",
|
||||
"output_config",
|
||||
"reasoning_effort",
|
||||
"safeguards",
|
||||
# TODO: Add Anthropic `metadata` support
|
||||
# "metadata",
|
||||
]
|
||||
|
||||
def should_filter_anthropic_beta_headers(self) -> bool:
|
||||
return self._resolved_provider != "anthropic"
|
||||
|
||||
def _remove_scope_from_cache_control(self, anthropic_messages_request: dict) -> None:
|
||||
"""
|
||||
Remove `scope` field from cache_control blocks.
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
from collections.abc import Coroutine
|
||||
from typing import Any, Final, cast
|
||||
from typing import Final, cast
|
||||
|
||||
import httpx
|
||||
from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI
|
||||
|
|
@ -19,7 +19,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
|
|||
"""
|
||||
|
||||
@staticmethod
|
||||
def _ensure_training_type(create_fine_tuning_job_data: dict[str, Any]) -> None:
|
||||
def _ensure_training_type(create_fine_tuning_job_data: dict[str, object]) -> None:
|
||||
"""
|
||||
Azure requires trainingType in extra_body. Default to 1 (supervised) if omitted.
|
||||
"""
|
||||
|
|
@ -66,7 +66,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
|
|||
max_retries: int | None,
|
||||
organization: str | None,
|
||||
client: OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None = None,
|
||||
) -> LiteLLMFineTuningJob | Coroutine[Any, Any, LiteLLMFineTuningJob]:
|
||||
) -> LiteLLMFineTuningJob | Coroutine[object, object, LiteLLMFineTuningJob]:
|
||||
self._ensure_training_type(create_fine_tuning_job_data)
|
||||
|
||||
openai_client: Final[OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None] = self.get_openai_client(
|
||||
|
|
@ -109,7 +109,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
|
|||
max_retries: int | None,
|
||||
organization: str | None,
|
||||
client: OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None = None,
|
||||
) -> LiteLLMFineTuningJob | Coroutine[Any, Any, LiteLLMFineTuningJob]:
|
||||
) -> LiteLLMFineTuningJob | Coroutine[object, object, LiteLLMFineTuningJob]:
|
||||
openai_client: Final[OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None] = self.get_openai_client(
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
|
|
@ -149,7 +149,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
|
|||
max_retries: int | None,
|
||||
organization: str | None,
|
||||
client: OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None = None,
|
||||
) -> LiteLLMFineTuningJob | Coroutine[Any, Any, LiteLLMFineTuningJob]:
|
||||
) -> LiteLLMFineTuningJob | Coroutine[object, object, LiteLLMFineTuningJob]:
|
||||
openai_client: Final[OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None] = self.get_openai_client(
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ class CodestralTextCompletionConfig(OpenAITextCompletionConfig):
|
|||
random_seed: int | None = None,
|
||||
stop: str | None = None,
|
||||
) -> None:
|
||||
locals_: Final = locals().copy()
|
||||
locals_: Final[dict[str, object]] = locals().copy()
|
||||
for key, value in locals_.items():
|
||||
if key != "self" and value is not None:
|
||||
setattr(self.__class__, key, value)
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ Authentication priority:
|
|||
|
||||
import os
|
||||
import re
|
||||
from typing import Any, Final, Literal
|
||||
from typing import Final, Literal
|
||||
from urllib.parse import urlsplit, urlunsplit
|
||||
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
|
@ -48,7 +48,7 @@ class DatabricksBase:
|
|||
]
|
||||
|
||||
@classmethod
|
||||
def redact_sensitive_data(cls, data: Any) -> Any:
|
||||
def redact_sensitive_data(cls, data: object) -> object:
|
||||
"""
|
||||
Redact sensitive information (tokens, secrets) from data before logging.
|
||||
|
||||
|
|
|
|||
|
|
@ -453,7 +453,7 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
|
|||
return normalized
|
||||
|
||||
@staticmethod
|
||||
def _finalize_gemini_live_setup(model: str, setup: dict[str, Any]) -> dict[str, Any]:
|
||||
def _finalize_gemini_live_setup(model: str, setup: dict[str, object]) -> dict[str, object]:
|
||||
generation_config: Final = setup.get("generationConfig")
|
||||
if isinstance(generation_config, dict):
|
||||
modalities: Final = generation_config.get("responseModalities")
|
||||
|
|
@ -1172,7 +1172,7 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
|
|||
def map_openai_event(
|
||||
self,
|
||||
key: str,
|
||||
value: Any,
|
||||
value: object,
|
||||
current_delta_type: ALL_DELTA_TYPES | None,
|
||||
) -> OpenAIRealtimeEventTypes | ResponsesAPIStreamEvents:
|
||||
if isinstance(value, dict):
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@ class JinaAIEmbeddingConfig(BaseEmbeddingConfig):
|
|||
def __init__(
|
||||
self,
|
||||
) -> None:
|
||||
locals_: Final = locals().copy()
|
||||
locals_: Final[dict[str, object]] = locals().copy()
|
||||
for key, value in locals_.items():
|
||||
if key != "self" and value is not None:
|
||||
setattr(self.__class__, key, value)
|
||||
|
|
|
|||
|
|
@ -170,7 +170,9 @@ class OpenrouterEmbeddingConfig(BaseEmbeddingConfig):
|
|||
optional_params[param] = value
|
||||
return optional_params
|
||||
|
||||
def get_error_class(self, error_message: str, status_code: int, headers: Any) -> Any:
|
||||
def get_error_class(
|
||||
self, error_message: str, status_code: int, headers: dict[str, str] | httpx.Headers
|
||||
) -> OpenRouterException:
|
||||
"""
|
||||
Get the error class for OpenRouter errors.
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -3,6 +3,8 @@ import binascii
|
|||
from collections import defaultdict
|
||||
from typing import TYPE_CHECKING, Any, Final, NoReturn
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.constants import request_timeout
|
||||
|
||||
REDUCTO_API_BASE: Final = "https://platform.reducto.ai"
|
||||
|
|
@ -62,7 +64,7 @@ def extract_file_id_or_bytes(
|
|||
return None, raw_bytes, mime
|
||||
|
||||
|
||||
def _extract_file_id_from_upload_response(response: Any) -> str:
|
||||
def _extract_file_id_from_upload_response(response: httpx.Response) -> str:
|
||||
try:
|
||||
payload: Final = response.json()
|
||||
except ValueError as exc:
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ from typing import TYPE_CHECKING, Any, Final
|
|||
|
||||
import httpx
|
||||
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.llms.openai import AllEmbeddingInputValues
|
||||
|
|
@ -160,7 +161,7 @@ class VercelAIGatewayEmbeddingConfig(BaseEmbeddingConfig):
|
|||
optional_params[param] = value
|
||||
return optional_params
|
||||
|
||||
def get_error_class(self, error_message: str, status_code: int, headers: Any) -> Any:
|
||||
def get_error_class(self, error_message: str, status_code: int, headers: Any) -> BaseLLMException:
|
||||
"""
|
||||
Get the error class for Vercel AI Gateway errors.
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -205,7 +205,7 @@ class VertexAgentEngineConfig(BaseConfig, VertexBase):
|
|||
session_id: Final = self._get_session_id(optional_params)
|
||||
|
||||
# Build the input
|
||||
input_data: Final[dict[str, Any]] = {
|
||||
input_data: Final[dict[str, str]] = {
|
||||
"message": prompt,
|
||||
"user_id": user_id,
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1327,7 +1327,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 2048,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"anthropic.claude-mythos-preview": {
|
||||
"input_cost_per_token": 0,
|
||||
|
|
@ -1381,7 +1381,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 2048,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-opus-4-7": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -1419,7 +1419,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 2048,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-opus-4-7": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -1531,7 +1531,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"anthropic.claude-fable-5-1": {
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
|
|
@ -1570,7 +1570,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-fable-5": {
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
|
|
@ -1608,7 +1608,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-fable-5-1": {
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
|
|
@ -1647,7 +1647,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-fable-5": {
|
||||
"cache_creation_input_token_cost": 1.375e-05,
|
||||
|
|
@ -1685,7 +1685,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-fable-5-1": {
|
||||
"cache_creation_input_token_cost": 1.375e-05,
|
||||
|
|
@ -1724,7 +1724,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-fable-5": {
|
||||
"cache_creation_input_token_cost": 1.375e-05,
|
||||
|
|
@ -1837,7 +1837,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-opus-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -1875,7 +1875,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-opus-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -1913,7 +1913,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-opus-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2063,7 +2063,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-opus-4-8": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2102,7 +2102,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-opus-4-8": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2141,7 +2141,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-opus-4-8": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2329,7 +2329,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-sonnet-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2368,7 +2368,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-sonnet-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2407,7 +2407,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-sonnet-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2556,7 +2556,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
|
|
@ -2591,7 +2591,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
|
|
@ -2626,7 +2626,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
|
|
@ -3740,6 +3740,21 @@
|
|||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"azure_ai/gpt-image-2": {
|
||||
"cache_read_input_image_token_cost": 2e-06,
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"input_cost_per_image_token": 8e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"mode": "image_generation",
|
||||
"output_cost_per_image_token": 3e-05,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
|
||||
"supported_endpoints": [
|
||||
"/v1/images/generations",
|
||||
"/v1/images/edits"
|
||||
],
|
||||
"supports_vision": true
|
||||
},
|
||||
"azure_ai/codex-mini": {
|
||||
"cache_read_input_token_cost": 3.75e-07,
|
||||
"deprecation_date": "2026-11-15",
|
||||
|
|
@ -11159,6 +11174,20 @@
|
|||
],
|
||||
"deprecation_date": "2026-10-01"
|
||||
},
|
||||
"azure_ai/MAI-Image-2.5-Pro": {
|
||||
"deprecation_date": "2026-10-01",
|
||||
"input_cost_per_image_token": 8e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.1085,
|
||||
"output_cost_per_image_token": 0.000106,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-mai-image-2-5-pro-and-mai-voice-2-flash-in-microsoft-foundry/4539446",
|
||||
"supported_endpoints": [
|
||||
"/v1/images/generations",
|
||||
"/v1/images/edits"
|
||||
]
|
||||
},
|
||||
"azure_ai/MAI-Image-2e": {
|
||||
"deprecation_date": "2026-08-15",
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -21888,6 +21917,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-coder": {
|
||||
"cache_read_input_token_cost": 1.4e-08,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-08,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -21902,6 +21932,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-r1": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 5.5e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -21957,6 +21988,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-v3.2": {
|
||||
"cache_read_input_token_cost": 2.8e-08,
|
||||
"input_cost_per_token": 2.8e-07,
|
||||
"input_cost_per_token_cache_hit": 2.8e-08,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -23796,6 +23828,25 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/deepseek-v4-pro-0813": {
|
||||
"cache_read_input_token_cost": 4.4e-08,
|
||||
"cache_read_input_token_cost_priority": 5.5e-08,
|
||||
"input_cost_per_token": 1.32e-06,
|
||||
"input_cost_per_token_priority": 1.65e-06,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.96e-06,
|
||||
"output_cost_per_token_priority": 4.95e-06,
|
||||
"source": "https://api.fireworks.ai/v1/serverless/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/models/firefunction-v2": {
|
||||
"input_cost_per_token": 9e-07,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
|
|
@ -24182,7 +24233,7 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf": {
|
||||
"input_cost_per_token": 1.2e-06,
|
||||
|
|
@ -24508,7 +24559,7 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/qwen3p7-plus": {
|
||||
"cache_read_input_token_cost": 8e-08,
|
||||
|
|
@ -35244,6 +35295,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"source": "https://console.groq.com/docs/model/qwen/qwen3.6-27b",
|
||||
"deprecation_date": "2026-09-14",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": false,
|
||||
|
|
@ -41472,6 +41524,7 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v3.2-exp": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"deprecation_date": "2026-09-28",
|
||||
"input_cost_per_token": 2.7e-07,
|
||||
"input_cost_per_token_cache_hit": 2e-08,
|
||||
|
|
@ -41494,6 +41547,7 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-r1": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 7e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -41537,21 +41591,21 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4-pro": {
|
||||
"input_cost_per_token": 9.5526e-07,
|
||||
"input_cost_per_token": 9.42906e-07,
|
||||
"input_cost_per_token_cache_hit": 4.4e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.91052e-06,
|
||||
"output_cost_per_token": 1.885812e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"cache_read_input_token_cost": 7.9605e-08,
|
||||
"cache_read_input_token_cost": 7.85755e-08,
|
||||
"supports_audio_input": false,
|
||||
"supports_pdf_input": false,
|
||||
"supports_vision": false,
|
||||
|
|
@ -41579,22 +41633,22 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4-pro-0813": {
|
||||
"input_cost_per_token": 1.32e-06,
|
||||
"input_cost_per_token_cache_hit": 4.4e-08,
|
||||
"input_cost_per_token": 5.7684e-07,
|
||||
"input_cost_per_token_cache_hit": 1.9272e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.96e-06,
|
||||
"output_cost_per_token": 1.73052e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"cache_read_input_token_cost": 4.4e-08,
|
||||
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},
|
||||
"cache_read_input_token_cost": 1.8354e-08,
|
||||
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":5.7684e-7,"output_cost_per_token":0.00000173052,"cache_read_input_token_cost":1.8354e-8},
|
||||
"supports_audio_input": false,
|
||||
"supports_pdf_input": false,
|
||||
"supports_vision": false,
|
||||
|
|
@ -44247,6 +44301,84 @@
|
|||
"supports_system_messages": true,
|
||||
"supports_native_structured_output": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.45e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/ap-south-1/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.41e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.545e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.236e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/eu-west-1/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.41e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/eu-west-2/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 2.3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.86e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/sa-east-1/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.45e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"qwen.qwen3-vl-235b-a22b": {
|
||||
"input_cost_per_token": 5.3e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
|
|
@ -46291,8 +46423,8 @@
|
|||
"together_ai/zai-org/GLM-4.6": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_tokens": 200000,
|
||||
"max_input_tokens": 202752,
|
||||
"max_tokens": 202752,
|
||||
"metadata": {
|
||||
"successor": "together_ai/zai-org/GLM-5.2"
|
||||
},
|
||||
|
|
@ -46308,8 +46440,8 @@
|
|||
"deprecation_date": "2026-04-02",
|
||||
"input_cost_per_token": 4.5e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_tokens": 200000,
|
||||
"max_input_tokens": 202752,
|
||||
"max_tokens": 202752,
|
||||
"metadata": {
|
||||
"successor": "together_ai/zai-org/GLM-5.2"
|
||||
},
|
||||
|
|
@ -47158,7 +47290,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-05,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -47192,7 +47324,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-05,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -47225,7 +47357,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-05,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -47276,7 +47408,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us-gov.nvidia.nemotron-nano-3-30b": {
|
||||
"input_cost_per_token": 7.2e-08,
|
||||
|
|
@ -64215,6 +64347,25 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/glm-5p3": {
|
||||
"cache_read_input_token_cost": 2.6e-07,
|
||||
"cache_read_input_token_cost_priority": 3.25e-07,
|
||||
"input_cost_per_token": 1.4e-06,
|
||||
"input_cost_per_token_priority": 1.75e-06,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.4e-06,
|
||||
"output_cost_per_token_priority": 5.5e-06,
|
||||
"source": "https://api.fireworks.ai/v1/serverless/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/routers/glm-5p3-fast": {
|
||||
"cache_read_input_token_cost": 3.9e-07,
|
||||
"input_cost_per_token": 2.1e-06,
|
||||
|
|
@ -64262,6 +64413,23 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"fireworks_ai/glm-5p3-flash": {
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"cache_read_input_token_cost_priority": 3.75e-08,
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"input_cost_per_token_priority": 1.875e-07,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_tokens": 1048576,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 5e-07,
|
||||
"output_cost_per_token_priority": 6.25e-07,
|
||||
"source": "https://api.fireworks.ai/v1/serverless/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/models/inkling": {
|
||||
"cache_read_input_token_cost": 1.7e-07,
|
||||
"input_cost_per_token": 1e-06,
|
||||
|
|
@ -67259,9 +67427,9 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4-flash": {
|
||||
"input_cost_per_token": 8.8606e-08,
|
||||
"output_cost_per_token": 1.77212e-07,
|
||||
"cache_read_input_token_cost": 1.77212e-08,
|
||||
"input_cost_per_token": 5.544e-08,
|
||||
"output_cost_per_token": 1.1088e-07,
|
||||
"cache_read_input_token_cost": 1.1088e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
|
|
@ -69010,12 +69178,12 @@
|
|||
"supports_reasoning": false
|
||||
},
|
||||
"openrouter/meta-llama/llama-3.1-70b-instruct": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"output_cost_per_token": 7.2e-07,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"mode": "chat",
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
|
|
@ -71317,15 +71485,15 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/~deepseek/deepseek-pro-latest": {
|
||||
"cache_read_input_token_cost": 4.4e-08,
|
||||
"input_cost_per_token": 1.32e-06,
|
||||
"cache_read_input_token_cost": 1.8354e-08,
|
||||
"input_cost_per_token": 5.7684e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},
|
||||
"output_cost_per_token": 3.96e-06,
|
||||
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":5.7684e-7,"output_cost_per_token":0.00000173052,"cache_read_input_token_cost":1.8354e-8},
|
||||
"output_cost_per_token": 1.73052e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
|
|
|
|||
|
|
@ -1154,7 +1154,7 @@ class MCPRequestHandler:
|
|||
|
||||
Failures surface with the status the standard pipeline would give them, mirroring
|
||||
``UserAPIKeyAuthExceptionHandler``: a disallowed route is the route gate's own 403, an
|
||||
over-budget identity is a 429, a sub-check that raised its own ``HTTPException``/
|
||||
over-budget identity is a 422, a sub-check that raised its own ``HTTPException``/
|
||||
``ProxyException`` keeps that status, a transient database outage is a retryable 503, and
|
||||
only a genuinely unresolvable failure (a blocked team/project raises a bare ``Exception``,
|
||||
same as the standard pipeline's fallback) becomes the fail-closed 401. Collapsing every
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ fetcher dispatches by ``discovery_mode``:
|
|||
pure-A2A fallback strategy returns 404 for these deployments.
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping
|
||||
from enum import Enum
|
||||
from typing import Any, Final
|
||||
from urllib.parse import urlencode
|
||||
|
|
@ -55,7 +56,7 @@ def _normalize_base_url(base_url: str) -> str:
|
|||
|
||||
|
||||
def _build_langgraph_platform_paths(
|
||||
params: dict[str, Any] | None,
|
||||
params: Mapping[str, object] | None,
|
||||
) -> tuple[str, ...]:
|
||||
"""Build the paths to try for LangGraph Platform discovery.
|
||||
|
||||
|
|
@ -71,7 +72,7 @@ def _build_langgraph_platform_paths(
|
|||
return tuple(f"{path}?{query}" for path in AGENT_CARD_WELL_KNOWN_PATHS)
|
||||
|
||||
|
||||
def _paths_for_mode(mode: DiscoveryMode, params: dict[str, Any] | None) -> tuple[str, ...]:
|
||||
def _paths_for_mode(mode: DiscoveryMode, params: Mapping[str, object] | None) -> tuple[str, ...]:
|
||||
if mode == DiscoveryMode.WELL_KNOWN_FALLBACK:
|
||||
return AGENT_CARD_WELL_KNOWN_PATHS
|
||||
if mode == DiscoveryMode.LANGGRAPH_PLATFORM:
|
||||
|
|
@ -83,7 +84,7 @@ async def fetch_well_known_card(
|
|||
base_url: str,
|
||||
*,
|
||||
discovery_mode: DiscoveryMode = DiscoveryMode.WELL_KNOWN_FALLBACK,
|
||||
params: dict[str, Any] | None = None,
|
||||
params: Mapping[str, object] | None = None,
|
||||
timeout: float = DEFAULT_DISCOVERY_TIMEOUT_SECONDS,
|
||||
headers: dict[str, str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
|
|
|
|||
|
|
@ -25,8 +25,9 @@ Config example::
|
|||
import asyncio
|
||||
import base64
|
||||
import hashlib
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Final
|
||||
from typing import Final
|
||||
|
||||
import httpx
|
||||
|
||||
|
|
@ -43,7 +44,7 @@ _TOKEN_EXPIRY_BUFFER_SECONDS: Final = 60
|
|||
_DEFAULT_TTL_SECONDS: Final = 3600
|
||||
|
||||
|
||||
def _resolve_secret(value: Any) -> str | None:
|
||||
def _resolve_secret(value: object) -> str | None:
|
||||
"""Resolve a config value, expanding ``os.environ/`` references."""
|
||||
if not isinstance(value, str):
|
||||
return None
|
||||
|
|
@ -75,7 +76,7 @@ class DatabricksAppOAuthConfig:
|
|||
|
||||
|
||||
def parse_databricks_oauth_config(
|
||||
litellm_params: dict[str, Any] | None,
|
||||
litellm_params: Mapping[str, object] | None,
|
||||
) -> DatabricksAppOAuthConfig | None:
|
||||
"""Build a Databricks App OAuth config from an agent's ``litellm_params``.
|
||||
|
||||
|
|
@ -191,7 +192,7 @@ class DatabricksAppOAuthTokenCache(InMemoryCache):
|
|||
except httpx.HTTPError as exc:
|
||||
raise ValueError(f"Databricks App OAuth token request failed: {exc}") from exc
|
||||
|
||||
body: Final = response.json()
|
||||
body: Final[object] = response.json()
|
||||
if not isinstance(body, dict):
|
||||
raise ValueError(
|
||||
f"Databricks App OAuth token response returned non-object JSON (got {type(body).__name__})"
|
||||
|
|
@ -215,7 +216,7 @@ databricks_app_oauth_token_cache: Final = DatabricksAppOAuthTokenCache()
|
|||
|
||||
|
||||
async def resolve_databricks_app_auth_header(
|
||||
litellm_params: dict[str, Any] | None,
|
||||
litellm_params: Mapping[str, object] | None,
|
||||
) -> dict[str, str] | None:
|
||||
"""Return ``{"Authorization": "Bearer <token>"}`` for a Databricks App agent.
|
||||
|
||||
|
|
|
|||
|
|
@ -9,7 +9,15 @@ from litellm._version import version as litellm_version
|
|||
from litellm.proxy.client.health import HealthManagementClient
|
||||
|
||||
from .commands.agents import agent_commands
|
||||
from .commands.auth import auth_group, context_secret_vault, get_stored_api_key, login, logout, whoami
|
||||
from .commands.auth import (
|
||||
CliContextObj,
|
||||
auth_group,
|
||||
context_secret_vault,
|
||||
get_stored_api_key,
|
||||
login,
|
||||
logout,
|
||||
whoami,
|
||||
)
|
||||
from .commands.autoroute.commands import autoroute_group
|
||||
from .commands.chat import chat
|
||||
from .commands.config import config_commands, get_config_value, hidden_command_names
|
||||
|
|
@ -126,7 +134,8 @@ def cli(ctx: click.Context, show_version: bool, base_url: str | None, api_key: s
|
|||
@click.pass_context
|
||||
def version(ctx: click.Context):
|
||||
"""Show the LiteLLM Proxy CLI and server version."""
|
||||
print_version(ctx.obj.get("base_url"), ctx.obj.get("api_key"))
|
||||
ctx_obj: Final[CliContextObj] = ctx.obj
|
||||
print_version(ctx_obj.get("base_url"), ctx_obj.get("api_key"))
|
||||
|
||||
|
||||
# Add authentication commands as top-level commands
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ if TYPE_CHECKING:
|
|||
from litellm.types.guardrails import Guardrail, LitellmParams
|
||||
|
||||
|
||||
def _get_config_value(litellm_params: Any, optional_params: Any, attribute_name: str) -> Any | None:
|
||||
def _get_config_value(litellm_params: "LitellmParams", optional_params: object, attribute_name: str) -> Any | None:
|
||||
if optional_params is not None:
|
||||
value: Final = (
|
||||
optional_params.get(attribute_name)
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@
|
|||
# +-------------------------------------------------------------+
|
||||
import os
|
||||
import uuid
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal, Optional
|
||||
from typing import TYPE_CHECKING, Final, Literal, Optional
|
||||
|
||||
import httpx
|
||||
from fastapi import HTTPException
|
||||
|
|
@ -63,7 +63,7 @@ class OnyxGuardrail(CustomGuardrail):
|
|||
|
||||
async def _validate_with_guard_server(
|
||||
self,
|
||||
payload: Any,
|
||||
payload: object,
|
||||
input_type: Literal["request", "response"],
|
||||
conversation_id: str,
|
||||
) -> dict:
|
||||
|
|
|
|||
|
|
@ -40,7 +40,7 @@ _UNMANAGED_RESPONSE_ID_DETAIL: Final = (
|
|||
_PROXY_ADMIN_ROLES: Final = frozenset({LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN.value})
|
||||
|
||||
|
||||
def _proxy_general_settings() -> Mapping[str, Any]:
|
||||
def _proxy_general_settings() -> Mapping[str, object]:
|
||||
from litellm.proxy.proxy_server import general_settings
|
||||
|
||||
return general_settings
|
||||
|
|
@ -107,7 +107,7 @@ def _is_responses_api_create_route(request_route: str | None) -> bool:
|
|||
class ResponsesIDSecurity(CustomLogger):
|
||||
def __init__(
|
||||
self,
|
||||
general_settings_reader: Callable[[], Mapping[str, Any]] = _proxy_general_settings,
|
||||
general_settings_reader: Callable[[], Mapping[str, object]] = _proxy_general_settings,
|
||||
signing_key_reader: Callable[[], str | None] = _proxy_signing_key,
|
||||
) -> None:
|
||||
self._general_settings_reader: Final = general_settings_reader
|
||||
|
|
@ -307,7 +307,7 @@ class ResponsesIDSecurity(CustomLogger):
|
|||
data: dict,
|
||||
user_api_key_dict: "UserAPIKeyAuth",
|
||||
response: LLMResponseTypes,
|
||||
) -> Any:
|
||||
) -> LLMResponseTypes:
|
||||
"""
|
||||
Queue response IDs for batch processing instead of writing directly to DB.
|
||||
|
||||
|
|
|
|||
|
|
@ -15,6 +15,7 @@ self-describing `StandardLoggingPayload`, so completions/responses can use it to
|
|||
"""
|
||||
|
||||
import uuid
|
||||
from collections.abc import Mapping
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Final
|
||||
|
||||
|
|
@ -48,7 +49,7 @@ class CallbackLogsReplayer:
|
|||
"""
|
||||
|
||||
@staticmethod
|
||||
def _epoch_to_datetime(value: Any) -> datetime:
|
||||
def _epoch_to_datetime(value: object) -> datetime:
|
||||
"""`StandardLoggingPayload` stores startTime/endTime as float epoch seconds."""
|
||||
if isinstance(value, (int, float)):
|
||||
return datetime.fromtimestamp(float(value), tz=timezone.utc)
|
||||
|
|
@ -114,7 +115,7 @@ class CallbackLogsReplayer:
|
|||
return logging_obj
|
||||
|
||||
@staticmethod
|
||||
def _response_obj_from_payload(payload: dict[str, Any]) -> dict[str, Any]:
|
||||
def _response_obj_from_payload(payload: Mapping[str, object]) -> dict[str, object]:
|
||||
"""Minimal response object so usage-derived spend-log fields resolve."""
|
||||
return {
|
||||
"id": payload.get("id"),
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""`/management/v1/spend_logs` facets."""
|
||||
|
||||
from datetime import datetime, timezone
|
||||
from typing import Annotated, Any, Final, Literal
|
||||
from typing import Annotated, Final, Literal
|
||||
|
||||
from fastapi import APIRouter, Depends, Query, Request
|
||||
|
||||
|
|
@ -39,7 +39,7 @@ async def _spend_log_scope_clause(
|
|||
user_api_key_dict: UserAPIKeyAuth,
|
||||
prisma_client: PrismaClient,
|
||||
next_param_index: int,
|
||||
) -> tuple[str | None, tuple[Any, ...]]:
|
||||
) -> tuple[str | None, tuple[str | list[str], ...]]:
|
||||
"""SQL predicate restricting the facet to spend logs this caller may read.
|
||||
|
||||
Returns ``(None, ())`` for a proxy admin. Mirrors the scoping ``/spend/logs/ui``
|
||||
|
|
@ -101,8 +101,8 @@ async def _list_spend_log_facet(
|
|||
)
|
||||
|
||||
column_sql: Final = "end_user" if column == "end_user" else '"user"'
|
||||
window_params: Final[tuple[Any, ...]] = (_as_utc(start_time), _as_utc(end_time))
|
||||
search_params: Final[tuple[Any, ...]] = (f"%{escape_like(q)}%",) if q else ()
|
||||
window_params: Final[tuple[datetime, datetime]] = (_as_utc(start_time), _as_utc(end_time))
|
||||
search_params: Final[tuple[str, ...]] = (f"%{escape_like(q)}%",) if q else ()
|
||||
search_clause: Final = (f"{column_sql} ILIKE ${len(window_params) + 1} ESCAPE '\\'",) if q else ()
|
||||
|
||||
scope_clause, scope_params = await _spend_log_scope_clause(
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ from collections.abc import Mapping, Sequence
|
|||
from collections.abc import Set as AbstractSet
|
||||
from dataclasses import dataclass
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Any, Final, Optional
|
||||
from typing import TYPE_CHECKING, Final, Optional
|
||||
|
||||
from fastapi import HTTPException, status
|
||||
from pydantic import TypeAdapter
|
||||
|
|
@ -230,7 +230,7 @@ def _dedupe_preserving_order(values: list[str]) -> list[str]:
|
|||
return result
|
||||
|
||||
|
||||
def _mcp_server_identifier_matches(server: Any, identifier: str) -> bool:
|
||||
def _mcp_server_identifier_matches(server: object, identifier: str) -> bool:
|
||||
return identifier in {
|
||||
getattr(server, "server_id", None),
|
||||
getattr(server, "alias", None),
|
||||
|
|
|
|||
|
|
@ -147,7 +147,7 @@ class GeminiPassthroughLoggingHandler:
|
|||
- Creates standard logging object
|
||||
- Logs in litellm callbacks
|
||||
"""
|
||||
kwargs: dict[str, Any] = {}
|
||||
kwargs: dict[str, object] = {}
|
||||
model = model or GeminiPassthroughLoggingHandler.extract_model_from_url(url_route)
|
||||
complete_streaming_response: Final = GeminiPassthroughLoggingHandler._build_complete_streaming_response(
|
||||
all_chunks=all_chunks,
|
||||
|
|
|
|||
|
|
@ -199,9 +199,13 @@ def _build_endpoints(raw: _ProvidersFile) -> list[_EndpointEntry]:
|
|||
return result
|
||||
|
||||
|
||||
_PROVIDERS_FILE_ADAPTER: Final = TypeAdapter(_ProvidersFile)
|
||||
_PROVIDER_CREATE_FIELDS_ADAPTER: Final = TypeAdapter(list[ProviderCreateInfo])
|
||||
|
||||
|
||||
def _load_endpoints() -> list[_EndpointEntry]:
|
||||
raw: Final[_ProvidersFile] = json.loads(
|
||||
files("litellm").joinpath("provider_endpoints_support_backup.json").read_text(encoding="utf-8")
|
||||
raw: Final = _PROVIDERS_FILE_ADAPTER.validate_python(
|
||||
json.loads(files("litellm").joinpath("provider_endpoints_support_backup.json").read_text(encoding="utf-8"))
|
||||
)
|
||||
return _build_endpoints(raw)
|
||||
|
||||
|
|
@ -398,7 +402,7 @@ async def get_provider_fields() -> list[ProviderCreateInfo]:
|
|||
)
|
||||
|
||||
with open(provider_create_fields_path, "r") as f:
|
||||
provider_create_fields: Final = json.load(f)
|
||||
provider_create_fields: Final = _PROVIDER_CREATE_FIELDS_ADAPTER.validate_python(json.load(f))
|
||||
|
||||
return provider_create_fields
|
||||
|
||||
|
|
|
|||
|
|
@ -454,6 +454,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
"stream": stream,
|
||||
"metadata": kwargs.get("metadata"),
|
||||
"service_tier": kwargs.get("service_tier"),
|
||||
"safety_identifier": responses_api_request.get("safety_identifier"),
|
||||
"web_search_options": web_search_options,
|
||||
"response_format": response_format,
|
||||
"reasoning_effort": reasoning.effort,
|
||||
|
|
|
|||
|
|
@ -411,6 +411,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
|
|||
output_config: AnthropicOutputConfig | None # Configuration for Claude's output behavior
|
||||
cache_control: dict[str, Any] | None # Automatic prompt caching
|
||||
reasoning_effort: str | None
|
||||
safeguards: ReadOnly[list[dict[str, object]] | None]
|
||||
|
||||
|
||||
class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False):
|
||||
|
|
@ -530,6 +531,7 @@ class AnthropicStopDetails(TypedDict, total=False):
|
|||
class MessageDelta(TypedDict, total=False):
|
||||
stop_reason: str | None
|
||||
stop_details: ReadOnly[AnthropicStopDetails]
|
||||
safeguard_results: ReadOnly[list[dict[str, object]]]
|
||||
|
||||
|
||||
class ServerToolUsage(TypedDict, total=False):
|
||||
|
|
@ -600,6 +602,7 @@ class MessageChunk(TypedDict, total=False):
|
|||
stop_reason: str | None
|
||||
stop_sequence: str | None
|
||||
usage: UsageDelta
|
||||
safeguard_results: ReadOnly[list[dict[str, object]]]
|
||||
|
||||
|
||||
class MessageStartBlock(TypedDict):
|
||||
|
|
|
|||
|
|
@ -97,3 +97,4 @@ class AnthropicMessagesResponse(TypedDict, total=False):
|
|||
type: Literal["message"] | None
|
||||
usage: AnthropicUsage | None
|
||||
context_management: NotRequired[ContextManagementResponse]
|
||||
safeguard_results: NotRequired[ReadOnly[list[dict[str, object]]]]
|
||||
|
|
|
|||
|
|
@ -255,6 +255,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
cache_creation_input_token_cost_ultrafast: ReadOnly[float | None] # OpenAI ultrafast service tier pricing
|
||||
cache_read_input_token_cost: float | None
|
||||
cache_read_input_audio_token_cost: ReadOnly[float | None]
|
||||
cache_read_input_image_token_cost: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_flex: float | None # OpenAI flex service tier pricing
|
||||
cache_read_input_token_cost_priority: float | None # OpenAI priority service tier pricing
|
||||
cache_read_input_token_cost_ultrafast: ReadOnly[float | None] # OpenAI ultrafast service tier pricing
|
||||
|
|
@ -3635,6 +3636,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
cache_read_input_token_cost_above_272k_tokens_priority: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_flex: float | None = None
|
||||
cache_read_input_audio_token_cost: float | None = None
|
||||
cache_read_input_image_token_cost: float | None = None
|
||||
input_cost_per_character_above_128k_tokens: float | None = None
|
||||
input_cost_per_audio_token: float | None = None
|
||||
input_cost_per_token_cache_hit: float | None = None
|
||||
|
|
|
|||
|
|
@ -1327,7 +1327,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 2048,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"anthropic.claude-mythos-preview": {
|
||||
"input_cost_per_token": 0,
|
||||
|
|
@ -1381,7 +1381,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 2048,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-opus-4-7": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -1419,7 +1419,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 2048,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-opus-4-7": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -1531,7 +1531,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"anthropic.claude-fable-5-1": {
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
|
|
@ -1570,7 +1570,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-fable-5": {
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
|
|
@ -1608,7 +1608,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-fable-5-1": {
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
|
|
@ -1647,7 +1647,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-fable-5": {
|
||||
"cache_creation_input_token_cost": 1.375e-05,
|
||||
|
|
@ -1685,7 +1685,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-fable-5-1": {
|
||||
"cache_creation_input_token_cost": 1.375e-05,
|
||||
|
|
@ -1724,7 +1724,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-fable-5": {
|
||||
"cache_creation_input_token_cost": 1.375e-05,
|
||||
|
|
@ -1837,7 +1837,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-opus-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -1875,7 +1875,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-opus-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -1913,7 +1913,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-opus-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2063,7 +2063,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-opus-4-8": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2102,7 +2102,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-opus-4-8": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2141,7 +2141,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-opus-4-8": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2329,7 +2329,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-sonnet-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2368,7 +2368,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-sonnet-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2407,7 +2407,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-sonnet-5": {
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
|
|
@ -2556,7 +2556,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
|
|
@ -2591,7 +2591,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
|
|
@ -2626,7 +2626,7 @@
|
|||
"supports_output_config": true,
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"eu.anthropic.claude-sonnet-4-6": {
|
||||
"supports_adaptive_thinking": true,
|
||||
|
|
@ -3740,6 +3740,21 @@
|
|||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"azure_ai/gpt-image-2": {
|
||||
"cache_read_input_image_token_cost": 2e-06,
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"input_cost_per_image_token": 8e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"mode": "image_generation",
|
||||
"output_cost_per_image_token": 3e-05,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
|
||||
"supported_endpoints": [
|
||||
"/v1/images/generations",
|
||||
"/v1/images/edits"
|
||||
],
|
||||
"supports_vision": true
|
||||
},
|
||||
"azure_ai/codex-mini": {
|
||||
"cache_read_input_token_cost": 3.75e-07,
|
||||
"deprecation_date": "2026-11-15",
|
||||
|
|
@ -11159,6 +11174,20 @@
|
|||
],
|
||||
"deprecation_date": "2026-10-01"
|
||||
},
|
||||
"azure_ai/MAI-Image-2.5-Pro": {
|
||||
"deprecation_date": "2026-10-01",
|
||||
"input_cost_per_image_token": 8e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.1085,
|
||||
"output_cost_per_image_token": 0.000106,
|
||||
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-mai-image-2-5-pro-and-mai-voice-2-flash-in-microsoft-foundry/4539446",
|
||||
"supported_endpoints": [
|
||||
"/v1/images/generations",
|
||||
"/v1/images/edits"
|
||||
]
|
||||
},
|
||||
"azure_ai/MAI-Image-2e": {
|
||||
"deprecation_date": "2026-08-15",
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -21888,6 +21917,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-coder": {
|
||||
"cache_read_input_token_cost": 1.4e-08,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-08,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -21902,6 +21932,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-r1": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 5.5e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -21957,6 +21988,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-v3.2": {
|
||||
"cache_read_input_token_cost": 2.8e-08,
|
||||
"input_cost_per_token": 2.8e-07,
|
||||
"input_cost_per_token_cache_hit": 2.8e-08,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -23796,6 +23828,25 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/deepseek-v4-pro-0813": {
|
||||
"cache_read_input_token_cost": 4.4e-08,
|
||||
"cache_read_input_token_cost_priority": 5.5e-08,
|
||||
"input_cost_per_token": 1.32e-06,
|
||||
"input_cost_per_token_priority": 1.65e-06,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
"max_tokens": 131072,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.96e-06,
|
||||
"output_cost_per_token_priority": 4.95e-06,
|
||||
"source": "https://api.fireworks.ai/v1/serverless/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/models/firefunction-v2": {
|
||||
"input_cost_per_token": 9e-07,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
|
|
@ -24182,7 +24233,7 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf": {
|
||||
"input_cost_per_token": 1.2e-06,
|
||||
|
|
@ -24508,7 +24559,7 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/qwen3p7-plus": {
|
||||
"cache_read_input_token_cost": 8e-08,
|
||||
|
|
@ -35244,6 +35295,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-06,
|
||||
"source": "https://console.groq.com/docs/model/qwen/qwen3.6-27b",
|
||||
"deprecation_date": "2026-09-14",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": false,
|
||||
|
|
@ -41472,6 +41524,7 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v3.2-exp": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"deprecation_date": "2026-09-28",
|
||||
"input_cost_per_token": 2.7e-07,
|
||||
"input_cost_per_token_cache_hit": 2e-08,
|
||||
|
|
@ -41494,6 +41547,7 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-r1": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 7e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -41537,21 +41591,21 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4-pro": {
|
||||
"input_cost_per_token": 9.5526e-07,
|
||||
"input_cost_per_token": 9.42906e-07,
|
||||
"input_cost_per_token_cache_hit": 4.4e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.91052e-06,
|
||||
"output_cost_per_token": 1.885812e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"cache_read_input_token_cost": 7.9605e-08,
|
||||
"cache_read_input_token_cost": 7.85755e-08,
|
||||
"supports_audio_input": false,
|
||||
"supports_pdf_input": false,
|
||||
"supports_vision": false,
|
||||
|
|
@ -41579,22 +41633,22 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4-pro-0813": {
|
||||
"input_cost_per_token": 1.32e-06,
|
||||
"input_cost_per_token_cache_hit": 4.4e-08,
|
||||
"input_cost_per_token": 5.7684e-07,
|
||||
"input_cost_per_token_cache_hit": 1.9272e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3.96e-06,
|
||||
"output_cost_per_token": 1.73052e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"cache_read_input_token_cost": 4.4e-08,
|
||||
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},
|
||||
"cache_read_input_token_cost": 1.8354e-08,
|
||||
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":5.7684e-7,"output_cost_per_token":0.00000173052,"cache_read_input_token_cost":1.8354e-8},
|
||||
"supports_audio_input": false,
|
||||
"supports_pdf_input": false,
|
||||
"supports_vision": false,
|
||||
|
|
@ -44247,6 +44301,84 @@
|
|||
"supports_system_messages": true,
|
||||
"supports_native_structured_output": true
|
||||
},
|
||||
"bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.45e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/ap-south-1/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.41e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.545e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.236e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/eu-west-1/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.41e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/eu-west-2/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 2.3e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.86e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"bedrock/sa-east-1/qwen.qwen3-next-80b-a3b": {
|
||||
"input_cost_per_token": 1.8e-07,
|
||||
"litellm_provider": "bedrock",
|
||||
"max_input_tokens": 128000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.45e-06,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_function_calling": true,
|
||||
"supports_native_structured_output": true,
|
||||
"supports_system_messages": true
|
||||
},
|
||||
"qwen.qwen3-vl-235b-a22b": {
|
||||
"input_cost_per_token": 5.3e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
|
|
@ -46291,8 +46423,8 @@
|
|||
"together_ai/zai-org/GLM-4.6": {
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_tokens": 200000,
|
||||
"max_input_tokens": 202752,
|
||||
"max_tokens": 202752,
|
||||
"metadata": {
|
||||
"successor": "together_ai/zai-org/GLM-5.2"
|
||||
},
|
||||
|
|
@ -46308,8 +46440,8 @@
|
|||
"deprecation_date": "2026-04-02",
|
||||
"input_cost_per_token": 4.5e-07,
|
||||
"litellm_provider": "together_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_tokens": 200000,
|
||||
"max_input_tokens": 202752,
|
||||
"max_tokens": 202752,
|
||||
"metadata": {
|
||||
"successor": "together_ai/zai-org/GLM-5.2"
|
||||
},
|
||||
|
|
@ -47158,7 +47290,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-05,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -47192,7 +47324,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-05,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -47225,7 +47357,7 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-05,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -47276,7 +47408,7 @@
|
|||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"us-gov.nvidia.nemotron-nano-3-30b": {
|
||||
"input_cost_per_token": 7.2e-08,
|
||||
|
|
@ -64215,6 +64347,25 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/glm-5p3": {
|
||||
"cache_read_input_token_cost": 2.6e-07,
|
||||
"cache_read_input_token_cost_priority": 3.25e-07,
|
||||
"input_cost_per_token": 1.4e-06,
|
||||
"input_cost_per_token_priority": 1.75e-06,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.4e-06,
|
||||
"output_cost_per_token_priority": 5.5e-06,
|
||||
"source": "https://api.fireworks.ai/v1/serverless/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": false
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/routers/glm-5p3-fast": {
|
||||
"cache_read_input_token_cost": 3.9e-07,
|
||||
"input_cost_per_token": 2.1e-06,
|
||||
|
|
@ -64262,6 +64413,23 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"fireworks_ai/glm-5p3-flash": {
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"cache_read_input_token_cost_priority": 3.75e-08,
|
||||
"input_cost_per_token": 1.5e-07,
|
||||
"input_cost_per_token_priority": 1.875e-07,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_tokens": 1048576,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 5e-07,
|
||||
"output_cost_per_token_priority": 6.25e-07,
|
||||
"source": "https://api.fireworks.ai/v1/serverless/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/models/inkling": {
|
||||
"cache_read_input_token_cost": 1.7e-07,
|
||||
"input_cost_per_token": 1e-06,
|
||||
|
|
@ -67259,9 +67427,9 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v4-flash": {
|
||||
"input_cost_per_token": 8.8606e-08,
|
||||
"output_cost_per_token": 1.77212e-07,
|
||||
"cache_read_input_token_cost": 1.77212e-08,
|
||||
"input_cost_per_token": 5.544e-08,
|
||||
"output_cost_per_token": 1.1088e-07,
|
||||
"cache_read_input_token_cost": 1.1088e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
|
|
@ -69010,12 +69178,12 @@
|
|||
"supports_reasoning": false
|
||||
},
|
||||
"openrouter/meta-llama/llama-3.1-70b-instruct": {
|
||||
"input_cost_per_token": 7.2e-07,
|
||||
"output_cost_per_token": 7.2e-07,
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 4e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"max_output_tokens": 16384,
|
||||
"max_tokens": 16384,
|
||||
"mode": "chat",
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
|
|
@ -71317,15 +71485,15 @@
|
|||
"supports_web_search": false
|
||||
},
|
||||
"openrouter/~deepseek/deepseek-pro-latest": {
|
||||
"cache_read_input_token_cost": 4.4e-08,
|
||||
"input_cost_per_token": 1.32e-06,
|
||||
"cache_read_input_token_cost": 1.8354e-08,
|
||||
"input_cost_per_token": 5.7684e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 384000,
|
||||
"max_tokens": 384000,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},
|
||||
"output_cost_per_token": 3.96e-06,
|
||||
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":5.7684e-7,"output_cost_per_token":0.00000173052,"cache_read_input_token_cost":1.8354e-8},
|
||||
"output_cost_per_token": 1.73052e-06,
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_audio_input": false,
|
||||
"supports_function_calling": true,
|
||||
|
|
|
|||
|
|
@ -137,6 +137,10 @@
|
|||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"cache_read_input_image_token_cost": {
|
||||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"cache_read_input_token_cost": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
|
|||
|
|
@ -122,7 +122,7 @@ E2E_FIXTURE_MODE=replay E2E_FIXTURE_DIR=/tmp/e2e-fixtures E2E_RESET_SPEND_LOGS=1
|
|||
|
||||
Point the proxy at bogus provider credentials for the replay run and it still has to pass: that is the whole proof that nothing left the process. Bundles are never committed. `tests/e2e/.fixtures` is gitignored because a bundle holds verbatim provider response bodies and hard-fails after seven days. CI records and replays this lane on a schedule in `.github/workflows/e2e_record_replay.yml`, publishing the bundle as a private `e2e-fixtures-bundle` artifact instead of committing it, selecting the tests with the `@pytest.mark.replayable` marker, and proving the bogus-credentials replay hermetic by counting provider egress with `.github/scripts/e2e_egress_sentinel.py`
|
||||
|
||||
Current limits: Bedrock cannot be mounted (SigV4 signs the Host header, so a rewritten api_base fails signature verification), deployments baked into the proxy's config file cannot be edge-wired (only `/model/new` registrations can carry the edge api_base), and a file upload routed by `custom_llm_provider` through the proxy's `files_settings` block never passes a deployment at all, so the batches `model_param` and `provider_fallback` scenarios keep uploading live in every mode
|
||||
Current limits: Bedrock cannot be mounted in record or replay (SigV4 signs the Host header, so a rewritten api_base fails signature verification); a test that needs to observe the Converse body registers its own `LiveEdge` with `provider_edge_bedrock.bedrock_signer` re-signing the forwarded request, and carries the `provider_edge_host` opt-in marker because the gateway must reach the pytest host, which the Buildkite ephemeral stack cannot (the GitHub changed-e2e lane, whose gateways run on the runner, sets `E2E_PROVIDER_EDGE_HOST_REACHABLE`). Deployments baked into the proxy's config file cannot be edge-wired (only `/model/new` registrations can carry the edge api_base), and a file upload routed by `custom_llm_provider` through the proxy's `files_settings` block never passes a deployment at all, so the batches `model_param` and `provider_fallback` scenarios keep uploading live in every mode
|
||||
|
||||
## Typing
|
||||
|
||||
|
|
|
|||
|
|
@ -31,6 +31,7 @@ from e2e_config import (
|
|||
MANAGED_FILES_OPT_IN_ENV,
|
||||
MCP_OAUTH_LIVE_OPT_IN_ENV,
|
||||
PROMPT_CACHING_OPT_IN_ENV,
|
||||
PROVIDER_EDGE_HOST_OPT_IN_ENV,
|
||||
PROXY_BASE_URL,
|
||||
REDIS_CHAOS_OPT_IN_ENV,
|
||||
WEEKLY_ANOMALY_OPT_IN_ENV,
|
||||
|
|
@ -59,6 +60,7 @@ OPT_IN_MARKERS: Final = MappingProxyType(
|
|||
"redis_chaos": REDIS_CHAOS_OPT_IN_ENV,
|
||||
"cli_determinism": CLI_DETERMINISM_OPT_IN_ENV,
|
||||
"mcp_oauth_live": MCP_OAUTH_LIVE_OPT_IN_ENV,
|
||||
"provider_edge_host": PROVIDER_EDGE_HOST_OPT_IN_ENV,
|
||||
}
|
||||
)
|
||||
|
||||
|
|
@ -143,6 +145,11 @@ def pytest_configure(config: pytest.Config) -> None:
|
|||
"mcp_oauth_live: real Linear OAuth consent via a captured browser session; deselected unless "
|
||||
"E2E_MCP_OAUTH_LIVE is set",
|
||||
)
|
||||
config.addinivalue_line(
|
||||
"markers",
|
||||
"provider_edge_host: routes provider traffic through the pytest host's edge in every fixture mode, so the "
|
||||
"gateway must reach the pytest host; deselected unless E2E_PROVIDER_EDGE_HOST_REACHABLE is set",
|
||||
)
|
||||
|
||||
|
||||
def pytest_sessionstart(session: pytest.Session) -> None:
|
||||
|
|
|
|||
|
|
@ -146,6 +146,7 @@ PROMPT_CACHING_OPT_IN_ENV = "E2E_PROMPT_CACHING_STACK"
|
|||
REDIS_CHAOS_OPT_IN_ENV = "E2E_REDIS_CHAOS"
|
||||
CLI_DETERMINISM_OPT_IN_ENV = "E2E_CLI_DETERMINISM"
|
||||
MCP_OAUTH_LIVE_OPT_IN_ENV: Final = "E2E_MCP_OAUTH_LIVE"
|
||||
PROVIDER_EDGE_HOST_OPT_IN_ENV: Final = "E2E_PROVIDER_EDGE_HOST_REACHABLE"
|
||||
ANOMALY_SESSIONS = int(os.environ.get("E2E_ANOMALY_SESSIONS", "6"))
|
||||
ANOMALY_TURNS_PER_SESSION = int(os.environ.get("E2E_ANOMALY_TURNS_PER_SESSION", "6"))
|
||||
ANOMALY_TURN_ATTEMPTS = int(os.environ.get("E2E_ANOMALY_TURN_ATTEMPTS", "3"))
|
||||
|
|
|
|||
|
|
@ -95,7 +95,7 @@ class UnauthorizedError(BaseModel):
|
|||
class RateLimitedError(BaseModel):
|
||||
kind: Literal["rate_limited"] = "rate_limited"
|
||||
retry_after_seconds: int | None = None
|
||||
# litellm overloads 429 for budget_exceeded too, so keep the body to tell them apart.
|
||||
# keep the body so callers can tell limiter kinds apart.
|
||||
body: str = ""
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -75,6 +75,7 @@ class ResponsesRequest(BaseModel):
|
|||
stream: bool = False
|
||||
tools: list[ResponsesFunctionTool] | None = None
|
||||
guardrails: list[str] | None = None
|
||||
safety_identifier: str | None = None
|
||||
cache: dict[str, bool] | None = {"no-cache": True}
|
||||
|
||||
|
||||
|
|
@ -316,6 +317,7 @@ class EndpointsClient:
|
|||
*,
|
||||
stream: bool = False,
|
||||
guardrails: list[str] | None = None,
|
||||
safety_identifier: str | None = None,
|
||||
) -> StreamingResponse:
|
||||
return self._send(
|
||||
"/v1/responses",
|
||||
|
|
@ -326,6 +328,7 @@ class EndpointsClient:
|
|||
instructions="You are a helpful assistant",
|
||||
stream=stream,
|
||||
guardrails=guardrails,
|
||||
safety_identifier=safety_identifier,
|
||||
),
|
||||
stream=stream,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -8,10 +8,14 @@ litellm-regression-tests/tests/test_inference_endpoints.py.
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import cast
|
||||
import threading
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from types import MappingProxyType
|
||||
from typing import Final, cast
|
||||
|
||||
import pytest
|
||||
from e2e_config import unique_marker
|
||||
from e2e_config import PROVIDER_EDGE_ADVERTISE_HOST, PROVIDER_EDGE_BIND_HOST, unique_marker
|
||||
from e2e_http import (
|
||||
assert_client_error,
|
||||
require_successful_call,
|
||||
|
|
@ -26,7 +30,9 @@ from endpoints_client import (
|
|||
ResponsesStreamEventType,
|
||||
)
|
||||
from lifecycle import ResourceManager
|
||||
from models import LiteLLMParamsBody
|
||||
from models import ChatBody, ChatMessage, LiteLLMParamsBody
|
||||
from provider_edge import LiveEdge, start_provider_edge
|
||||
from provider_edge_bedrock import bedrock_signer
|
||||
from pydantic import BaseModel, ValidationError
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
|
@ -39,6 +45,33 @@ class _OptionalResponsesBody(BaseModel):
|
|||
|
||||
|
||||
BEDROCK_CONVERSE_BACKEND = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"
|
||||
BEDROCK_EDGE_REGION: Final = "us-east-1"
|
||||
BEDROCK_EDGE_MOUNT: Final = f"bedrock/{BEDROCK_EDGE_REGION}"
|
||||
|
||||
|
||||
class ConverseRequestBody(BaseModel):
|
||||
additionalModelRequestFields: dict[str, str] | None = None
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class ConverseRequestCapture:
|
||||
"""The Converse bodies the proxy actually sent upstream, as seen by a live
|
||||
edge sitting between the proxy and Bedrock."""
|
||||
|
||||
_bodies: list[ConverseRequestBody] = field(default_factory=list)
|
||||
_lock: threading.Lock = field(default_factory=threading.Lock)
|
||||
|
||||
def observe(self, url: str, headers: Mapping[str, str], body: bytes | None) -> None:
|
||||
if body is None or "/converse" not in url:
|
||||
return
|
||||
with self._lock:
|
||||
self._bodies.append(ConverseRequestBody.model_validate_json(body))
|
||||
|
||||
@property
|
||||
def bodies(self) -> tuple[ConverseRequestBody, ...]:
|
||||
with self._lock:
|
||||
return tuple(self._bodies)
|
||||
|
||||
|
||||
WEATHER_TOOL = ResponsesFunctionTool(
|
||||
name="get_weather",
|
||||
|
|
@ -295,6 +328,53 @@ class TestResponses:
|
|||
arguments = WeatherArguments.model_validate(raw_arguments)
|
||||
assert arguments.location, f"function call arguments missing location: {function_call.arguments}"
|
||||
|
||||
@pytest.mark.provider_edge_host
|
||||
@pytest.mark.parametrize("endpoint", ["/v1/responses", "/v1/chat/completions"])
|
||||
def test_bedrock_forwards_allowed_safety_identifier_as_additional_model_request_field(
|
||||
self, endpoints_client: EndpointsClient, resources: ResourceManager, endpoint: str
|
||||
) -> None:
|
||||
capture: Final = ConverseRequestCapture()
|
||||
edge: Final = start_provider_edge(
|
||||
LiveEdge(observe_request=capture.observe, sign=bedrock_signer(BEDROCK_EDGE_REGION)),
|
||||
mounts=MappingProxyType({BEDROCK_EDGE_MOUNT: f"https://bedrock-runtime.{BEDROCK_EDGE_REGION}.amazonaws.com"}),
|
||||
bind_host=PROVIDER_EDGE_BIND_HOST,
|
||||
advertise_host=PROVIDER_EDGE_ADVERTISE_HOST,
|
||||
)
|
||||
resources.defer(edge.shutdown)
|
||||
model: Final = f"e2e-responses-{unique_marker()}"
|
||||
model_id: Final = endpoints_client.create_model(
|
||||
model,
|
||||
LiteLLMParamsBody(
|
||||
model=BEDROCK_CONVERSE_BACKEND,
|
||||
api_base=edge.edge.api_base(BEDROCK_EDGE_MOUNT),
|
||||
aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID",
|
||||
aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY",
|
||||
aws_region_name=BEDROCK_EDGE_REGION,
|
||||
allowed_openai_params=["safety_identifier"],
|
||||
),
|
||||
)
|
||||
resources.defer(lambda: endpoints_client.delete_model(model_id))
|
||||
key: Final = resources.key()
|
||||
safety_identifier: Final = f"end-user-{unique_marker()}"
|
||||
|
||||
if endpoint == "/v1/responses":
|
||||
endpoints_client.responses(key, model, "reply with one word", safety_identifier=safety_identifier)
|
||||
else:
|
||||
endpoints_client.proxy.chat(
|
||||
key,
|
||||
ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content="reply with one word")],
|
||||
safety_identifier=safety_identifier,
|
||||
),
|
||||
)
|
||||
|
||||
forwarded: Final = tuple(body.additionalModelRequestFields for body in capture.bodies)
|
||||
assert forwarded, f"{endpoint} produced no Bedrock Converse request"
|
||||
assert forwarded == ({"safety_identifier": safety_identifier},) * len(forwarded), (
|
||||
f"{endpoint} did not forward safety_identifier to Bedrock Converse on every attempt: {capture.bodies}"
|
||||
)
|
||||
|
||||
@pytest.mark.skip(reason="stage red: product gap, /v1/responses 500s (aresponses TypeError) on missing input instead of 400")
|
||||
@pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works")
|
||||
def test_missing_input_returns_error(
|
||||
|
|
|
|||
|
|
@ -96,8 +96,8 @@ def _spend_until_budget_blocks(client: ManagementClient, key: str) -> None:
|
|||
for _ in range(40):
|
||||
outcome = client.chat_status(key, SPEND_MODEL, f"spend {unique_marker()}")
|
||||
if _is_budget_block(outcome):
|
||||
assert outcome.status_code == 429, (
|
||||
f"budget refusal must be 429, got {outcome.status_code}: {outcome.body[:200]}"
|
||||
assert outcome.status_code == 422, (
|
||||
f"budget refusal must be 422, got {outcome.status_code}: {outcome.body[:200]}"
|
||||
)
|
||||
return
|
||||
assert outcome.ok, f"paid call failed before the budget tripped ({outcome.status_code}): {outcome.body[:300]}"
|
||||
|
|
|
|||
|
|
@ -298,6 +298,7 @@ class ChatBody(BaseModel):
|
|||
max_completion_tokens: int | None = None
|
||||
temperature: float | None = None
|
||||
user: str | None = None
|
||||
safety_identifier: str | None = None
|
||||
metadata: ChatMetadata | None = None
|
||||
reasoning_effort: str | None = None
|
||||
thinking: ThinkingParam | None = None
|
||||
|
|
@ -976,6 +977,7 @@ class LiteLLMParamsBody(BaseModel):
|
|||
api_base: str | None = None
|
||||
api_version: str | None = None
|
||||
realtime_protocol: str | None = None
|
||||
allowed_openai_params: list[str] | None = None
|
||||
aws_access_key_id: str | None = None
|
||||
aws_secret_access_key: str | None = None
|
||||
aws_region_name: str | None = None
|
||||
|
|
|
|||
|
|
@ -99,6 +99,7 @@ from provider_cache import (
|
|||
SIGNATURE_HEADERS,
|
||||
CacheEdge,
|
||||
MountPolicy,
|
||||
RequestSigner,
|
||||
is_bedrock,
|
||||
scoped_edge_base,
|
||||
split_test_segment,
|
||||
|
|
@ -539,6 +540,7 @@ class ReplayEdge:
|
|||
@dataclass(frozen=True, slots=True)
|
||||
class LiveEdge:
|
||||
observe_request: Callable[[str, Mapping[str, str], bytes | None], None] | None = None
|
||||
sign: RequestSigner | None = None
|
||||
|
||||
|
||||
type EdgeBackend = RecordEdge | ReplayEdge | LiveEdge | CacheEdge
|
||||
|
|
@ -788,14 +790,16 @@ def _handle_live(
|
|||
method: str, url: str, headers: Mapping[str, str], body: bytes | None, timeout: float,
|
||||
cache: CacheEdge | None = None, mount: str = "", test_key: str | None = None,
|
||||
observe_request: Callable[[str, Mapping[str, str], bytes | None], None] | None = None,
|
||||
sign: RequestSigner | None = None,
|
||||
) -> EdgeOutcome:
|
||||
forwarded: Final = {
|
||||
name: value for name, value in headers.items() if name.lower() not in _REQUEST_DROPPED_HEADERS
|
||||
}
|
||||
if observe_request is not None:
|
||||
observe_request(url, forwarded, body)
|
||||
outbound: Final = forwarded if sign is None else sign(method, url, forwarded, body)
|
||||
head: Final = (
|
||||
forward_stream(method, url, headers=forwarded, body=body, timeout=timeout)
|
||||
forward_stream(method, url, headers=outbound, body=body, timeout=timeout)
|
||||
if cache is None else cache.forward(mount, method, url, forwarded, body, timeout, test_key=test_key)
|
||||
)
|
||||
match head:
|
||||
|
|
@ -871,10 +875,10 @@ def handle_edge_request(
|
|||
method, _upstream_url(upstream_base, upstream_path, split.query), headers, body, timeout,
|
||||
backend, mount, test_key,
|
||||
)
|
||||
case LiveEdge(observe_request=observe_request):
|
||||
case LiveEdge(observe_request=observe_request, sign=sign):
|
||||
return _handle_live(
|
||||
method, _upstream_url(upstream_base, upstream_path, split.query), headers, body, timeout,
|
||||
observe_request=observe_request,
|
||||
observe_request=observe_request, sign=sign,
|
||||
)
|
||||
case RecordEdge():
|
||||
return _handle_record(
|
||||
|
|
|
|||
|
|
@ -13,3 +13,4 @@ markers =
|
|||
cli_determinism: drives the real claude CLI for several seconds; deselected unless E2E_CLI_DETERMINISM is set
|
||||
redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set
|
||||
mcp_oauth_live: real Linear OAuth consent via a captured browser session; deselected unless E2E_MCP_OAUTH_LIVE is set
|
||||
provider_edge_host: routes provider traffic through the pytest host's edge in every fixture mode, so the gateway must reach the pytest host; deselected unless E2E_PROVIDER_EDGE_HOST_REACHABLE is set
|
||||
|
|
|
|||
|
|
@ -46,10 +46,10 @@ def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") ->
|
|||
pytest.fail("budget never enforced within the call budget")
|
||||
|
||||
|
||||
def _assert_blocked_429(client: BudgetClient, key: str) -> StreamingResponse:
|
||||
def _assert_blocked_422(client: BudgetClient, key: str) -> StreamingResponse:
|
||||
blocked = _assert_budget_blocks(client, key)
|
||||
assert blocked.status_code == 429, (
|
||||
f"budget refusal must be 429, got {blocked.status_code}: {blocked.body[:200]}"
|
||||
assert blocked.status_code == 422, (
|
||||
f"budget refusal must be 422, got {blocked.status_code}: {blocked.body[:200]}"
|
||||
)
|
||||
return blocked
|
||||
|
||||
|
|
@ -60,7 +60,7 @@ class TestBudgetBlocksPerLevel:
|
|||
key = client.generate_key(max_budget=TINY_CAP)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
_assert_blocked_429(client, key)
|
||||
_assert_blocked_422(client, key)
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.team.blocks_over_limit")
|
||||
def test_team_budget_blocks_every_team_key(self, client: BudgetClient, resources: ResourceManager) -> None:
|
||||
|
|
@ -71,10 +71,10 @@ class TestBudgetBlocksPerLevel:
|
|||
sibling_key = client.generate_key(team_id=team_id)
|
||||
resources.defer(lambda: client.delete_key(sibling_key))
|
||||
|
||||
_assert_blocked_429(client, spender_key)
|
||||
_assert_blocked_422(client, spender_key)
|
||||
sibling = _chat(client, sibling_key)
|
||||
assert is_budget_block(sibling) and sibling.status_code == 429, (
|
||||
f"a sibling key on the capped team must get the same 429 budget_exceeded, "
|
||||
assert is_budget_block(sibling) and sibling.status_code == 422, (
|
||||
f"a sibling key on the capped team must get the same 422 budget_exceeded, "
|
||||
f"got {sibling.status_code}: {sibling.body[:200]}"
|
||||
)
|
||||
|
||||
|
|
@ -99,10 +99,10 @@ class TestBudgetBlocksPerLevel:
|
|||
team_key = client.generate_key(team_id=team_id, user_id=user_id)
|
||||
resources.defer(lambda: client.delete_key(team_key))
|
||||
|
||||
_assert_blocked_429(client, first_key)
|
||||
_assert_blocked_422(client, first_key)
|
||||
second = _chat(client, second_key)
|
||||
assert is_budget_block(second) and second.status_code == 429, (
|
||||
f"the second personal key of a user over budget must get the same 429 budget_exceeded, "
|
||||
assert is_budget_block(second) and second.status_code == 422, (
|
||||
f"the second personal key of a user over budget must get the same 422 budget_exceeded, "
|
||||
f"got {second.status_code}: {second.body[:200]}"
|
||||
)
|
||||
team_result = _chat(client, team_key)
|
||||
|
|
@ -133,7 +133,7 @@ class TestBudgetBlocksPerLevel:
|
|||
key = client.generate_key(team_id=team_id)
|
||||
resources.defer(lambda: client.delete_key(key))
|
||||
|
||||
blocked = _assert_blocked_429(client, key)
|
||||
blocked = _assert_blocked_422(client, key)
|
||||
assert f"Organization={org_id}" in blocked.body, (
|
||||
f"refusal must name the org as the blocker, got: {blocked.body[:200]}"
|
||||
)
|
||||
|
|
@ -155,7 +155,7 @@ class TestBudgetBlocksPerLevel:
|
|||
teammate_key = client.generate_key(team_id=team_id, user_id=teammate_id)
|
||||
resources.defer(lambda: client.delete_key(teammate_key))
|
||||
|
||||
_assert_blocked_429(client, member_key)
|
||||
_assert_blocked_422(client, member_key)
|
||||
require_successful_call(_chat(client, teammate_key))
|
||||
|
||||
|
||||
|
|
@ -176,7 +176,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
|
|||
control_key = client.generate_key(user_id=user_id)
|
||||
resources.defer(lambda: client.delete_key(control_key))
|
||||
|
||||
_assert_blocked_429(client, capped_key)
|
||||
_assert_blocked_422(client, capped_key)
|
||||
require_successful_call(_chat(client, control_key))
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
|
||||
|
|
@ -188,7 +188,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
|
|||
control_key = client.generate_key(team_id=team_id)
|
||||
resources.defer(lambda: client.delete_key(control_key))
|
||||
|
||||
_assert_blocked_429(client, capped_key)
|
||||
_assert_blocked_422(client, capped_key)
|
||||
require_successful_call(_chat(client, control_key))
|
||||
|
||||
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
|
||||
|
|
@ -205,5 +205,5 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
|
|||
control_key = client.generate_key(team_id=team_id, user_id=member_id)
|
||||
resources.defer(lambda: client.delete_key(control_key))
|
||||
|
||||
_assert_blocked_429(client, capped_key)
|
||||
_assert_blocked_422(client, capped_key)
|
||||
require_successful_call(_chat(client, control_key))
|
||||
|
|
|
|||
|
|
@ -102,7 +102,7 @@ def test_long_window_blocks_after_short_window_resets(client: BudgetClient, reso
|
|||
|
||||
# 1. drive the key to get blocked by SHORT_WINDOW, assert it's budget error
|
||||
blocked = _drive_to_block(client, key)
|
||||
assert blocked.status_code == 429, f"budget block was not a 429: {blocked.status_code} {blocked.body[:200]}"
|
||||
assert blocked.status_code == 422, f"budget block was not a 422: {blocked.status_code} {blocked.body[:200]}"
|
||||
|
||||
# 2. check the reset times of both budget windows after we drove to being blocked
|
||||
blocked_reset_at = window_reset_at(client.key_budget_windows(key), SHORT_WINDOW)
|
||||
|
|
|
|||
|
|
@ -101,7 +101,7 @@ def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient,
|
|||
|
||||
# 1. drive the key to being blocked, assert its blocked by budget budget_exceeded
|
||||
blocked = _drive_to_block(client, key)
|
||||
assert blocked.status_code == 429, f"budget block was not a 429: {blocked.status_code} {blocked.body[:200]}"
|
||||
assert blocked.status_code == 422, f"budget block was not a 422: {blocked.status_code} {blocked.body[:200]}"
|
||||
|
||||
# 2. check the the teams budget windows
|
||||
blocked_reset_at = window_reset_at(client.team_budget_windows(team_id), SHORT_WINDOW)
|
||||
|
|
|
|||
|
|
@ -97,7 +97,7 @@ def test_zero_false_and_empty_values_are_not_treated_as_omission(gateway: Gatewa
|
|||
"POST", "/v1/chat/completions",
|
||||
{"model": models[0], "messages": [{"role": "user", "content": "zero budget"}]}, key=key,
|
||||
)
|
||||
assert denied.status_code == 429, denied.text
|
||||
assert denied.status_code == 422, denied.text
|
||||
assert denied.json()["error"]["type"] == "budget_exceeded"
|
||||
gateway.post("/key/update", {"key": key, "max_budget": 1, "models": [], "metadata": {}})
|
||||
info: Final = object_value(gateway.get("/key/info", {"key": key})["info"])
|
||||
|
|
@ -127,7 +127,7 @@ def test_zero_false_and_empty_values_are_not_treated_as_omission(gateway: Gatewa
|
|||
"POST", "/v1/chat/completions",
|
||||
{"model": models[0], "messages": [{"role": "user", "content": "updated zero budget"}]}, key=key,
|
||||
)
|
||||
assert zero_after_update.status_code == 429, zero_after_update.text
|
||||
assert zero_after_update.status_code == 422, zero_after_update.text
|
||||
assert zero_after_update.json()["error"]["type"] == "budget_exceeded"
|
||||
gateway.post("/key/update", {"key": key, "max_budget": None})
|
||||
assert read_rows(
|
||||
|
|
|
|||
|
|
@ -185,7 +185,7 @@ def test_key_budget_at_boundary_blocks_provider_then_explicit_reset_restores(gat
|
|||
{"model": model, "messages": [{"role": "user", "content": f"over budget {uuid.uuid4().hex}"}]},
|
||||
key=key,
|
||||
)
|
||||
assert denied.status_code == 429 and denied.json()["error"]["type"] == "budget_exceeded", denied.text
|
||||
assert denied.status_code == 422 and denied.json()["error"]["type"] == "budget_exceeded", denied.text
|
||||
assert upstream.get("/__observations").json()["requests"] == []
|
||||
assert gateway.chat(model, key=control, text=f"control {uuid.uuid4().hex}")["usage"]["total_tokens"] == 40
|
||||
gateway.post("/key/update", {"key": key, "spend": 0})
|
||||
|
|
@ -205,7 +205,7 @@ def test_key_budget_at_boundary_blocks_provider_then_explicit_reset_restores(gat
|
|||
{"model": model, "messages": [{"role": "user", "content": f"boundary again {uuid.uuid4().hex}"}]},
|
||||
key=key,
|
||||
)
|
||||
assert denied_again.status_code == 429 and denied_again.json()["error"]["type"] == "budget_exceeded", (
|
||||
assert denied_again.status_code == 422 and denied_again.json()["error"]["type"] == "budget_exceeded", (
|
||||
denied_again.text
|
||||
)
|
||||
assert upstream.get("/__observations").json()["requests"] == []
|
||||
|
|
|
|||
|
|
@ -133,3 +133,9 @@ meta.llama3-2-11b-instruct-v1:0
|
|||
us.meta.llama3-2-11b-instruct-v1:0
|
||||
meta.llama3-2-90b-instruct-v1:0
|
||||
us.meta.llama3-2-90b-instruct-v1:0
|
||||
bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b
|
||||
bedrock/ap-south-1/qwen.qwen3-next-80b-a3b
|
||||
bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b
|
||||
bedrock/eu-west-1/qwen.qwen3-next-80b-a3b
|
||||
bedrock/eu-west-2/qwen.qwen3-next-80b-a3b
|
||||
bedrock/sa-east-1/qwen.qwen3-next-80b-a3b
|
||||
|
|
|
|||
|
|
@ -605,6 +605,38 @@ class TestNonStreaming:
|
|||
assert result["error"]["message"] == "Bad request"
|
||||
|
||||
|
||||
class TestStreaming:
|
||||
"""Streaming requests must ask AgentCore for a stream, not a single send."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_streaming_request_uses_message_stream_method_and_yields_sse_events(self, httpx_transport):
|
||||
from litellm.a2a_protocol.providers.bedrock_agentcore.config import (
|
||||
BedrockAgentCoreA2AConfig,
|
||||
)
|
||||
|
||||
sse_body = (
|
||||
'data: {"jsonrpc": "2.0", "id": "req-001", "result": {"kind": "task", "id": "t1"}}\n\n'
|
||||
'data: {"jsonrpc": "2.0", "id": "req-001", "result": {"kind": "status-update", "final": true}}\n\n'
|
||||
)
|
||||
with respx.mock(assert_all_called=True) as router:
|
||||
route = router.post(url__regex=r".*/invocations.*").mock(
|
||||
return_value=httpx.Response(200, headers={"content-type": "text/event-stream"}, text=sse_body)
|
||||
)
|
||||
events = [
|
||||
event
|
||||
async for event in BedrockAgentCoreA2AConfig().handle_streaming(
|
||||
request_id="req-001",
|
||||
params=SAMPLE_PARAMS,
|
||||
litellm_params=SAMPLE_LITELLM_PARAMS,
|
||||
)
|
||||
]
|
||||
|
||||
sent_body = json.loads(route.calls.last.request.content)
|
||||
assert sent_body["method"] == "message/stream", sent_body
|
||||
assert sent_body["params"]["message"]["messageId"] == "msg-001"
|
||||
assert [event["result"]["kind"] for event in events] == ["task", "status-update"]
|
||||
|
||||
|
||||
class TestConfigManager:
|
||||
"""Test that config manager routes 'bedrock' correctly."""
|
||||
|
||||
|
|
|
|||
|
|
@ -3033,7 +3033,7 @@ def test_get_error_information_budget_exceeded_structured_fields():
|
|||
assert result["error_budget_entity_id"] == "repro-user"
|
||||
assert result["error_budget_limit"] == 1e-06
|
||||
assert result["error_budget_spend"] == 3.4e-05
|
||||
assert result["error_code"] == "429"
|
||||
assert result["error_code"] == "422"
|
||||
assert result["error_class"] == "BudgetExceededError"
|
||||
assert result["error_rate_limit_type"] == "budget"
|
||||
|
||||
|
|
@ -6407,7 +6407,7 @@ def test_get_error_information_keeps_traceback_for_unmapped_provider_4xx():
|
|||
|
||||
|
||||
def test_get_error_information_skips_traceback_for_budget_rejection_with_provider():
|
||||
"""A key-over-budget 429 is the proxy's own rejection even after the auth
|
||||
"""A key-over-budget 422 is the proxy's own rejection even after the auth
|
||||
handler stamps the requested model's provider onto it, so it stays cheap."""
|
||||
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
|
||||
|
||||
|
|
@ -6416,7 +6416,7 @@ def test_get_error_information_skips_traceback_for_budget_rejection_with_provide
|
|||
litellm.BudgetExceededError(current_cost=0.01, max_budget=0.0, llm_provider="anthropic")
|
||||
)
|
||||
result = StandardLoggingPayloadSetup.get_error_information(over_budget)
|
||||
assert result["error_code"] == "429"
|
||||
assert result["error_code"] == "422"
|
||||
assert result["llm_provider"] == "anthropic"
|
||||
assert result["traceback"] == ""
|
||||
|
||||
|
|
|
|||
|
|
@ -110,6 +110,19 @@ class TestOutputConfigStrippedFromCompletionKwargs:
|
|||
"reject it with 400 'Extra inputs are not permitted'"
|
||||
)
|
||||
|
||||
def test_safeguards_is_stripped_for_non_anthropic_target(self):
|
||||
extra_kwargs = {
|
||||
"custom_llm_provider": "azure",
|
||||
"safeguards": [{"type": "dangerous_tool_use", "classifier_context": {"v": 1}}],
|
||||
}
|
||||
|
||||
result = _call_prepare(extra_kwargs=extra_kwargs)
|
||||
|
||||
completion_kwargs = result[0] if isinstance(result, tuple) else result
|
||||
assert "safeguards" not in completion_kwargs, (
|
||||
"safeguards is an Anthropic-only field; OpenAI-format backends reject it with 400"
|
||||
)
|
||||
|
||||
def test_output_config_format_translated_to_response_format(self):
|
||||
"""When ``output_config`` carries structured-output ``format``, the
|
||||
translator now maps it to OpenAI's ``response_format`` so non-Anthropic
|
||||
|
|
|
|||
|
|
@ -1438,3 +1438,109 @@ async def test_anthropic_messages_leaves_non_provider_failures_unmapped():
|
|||
)
|
||||
|
||||
assert "Traceback" not in str(excinfo.value)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_messages_forwards_safeguards_and_unknown_beta_to_anthropic():
|
||||
"""Shapes are what Claude Code 2.1.278 sends and api.anthropic.com returns, captured 2026-09-21."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
|
||||
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
client_betas = "dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14"
|
||||
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": {}}}]
|
||||
captured: dict[str, object] = {}
|
||||
|
||||
def upstream_records_the_request(request: httpx.Request) -> httpx.Response:
|
||||
captured["body"] = json.loads(request.content)
|
||||
captured["anthropic-beta"] = request.headers.get("anthropic-beta")
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "claude-haiku-4-5",
|
||||
"content": [{"type": "text", "text": "ok"}],
|
||||
"stop_reason": "end_turn",
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
"safeguard_results": safeguard_results,
|
||||
},
|
||||
request=request,
|
||||
)
|
||||
|
||||
upstream = AsyncHTTPHandler()
|
||||
upstream.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream_records_the_request))
|
||||
|
||||
response = await handler.anthropic_messages(
|
||||
max_tokens=16,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="anthropic/claude-haiku-4-5",
|
||||
custom_llm_provider="anthropic",
|
||||
api_key="sk-test",
|
||||
client=upstream,
|
||||
safeguards=safeguards,
|
||||
extra_headers={"anthropic-beta": client_betas},
|
||||
)
|
||||
|
||||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert set(captured["anthropic-beta"].split(",")) == set(client_betas.split(","))
|
||||
assert response["safeguard_results"] == safeguard_results
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_messages_streaming_forwards_safeguards_and_keeps_safeguard_results():
|
||||
"""Shapes are what Claude Code 2.1.278 sends and api.anthropic.com returns, captured 2026-09-21."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages import handler
|
||||
|
||||
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
|
||||
tool_verdicts = {"toolu_01": {"type": "evaluated", "outcome": "not_flagged"}}
|
||||
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": tool_verdicts}}]
|
||||
captured: dict[str, object] = {}
|
||||
message_start = {
|
||||
"type": "message_start",
|
||||
"message": {
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "claude-haiku-4-5",
|
||||
"content": [],
|
||||
"stop_reason": None,
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 1, "output_tokens": 0},
|
||||
"safeguard_results": safeguard_results,
|
||||
},
|
||||
}
|
||||
message_delta = {
|
||||
"type": "message_delta",
|
||||
"delta": {"stop_reason": "end_turn", "stop_sequence": None, "safeguard_results": safeguard_results},
|
||||
"usage": {"output_tokens": 1},
|
||||
}
|
||||
sse = "".join(
|
||||
f"event: {event['type']}\ndata: {json.dumps(event)}\n\n"
|
||||
for event in (message_start, message_delta, {"type": "message_stop"})
|
||||
)
|
||||
|
||||
def upstream_streams_safeguard_results(request: httpx.Request) -> httpx.Response:
|
||||
captured["body"] = json.loads(request.content)
|
||||
return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=sse.encode(), request=request)
|
||||
|
||||
upstream = AsyncHTTPHandler()
|
||||
upstream.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream_streams_safeguard_results))
|
||||
|
||||
stream = await handler.anthropic_messages(
|
||||
max_tokens=16,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="anthropic/claude-haiku-4-5",
|
||||
custom_llm_provider="anthropic",
|
||||
api_key="sk-test",
|
||||
client=upstream,
|
||||
stream=True,
|
||||
safeguards=safeguards,
|
||||
)
|
||||
raw = b"".join([chunk async for chunk in stream]).decode()
|
||||
events = [json.loads(line[len("data: ") :]) for line in raw.splitlines() if line.startswith("data: ")]
|
||||
|
||||
assert captured["body"]["safeguards"] == safeguards
|
||||
assert events[0]["message"]["safeguard_results"] == safeguard_results
|
||||
assert [e for e in events if e["type"] == "message_delta"][0]["delta"]["safeguard_results"] == safeguard_results
|
||||
|
|
|
|||
|
|
@ -453,6 +453,38 @@ class TestAzureMAIImageGeneration:
|
|||
)
|
||||
assert round(cost, 10) == round(expected_cost, 10)
|
||||
|
||||
def test_mai_image_pro_edit_cost_splits_text_and_image_input(self, monkeypatch):
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
model = "azure_ai/MAI-Image-2.5-Pro"
|
||||
model_info = litellm.get_model_info(model=model, custom_llm_provider="azure_ai")
|
||||
text_tokens = 37
|
||||
image_tokens = 1024
|
||||
output_image_tokens = 1024
|
||||
|
||||
image_response = ImageResponse(
|
||||
data=[ImageObject(b64_json="img1")],
|
||||
usage=ImageUsage(
|
||||
input_tokens=text_tokens + image_tokens,
|
||||
input_tokens_details=ImageUsageInputTokensDetails(
|
||||
text_tokens=text_tokens,
|
||||
image_tokens=image_tokens,
|
||||
),
|
||||
output_tokens=output_image_tokens,
|
||||
total_tokens=text_tokens + image_tokens + output_image_tokens,
|
||||
),
|
||||
)
|
||||
|
||||
cost = azure_ai_image_cost_calculator(model=model, image_response=image_response)
|
||||
|
||||
expected_cost = (
|
||||
text_tokens * model_info["input_cost_per_token"]
|
||||
+ image_tokens * model_info["input_cost_per_image_token"]
|
||||
+ output_image_tokens * model_info["output_cost_per_image_token"]
|
||||
)
|
||||
assert round(cost, 10) == round(expected_cost, 10)
|
||||
assert model_info["input_cost_per_image_token"] != model_info["input_cost_per_token"]
|
||||
|
||||
def test_mai_image_cost_calculator_falls_back_to_flat_image_pricing(self, monkeypatch):
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
|
|
|||
|
|
@ -6339,15 +6339,15 @@ class TestMCPDcrBridgeDelegateAdmission:
|
|||
)
|
||||
return exc_info.value
|
||||
|
||||
async def test_over_budget_admission_surfaces_429_not_401(self):
|
||||
"""A validly-authenticated but over-budget identity surfaces the standard pipeline's 429, not
|
||||
async def test_over_budget_admission_surfaces_422_not_401(self):
|
||||
"""A validly-authenticated but over-budget identity surfaces the standard pipeline's 422, not
|
||||
a misleading 401. Flattening budget to 401 told the caller their credential was invalid, which
|
||||
on a DCR client reads as broken auth and triggers a re-authorize that cannot fix a budget
|
||||
problem. Regression for the status-flattening finding on the live-policy gate."""
|
||||
import litellm
|
||||
|
||||
mapped = await self._enforce_with_gate_error(litellm.BudgetExceededError(current_cost=10.0, max_budget=1.0))
|
||||
assert mapped.status_code == 429
|
||||
assert mapped.status_code == 422
|
||||
|
||||
async def test_db_outage_during_policy_surfaces_503_not_401(self):
|
||||
"""A transient database outage during the live-policy gate surfaces a retryable 503, not a 401
|
||||
|
|
|
|||
|
|
@ -448,7 +448,7 @@ async def test_handle_authentication_error_budget_exceeded():
|
|||
)
|
||||
|
||||
assert exc_info.value.type == ProxyErrorTypes.budget_exceeded
|
||||
assert int(exc_info.value.code) == status.HTTP_429_TOO_MANY_REQUESTS
|
||||
assert int(exc_info.value.code) == status.HTTP_422_UNPROCESSABLE_CONTENT
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -687,7 +687,7 @@ def _http_request(client_host: str | None = "10.1.2.3", headers: dict[str, str]
|
|||
{"allow_requests_on_db_unavailable": False},
|
||||
{},
|
||||
"10.1.2.3",
|
||||
id="429_budget_exceeded",
|
||||
id="422_budget_exceeded",
|
||||
),
|
||||
],
|
||||
)
|
||||
|
|
@ -697,7 +697,7 @@ async def test_auth_failure_logs_requester_ip_address(
|
|||
request_kwargs: dict[str, dict[str, str]],
|
||||
expected_ip: str,
|
||||
) -> None:
|
||||
"""401s and budget 429s are rejected before `add_litellm_data_to_request` stamps
|
||||
"""401s and budget 422s are rejected before `add_litellm_data_to_request` stamps
|
||||
the caller IP, so without this the failure logs (spend logs, prometheus client_ip)
|
||||
had no IP, and a 401 rarely carries a key or user identity either."""
|
||||
with (
|
||||
|
|
|
|||
|
|
@ -75,7 +75,7 @@ async def test_over_first_window_raises():
|
|||
await _virtual_key_multi_budget_check(valid_token=token)
|
||||
|
||||
err = exc_info.value
|
||||
assert err.status_code == 429
|
||||
assert err.status_code == 422
|
||||
assert "24h" in str(err)
|
||||
assert "Key over" in str(err)
|
||||
|
||||
|
|
@ -107,7 +107,7 @@ async def test_over_second_window_raises():
|
|||
await _virtual_key_multi_budget_check(valid_token=token)
|
||||
|
||||
err = exc_info.value
|
||||
assert err.status_code == 429
|
||||
assert err.status_code == 422
|
||||
assert "30d" in str(err)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -8156,7 +8156,7 @@ async def test_reset_key_spend_resets_budget_windows(monkeypatch):
|
|||
counter without also advancing reset_at is not durable either: the very
|
||||
next request would re-sum the unchanged historical spend and put the
|
||||
counter right back above the window's max_budget, so
|
||||
_virtual_key_multi_budget_check kept raising BudgetExceededError (429) on
|
||||
_virtual_key_multi_budget_check kept raising BudgetExceededError (422) on
|
||||
every request even though the key's own reported spend read $0.
|
||||
"""
|
||||
mock_prisma_client = MagicMock()
|
||||
|
|
@ -16593,7 +16593,7 @@ async def test_info_key_fn_reads_the_configured_budget_model_key(monkeypatch):
|
|||
|
||||
It used to probe a second, provider-stripped key because the counter was
|
||||
written under the request model instead, which is what let a key report zero
|
||||
usage while being blocked at 429.
|
||||
usage while being blocked at 422.
|
||||
"""
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
|
|
|
|||
|
|
@ -2099,7 +2099,7 @@ class TestCursorVariantPerModelBudgetEnforcement:
|
|||
|
||||
response = _post_cursor_with_real_auth(valid_token, attrs, request_model="claude-opus-5-thinking-high")
|
||||
|
||||
assert response.status_code == 429, response.text
|
||||
assert response.status_code == 422, response.text
|
||||
error = response.json()["error"]
|
||||
assert error["type"] == "budget_exceeded"
|
||||
assert "exceeded budget for model=claude-opus-5" in error["message"]
|
||||
|
|
@ -2110,8 +2110,8 @@ class TestCursorVariantPerModelBudgetEnforcement:
|
|||
base_response = _post_cursor_with_real_auth(valid_token, attrs, request_model="claude-opus-5")
|
||||
alias_response = _post_cursor_with_real_auth(valid_token, attrs, request_model="claude-opus-5-fast")
|
||||
|
||||
assert base_response.status_code == 429, base_response.text
|
||||
assert alias_response.status_code == 429, alias_response.text
|
||||
assert base_response.status_code == 422, base_response.text
|
||||
assert alias_response.status_code == 422, alias_response.text
|
||||
assert alias_response.json() == base_response.json()
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -495,7 +495,7 @@ class TestProxyBaseLLMRequestProcessing:
|
|||
)
|
||||
|
||||
assert exc_info.value.type == ProxyErrorTypes.budget_exceeded
|
||||
assert exc_info.value.code == "429"
|
||||
assert exc_info.value.code == "422"
|
||||
tag_budget_check.assert_awaited_once()
|
||||
_, call_kwargs = tag_budget_check.call_args
|
||||
assert call_kwargs["tags"] == ("guardrail-tag",)
|
||||
|
|
@ -702,7 +702,7 @@ class TestProxyBaseLLMRequestProcessing:
|
|||
)
|
||||
|
||||
assert exc_info.value.type == ProxyErrorTypes.budget_exceeded
|
||||
assert exc_info.value.code == "429"
|
||||
assert exc_info.value.code == "422"
|
||||
assert "guardrail-tag" in exc_info.value.message
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
|
|||
|
|
@ -10872,7 +10872,7 @@ async def test_realtime_session_rejected_in_pre_call_releases_the_budget_reserva
|
|||
"""A rate-limit or guardrail rejection happens before route_request, so the
|
||||
relay never runs and no success log can own the reservation. The endpoint
|
||||
must release it on that exit too, or the key stays pinned at the reserved
|
||||
amount and its next requests 429 with budget_exceeded while /key/info shows
|
||||
amount and its next requests 422 with budget_exceeded while /key/info shows
|
||||
spend 0 (reproduced live with rpm_limit=1). The client still gets the
|
||||
pre-call error event and the 1011 close it got before."""
|
||||
reservation: Final = {"reserved_cost": 0.55, "input_cost": 0.0, "finalized": False, "entries": []}
|
||||
|
|
|
|||
|
|
@ -1248,6 +1248,16 @@ class TestFunctionCallTransformation:
|
|||
assert "tool_choice" not in result
|
||||
assert "tools" not in result
|
||||
|
||||
def test_safety_identifier_forwarded_to_chat_completion_request(self) -> None:
|
||||
result: Final = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
|
||||
model="bedrock/global.openai.gpt-5.6-luna",
|
||||
input="hi",
|
||||
responses_api_request={"safety_identifier": "user-7f3a"},
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
assert result["safety_identifier"] == "user-7f3a"
|
||||
|
||||
def test_parallel_tool_calls_dropped_when_no_chat_tools_remain(self) -> None:
|
||||
transform: Final = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request
|
||||
codex_tool_search: Final = {
|
||||
|
|
|
|||
|
|
@ -4230,3 +4230,30 @@ def test_completion_cost_prices_responses_websocket_turns_per_service_tier():
|
|||
assert ws_cost == pytest.approx(_http_cost(100, 40, "default") + _http_cost(60, 10, "priority"))
|
||||
assert ws_cost != pytest.approx(_http_cost(160, 50, "default"))
|
||||
assert ws_cost != pytest.approx(_http_cost(160, 50, "priority"))
|
||||
|
||||
|
||||
QWEN3_NEXT_REGIONS: Final = ("ap-northeast-1", "ap-south-1", "ap-southeast-2", "eu-west-1", "eu-west-2", "sa-east-1")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("region", QWEN3_NEXT_REGIONS)
|
||||
def test_cost_per_token_bedrock_qwen3_next_uses_regional_entry_not_us_rate(
|
||||
monkeypatch: pytest.MonkeyPatch, region: str
|
||||
) -> None:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
regional: Final = litellm.model_cost[f"bedrock/{region}/qwen.qwen3-next-80b-a3b"]
|
||||
us: Final = litellm.model_cost["qwen.qwen3-next-80b-a3b"]
|
||||
assert regional["input_cost_per_token"] != us["input_cost_per_token"]
|
||||
assert regional["output_cost_per_token"] != us["output_cost_per_token"]
|
||||
|
||||
prompt_tokens, completion_tokens = 1000, 500
|
||||
prompt_usd, completion_usd = cost_per_token(
|
||||
model=f"bedrock/{region}/qwen.qwen3-next-80b-a3b",
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
assert prompt_usd == pytest.approx(prompt_tokens * regional["input_cost_per_token"])
|
||||
assert completion_usd == pytest.approx(completion_tokens * regional["output_cost_per_token"])
|
||||
|
|
|
|||
|
|
@ -1397,13 +1397,18 @@ class TestBudgetExceededErrorSurfacesUnifiedFields:
|
|||
assert e.llm_provider == "anthropic"
|
||||
|
||||
def test_should_keep_existing_status_code_and_message(self):
|
||||
# Backward-compat guard: existing callers depend on `status_code=429`
|
||||
# Backward-compat guard: existing callers depend on `status_code=422`
|
||||
# and the canonical message format.
|
||||
e = litellm.BudgetExceededError(current_cost=0.000109, max_budget=0.0001)
|
||||
assert e.status_code == 429
|
||||
assert e.status_code == 422
|
||||
assert "Current cost: 0.000109" in e.message
|
||||
assert "Max budget: 0.0001" in e.message
|
||||
|
||||
def test_should_honor_budget_exceeded_status_code_override(self, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setattr(litellm, "budget_exceeded_status_code", 429)
|
||||
e = litellm.BudgetExceededError(current_cost=0.5, max_budget=0.1)
|
||||
assert e.status_code == 429
|
||||
|
||||
def test_should_still_be_catchable_as_exception_not_rate_limit_error(self):
|
||||
# Critical: we deliberately did NOT make BudgetExceededError a
|
||||
# RateLimitError subclass. Existing `except BudgetExceededError:`
|
||||
|
|
@ -1424,7 +1429,7 @@ class TestBudgetExceededErrorSurfacesUnifiedFields:
|
|||
info = StandardLoggingPayloadSetup.get_error_information(e)
|
||||
assert info["error_rate_limit_category"] == "litellm_rate_limit"
|
||||
assert info["error_rate_limit_type"] == "budget"
|
||||
assert info["error_code"] == "429"
|
||||
assert info["error_code"] == "422"
|
||||
assert info["error_class"] == "BudgetExceededError"
|
||||
|
||||
def test_should_propagate_llm_provider_to_standard_logging_payload(self):
|
||||
|
|
|
|||
|
|
@ -652,6 +652,7 @@ def validate_model_cost_values(model_data, exceptions=None):
|
|||
"cache_creation_input_audio_token_cost",
|
||||
"cache_read_input_token_cost",
|
||||
"cache_read_input_audio_token_cost",
|
||||
"cache_read_input_image_token_cost",
|
||||
"input_dbu_cost_per_token",
|
||||
"output_db_cost_per_token",
|
||||
"output_dbu_cost_per_token",
|
||||
|
|
@ -740,6 +741,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_read_input_token_cost_above_512k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": {"type": "number"},
|
||||
"cache_read_input_audio_token_cost": {"type": "number"},
|
||||
"cache_read_input_image_token_cost": {"type": "number"},
|
||||
"audio_transcription_config": {"type": "string"},
|
||||
"deprecation_date": {"type": "string"},
|
||||
"input_cost_per_audio_per_second": {"type": "number"},
|
||||
|
|
|
|||
4
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
4
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -30908,6 +30908,8 @@ export interface components {
|
|||
cache_creation_input_token_cost_ultrafast?: number | null;
|
||||
/** Cache Read Input Audio Token Cost */
|
||||
cache_read_input_audio_token_cost?: number | null;
|
||||
/** Cache Read Input Image Token Cost */
|
||||
cache_read_input_image_token_cost?: number | null;
|
||||
/** Cache Read Input Token Cost */
|
||||
cache_read_input_token_cost?: number | null;
|
||||
/** Cache Read Input Token Cost Above 200K Tokens */
|
||||
|
|
@ -41635,6 +41637,8 @@ export interface components {
|
|||
cache_creation_input_token_cost_ultrafast?: number | null;
|
||||
/** Cache Read Input Audio Token Cost */
|
||||
cache_read_input_audio_token_cost?: number | null;
|
||||
/** Cache Read Input Image Token Cost */
|
||||
cache_read_input_image_token_cost?: number | null;
|
||||
/** Cache Read Input Token Cost */
|
||||
cache_read_input_token_cost?: number | null;
|
||||
/** Cache Read Input Token Cost Above 200K Tokens */
|
||||
|
|
|
|||
|
|
@ -44,6 +44,7 @@ bedrock/ap-northeast-1/minimax.minimax-m2.5
|
|||
bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking
|
||||
bedrock/ap-northeast-1/moonshotai.kimi-k2.5
|
||||
bedrock/ap-northeast-1/qwen.qwen3-coder-next
|
||||
bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b
|
||||
bedrock/moonshotai.kimi-k2-thinking
|
||||
bedrock/moonshotai.kimi-k2.5
|
||||
bedrock/ap-south-1/meta.llama3-70b-instruct-v1:0
|
||||
|
|
@ -54,6 +55,7 @@ bedrock/ap-south-1/minimax.minimax-m2.5
|
|||
bedrock/ap-south-1/moonshotai.kimi-k2-thinking
|
||||
bedrock/ap-south-1/moonshotai.kimi-k2.5
|
||||
bedrock/ap-south-1/qwen.qwen3-coder-next
|
||||
bedrock/ap-south-1/qwen.qwen3-next-80b-a3b
|
||||
bedrock/ap-southeast-2/minimax.minimax-m2.5
|
||||
bedrock/ap-southeast-3/deepseek.v3.2
|
||||
bedrock/ap-southeast-3/minimax.minimax-m2.1
|
||||
|
|
@ -83,11 +85,13 @@ bedrock/eu-west-1/meta.llama3-8b-instruct-v1:0
|
|||
bedrock/eu-west-1/minimax.minimax-m2.1
|
||||
bedrock/eu-west-1/minimax.minimax-m2.5
|
||||
bedrock/eu-west-1/qwen.qwen3-coder-next
|
||||
bedrock/eu-west-1/qwen.qwen3-next-80b-a3b
|
||||
bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0
|
||||
bedrock/eu-west-2/meta.llama3-8b-instruct-v1:0
|
||||
bedrock/eu-west-2/minimax.minimax-m2.1
|
||||
bedrock/eu-west-2/minimax.minimax-m2.5
|
||||
bedrock/eu-west-2/qwen.qwen3-coder-next
|
||||
bedrock/eu-west-2/qwen.qwen3-next-80b-a3b
|
||||
bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2
|
||||
bedrock/eu-west-3/mistral.mistral-large-2402-v1:0
|
||||
bedrock/eu-west-3/mistral.mixtral-8x7b-instruct-v0:1
|
||||
|
|
@ -103,6 +107,7 @@ bedrock/sa-east-1/minimax.minimax-m2.5
|
|||
bedrock/sa-east-1/moonshotai.kimi-k2-thinking
|
||||
bedrock/sa-east-1/moonshotai.kimi-k2.5
|
||||
bedrock/sa-east-1/qwen.qwen3-coder-next
|
||||
bedrock/sa-east-1/qwen.qwen3-next-80b-a3b
|
||||
bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1
|
||||
bedrock/us-east-1/1-month-commitment/anthropic.claude-v1
|
||||
bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1
|
||||
|
|
@ -240,3 +245,4 @@ bedrock/us-gov-east-1/anthropic.claude-sonnet-5
|
|||
bedrock/us-gov-east-1/anthropic.claude-opus-4-8
|
||||
bedrock/us-gov-east-1/anthropic.claude-opus-5
|
||||
bedrock/us-gov-east-1/anthropic.claude-fable-5-1
|
||||
bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue