chore: merge upstream main into neuraltrust guardrail PR

This commit is contained in:
albertbausili 2026-09-28 11:53:41 +02:00
commit c9c268fa90
85 changed files with 1107 additions and 243 deletions

View file

@ -175,6 +175,8 @@ jobs:
env:
TESTS: ${{ needs.detect.outputs.tests }}
E2E_FIXTURE_MODE: live
E2E_PROVIDER_EDGE_HOST_REACHABLE: '1'
COLUMNS: '400'
run: |
umask 077
read -r -a test_files <<< "${TESTS}"
@ -189,6 +191,7 @@ jobs:
uv run --no-sync python .github/e2e-stack/assert_tests_ran.py "${report}" "${test_files[@]}"
verified=$?
set -e
grep -E '^(FAILED|ERROR) ' "${log}" || true
grep -E '^=+ .* in [0-9.]+s( \([0-9:]+\))? =+$' "${log}" | tail -n 1
echo "::endgroup::"
if [ "${status}" = "5" ]; then

View file

@ -7,6 +7,9 @@ metadata:
{{- include "litellm.commonLabels" . | nindent 4 }}
app.kubernetes.io/component: backend
spec:
{{- if and (not .Values.backend.hpa.enabled) (not (kindIs "invalid" .Values.backend.replicaCount)) }}
replicas: {{ .Values.backend.replicaCount }}
{{- end }}
{{- with .Values.backend.strategy }}
strategy:
{{- toYaml . | nindent 4 }}

View file

@ -7,6 +7,9 @@ metadata:
{{- include "litellm.commonLabels" . | nindent 4 }}
app.kubernetes.io/component: gateway
spec:
{{- if and (not .Values.gateway.hpa.enabled) (not (kindIs "invalid" .Values.gateway.replicaCount)) }}
replicas: {{ .Values.gateway.replicaCount }}
{{- end }}
{{- with .Values.gateway.strategy }}
strategy:
{{- toYaml . | nindent 4 }}

View file

@ -7,6 +7,9 @@ metadata:
{{- include "litellm.commonLabels" . | nindent 4 }}
app.kubernetes.io/component: ui
spec:
{{- if and (not .Values.ui.hpa.enabled) (not (kindIs "invalid" .Values.ui.replicaCount)) }}
replicas: {{ .Values.ui.replicaCount }}
{{- end }}
{{- with .Values.ui.strategy }}
strategy:
{{- toYaml . | nindent 4 }}

View file

@ -0,0 +1,100 @@
suite: test fixed replica count when HPA is disabled
templates:
- gateway/deployment.yaml
- gateway/configmap.yaml
- backend/deployment.yaml
- ui/deployment.yaml
values:
- ./values/required.yaml
tests:
- it: gateway renders replicaCount into spec.replicas when its HPA is disabled
template: gateway/deployment.yaml
set:
gateway.hpa.enabled: false
gateway.replicaCount: 3
asserts:
- isKind:
of: Deployment
- equal:
path: spec.replicas
value: 3
- it: backend renders replicaCount into spec.replicas when its HPA is disabled
template: backend/deployment.yaml
set:
backend.hpa.enabled: false
backend.replicaCount: 2
asserts:
- equal:
path: spec.replicas
value: 2
- it: ui renders replicaCount into spec.replicas when its HPA is disabled
template: ui/deployment.yaml
set:
ui.hpa.enabled: false
ui.replicaCount: 2
asserts:
- equal:
path: spec.replicas
value: 2
- it: replicaCount 0 scales the gateway to zero instead of being treated as unset
template: gateway/deployment.yaml
set:
gateway.hpa.enabled: false
gateway.replicaCount: 0
asserts:
- equal:
path: spec.replicas
value: 0
- it: a component with HPA disabled but no replicaCount set keeps omitting spec.replicas, so upgrades do not reset a hand-scaled Deployment
set:
gateway.hpa.enabled: false
backend.hpa.enabled: false
ui.hpa.enabled: false
asserts:
- notExists:
path: spec.replicas
template: gateway/deployment.yaml
- notExists:
path: spec.replicas
template: backend/deployment.yaml
- notExists:
path: spec.replicas
template: ui/deployment.yaml
- it: every component omits spec.replicas when its HPA is enabled, so the autoscaler owns the count
set:
gateway.hpa.enabled: true
gateway.replicaCount: 3
backend.hpa.enabled: true
backend.replicaCount: 3
ui.hpa.enabled: true
ui.replicaCount: 3
asserts:
- notExists:
path: spec.replicas
template: gateway/deployment.yaml
- notExists:
path: spec.replicas
template: backend/deployment.yaml
- notExists:
path: spec.replicas
template: ui/deployment.yaml
- it: a component with HPA disabled renders replicas while a sibling with HPA enabled does not
set:
gateway.hpa.enabled: false
gateway.replicaCount: 4
backend.hpa.enabled: true
backend.replicaCount: 4
asserts:
- equal:
path: spec.replicas
value: 4
template: gateway/deployment.yaml
- notExists:
path: spec.replicas
template: backend/deployment.yaml

View file

@ -397,6 +397,11 @@ gateway:
# failureThreshold: 30
# periodSeconds: 10
startupProbe: {}
# Optional fixed pod count, rendered into the Deployment's spec.replicas only
# when hpa.enabled is false. Unset by default so an existing Deployment keeps
# its current count; with the HPA on, the autoscaler owns the count, e.g.:
# replicaCount: 3
replicaCount:
hpa:
enabled: true
minReplicas: 1
@ -524,6 +529,8 @@ backend:
strategy: {}
# Optional startupProbe; same shape as gateway.startupProbe. Empty by default.
startupProbe: {}
# Same semantics as gateway.replicaCount.
replicaCount:
hpa:
enabled: true
minReplicas: 1
@ -590,6 +597,8 @@ ui:
strategy: {}
# Optional startupProbe; same shape as gateway.startupProbe. Empty by default.
startupProbe: {}
# Same semantics as gateway.replicaCount.
replicaCount:
hpa:
enabled: false
minReplicas: 1

View file

@ -400,6 +400,7 @@ default_redis_batch_cache_expiry: Optional[float] = None
model_alias_map: Dict[str, str] = {}
model_group_settings: Optional["ModelGroupSettings"] = None
max_budget: float = 0.0 # set the max budget across all providers
budget_exceeded_status_code: int = 422 # set to 429 to restore the pre-422 budget_exceeded response code
budget_duration: Optional[str] = (
None # proxy only - resets budget after fixed duration. You can set duration as seconds ("30s"), minutes ("30m"), hours ("30h"), days ("30d").
)

View file

@ -5,6 +5,7 @@ import logging
import os
import re
import sys
from collections.abc import Sequence
from datetime import datetime
from logging import Formatter
from typing import Any, Final, TextIO
@ -186,7 +187,8 @@ class SecretRedactionFilter(logging.Filter):
record.stack_info = _redact_string(record.stack_info) # rebind-ok: a Filter scrubs records in place
# Redact extra fields passed via logger.debug("msg", extra={...})
for key, value in list(record.__dict__.items()):
record_items: Final[Sequence[tuple[str, object]]] = list(record.__dict__.items())
for key, value in record_items:
if key in _STANDARD_RECORD_ATTRS:
continue
if isinstance(value, str):
@ -507,7 +509,7 @@ handler.addFilter(_secret_filter)
handler.addFilter(_correlation_filter)
def _try_parse_json_message(message: str) -> dict[str, Any] | None:
def _try_parse_json_message(message: str) -> dict[str, object] | None:
"""
Try to parse a log message as JSON. Returns parsed dict if valid, else None.
Handles messages that are entirely valid JSON (e.g. json.dumps output).
@ -585,7 +587,7 @@ class JsonFormatter(Formatter):
def format(self, record):
message_str: Final = record.getMessage()
json_record: Final[dict[str, Any]] = {
json_record: Final[dict[str, object]] = {
"message": message_str,
"level": record.levelname,
"timestamp": self.formatTime(record),

View file

@ -98,7 +98,7 @@ class BedrockAgentCoreA2AHandler:
request_id=request_id,
params=params,
litellm_params=litellm_params,
method="message/send",
method="message/stream",
stream=True,
agent_extra_headers=agent_extra_headers,
)

View file

@ -5,7 +5,7 @@ A2A Streaming Iterator with token tracking and logging support.
import asyncio
from collections.abc import AsyncIterator
from datetime import datetime
from typing import TYPE_CHECKING, Any, Final
from typing import TYPE_CHECKING, Final
import litellm
from litellm._logging import verbose_logger
@ -15,7 +15,7 @@ from litellm.litellm_core_utils.asyncify import asyncify
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
if TYPE_CHECKING:
from a2a.types import SendStreamingMessageRequest, SendStreamingMessageResponse
from a2a.compat.v0_3.types import SendStreamingMessageRequest, SendStreamingMessageResponse
class A2AStreamingIterator:
@ -39,9 +39,9 @@ class A2AStreamingIterator:
self.start_time = datetime.now()
# Collect chunks for token counting
self.chunks: list[Any] = []
self.chunks: list[SendStreamingMessageResponse] = []
self.collected_text_parts: list[str] = []
self.final_chunk: Any | None = None
self.final_chunk: SendStreamingMessageResponse | None = None
def __aiter__(self):
return self
@ -69,7 +69,7 @@ class A2AStreamingIterator:
await self._handle_stream_complete()
raise
def _collect_text_from_chunk(self, chunk: Any) -> None:
def _collect_text_from_chunk(self, chunk: "SendStreamingMessageResponse") -> None:
"""Extract text from a streaming chunk and add to collected parts."""
try:
chunk_dict: Final = chunk.model_dump(mode="json", exclude_none=True) if hasattr(chunk, "model_dump") else {}
@ -79,7 +79,7 @@ class A2AStreamingIterator:
except Exception:
verbose_logger.debug("Failed to extract text from A2A streaming chunk")
def _is_completed_chunk(self, chunk: Any) -> bool:
def _is_completed_chunk(self, chunk: "SendStreamingMessageResponse") -> bool:
"""Check if chunk indicates stream completion."""
try:
chunk_dict: Final = chunk.model_dump(mode="json", exclude_none=True) if hasattr(chunk, "model_dump") else {}

View file

@ -16,6 +16,7 @@ from typing import Any, Final
import httpx
import openai
import litellm
from litellm.types.utils import LiteLLMCommonStrings
from litellm.types.vector_stores import VectorStoreSearchFailure
@ -1002,7 +1003,7 @@ class BudgetExceededError(Exception):
):
self.current_cost = current_cost
self.max_budget = max_budget
self.status_code = 429
self.status_code = litellm.budget_exceeded_status_code
self.llm_provider = llm_provider or ""
self.entity_type = entity_type
self.entity_id = entity_id

View file

@ -15,6 +15,8 @@ from .destinations import FocusTimeWindow
if TYPE_CHECKING:
from apscheduler.schedulers.asyncio import AsyncIOScheduler
from litellm.proxy.db.db_transaction_queue.pod_lock_manager import PodLockManager
from .export_engine import FocusExportEngine
else:
AsyncIOScheduler = Any
@ -111,7 +113,7 @@ class FocusLogger(CustomLogger):
"""Entry point for scheduler jobs to run export cycle with locking."""
from litellm.proxy.proxy_server import proxy_logging_obj
pod_lock_manager = None
pod_lock_manager: PodLockManager | None = None
if proxy_logging_obj is not None:
writer: Final = getattr(proxy_logging_obj, "db_spend_update_writer", None)
if writer is not None:

View file

@ -58,7 +58,7 @@ class GenericPromptManager(CustomPromptManagement):
api_key: str | None = None,
timeout: int = 30,
prompt_id: str | None = None,
additional_provider_specific_query_params: dict[str, Any] | None = None,
additional_provider_specific_query_params: Mapping[str, object] | None = None,
**kwargs,
):
"""

View file

@ -21,7 +21,7 @@ from __future__ import annotations
import os
from datetime import datetime, timedelta, timezone
from typing import TYPE_CHECKING, Any, Final
from typing import TYPE_CHECKING, Any, Final, Protocol
import litellm
from litellm._logging import verbose_proxy_logger
@ -35,6 +35,17 @@ else:
AsyncIOScheduler = Any
class _PodLockManager(Protocol):
"""The subset of PodLockManager this logger drives to serialize the export across pods."""
@property
def redis_cache(self) -> object: ...
async def acquire_lock(self, cronjob_id: str) -> bool | None: ...
async def release_lock(self, cronjob_id: str) -> None: ...
def _parse_metrics_marker(
marker: object | None,
) -> datetime | None:
@ -226,9 +237,9 @@ class MavvrikFocusLogger(FocusLogger):
"""Scheduler entry point — uses Mavvrik-specific pod-lock key."""
from litellm.proxy.proxy_server import proxy_logging_obj # noqa: PLC0415
pod_lock_manager = None
pod_lock_manager: _PodLockManager | None = None
if proxy_logging_obj is not None:
writer: Final = getattr(proxy_logging_obj, "db_spend_update_writer", None)
writer: Final[object] = getattr(proxy_logging_obj, "db_spend_update_writer", None)
if writer is not None:
pod_lock_manager = getattr(writer, "pod_lock_manager", None)

View file

@ -9,9 +9,12 @@ this preset registers a custom exporter (``kind="agentops"``) that mints the JWT
worker thread, off any event loop — and caches it for the process lifetime.
"""
from collections.abc import Sequence
from typing import Any, Final
import httpx
from opentelemetry.sdk.trace import ReadableSpan
from opentelemetry.sdk.trace.export import SpanExporter, SpanExportResult
from pydantic import Field
from pydantic_settings import BaseSettings, SettingsConfigDict
@ -71,7 +74,7 @@ def agentops_preset(
)
def _build_agentops_exporter(spec: ExporterSpec) -> Any:
def _build_agentops_exporter(spec: ExporterSpec) -> SpanExporter:
"""Factory for the ``agentops`` exporter kind: a lazy-auth OTLP/HTTP exporter."""
from opentelemetry.exporter.otlp.proto.http.trace_exporter import (
OTLPSpanExporter,
@ -106,7 +109,7 @@ def _build_agentops_exporter(spec: ExporterSpec) -> Any:
except Exception as e:
verbose_logger.debug("AgentOps JWT fetch failed: %s", e)
def export(self, spans: Any) -> Any:
def export(self, spans: Sequence[ReadableSpan]) -> SpanExportResult:
self._ensure_authenticated()
return super().export(spans)

View file

@ -8,13 +8,16 @@ identity unconditionally.
"""
from collections.abc import Callable, Iterator
from contextlib import contextmanager
from contextlib import AbstractContextManager, contextmanager
from functools import cache
from typing import Any, Final
from typing import TYPE_CHECKING, Final
if TYPE_CHECKING:
from opentelemetry.trace import Span
@cache
def _otel_runtime() -> "tuple[Callable[[str], Any], Callable[..., None]] | None":
def _otel_runtime() -> "tuple[Callable[[str], AbstractContextManager[Span | None]], Callable[..., None]] | None":
"""Resolve the SDK-backed hooks once and cache the outcome, absence included.
CPython never caches a failed import, so without this memoization every call
@ -29,7 +32,7 @@ def _otel_runtime() -> "tuple[Callable[[str], Any], Callable[..., None]] | None"
@contextmanager
def phase_span(name: str) -> "Iterator[Any]":
def phase_span(name: str) -> "Iterator[Span | None]":
"""Run a request phase inside a live active span so its DB/service calls nest.
Yields ``None`` (a plain no-op) when the OTel SDK is unavailable or V2 is not
@ -43,7 +46,7 @@ def phase_span(name: str) -> "Iterator[Any]":
yield span
def seed_request_identity(user_api_key_dict: Any, model: Any = None) -> None:
def seed_request_identity(user_api_key_dict: object, model: object = None) -> None:
"""Seed request-identity Baggage at the auth boundary (no-op without V2)."""
runtime: Final = _otel_runtime()
if runtime is None:

View file

@ -3,7 +3,13 @@ from __future__ import annotations
import time
from collections import OrderedDict
from threading import RLock
from typing import Any, Final
from typing import Final, Protocol
class _RemovableMetric(Protocol):
"""The one prometheus-client metric method this tracker calls."""
def remove(self, *labelvalues: object) -> None: ...
class BoundedPrometheusSeriesTracker:
@ -21,7 +27,7 @@ class BoundedPrometheusSeriesTracker:
def track_series(
self,
metric: Any,
metric: _RemovableMetric,
metric_name: str,
label_values: tuple[str | None, ...],
max_series: int | None,
@ -60,7 +66,7 @@ class BoundedPrometheusSeriesTracker:
break
del series[tracked_label_values]
def remove_series(self, metric: object, label_values: tuple[str | None, ...]) -> bool:
def remove_series(self, metric: _RemovableMetric, label_values: tuple[str | None, ...]) -> bool:
"""Drop one child series, True when it is gone (removed or never existed)."""
return self._remove_metric_child(metric, label_values)
@ -82,7 +88,7 @@ class BoundedPrometheusSeriesTracker:
def _remove_metric_series(
self,
metric: Any,
metric: _RemovableMetric,
series: OrderedDict[tuple[str | None, ...], float],
label_values: tuple[str | None, ...],
) -> None:
@ -90,7 +96,7 @@ class BoundedPrometheusSeriesTracker:
series.pop(label_values, None)
@staticmethod
def _remove_metric_child(metric: Any, label_values: tuple[str | None, ...]) -> bool:
def _remove_metric_child(metric: _RemovableMetric, label_values: tuple[str | None, ...]) -> bool:
"""
Remove the Prometheus child for ``label_values`` and report whether the
tracker should commit the matching state change.

View file

@ -406,7 +406,7 @@ class VectorStorePreCallHook(CustomLogger):
request_data: dict,
response_chunk: Any,
call_type: CallTypes | None,
) -> Any | None:
) -> object | None:
"""
Add search results to the final streaming chunk.

View file

@ -4,6 +4,7 @@ imported_openAIResponse = True
try:
import io
import logging
from collections.abc import Mapping
from typing import Any, Literal, Protocol, TypeVar
from wandb.sdk.data_types import trace_tree
@ -43,7 +44,7 @@ try:
@staticmethod
def results_to_trace_tree(
request: dict[str, Any],
request: Mapping[str, object],
response: OpenAIResponse,
results: list[trace_tree.Result],
time_elapsed: float,
@ -73,7 +74,7 @@ try:
def _resolve_edit(
self,
request: dict[str, Any],
request: Mapping[str, object],
response: OpenAIResponse,
time_elapsed: float,
) -> trace_tree.WBTraceTree:
@ -91,7 +92,7 @@ try:
def _resolve_completion(
self,
request: dict[str, Any],
request: Mapping[str, object],
response: OpenAIResponse,
time_elapsed: float,
) -> trace_tree.WBTraceTree:
@ -134,7 +135,7 @@ try:
def _request_response_result_to_trace(
self,
request: dict[str, Any],
request: Mapping[str, object],
response: OpenAIResponse,
request_str: str,
choices: list[str],

View file

@ -50,7 +50,7 @@ _INTERACTIONS_MODALITY_FIELDS: Final[Mapping[str, str]] = MappingProxyType(
)
def _modality_field(entry: Mapping[str, Any]) -> str | None:
def _modality_field(entry: Mapping[str, object]) -> str | None:
return _INTERACTIONS_MODALITY_FIELDS.get(str(entry.get("modality", "")).lower())
@ -58,7 +58,7 @@ def _token_count(value: object) -> int:
return value if isinstance(value, int) else 0
def _modality_token_sums(entries: Sequence[Mapping[str, Any]]) -> Mapping[str, int]:
def _modality_token_sums(entries: Sequence[Mapping[str, object]]) -> Mapping[str, int]:
fields: Final = frozenset(field for entry in entries if (field := _modality_field(entry)) is not None)
return MappingProxyType(
{
@ -68,7 +68,7 @@ def _modality_token_sums(entries: Sequence[Mapping[str, Any]]) -> Mapping[str, i
)
def _google_search_query_count(usage_object: Mapping[str, Any]) -> int:
def _google_search_query_count(usage_object: Mapping[str, object]) -> int:
entries: Final = usage_object.get("grounding_tool_count")
if not isinstance(entries, Sequence):
return 0

View file

@ -85,7 +85,7 @@ def safe_json_structure(
def safe_dumps(
data: Any,
data: object,
max_depth: int = DEFAULT_MAX_RECURSE_DEPTH,
value_transform: Callable[[str | None, str], str] | None = None,
) -> str:

View file

@ -35,7 +35,7 @@ if TYPE_CHECKING:
from litellm.router import Router
# Anthropic-only keys already mapped by the translator; strip on extra_kwargs re-merge.
ANTHROPIC_ONLY_REQUEST_KEYS: Final[frozenset[str]] = frozenset({"output_config"})
ANTHROPIC_ONLY_REQUEST_KEYS: Final[frozenset[str]] = frozenset({"output_config", "safeguards"})
_AnthropicMessages: TypeAlias = "list[dict[str, object]]"
_AnthropicSystem: TypeAlias = "str | list[dict[str, object]] | None"

View file

@ -79,10 +79,14 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
"speed",
"output_config",
"reasoning_effort",
"safeguards",
# TODO: Add Anthropic `metadata` support
# "metadata",
]
def should_filter_anthropic_beta_headers(self) -> bool:
return self._resolved_provider != "anthropic"
def _remove_scope_from_cache_control(self, anthropic_messages_request: dict) -> None:
"""
Remove `scope` field from cache_control blocks.

View file

@ -1,5 +1,5 @@
from collections.abc import Coroutine
from typing import Any, Final, cast
from typing import Final, cast
import httpx
from openai import AsyncAzureOpenAI, AsyncOpenAI, AzureOpenAI, OpenAI
@ -19,7 +19,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
"""
@staticmethod
def _ensure_training_type(create_fine_tuning_job_data: dict[str, Any]) -> None:
def _ensure_training_type(create_fine_tuning_job_data: dict[str, object]) -> None:
"""
Azure requires trainingType in extra_body. Default to 1 (supervised) if omitted.
"""
@ -66,7 +66,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
max_retries: int | None,
organization: str | None,
client: OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None = None,
) -> LiteLLMFineTuningJob | Coroutine[Any, Any, LiteLLMFineTuningJob]:
) -> LiteLLMFineTuningJob | Coroutine[object, object, LiteLLMFineTuningJob]:
self._ensure_training_type(create_fine_tuning_job_data)
openai_client: Final[OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None] = self.get_openai_client(
@ -109,7 +109,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
max_retries: int | None,
organization: str | None,
client: OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None = None,
) -> LiteLLMFineTuningJob | Coroutine[Any, Any, LiteLLMFineTuningJob]:
) -> LiteLLMFineTuningJob | Coroutine[object, object, LiteLLMFineTuningJob]:
openai_client: Final[OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None] = self.get_openai_client(
api_key=api_key,
api_base=api_base,
@ -149,7 +149,7 @@ class AzureOpenAIFineTuningAPI(OpenAIFineTuningAPI, BaseAzureLLM):
max_retries: int | None,
organization: str | None,
client: OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None = None,
) -> LiteLLMFineTuningJob | Coroutine[Any, Any, LiteLLMFineTuningJob]:
) -> LiteLLMFineTuningJob | Coroutine[object, object, LiteLLMFineTuningJob]:
openai_client: Final[OpenAI | AsyncOpenAI | AzureOpenAI | AsyncAzureOpenAI | None] = self.get_openai_client(
api_key=api_key,
api_base=api_base,

View file

@ -29,7 +29,7 @@ class CodestralTextCompletionConfig(OpenAITextCompletionConfig):
random_seed: int | None = None,
stop: str | None = None,
) -> None:
locals_: Final = locals().copy()
locals_: Final[dict[str, object]] = locals().copy()
for key, value in locals_.items():
if key != "self" and value is not None:
setattr(self.__class__, key, value)

View file

@ -12,7 +12,7 @@ Authentication priority:
import os
import re
from typing import Any, Final, Literal
from typing import Final, Literal
from urllib.parse import urlsplit, urlunsplit
from litellm.llms.base_llm.chat.transformation import BaseLLMException
@ -48,7 +48,7 @@ class DatabricksBase:
]
@classmethod
def redact_sensitive_data(cls, data: Any) -> Any:
def redact_sensitive_data(cls, data: object) -> object:
"""
Redact sensitive information (tokens, secrets) from data before logging.

View file

@ -453,7 +453,7 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
return normalized
@staticmethod
def _finalize_gemini_live_setup(model: str, setup: dict[str, Any]) -> dict[str, Any]:
def _finalize_gemini_live_setup(model: str, setup: dict[str, object]) -> dict[str, object]:
generation_config: Final = setup.get("generationConfig")
if isinstance(generation_config, dict):
modalities: Final = generation_config.get("responseModalities")
@ -1172,7 +1172,7 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
def map_openai_event(
self,
key: str,
value: Any,
value: object,
current_delta_type: ALL_DELTA_TYPES | None,
) -> OpenAIRealtimeEventTypes | ResponsesAPIStreamEvents:
if isinstance(value, dict):

View file

@ -31,7 +31,7 @@ class JinaAIEmbeddingConfig(BaseEmbeddingConfig):
def __init__(
self,
) -> None:
locals_: Final = locals().copy()
locals_: Final[dict[str, object]] = locals().copy()
for key, value in locals_.items():
if key != "self" and value is not None:
setattr(self.__class__, key, value)

View file

@ -170,7 +170,9 @@ class OpenrouterEmbeddingConfig(BaseEmbeddingConfig):
optional_params[param] = value
return optional_params
def get_error_class(self, error_message: str, status_code: int, headers: Any) -> Any:
def get_error_class(
self, error_message: str, status_code: int, headers: dict[str, str] | httpx.Headers
) -> OpenRouterException:
"""
Get the error class for OpenRouter errors.
"""

View file

@ -3,6 +3,8 @@ import binascii
from collections import defaultdict
from typing import TYPE_CHECKING, Any, Final, NoReturn
import httpx
from litellm.constants import request_timeout
REDUCTO_API_BASE: Final = "https://platform.reducto.ai"
@ -62,7 +64,7 @@ def extract_file_id_or_bytes(
return None, raw_bytes, mime
def _extract_file_id_from_upload_response(response: Any) -> str:
def _extract_file_id_from_upload_response(response: httpx.Response) -> str:
try:
payload: Final = response.json()
except ValueError as exc:

View file

@ -11,6 +11,7 @@ from typing import TYPE_CHECKING, Any, Final
import httpx
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import AllEmbeddingInputValues
@ -160,7 +161,7 @@ class VercelAIGatewayEmbeddingConfig(BaseEmbeddingConfig):
optional_params[param] = value
return optional_params
def get_error_class(self, error_message: str, status_code: int, headers: Any) -> Any:
def get_error_class(self, error_message: str, status_code: int, headers: Any) -> BaseLLMException:
"""
Get the error class for Vercel AI Gateway errors.
"""

View file

@ -205,7 +205,7 @@ class VertexAgentEngineConfig(BaseConfig, VertexBase):
session_id: Final = self._get_session_id(optional_params)
# Build the input
input_data: Final[dict[str, Any]] = {
input_data: Final[dict[str, str]] = {
"message": prompt,
"user_id": user_id,
}

View file

@ -1327,7 +1327,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 2048,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"anthropic.claude-mythos-preview": {
"input_cost_per_token": 0,
@ -1381,7 +1381,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 2048,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-opus-4-7": {
"bedrock_converse_supports_strict_tools": false,
@ -1419,7 +1419,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 2048,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-opus-4-7": {
"bedrock_converse_supports_strict_tools": false,
@ -1531,7 +1531,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"anthropic.claude-fable-5-1": {
"cache_creation_input_token_cost": 1.25e-05,
@ -1570,7 +1570,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-fable-5": {
"cache_creation_input_token_cost": 1.25e-05,
@ -1608,7 +1608,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-fable-5-1": {
"cache_creation_input_token_cost": 1.25e-05,
@ -1647,7 +1647,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-fable-5": {
"cache_creation_input_token_cost": 1.375e-05,
@ -1685,7 +1685,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-fable-5-1": {
"cache_creation_input_token_cost": 1.375e-05,
@ -1724,7 +1724,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-fable-5": {
"cache_creation_input_token_cost": 1.375e-05,
@ -1837,7 +1837,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-opus-5": {
"bedrock_converse_supports_strict_tools": false,
@ -1875,7 +1875,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-opus-5": {
"bedrock_converse_supports_strict_tools": false,
@ -1913,7 +1913,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-opus-5": {
"bedrock_converse_supports_strict_tools": false,
@ -2063,7 +2063,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-opus-4-8": {
"bedrock_converse_supports_strict_tools": false,
@ -2102,7 +2102,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-opus-4-8": {
"bedrock_converse_supports_strict_tools": false,
@ -2141,7 +2141,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-opus-4-8": {
"bedrock_converse_supports_strict_tools": false,
@ -2329,7 +2329,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-sonnet-5": {
"bedrock_converse_supports_strict_tools": false,
@ -2368,7 +2368,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-sonnet-5": {
"bedrock_converse_supports_strict_tools": false,
@ -2407,7 +2407,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-sonnet-5": {
"bedrock_converse_supports_strict_tools": false,
@ -2556,7 +2556,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
@ -2591,7 +2591,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
@ -2626,7 +2626,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
@ -3740,6 +3740,21 @@
"supports_vision": true,
"supports_web_search": true
},
"azure_ai/gpt-image-2": {
"cache_read_input_image_token_cost": 2e-06,
"cache_read_input_token_cost": 1.25e-06,
"input_cost_per_image_token": 8e-06,
"input_cost_per_token": 5e-06,
"litellm_provider": "azure_ai",
"mode": "image_generation",
"output_cost_per_image_token": 3e-05,
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
"supported_endpoints": [
"/v1/images/generations",
"/v1/images/edits"
],
"supports_vision": true
},
"azure_ai/codex-mini": {
"cache_read_input_token_cost": 3.75e-07,
"deprecation_date": "2026-11-15",
@ -11159,6 +11174,20 @@
],
"deprecation_date": "2026-10-01"
},
"azure_ai/MAI-Image-2.5-Pro": {
"deprecation_date": "2026-10-01",
"input_cost_per_image_token": 8e-06,
"input_cost_per_token": 5e-06,
"litellm_provider": "azure_ai",
"mode": "image_generation",
"output_cost_per_image": 0.1085,
"output_cost_per_image_token": 0.000106,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-mai-image-2-5-pro-and-mai-voice-2-flash-in-microsoft-foundry/4539446",
"supported_endpoints": [
"/v1/images/generations",
"/v1/images/edits"
]
},
"azure_ai/MAI-Image-2e": {
"deprecation_date": "2026-08-15",
"input_cost_per_token": 5e-06,
@ -21888,6 +21917,7 @@
"supports_tool_choice": true
},
"deepseek/deepseek-coder": {
"cache_read_input_token_cost": 1.4e-08,
"input_cost_per_token": 1.4e-07,
"input_cost_per_token_cache_hit": 1.4e-08,
"litellm_provider": "deepseek",
@ -21902,6 +21932,7 @@
"supports_tool_choice": true
},
"deepseek/deepseek-r1": {
"cache_read_input_token_cost": 1.4e-07,
"input_cost_per_token": 5.5e-07,
"input_cost_per_token_cache_hit": 1.4e-07,
"litellm_provider": "deepseek",
@ -21957,6 +21988,7 @@
"supports_tool_choice": true
},
"deepseek/deepseek-v3.2": {
"cache_read_input_token_cost": 2.8e-08,
"input_cost_per_token": 2.8e-07,
"input_cost_per_token_cache_hit": 2.8e-08,
"litellm_provider": "deepseek",
@ -23796,6 +23828,25 @@
"supports_tool_choice": true,
"supports_vision": false
},
"fireworks_ai/deepseek-v4-pro-0813": {
"cache_read_input_token_cost": 4.4e-08,
"cache_read_input_token_cost_priority": 5.5e-08,
"input_cost_per_token": 1.32e-06,
"input_cost_per_token_priority": 1.65e-06,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 3.96e-06,
"output_cost_per_token_priority": 4.95e-06,
"source": "https://api.fireworks.ai/v1/serverless/models",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
},
"fireworks_ai/accounts/fireworks/models/firefunction-v2": {
"input_cost_per_token": 9e-07,
"litellm_provider": "fireworks_ai",
@ -24182,7 +24233,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
"supports_vision": false
},
"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf": {
"input_cost_per_token": 1.2e-06,
@ -24508,7 +24559,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
"supports_vision": false
},
"fireworks_ai/qwen3p7-plus": {
"cache_read_input_token_cost": 8e-08,
@ -35244,6 +35295,7 @@
"mode": "chat",
"output_cost_per_token": 3e-06,
"source": "https://console.groq.com/docs/model/qwen/qwen3.6-27b",
"deprecation_date": "2026-09-14",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": false,
@ -41472,6 +41524,7 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v3.2-exp": {
"cache_read_input_token_cost": 2e-08,
"deprecation_date": "2026-09-28",
"input_cost_per_token": 2.7e-07,
"input_cost_per_token_cache_hit": 2e-08,
@ -41494,6 +41547,7 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-r1": {
"cache_read_input_token_cost": 1.4e-07,
"input_cost_per_token": 7e-07,
"input_cost_per_token_cache_hit": 1.4e-07,
"litellm_provider": "openrouter",
@ -41537,21 +41591,21 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v4-pro": {
"input_cost_per_token": 9.5526e-07,
"input_cost_per_token": 9.42906e-07,
"input_cost_per_token_cache_hit": 4.4e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
"max_tokens": 384000,
"mode": "chat",
"output_cost_per_token": 1.91052e-06,
"output_cost_per_token": 1.885812e-06,
"source": "https://openrouter.ai/api/v1/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"cache_read_input_token_cost": 7.9605e-08,
"cache_read_input_token_cost": 7.85755e-08,
"supports_audio_input": false,
"supports_pdf_input": false,
"supports_vision": false,
@ -41579,22 +41633,22 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v4-pro-0813": {
"input_cost_per_token": 1.32e-06,
"input_cost_per_token_cache_hit": 4.4e-08,
"input_cost_per_token": 5.7684e-07,
"input_cost_per_token_cache_hit": 1.9272e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
"max_tokens": 384000,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 3.96e-06,
"output_cost_per_token": 1.73052e-06,
"source": "https://openrouter.ai/api/v1/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"cache_read_input_token_cost": 4.4e-08,
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},
"cache_read_input_token_cost": 1.8354e-08,
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":5.7684e-7,"output_cost_per_token":0.00000173052,"cache_read_input_token_cost":1.8354e-8},
"supports_audio_input": false,
"supports_pdf_input": false,
"supports_vision": false,
@ -44247,6 +44301,84 @@
"supports_system_messages": true,
"supports_native_structured_output": true
},
"bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.45e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/ap-south-1/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.41e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.545e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.236e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/eu-west-1/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.41e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/eu-west-2/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 2.3e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.86e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/sa-east-1/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.45e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"qwen.qwen3-vl-235b-a22b": {
"input_cost_per_token": 5.3e-07,
"litellm_provider": "bedrock_converse",
@ -46291,8 +46423,8 @@
"together_ai/zai-org/GLM-4.6": {
"input_cost_per_token": 6e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 200000,
"max_tokens": 200000,
"max_input_tokens": 202752,
"max_tokens": 202752,
"metadata": {
"successor": "together_ai/zai-org/GLM-5.2"
},
@ -46308,8 +46440,8 @@
"deprecation_date": "2026-04-02",
"input_cost_per_token": 4.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 200000,
"max_tokens": 200000,
"max_input_tokens": 202752,
"max_tokens": 202752,
"metadata": {
"successor": "together_ai/zai-org/GLM-5.2"
},
@ -47158,7 +47290,7 @@
"mode": "chat",
"output_cost_per_token": 1.2e-05,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
@ -47192,7 +47324,7 @@
"mode": "chat",
"output_cost_per_token": 3e-05,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
@ -47225,7 +47357,7 @@
"mode": "chat",
"output_cost_per_token": 3e-05,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
@ -47276,7 +47408,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us-gov.nvidia.nemotron-nano-3-30b": {
"input_cost_per_token": 7.2e-08,
@ -64215,6 +64347,25 @@
"supports_tool_choice": true,
"supports_vision": false
},
"fireworks_ai/glm-5p3": {
"cache_read_input_token_cost": 2.6e-07,
"cache_read_input_token_cost_priority": 3.25e-07,
"input_cost_per_token": 1.4e-06,
"input_cost_per_token_priority": 1.75e-06,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 4.4e-06,
"output_cost_per_token_priority": 5.5e-06,
"source": "https://api.fireworks.ai/v1/serverless/models",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
},
"fireworks_ai/accounts/fireworks/routers/glm-5p3-fast": {
"cache_read_input_token_cost": 3.9e-07,
"input_cost_per_token": 2.1e-06,
@ -64262,6 +64413,23 @@
"supports_tool_choice": true,
"supports_vision": true
},
"fireworks_ai/glm-5p3-flash": {
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_priority": 3.75e-08,
"input_cost_per_token": 1.5e-07,
"input_cost_per_token_priority": 1.875e-07,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 5e-07,
"output_cost_per_token_priority": 6.25e-07,
"source": "https://api.fireworks.ai/v1/serverless/models",
"supports_function_calling": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
},
"fireworks_ai/accounts/fireworks/models/inkling": {
"cache_read_input_token_cost": 1.7e-07,
"input_cost_per_token": 1e-06,
@ -67259,9 +67427,9 @@
"supports_web_search": true
},
"openrouter/deepseek/deepseek-v4-flash": {
"input_cost_per_token": 8.8606e-08,
"output_cost_per_token": 1.77212e-07,
"cache_read_input_token_cost": 1.77212e-08,
"input_cost_per_token": 5.544e-08,
"output_cost_per_token": 1.1088e-07,
"cache_read_input_token_cost": 1.1088e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
@ -69010,12 +69178,12 @@
"supports_reasoning": false
},
"openrouter/meta-llama/llama-3.1-70b-instruct": {
"input_cost_per_token": 7.2e-07,
"output_cost_per_token": 7.2e-07,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 131072,
"max_output_tokens": 8192,
"max_tokens": 8192,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
@ -71317,15 +71485,15 @@
"supports_web_search": false
},
"openrouter/~deepseek/deepseek-pro-latest": {
"cache_read_input_token_cost": 4.4e-08,
"input_cost_per_token": 1.32e-06,
"cache_read_input_token_cost": 1.8354e-08,
"input_cost_per_token": 5.7684e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
"max_tokens": 384000,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},
"output_cost_per_token": 3.96e-06,
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":5.7684e-7,"output_cost_per_token":0.00000173052,"cache_read_input_token_cost":1.8354e-8},
"output_cost_per_token": 1.73052e-06,
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
"supports_function_calling": true,

View file

@ -1154,7 +1154,7 @@ class MCPRequestHandler:
Failures surface with the status the standard pipeline would give them, mirroring
``UserAPIKeyAuthExceptionHandler``: a disallowed route is the route gate's own 403, an
over-budget identity is a 429, a sub-check that raised its own ``HTTPException``/
over-budget identity is a 422, a sub-check that raised its own ``HTTPException``/
``ProxyException`` keeps that status, a transient database outage is a retryable 503, and
only a genuinely unresolvable failure (a blocked team/project raises a bare ``Exception``,
same as the standard pipeline's fallback) becomes the fail-closed 401. Collapsing every

View file

@ -14,6 +14,7 @@ fetcher dispatches by ``discovery_mode``:
pure-A2A fallback strategy returns 404 for these deployments.
"""
from collections.abc import Mapping
from enum import Enum
from typing import Any, Final
from urllib.parse import urlencode
@ -55,7 +56,7 @@ def _normalize_base_url(base_url: str) -> str:
def _build_langgraph_platform_paths(
params: dict[str, Any] | None,
params: Mapping[str, object] | None,
) -> tuple[str, ...]:
"""Build the paths to try for LangGraph Platform discovery.
@ -71,7 +72,7 @@ def _build_langgraph_platform_paths(
return tuple(f"{path}?{query}" for path in AGENT_CARD_WELL_KNOWN_PATHS)
def _paths_for_mode(mode: DiscoveryMode, params: dict[str, Any] | None) -> tuple[str, ...]:
def _paths_for_mode(mode: DiscoveryMode, params: Mapping[str, object] | None) -> tuple[str, ...]:
if mode == DiscoveryMode.WELL_KNOWN_FALLBACK:
return AGENT_CARD_WELL_KNOWN_PATHS
if mode == DiscoveryMode.LANGGRAPH_PLATFORM:
@ -83,7 +84,7 @@ async def fetch_well_known_card(
base_url: str,
*,
discovery_mode: DiscoveryMode = DiscoveryMode.WELL_KNOWN_FALLBACK,
params: dict[str, Any] | None = None,
params: Mapping[str, object] | None = None,
timeout: float = DEFAULT_DISCOVERY_TIMEOUT_SECONDS,
headers: dict[str, str] | None = None,
) -> dict[str, Any]:

View file

@ -25,8 +25,9 @@ Config example::
import asyncio
import base64
import hashlib
from collections.abc import Mapping
from dataclasses import dataclass
from typing import Any, Final
from typing import Final
import httpx
@ -43,7 +44,7 @@ _TOKEN_EXPIRY_BUFFER_SECONDS: Final = 60
_DEFAULT_TTL_SECONDS: Final = 3600
def _resolve_secret(value: Any) -> str | None:
def _resolve_secret(value: object) -> str | None:
"""Resolve a config value, expanding ``os.environ/`` references."""
if not isinstance(value, str):
return None
@ -75,7 +76,7 @@ class DatabricksAppOAuthConfig:
def parse_databricks_oauth_config(
litellm_params: dict[str, Any] | None,
litellm_params: Mapping[str, object] | None,
) -> DatabricksAppOAuthConfig | None:
"""Build a Databricks App OAuth config from an agent's ``litellm_params``.
@ -191,7 +192,7 @@ class DatabricksAppOAuthTokenCache(InMemoryCache):
except httpx.HTTPError as exc:
raise ValueError(f"Databricks App OAuth token request failed: {exc}") from exc
body: Final = response.json()
body: Final[object] = response.json()
if not isinstance(body, dict):
raise ValueError(
f"Databricks App OAuth token response returned non-object JSON (got {type(body).__name__})"
@ -215,7 +216,7 @@ databricks_app_oauth_token_cache: Final = DatabricksAppOAuthTokenCache()
async def resolve_databricks_app_auth_header(
litellm_params: dict[str, Any] | None,
litellm_params: Mapping[str, object] | None,
) -> dict[str, str] | None:
"""Return ``{"Authorization": "Bearer <token>"}`` for a Databricks App agent.

View file

@ -9,7 +9,15 @@ from litellm._version import version as litellm_version
from litellm.proxy.client.health import HealthManagementClient
from .commands.agents import agent_commands
from .commands.auth import auth_group, context_secret_vault, get_stored_api_key, login, logout, whoami
from .commands.auth import (
CliContextObj,
auth_group,
context_secret_vault,
get_stored_api_key,
login,
logout,
whoami,
)
from .commands.autoroute.commands import autoroute_group
from .commands.chat import chat
from .commands.config import config_commands, get_config_value, hidden_command_names
@ -126,7 +134,8 @@ def cli(ctx: click.Context, show_version: bool, base_url: str | None, api_key: s
@click.pass_context
def version(ctx: click.Context):
"""Show the LiteLLM Proxy CLI and server version."""
print_version(ctx.obj.get("base_url"), ctx.obj.get("api_key"))
ctx_obj: Final[CliContextObj] = ctx.obj
print_version(ctx_obj.get("base_url"), ctx_obj.get("api_key"))
# Add authentication commands as top-level commands

View file

@ -8,7 +8,7 @@ if TYPE_CHECKING:
from litellm.types.guardrails import Guardrail, LitellmParams
def _get_config_value(litellm_params: Any, optional_params: Any, attribute_name: str) -> Any | None:
def _get_config_value(litellm_params: "LitellmParams", optional_params: object, attribute_name: str) -> Any | None:
if optional_params is not None:
value: Final = (
optional_params.get(attribute_name)

View file

@ -6,7 +6,7 @@
# +-------------------------------------------------------------+
import os
import uuid
from typing import TYPE_CHECKING, Any, Final, Literal, Optional
from typing import TYPE_CHECKING, Final, Literal, Optional
import httpx
from fastapi import HTTPException
@ -63,7 +63,7 @@ class OnyxGuardrail(CustomGuardrail):
async def _validate_with_guard_server(
self,
payload: Any,
payload: object,
input_type: Literal["request", "response"],
conversation_id: str,
) -> dict:

View file

@ -40,7 +40,7 @@ _UNMANAGED_RESPONSE_ID_DETAIL: Final = (
_PROXY_ADMIN_ROLES: Final = frozenset({LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN.value})
def _proxy_general_settings() -> Mapping[str, Any]:
def _proxy_general_settings() -> Mapping[str, object]:
from litellm.proxy.proxy_server import general_settings
return general_settings
@ -107,7 +107,7 @@ def _is_responses_api_create_route(request_route: str | None) -> bool:
class ResponsesIDSecurity(CustomLogger):
def __init__(
self,
general_settings_reader: Callable[[], Mapping[str, Any]] = _proxy_general_settings,
general_settings_reader: Callable[[], Mapping[str, object]] = _proxy_general_settings,
signing_key_reader: Callable[[], str | None] = _proxy_signing_key,
) -> None:
self._general_settings_reader: Final = general_settings_reader
@ -307,7 +307,7 @@ class ResponsesIDSecurity(CustomLogger):
data: dict,
user_api_key_dict: "UserAPIKeyAuth",
response: LLMResponseTypes,
) -> Any:
) -> LLMResponseTypes:
"""
Queue response IDs for batch processing instead of writing directly to DB.

View file

@ -15,6 +15,7 @@ self-describing `StandardLoggingPayload`, so completions/responses can use it to
"""
import uuid
from collections.abc import Mapping
from datetime import datetime, timezone
from typing import Any, Final
@ -48,7 +49,7 @@ class CallbackLogsReplayer:
"""
@staticmethod
def _epoch_to_datetime(value: Any) -> datetime:
def _epoch_to_datetime(value: object) -> datetime:
"""`StandardLoggingPayload` stores startTime/endTime as float epoch seconds."""
if isinstance(value, (int, float)):
return datetime.fromtimestamp(float(value), tz=timezone.utc)
@ -114,7 +115,7 @@ class CallbackLogsReplayer:
return logging_obj
@staticmethod
def _response_obj_from_payload(payload: dict[str, Any]) -> dict[str, Any]:
def _response_obj_from_payload(payload: Mapping[str, object]) -> dict[str, object]:
"""Minimal response object so usage-derived spend-log fields resolve."""
return {
"id": payload.get("id"),

View file

@ -1,7 +1,7 @@
"""`/management/v1/spend_logs` facets."""
from datetime import datetime, timezone
from typing import Annotated, Any, Final, Literal
from typing import Annotated, Final, Literal
from fastapi import APIRouter, Depends, Query, Request
@ -39,7 +39,7 @@ async def _spend_log_scope_clause(
user_api_key_dict: UserAPIKeyAuth,
prisma_client: PrismaClient,
next_param_index: int,
) -> tuple[str | None, tuple[Any, ...]]:
) -> tuple[str | None, tuple[str | list[str], ...]]:
"""SQL predicate restricting the facet to spend logs this caller may read.
Returns ``(None, ())`` for a proxy admin. Mirrors the scoping ``/spend/logs/ui``
@ -101,8 +101,8 @@ async def _list_spend_log_facet(
)
column_sql: Final = "end_user" if column == "end_user" else '"user"'
window_params: Final[tuple[Any, ...]] = (_as_utc(start_time), _as_utc(end_time))
search_params: Final[tuple[Any, ...]] = (f"%{escape_like(q)}%",) if q else ()
window_params: Final[tuple[datetime, datetime]] = (_as_utc(start_time), _as_utc(end_time))
search_params: Final[tuple[str, ...]] = (f"%{escape_like(q)}%",) if q else ()
search_clause: Final = (f"{column_sql} ILIKE ${len(window_params) + 1} ESCAPE '\\'",) if q else ()
scope_clause, scope_params = await _spend_log_scope_clause(

View file

@ -8,7 +8,7 @@ from collections.abc import Mapping, Sequence
from collections.abc import Set as AbstractSet
from dataclasses import dataclass
from types import MappingProxyType
from typing import TYPE_CHECKING, Any, Final, Optional
from typing import TYPE_CHECKING, Final, Optional
from fastapi import HTTPException, status
from pydantic import TypeAdapter
@ -230,7 +230,7 @@ def _dedupe_preserving_order(values: list[str]) -> list[str]:
return result
def _mcp_server_identifier_matches(server: Any, identifier: str) -> bool:
def _mcp_server_identifier_matches(server: object, identifier: str) -> bool:
return identifier in {
getattr(server, "server_id", None),
getattr(server, "alias", None),

View file

@ -147,7 +147,7 @@ class GeminiPassthroughLoggingHandler:
- Creates standard logging object
- Logs in litellm callbacks
"""
kwargs: dict[str, Any] = {}
kwargs: dict[str, object] = {}
model = model or GeminiPassthroughLoggingHandler.extract_model_from_url(url_route)
complete_streaming_response: Final = GeminiPassthroughLoggingHandler._build_complete_streaming_response(
all_chunks=all_chunks,

View file

@ -199,9 +199,13 @@ def _build_endpoints(raw: _ProvidersFile) -> list[_EndpointEntry]:
return result
_PROVIDERS_FILE_ADAPTER: Final = TypeAdapter(_ProvidersFile)
_PROVIDER_CREATE_FIELDS_ADAPTER: Final = TypeAdapter(list[ProviderCreateInfo])
def _load_endpoints() -> list[_EndpointEntry]:
raw: Final[_ProvidersFile] = json.loads(
files("litellm").joinpath("provider_endpoints_support_backup.json").read_text(encoding="utf-8")
raw: Final = _PROVIDERS_FILE_ADAPTER.validate_python(
json.loads(files("litellm").joinpath("provider_endpoints_support_backup.json").read_text(encoding="utf-8"))
)
return _build_endpoints(raw)
@ -398,7 +402,7 @@ async def get_provider_fields() -> list[ProviderCreateInfo]:
)
with open(provider_create_fields_path, "r") as f:
provider_create_fields: Final = json.load(f)
provider_create_fields: Final = _PROVIDER_CREATE_FIELDS_ADAPTER.validate_python(json.load(f))
return provider_create_fields

View file

@ -454,6 +454,7 @@ class LiteLLMCompletionResponsesConfig:
"stream": stream,
"metadata": kwargs.get("metadata"),
"service_tier": kwargs.get("service_tier"),
"safety_identifier": responses_api_request.get("safety_identifier"),
"web_search_options": web_search_options,
"response_format": response_format,
"reasoning_effort": reasoning.effort,

View file

@ -411,6 +411,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
output_config: AnthropicOutputConfig | None # Configuration for Claude's output behavior
cache_control: dict[str, Any] | None # Automatic prompt caching
reasoning_effort: str | None
safeguards: ReadOnly[list[dict[str, object]] | None]
class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False):
@ -530,6 +531,7 @@ class AnthropicStopDetails(TypedDict, total=False):
class MessageDelta(TypedDict, total=False):
stop_reason: str | None
stop_details: ReadOnly[AnthropicStopDetails]
safeguard_results: ReadOnly[list[dict[str, object]]]
class ServerToolUsage(TypedDict, total=False):
@ -600,6 +602,7 @@ class MessageChunk(TypedDict, total=False):
stop_reason: str | None
stop_sequence: str | None
usage: UsageDelta
safeguard_results: ReadOnly[list[dict[str, object]]]
class MessageStartBlock(TypedDict):

View file

@ -97,3 +97,4 @@ class AnthropicMessagesResponse(TypedDict, total=False):
type: Literal["message"] | None
usage: AnthropicUsage | None
context_management: NotRequired[ContextManagementResponse]
safeguard_results: NotRequired[ReadOnly[list[dict[str, object]]]]

View file

@ -255,6 +255,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
cache_creation_input_token_cost_ultrafast: ReadOnly[float | None] # OpenAI ultrafast service tier pricing
cache_read_input_token_cost: float | None
cache_read_input_audio_token_cost: ReadOnly[float | None]
cache_read_input_image_token_cost: ReadOnly[float | None]
cache_read_input_token_cost_flex: float | None # OpenAI flex service tier pricing
cache_read_input_token_cost_priority: float | None # OpenAI priority service tier pricing
cache_read_input_token_cost_ultrafast: ReadOnly[float | None] # OpenAI ultrafast service tier pricing
@ -3635,6 +3636,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
cache_read_input_token_cost_above_272k_tokens_priority: float | None = None
cache_read_input_token_cost_above_272k_tokens_flex: float | None = None
cache_read_input_audio_token_cost: float | None = None
cache_read_input_image_token_cost: float | None = None
input_cost_per_character_above_128k_tokens: float | None = None
input_cost_per_audio_token: float | None = None
input_cost_per_token_cache_hit: float | None = None

View file

@ -1327,7 +1327,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 2048,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"anthropic.claude-mythos-preview": {
"input_cost_per_token": 0,
@ -1381,7 +1381,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 2048,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-opus-4-7": {
"bedrock_converse_supports_strict_tools": false,
@ -1419,7 +1419,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 2048,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-opus-4-7": {
"bedrock_converse_supports_strict_tools": false,
@ -1531,7 +1531,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"anthropic.claude-fable-5-1": {
"cache_creation_input_token_cost": 1.25e-05,
@ -1570,7 +1570,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-fable-5": {
"cache_creation_input_token_cost": 1.25e-05,
@ -1608,7 +1608,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-fable-5-1": {
"cache_creation_input_token_cost": 1.25e-05,
@ -1647,7 +1647,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-fable-5": {
"cache_creation_input_token_cost": 1.375e-05,
@ -1685,7 +1685,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-fable-5-1": {
"cache_creation_input_token_cost": 1.375e-05,
@ -1724,7 +1724,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-fable-5": {
"cache_creation_input_token_cost": 1.375e-05,
@ -1837,7 +1837,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-opus-5": {
"bedrock_converse_supports_strict_tools": false,
@ -1875,7 +1875,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-opus-5": {
"bedrock_converse_supports_strict_tools": false,
@ -1913,7 +1913,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-opus-5": {
"bedrock_converse_supports_strict_tools": false,
@ -2063,7 +2063,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-opus-4-8": {
"bedrock_converse_supports_strict_tools": false,
@ -2102,7 +2102,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-opus-4-8": {
"bedrock_converse_supports_strict_tools": false,
@ -2141,7 +2141,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-opus-4-8": {
"bedrock_converse_supports_strict_tools": false,
@ -2329,7 +2329,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-sonnet-5": {
"bedrock_converse_supports_strict_tools": false,
@ -2368,7 +2368,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-sonnet-5": {
"bedrock_converse_supports_strict_tools": false,
@ -2407,7 +2407,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-sonnet-5": {
"bedrock_converse_supports_strict_tools": false,
@ -2556,7 +2556,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"global.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
@ -2591,7 +2591,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
@ -2626,7 +2626,7 @@
"supports_output_config": true,
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"eu.anthropic.claude-sonnet-4-6": {
"supports_adaptive_thinking": true,
@ -3740,6 +3740,21 @@
"supports_vision": true,
"supports_web_search": true
},
"azure_ai/gpt-image-2": {
"cache_read_input_image_token_cost": 2e-06,
"cache_read_input_token_cost": 1.25e-06,
"input_cost_per_image_token": 8e-06,
"input_cost_per_token": 5e-06,
"litellm_provider": "azure_ai",
"mode": "image_generation",
"output_cost_per_image_token": 3e-05,
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
"supported_endpoints": [
"/v1/images/generations",
"/v1/images/edits"
],
"supports_vision": true
},
"azure_ai/codex-mini": {
"cache_read_input_token_cost": 3.75e-07,
"deprecation_date": "2026-11-15",
@ -11159,6 +11174,20 @@
],
"deprecation_date": "2026-10-01"
},
"azure_ai/MAI-Image-2.5-Pro": {
"deprecation_date": "2026-10-01",
"input_cost_per_image_token": 8e-06,
"input_cost_per_token": 5e-06,
"litellm_provider": "azure_ai",
"mode": "image_generation",
"output_cost_per_image": 0.1085,
"output_cost_per_image_token": 0.000106,
"source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-mai-image-2-5-pro-and-mai-voice-2-flash-in-microsoft-foundry/4539446",
"supported_endpoints": [
"/v1/images/generations",
"/v1/images/edits"
]
},
"azure_ai/MAI-Image-2e": {
"deprecation_date": "2026-08-15",
"input_cost_per_token": 5e-06,
@ -21888,6 +21917,7 @@
"supports_tool_choice": true
},
"deepseek/deepseek-coder": {
"cache_read_input_token_cost": 1.4e-08,
"input_cost_per_token": 1.4e-07,
"input_cost_per_token_cache_hit": 1.4e-08,
"litellm_provider": "deepseek",
@ -21902,6 +21932,7 @@
"supports_tool_choice": true
},
"deepseek/deepseek-r1": {
"cache_read_input_token_cost": 1.4e-07,
"input_cost_per_token": 5.5e-07,
"input_cost_per_token_cache_hit": 1.4e-07,
"litellm_provider": "deepseek",
@ -21957,6 +21988,7 @@
"supports_tool_choice": true
},
"deepseek/deepseek-v3.2": {
"cache_read_input_token_cost": 2.8e-08,
"input_cost_per_token": 2.8e-07,
"input_cost_per_token_cache_hit": 2.8e-08,
"litellm_provider": "deepseek",
@ -23796,6 +23828,25 @@
"supports_tool_choice": true,
"supports_vision": false
},
"fireworks_ai/deepseek-v4-pro-0813": {
"cache_read_input_token_cost": 4.4e-08,
"cache_read_input_token_cost_priority": 5.5e-08,
"input_cost_per_token": 1.32e-06,
"input_cost_per_token_priority": 1.65e-06,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 131072,
"max_tokens": 131072,
"mode": "chat",
"output_cost_per_token": 3.96e-06,
"output_cost_per_token_priority": 4.95e-06,
"source": "https://api.fireworks.ai/v1/serverless/models",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
},
"fireworks_ai/accounts/fireworks/models/firefunction-v2": {
"input_cost_per_token": 9e-07,
"litellm_provider": "fireworks_ai",
@ -24182,7 +24233,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
"supports_vision": false
},
"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf": {
"input_cost_per_token": 1.2e-06,
@ -24508,7 +24559,7 @@
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
"supports_vision": false
},
"fireworks_ai/qwen3p7-plus": {
"cache_read_input_token_cost": 8e-08,
@ -35244,6 +35295,7 @@
"mode": "chat",
"output_cost_per_token": 3e-06,
"source": "https://console.groq.com/docs/model/qwen/qwen3.6-27b",
"deprecation_date": "2026-09-14",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": false,
@ -41472,6 +41524,7 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v3.2-exp": {
"cache_read_input_token_cost": 2e-08,
"deprecation_date": "2026-09-28",
"input_cost_per_token": 2.7e-07,
"input_cost_per_token_cache_hit": 2e-08,
@ -41494,6 +41547,7 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-r1": {
"cache_read_input_token_cost": 1.4e-07,
"input_cost_per_token": 7e-07,
"input_cost_per_token_cache_hit": 1.4e-07,
"litellm_provider": "openrouter",
@ -41537,21 +41591,21 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v4-pro": {
"input_cost_per_token": 9.5526e-07,
"input_cost_per_token": 9.42906e-07,
"input_cost_per_token_cache_hit": 4.4e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
"max_tokens": 384000,
"mode": "chat",
"output_cost_per_token": 1.91052e-06,
"output_cost_per_token": 1.885812e-06,
"source": "https://openrouter.ai/api/v1/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"cache_read_input_token_cost": 7.9605e-08,
"cache_read_input_token_cost": 7.85755e-08,
"supports_audio_input": false,
"supports_pdf_input": false,
"supports_vision": false,
@ -41579,22 +41633,22 @@
"supports_web_search": false
},
"openrouter/deepseek/deepseek-v4-pro-0813": {
"input_cost_per_token": 1.32e-06,
"input_cost_per_token_cache_hit": 4.4e-08,
"input_cost_per_token": 5.7684e-07,
"input_cost_per_token_cache_hit": 1.9272e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
"max_tokens": 384000,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 3.96e-06,
"output_cost_per_token": 1.73052e-06,
"source": "https://openrouter.ai/api/v1/models",
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"cache_read_input_token_cost": 4.4e-08,
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},
"cache_read_input_token_cost": 1.8354e-08,
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":5.7684e-7,"output_cost_per_token":0.00000173052,"cache_read_input_token_cost":1.8354e-8},
"supports_audio_input": false,
"supports_pdf_input": false,
"supports_vision": false,
@ -44247,6 +44301,84 @@
"supports_system_messages": true,
"supports_native_structured_output": true
},
"bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.45e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/ap-south-1/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.41e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.545e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.236e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/eu-west-1/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.41e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/eu-west-2/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 2.3e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.86e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"bedrock/sa-east-1/qwen.qwen3-next-80b-a3b": {
"input_cost_per_token": 1.8e-07,
"litellm_provider": "bedrock",
"max_input_tokens": 128000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 1.45e-06,
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_function_calling": true,
"supports_native_structured_output": true,
"supports_system_messages": true
},
"qwen.qwen3-vl-235b-a22b": {
"input_cost_per_token": 5.3e-07,
"litellm_provider": "bedrock_converse",
@ -46291,8 +46423,8 @@
"together_ai/zai-org/GLM-4.6": {
"input_cost_per_token": 6e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 200000,
"max_tokens": 200000,
"max_input_tokens": 202752,
"max_tokens": 202752,
"metadata": {
"successor": "together_ai/zai-org/GLM-5.2"
},
@ -46308,8 +46440,8 @@
"deprecation_date": "2026-04-02",
"input_cost_per_token": 4.5e-07,
"litellm_provider": "together_ai",
"max_input_tokens": 200000,
"max_tokens": 200000,
"max_input_tokens": 202752,
"max_tokens": 202752,
"metadata": {
"successor": "together_ai/zai-org/GLM-5.2"
},
@ -47158,7 +47290,7 @@
"mode": "chat",
"output_cost_per_token": 1.2e-05,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
@ -47192,7 +47324,7 @@
"mode": "chat",
"output_cost_per_token": 3e-05,
"prompt_cache_min_tokens": 1024,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
@ -47225,7 +47357,7 @@
"mode": "chat",
"output_cost_per_token": 3e-05,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
"source": "https://aws.amazon.com/bedrock/pricing/",
"supports_adaptive_thinking": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
@ -47276,7 +47408,7 @@
"bedrock_output_config_effort_ceiling": "xhigh",
"supports_parallel_tool_use_config": true,
"prompt_cache_min_tokens": 512,
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
"source": "https://aws.amazon.com/bedrock/pricing/"
},
"us-gov.nvidia.nemotron-nano-3-30b": {
"input_cost_per_token": 7.2e-08,
@ -64215,6 +64347,25 @@
"supports_tool_choice": true,
"supports_vision": false
},
"fireworks_ai/glm-5p3": {
"cache_read_input_token_cost": 2.6e-07,
"cache_read_input_token_cost_priority": 3.25e-07,
"input_cost_per_token": 1.4e-06,
"input_cost_per_token_priority": 1.75e-06,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 1048576,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 4.4e-06,
"output_cost_per_token_priority": 5.5e-06,
"source": "https://api.fireworks.ai/v1/serverless/models",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": false
},
"fireworks_ai/accounts/fireworks/routers/glm-5p3-fast": {
"cache_read_input_token_cost": 3.9e-07,
"input_cost_per_token": 2.1e-06,
@ -64262,6 +64413,23 @@
"supports_tool_choice": true,
"supports_vision": true
},
"fireworks_ai/glm-5p3-flash": {
"cache_read_input_token_cost": 3e-08,
"cache_read_input_token_cost_priority": 3.75e-08,
"input_cost_per_token": 1.5e-07,
"input_cost_per_token_priority": 1.875e-07,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 1048576,
"max_tokens": 1048576,
"mode": "chat",
"output_cost_per_token": 5e-07,
"output_cost_per_token_priority": 6.25e-07,
"source": "https://api.fireworks.ai/v1/serverless/models",
"supports_function_calling": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true
},
"fireworks_ai/accounts/fireworks/models/inkling": {
"cache_read_input_token_cost": 1.7e-07,
"input_cost_per_token": 1e-06,
@ -67259,9 +67427,9 @@
"supports_web_search": true
},
"openrouter/deepseek/deepseek-v4-flash": {
"input_cost_per_token": 8.8606e-08,
"output_cost_per_token": 1.77212e-07,
"cache_read_input_token_cost": 1.77212e-08,
"input_cost_per_token": 5.544e-08,
"output_cost_per_token": 1.1088e-07,
"cache_read_input_token_cost": 1.1088e-08,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
@ -69010,12 +69178,12 @@
"supports_reasoning": false
},
"openrouter/meta-llama/llama-3.1-70b-instruct": {
"input_cost_per_token": 7.2e-07,
"output_cost_per_token": 7.2e-07,
"input_cost_per_token": 4e-07,
"output_cost_per_token": 4e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 131072,
"max_output_tokens": 8192,
"max_tokens": 8192,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
@ -71317,15 +71485,15 @@
"supports_web_search": false
},
"openrouter/~deepseek/deepseek-pro-latest": {
"cache_read_input_token_cost": 4.4e-08,
"input_cost_per_token": 1.32e-06,
"cache_read_input_token_cost": 1.8354e-08,
"input_cost_per_token": 5.7684e-07,
"litellm_provider": "openrouter",
"max_input_tokens": 1048576,
"max_output_tokens": 384000,
"max_tokens": 384000,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},
"output_cost_per_token": 3.96e-06,
"off_peak_pricing": {"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":5.7684e-7,"output_cost_per_token":0.00000173052,"cache_read_input_token_cost":1.8354e-8},
"output_cost_per_token": 1.73052e-06,
"source": "https://openrouter.ai/api/v1/models",
"supports_audio_input": false,
"supports_function_calling": true,

View file

@ -137,6 +137,10 @@
"type": "number",
"minimum": 0
},
"cache_read_input_image_token_cost": {
"type": "number",
"minimum": 0
},
"cache_read_input_token_cost": {
"type": "number",
"minimum": 0,

View file

@ -122,7 +122,7 @@ E2E_FIXTURE_MODE=replay E2E_FIXTURE_DIR=/tmp/e2e-fixtures E2E_RESET_SPEND_LOGS=1
Point the proxy at bogus provider credentials for the replay run and it still has to pass: that is the whole proof that nothing left the process. Bundles are never committed. `tests/e2e/.fixtures` is gitignored because a bundle holds verbatim provider response bodies and hard-fails after seven days. CI records and replays this lane on a schedule in `.github/workflows/e2e_record_replay.yml`, publishing the bundle as a private `e2e-fixtures-bundle` artifact instead of committing it, selecting the tests with the `@pytest.mark.replayable` marker, and proving the bogus-credentials replay hermetic by counting provider egress with `.github/scripts/e2e_egress_sentinel.py`
Current limits: Bedrock cannot be mounted (SigV4 signs the Host header, so a rewritten api_base fails signature verification), deployments baked into the proxy's config file cannot be edge-wired (only `/model/new` registrations can carry the edge api_base), and a file upload routed by `custom_llm_provider` through the proxy's `files_settings` block never passes a deployment at all, so the batches `model_param` and `provider_fallback` scenarios keep uploading live in every mode
Current limits: Bedrock cannot be mounted in record or replay (SigV4 signs the Host header, so a rewritten api_base fails signature verification); a test that needs to observe the Converse body registers its own `LiveEdge` with `provider_edge_bedrock.bedrock_signer` re-signing the forwarded request, and carries the `provider_edge_host` opt-in marker because the gateway must reach the pytest host, which the Buildkite ephemeral stack cannot (the GitHub changed-e2e lane, whose gateways run on the runner, sets `E2E_PROVIDER_EDGE_HOST_REACHABLE`). Deployments baked into the proxy's config file cannot be edge-wired (only `/model/new` registrations can carry the edge api_base), and a file upload routed by `custom_llm_provider` through the proxy's `files_settings` block never passes a deployment at all, so the batches `model_param` and `provider_fallback` scenarios keep uploading live in every mode
## Typing

View file

@ -31,6 +31,7 @@ from e2e_config import (
MANAGED_FILES_OPT_IN_ENV,
MCP_OAUTH_LIVE_OPT_IN_ENV,
PROMPT_CACHING_OPT_IN_ENV,
PROVIDER_EDGE_HOST_OPT_IN_ENV,
PROXY_BASE_URL,
REDIS_CHAOS_OPT_IN_ENV,
WEEKLY_ANOMALY_OPT_IN_ENV,
@ -59,6 +60,7 @@ OPT_IN_MARKERS: Final = MappingProxyType(
"redis_chaos": REDIS_CHAOS_OPT_IN_ENV,
"cli_determinism": CLI_DETERMINISM_OPT_IN_ENV,
"mcp_oauth_live": MCP_OAUTH_LIVE_OPT_IN_ENV,
"provider_edge_host": PROVIDER_EDGE_HOST_OPT_IN_ENV,
}
)
@ -143,6 +145,11 @@ def pytest_configure(config: pytest.Config) -> None:
"mcp_oauth_live: real Linear OAuth consent via a captured browser session; deselected unless "
"E2E_MCP_OAUTH_LIVE is set",
)
config.addinivalue_line(
"markers",
"provider_edge_host: routes provider traffic through the pytest host's edge in every fixture mode, so the "
"gateway must reach the pytest host; deselected unless E2E_PROVIDER_EDGE_HOST_REACHABLE is set",
)
def pytest_sessionstart(session: pytest.Session) -> None:

View file

@ -146,6 +146,7 @@ PROMPT_CACHING_OPT_IN_ENV = "E2E_PROMPT_CACHING_STACK"
REDIS_CHAOS_OPT_IN_ENV = "E2E_REDIS_CHAOS"
CLI_DETERMINISM_OPT_IN_ENV = "E2E_CLI_DETERMINISM"
MCP_OAUTH_LIVE_OPT_IN_ENV: Final = "E2E_MCP_OAUTH_LIVE"
PROVIDER_EDGE_HOST_OPT_IN_ENV: Final = "E2E_PROVIDER_EDGE_HOST_REACHABLE"
ANOMALY_SESSIONS = int(os.environ.get("E2E_ANOMALY_SESSIONS", "6"))
ANOMALY_TURNS_PER_SESSION = int(os.environ.get("E2E_ANOMALY_TURNS_PER_SESSION", "6"))
ANOMALY_TURN_ATTEMPTS = int(os.environ.get("E2E_ANOMALY_TURN_ATTEMPTS", "3"))

View file

@ -95,7 +95,7 @@ class UnauthorizedError(BaseModel):
class RateLimitedError(BaseModel):
kind: Literal["rate_limited"] = "rate_limited"
retry_after_seconds: int | None = None
# litellm overloads 429 for budget_exceeded too, so keep the body to tell them apart.
# keep the body so callers can tell limiter kinds apart.
body: str = ""

View file

@ -75,6 +75,7 @@ class ResponsesRequest(BaseModel):
stream: bool = False
tools: list[ResponsesFunctionTool] | None = None
guardrails: list[str] | None = None
safety_identifier: str | None = None
cache: dict[str, bool] | None = {"no-cache": True}
@ -316,6 +317,7 @@ class EndpointsClient:
*,
stream: bool = False,
guardrails: list[str] | None = None,
safety_identifier: str | None = None,
) -> StreamingResponse:
return self._send(
"/v1/responses",
@ -326,6 +328,7 @@ class EndpointsClient:
instructions="You are a helpful assistant",
stream=stream,
guardrails=guardrails,
safety_identifier=safety_identifier,
),
stream=stream,
)

View file

@ -8,10 +8,14 @@ litellm-regression-tests/tests/test_inference_endpoints.py.
from __future__ import annotations
import json
from typing import cast
import threading
from collections.abc import Mapping
from dataclasses import dataclass, field
from types import MappingProxyType
from typing import Final, cast
import pytest
from e2e_config import unique_marker
from e2e_config import PROVIDER_EDGE_ADVERTISE_HOST, PROVIDER_EDGE_BIND_HOST, unique_marker
from e2e_http import (
assert_client_error,
require_successful_call,
@ -26,7 +30,9 @@ from endpoints_client import (
ResponsesStreamEventType,
)
from lifecycle import ResourceManager
from models import LiteLLMParamsBody
from models import ChatBody, ChatMessage, LiteLLMParamsBody
from provider_edge import LiveEdge, start_provider_edge
from provider_edge_bedrock import bedrock_signer
from pydantic import BaseModel, ValidationError
pytestmark = pytest.mark.e2e
@ -39,6 +45,33 @@ class _OptionalResponsesBody(BaseModel):
BEDROCK_CONVERSE_BACKEND = "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0"
BEDROCK_EDGE_REGION: Final = "us-east-1"
BEDROCK_EDGE_MOUNT: Final = f"bedrock/{BEDROCK_EDGE_REGION}"
class ConverseRequestBody(BaseModel):
additionalModelRequestFields: dict[str, str] | None = None
@dataclass(slots=True)
class ConverseRequestCapture:
"""The Converse bodies the proxy actually sent upstream, as seen by a live
edge sitting between the proxy and Bedrock."""
_bodies: list[ConverseRequestBody] = field(default_factory=list)
_lock: threading.Lock = field(default_factory=threading.Lock)
def observe(self, url: str, headers: Mapping[str, str], body: bytes | None) -> None:
if body is None or "/converse" not in url:
return
with self._lock:
self._bodies.append(ConverseRequestBody.model_validate_json(body))
@property
def bodies(self) -> tuple[ConverseRequestBody, ...]:
with self._lock:
return tuple(self._bodies)
WEATHER_TOOL = ResponsesFunctionTool(
name="get_weather",
@ -295,6 +328,53 @@ class TestResponses:
arguments = WeatherArguments.model_validate(raw_arguments)
assert arguments.location, f"function call arguments missing location: {function_call.arguments}"
@pytest.mark.provider_edge_host
@pytest.mark.parametrize("endpoint", ["/v1/responses", "/v1/chat/completions"])
def test_bedrock_forwards_allowed_safety_identifier_as_additional_model_request_field(
self, endpoints_client: EndpointsClient, resources: ResourceManager, endpoint: str
) -> None:
capture: Final = ConverseRequestCapture()
edge: Final = start_provider_edge(
LiveEdge(observe_request=capture.observe, sign=bedrock_signer(BEDROCK_EDGE_REGION)),
mounts=MappingProxyType({BEDROCK_EDGE_MOUNT: f"https://bedrock-runtime.{BEDROCK_EDGE_REGION}.amazonaws.com"}),
bind_host=PROVIDER_EDGE_BIND_HOST,
advertise_host=PROVIDER_EDGE_ADVERTISE_HOST,
)
resources.defer(edge.shutdown)
model: Final = f"e2e-responses-{unique_marker()}"
model_id: Final = endpoints_client.create_model(
model,
LiteLLMParamsBody(
model=BEDROCK_CONVERSE_BACKEND,
api_base=edge.edge.api_base(BEDROCK_EDGE_MOUNT),
aws_access_key_id="os.environ/AWS_ACCESS_KEY_ID",
aws_secret_access_key="os.environ/AWS_SECRET_ACCESS_KEY",
aws_region_name=BEDROCK_EDGE_REGION,
allowed_openai_params=["safety_identifier"],
),
)
resources.defer(lambda: endpoints_client.delete_model(model_id))
key: Final = resources.key()
safety_identifier: Final = f"end-user-{unique_marker()}"
if endpoint == "/v1/responses":
endpoints_client.responses(key, model, "reply with one word", safety_identifier=safety_identifier)
else:
endpoints_client.proxy.chat(
key,
ChatBody(
model=model,
messages=[ChatMessage(role="user", content="reply with one word")],
safety_identifier=safety_identifier,
),
)
forwarded: Final = tuple(body.additionalModelRequestFields for body in capture.bodies)
assert forwarded, f"{endpoint} produced no Bedrock Converse request"
assert forwarded == ({"safety_identifier": safety_identifier},) * len(forwarded), (
f"{endpoint} did not forward safety_identifier to Bedrock Converse on every attempt: {capture.bodies}"
)
@pytest.mark.skip(reason="stage red: product gap, /v1/responses 500s (aresponses TypeError) on missing input instead of 400")
@pytest.mark.covers("llm.responses.openai.input_validation.nonstream.works")
def test_missing_input_returns_error(

View file

@ -96,8 +96,8 @@ def _spend_until_budget_blocks(client: ManagementClient, key: str) -> None:
for _ in range(40):
outcome = client.chat_status(key, SPEND_MODEL, f"spend {unique_marker()}")
if _is_budget_block(outcome):
assert outcome.status_code == 429, (
f"budget refusal must be 429, got {outcome.status_code}: {outcome.body[:200]}"
assert outcome.status_code == 422, (
f"budget refusal must be 422, got {outcome.status_code}: {outcome.body[:200]}"
)
return
assert outcome.ok, f"paid call failed before the budget tripped ({outcome.status_code}): {outcome.body[:300]}"

View file

@ -298,6 +298,7 @@ class ChatBody(BaseModel):
max_completion_tokens: int | None = None
temperature: float | None = None
user: str | None = None
safety_identifier: str | None = None
metadata: ChatMetadata | None = None
reasoning_effort: str | None = None
thinking: ThinkingParam | None = None
@ -976,6 +977,7 @@ class LiteLLMParamsBody(BaseModel):
api_base: str | None = None
api_version: str | None = None
realtime_protocol: str | None = None
allowed_openai_params: list[str] | None = None
aws_access_key_id: str | None = None
aws_secret_access_key: str | None = None
aws_region_name: str | None = None

View file

@ -99,6 +99,7 @@ from provider_cache import (
SIGNATURE_HEADERS,
CacheEdge,
MountPolicy,
RequestSigner,
is_bedrock,
scoped_edge_base,
split_test_segment,
@ -539,6 +540,7 @@ class ReplayEdge:
@dataclass(frozen=True, slots=True)
class LiveEdge:
observe_request: Callable[[str, Mapping[str, str], bytes | None], None] | None = None
sign: RequestSigner | None = None
type EdgeBackend = RecordEdge | ReplayEdge | LiveEdge | CacheEdge
@ -788,14 +790,16 @@ def _handle_live(
method: str, url: str, headers: Mapping[str, str], body: bytes | None, timeout: float,
cache: CacheEdge | None = None, mount: str = "", test_key: str | None = None,
observe_request: Callable[[str, Mapping[str, str], bytes | None], None] | None = None,
sign: RequestSigner | None = None,
) -> EdgeOutcome:
forwarded: Final = {
name: value for name, value in headers.items() if name.lower() not in _REQUEST_DROPPED_HEADERS
}
if observe_request is not None:
observe_request(url, forwarded, body)
outbound: Final = forwarded if sign is None else sign(method, url, forwarded, body)
head: Final = (
forward_stream(method, url, headers=forwarded, body=body, timeout=timeout)
forward_stream(method, url, headers=outbound, body=body, timeout=timeout)
if cache is None else cache.forward(mount, method, url, forwarded, body, timeout, test_key=test_key)
)
match head:
@ -871,10 +875,10 @@ def handle_edge_request(
method, _upstream_url(upstream_base, upstream_path, split.query), headers, body, timeout,
backend, mount, test_key,
)
case LiveEdge(observe_request=observe_request):
case LiveEdge(observe_request=observe_request, sign=sign):
return _handle_live(
method, _upstream_url(upstream_base, upstream_path, split.query), headers, body, timeout,
observe_request=observe_request,
observe_request=observe_request, sign=sign,
)
case RecordEdge():
return _handle_record(

View file

@ -13,3 +13,4 @@ markers =
cli_determinism: drives the real claude CLI for several seconds; deselected unless E2E_CLI_DETERMINISM is set
redis_chaos: load test that pauses the proxy's Redis outright mid-run; needs a proxy booted from gateway/redis_chaos_ci_config.yml on the same host, and is deselected unless E2E_REDIS_CHAOS is set
mcp_oauth_live: real Linear OAuth consent via a captured browser session; deselected unless E2E_MCP_OAUTH_LIVE is set
provider_edge_host: routes provider traffic through the pytest host's edge in every fixture mode, so the gateway must reach the pytest host; deselected unless E2E_PROVIDER_EDGE_HOST_REACHABLE is set

View file

@ -46,10 +46,10 @@ def _assert_budget_blocks(client: BudgetClient, key: str, *, user: str = "") ->
pytest.fail("budget never enforced within the call budget")
def _assert_blocked_429(client: BudgetClient, key: str) -> StreamingResponse:
def _assert_blocked_422(client: BudgetClient, key: str) -> StreamingResponse:
blocked = _assert_budget_blocks(client, key)
assert blocked.status_code == 429, (
f"budget refusal must be 429, got {blocked.status_code}: {blocked.body[:200]}"
assert blocked.status_code == 422, (
f"budget refusal must be 422, got {blocked.status_code}: {blocked.body[:200]}"
)
return blocked
@ -60,7 +60,7 @@ class TestBudgetBlocksPerLevel:
key = client.generate_key(max_budget=TINY_CAP)
resources.defer(lambda: client.delete_key(key))
_assert_blocked_429(client, key)
_assert_blocked_422(client, key)
@pytest.mark.covers("quota_management.budget.team.blocks_over_limit")
def test_team_budget_blocks_every_team_key(self, client: BudgetClient, resources: ResourceManager) -> None:
@ -71,10 +71,10 @@ class TestBudgetBlocksPerLevel:
sibling_key = client.generate_key(team_id=team_id)
resources.defer(lambda: client.delete_key(sibling_key))
_assert_blocked_429(client, spender_key)
_assert_blocked_422(client, spender_key)
sibling = _chat(client, sibling_key)
assert is_budget_block(sibling) and sibling.status_code == 429, (
f"a sibling key on the capped team must get the same 429 budget_exceeded, "
assert is_budget_block(sibling) and sibling.status_code == 422, (
f"a sibling key on the capped team must get the same 422 budget_exceeded, "
f"got {sibling.status_code}: {sibling.body[:200]}"
)
@ -99,10 +99,10 @@ class TestBudgetBlocksPerLevel:
team_key = client.generate_key(team_id=team_id, user_id=user_id)
resources.defer(lambda: client.delete_key(team_key))
_assert_blocked_429(client, first_key)
_assert_blocked_422(client, first_key)
second = _chat(client, second_key)
assert is_budget_block(second) and second.status_code == 429, (
f"the second personal key of a user over budget must get the same 429 budget_exceeded, "
assert is_budget_block(second) and second.status_code == 422, (
f"the second personal key of a user over budget must get the same 422 budget_exceeded, "
f"got {second.status_code}: {second.body[:200]}"
)
team_result = _chat(client, team_key)
@ -133,7 +133,7 @@ class TestBudgetBlocksPerLevel:
key = client.generate_key(team_id=team_id)
resources.defer(lambda: client.delete_key(key))
blocked = _assert_blocked_429(client, key)
blocked = _assert_blocked_422(client, key)
assert f"Organization={org_id}" in blocked.body, (
f"refusal must name the org as the blocker, got: {blocked.body[:200]}"
)
@ -155,7 +155,7 @@ class TestBudgetBlocksPerLevel:
teammate_key = client.generate_key(team_id=team_id, user_id=teammate_id)
resources.defer(lambda: client.delete_key(teammate_key))
_assert_blocked_429(client, member_key)
_assert_blocked_422(client, member_key)
require_successful_call(_chat(client, teammate_key))
@ -176,7 +176,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
control_key = client.generate_key(user_id=user_id)
resources.defer(lambda: client.delete_key(control_key))
_assert_blocked_429(client, capped_key)
_assert_blocked_422(client, capped_key)
require_successful_call(_chat(client, control_key))
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
@ -188,7 +188,7 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
control_key = client.generate_key(team_id=team_id)
resources.defer(lambda: client.delete_key(control_key))
_assert_blocked_429(client, capped_key)
_assert_blocked_422(client, capped_key)
require_successful_call(_chat(client, control_key))
@pytest.mark.covers("quota_management.budget.key.blocks_over_limit")
@ -205,5 +205,5 @@ class TestKeyBudgetBlocksAcrossKeyKinds:
control_key = client.generate_key(team_id=team_id, user_id=member_id)
resources.defer(lambda: client.delete_key(control_key))
_assert_blocked_429(client, capped_key)
_assert_blocked_422(client, capped_key)
require_successful_call(_chat(client, control_key))

View file

@ -102,7 +102,7 @@ def test_long_window_blocks_after_short_window_resets(client: BudgetClient, reso
# 1. drive the key to get blocked by SHORT_WINDOW, assert it's budget error
blocked = _drive_to_block(client, key)
assert blocked.status_code == 429, f"budget block was not a 429: {blocked.status_code} {blocked.body[:200]}"
assert blocked.status_code == 422, f"budget block was not a 422: {blocked.status_code} {blocked.body[:200]}"
# 2. check the reset times of both budget windows after we drove to being blocked
blocked_reset_at = window_reset_at(client.key_budget_windows(key), SHORT_WINDOW)

View file

@ -101,7 +101,7 @@ def test_team_long_window_blocks_after_short_window_resets(client: BudgetClient,
# 1. drive the key to being blocked, assert its blocked by budget budget_exceeded
blocked = _drive_to_block(client, key)
assert blocked.status_code == 429, f"budget block was not a 429: {blocked.status_code} {blocked.body[:200]}"
assert blocked.status_code == 422, f"budget block was not a 422: {blocked.status_code} {blocked.body[:200]}"
# 2. check the the teams budget windows
blocked_reset_at = window_reset_at(client.team_budget_windows(team_id), SHORT_WINDOW)

View file

@ -97,7 +97,7 @@ def test_zero_false_and_empty_values_are_not_treated_as_omission(gateway: Gatewa
"POST", "/v1/chat/completions",
{"model": models[0], "messages": [{"role": "user", "content": "zero budget"}]}, key=key,
)
assert denied.status_code == 429, denied.text
assert denied.status_code == 422, denied.text
assert denied.json()["error"]["type"] == "budget_exceeded"
gateway.post("/key/update", {"key": key, "max_budget": 1, "models": [], "metadata": {}})
info: Final = object_value(gateway.get("/key/info", {"key": key})["info"])
@ -127,7 +127,7 @@ def test_zero_false_and_empty_values_are_not_treated_as_omission(gateway: Gatewa
"POST", "/v1/chat/completions",
{"model": models[0], "messages": [{"role": "user", "content": "updated zero budget"}]}, key=key,
)
assert zero_after_update.status_code == 429, zero_after_update.text
assert zero_after_update.status_code == 422, zero_after_update.text
assert zero_after_update.json()["error"]["type"] == "budget_exceeded"
gateway.post("/key/update", {"key": key, "max_budget": None})
assert read_rows(

View file

@ -185,7 +185,7 @@ def test_key_budget_at_boundary_blocks_provider_then_explicit_reset_restores(gat
{"model": model, "messages": [{"role": "user", "content": f"over budget {uuid.uuid4().hex}"}]},
key=key,
)
assert denied.status_code == 429 and denied.json()["error"]["type"] == "budget_exceeded", denied.text
assert denied.status_code == 422 and denied.json()["error"]["type"] == "budget_exceeded", denied.text
assert upstream.get("/__observations").json()["requests"] == []
assert gateway.chat(model, key=control, text=f"control {uuid.uuid4().hex}")["usage"]["total_tokens"] == 40
gateway.post("/key/update", {"key": key, "spend": 0})
@ -205,7 +205,7 @@ def test_key_budget_at_boundary_blocks_provider_then_explicit_reset_restores(gat
{"model": model, "messages": [{"role": "user", "content": f"boundary again {uuid.uuid4().hex}"}]},
key=key,
)
assert denied_again.status_code == 429 and denied_again.json()["error"]["type"] == "budget_exceeded", (
assert denied_again.status_code == 422 and denied_again.json()["error"]["type"] == "budget_exceeded", (
denied_again.text
)
assert upstream.get("/__observations").json()["requests"] == []

View file

@ -133,3 +133,9 @@ meta.llama3-2-11b-instruct-v1:0
us.meta.llama3-2-11b-instruct-v1:0
meta.llama3-2-90b-instruct-v1:0
us.meta.llama3-2-90b-instruct-v1:0
bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b
bedrock/ap-south-1/qwen.qwen3-next-80b-a3b
bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b
bedrock/eu-west-1/qwen.qwen3-next-80b-a3b
bedrock/eu-west-2/qwen.qwen3-next-80b-a3b
bedrock/sa-east-1/qwen.qwen3-next-80b-a3b

View file

@ -605,6 +605,38 @@ class TestNonStreaming:
assert result["error"]["message"] == "Bad request"
class TestStreaming:
"""Streaming requests must ask AgentCore for a stream, not a single send."""
@pytest.mark.asyncio
async def test_streaming_request_uses_message_stream_method_and_yields_sse_events(self, httpx_transport):
from litellm.a2a_protocol.providers.bedrock_agentcore.config import (
BedrockAgentCoreA2AConfig,
)
sse_body = (
'data: {"jsonrpc": "2.0", "id": "req-001", "result": {"kind": "task", "id": "t1"}}\n\n'
'data: {"jsonrpc": "2.0", "id": "req-001", "result": {"kind": "status-update", "final": true}}\n\n'
)
with respx.mock(assert_all_called=True) as router:
route = router.post(url__regex=r".*/invocations.*").mock(
return_value=httpx.Response(200, headers={"content-type": "text/event-stream"}, text=sse_body)
)
events = [
event
async for event in BedrockAgentCoreA2AConfig().handle_streaming(
request_id="req-001",
params=SAMPLE_PARAMS,
litellm_params=SAMPLE_LITELLM_PARAMS,
)
]
sent_body = json.loads(route.calls.last.request.content)
assert sent_body["method"] == "message/stream", sent_body
assert sent_body["params"]["message"]["messageId"] == "msg-001"
assert [event["result"]["kind"] for event in events] == ["task", "status-update"]
class TestConfigManager:
"""Test that config manager routes 'bedrock' correctly."""

View file

@ -3033,7 +3033,7 @@ def test_get_error_information_budget_exceeded_structured_fields():
assert result["error_budget_entity_id"] == "repro-user"
assert result["error_budget_limit"] == 1e-06
assert result["error_budget_spend"] == 3.4e-05
assert result["error_code"] == "429"
assert result["error_code"] == "422"
assert result["error_class"] == "BudgetExceededError"
assert result["error_rate_limit_type"] == "budget"
@ -6407,7 +6407,7 @@ def test_get_error_information_keeps_traceback_for_unmapped_provider_4xx():
def test_get_error_information_skips_traceback_for_budget_rejection_with_provider():
"""A key-over-budget 429 is the proxy's own rejection even after the auth
"""A key-over-budget 422 is the proxy's own rejection even after the auth
handler stamps the requested model's provider onto it, so it stays cheap."""
from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup
@ -6416,7 +6416,7 @@ def test_get_error_information_skips_traceback_for_budget_rejection_with_provide
litellm.BudgetExceededError(current_cost=0.01, max_budget=0.0, llm_provider="anthropic")
)
result = StandardLoggingPayloadSetup.get_error_information(over_budget)
assert result["error_code"] == "429"
assert result["error_code"] == "422"
assert result["llm_provider"] == "anthropic"
assert result["traceback"] == ""

View file

@ -110,6 +110,19 @@ class TestOutputConfigStrippedFromCompletionKwargs:
"reject it with 400 'Extra inputs are not permitted'"
)
def test_safeguards_is_stripped_for_non_anthropic_target(self):
extra_kwargs = {
"custom_llm_provider": "azure",
"safeguards": [{"type": "dangerous_tool_use", "classifier_context": {"v": 1}}],
}
result = _call_prepare(extra_kwargs=extra_kwargs)
completion_kwargs = result[0] if isinstance(result, tuple) else result
assert "safeguards" not in completion_kwargs, (
"safeguards is an Anthropic-only field; OpenAI-format backends reject it with 400"
)
def test_output_config_format_translated_to_response_format(self):
"""When ``output_config`` carries structured-output ``format``, the
translator now maps it to OpenAI's ``response_format`` so non-Anthropic

View file

@ -1438,3 +1438,109 @@ async def test_anthropic_messages_leaves_non_provider_failures_unmapped():
)
assert "Traceback" not in str(excinfo.value)
@pytest.mark.asyncio
async def test_anthropic_messages_forwards_safeguards_and_unknown_beta_to_anthropic():
"""Shapes are what Claude Code 2.1.278 sends and api.anthropic.com returns, captured 2026-09-21."""
from litellm.llms.anthropic.experimental_pass_through.messages import handler
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
client_betas = "dangerous-tool-use-2026-09-03,interleaved-thinking-2025-05-14"
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": {}}}]
captured: dict[str, object] = {}
def upstream_records_the_request(request: httpx.Request) -> httpx.Response:
captured["body"] = json.loads(request.content)
captured["anthropic-beta"] = request.headers.get("anthropic-beta")
return httpx.Response(
200,
json={
"id": "msg_1",
"type": "message",
"role": "assistant",
"model": "claude-haiku-4-5",
"content": [{"type": "text", "text": "ok"}],
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 1, "output_tokens": 1},
"safeguard_results": safeguard_results,
},
request=request,
)
upstream = AsyncHTTPHandler()
upstream.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream_records_the_request))
response = await handler.anthropic_messages(
max_tokens=16,
messages=[{"role": "user", "content": "hi"}],
model="anthropic/claude-haiku-4-5",
custom_llm_provider="anthropic",
api_key="sk-test",
client=upstream,
safeguards=safeguards,
extra_headers={"anthropic-beta": client_betas},
)
assert captured["body"]["safeguards"] == safeguards
assert set(captured["anthropic-beta"].split(",")) == set(client_betas.split(","))
assert response["safeguard_results"] == safeguard_results
@pytest.mark.asyncio
async def test_anthropic_messages_streaming_forwards_safeguards_and_keeps_safeguard_results():
"""Shapes are what Claude Code 2.1.278 sends and api.anthropic.com returns, captured 2026-09-21."""
from litellm.llms.anthropic.experimental_pass_through.messages import handler
safeguards = [{"type": "dangerous_tool_use", "classifier_context": {"v": 1, "permission_mode": "auto"}}]
tool_verdicts = {"toolu_01": {"type": "evaluated", "outcome": "not_flagged"}}
safeguard_results = [{"type": "dangerous_tool_use", "status": {"type": "available", "tool_uses": tool_verdicts}}]
captured: dict[str, object] = {}
message_start = {
"type": "message_start",
"message": {
"id": "msg_1",
"type": "message",
"role": "assistant",
"model": "claude-haiku-4-5",
"content": [],
"stop_reason": None,
"stop_sequence": None,
"usage": {"input_tokens": 1, "output_tokens": 0},
"safeguard_results": safeguard_results,
},
}
message_delta = {
"type": "message_delta",
"delta": {"stop_reason": "end_turn", "stop_sequence": None, "safeguard_results": safeguard_results},
"usage": {"output_tokens": 1},
}
sse = "".join(
f"event: {event['type']}\ndata: {json.dumps(event)}\n\n"
for event in (message_start, message_delta, {"type": "message_stop"})
)
def upstream_streams_safeguard_results(request: httpx.Request) -> httpx.Response:
captured["body"] = json.loads(request.content)
return httpx.Response(200, headers={"content-type": "text/event-stream"}, content=sse.encode(), request=request)
upstream = AsyncHTTPHandler()
upstream.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream_streams_safeguard_results))
stream = await handler.anthropic_messages(
max_tokens=16,
messages=[{"role": "user", "content": "hi"}],
model="anthropic/claude-haiku-4-5",
custom_llm_provider="anthropic",
api_key="sk-test",
client=upstream,
stream=True,
safeguards=safeguards,
)
raw = b"".join([chunk async for chunk in stream]).decode()
events = [json.loads(line[len("data: ") :]) for line in raw.splitlines() if line.startswith("data: ")]
assert captured["body"]["safeguards"] == safeguards
assert events[0]["message"]["safeguard_results"] == safeguard_results
assert [e for e in events if e["type"] == "message_delta"][0]["delta"]["safeguard_results"] == safeguard_results

View file

@ -453,6 +453,38 @@ class TestAzureMAIImageGeneration:
)
assert round(cost, 10) == round(expected_cost, 10)
def test_mai_image_pro_edit_cost_splits_text_and_image_input(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "azure_ai/MAI-Image-2.5-Pro"
model_info = litellm.get_model_info(model=model, custom_llm_provider="azure_ai")
text_tokens = 37
image_tokens = 1024
output_image_tokens = 1024
image_response = ImageResponse(
data=[ImageObject(b64_json="img1")],
usage=ImageUsage(
input_tokens=text_tokens + image_tokens,
input_tokens_details=ImageUsageInputTokensDetails(
text_tokens=text_tokens,
image_tokens=image_tokens,
),
output_tokens=output_image_tokens,
total_tokens=text_tokens + image_tokens + output_image_tokens,
),
)
cost = azure_ai_image_cost_calculator(model=model, image_response=image_response)
expected_cost = (
text_tokens * model_info["input_cost_per_token"]
+ image_tokens * model_info["input_cost_per_image_token"]
+ output_image_tokens * model_info["output_cost_per_image_token"]
)
assert round(cost, 10) == round(expected_cost, 10)
assert model_info["input_cost_per_image_token"] != model_info["input_cost_per_token"]
def test_mai_image_cost_calculator_falls_back_to_flat_image_pricing(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")

View file

@ -6339,15 +6339,15 @@ class TestMCPDcrBridgeDelegateAdmission:
)
return exc_info.value
async def test_over_budget_admission_surfaces_429_not_401(self):
"""A validly-authenticated but over-budget identity surfaces the standard pipeline's 429, not
async def test_over_budget_admission_surfaces_422_not_401(self):
"""A validly-authenticated but over-budget identity surfaces the standard pipeline's 422, not
a misleading 401. Flattening budget to 401 told the caller their credential was invalid, which
on a DCR client reads as broken auth and triggers a re-authorize that cannot fix a budget
problem. Regression for the status-flattening finding on the live-policy gate."""
import litellm
mapped = await self._enforce_with_gate_error(litellm.BudgetExceededError(current_cost=10.0, max_budget=1.0))
assert mapped.status_code == 429
assert mapped.status_code == 422
async def test_db_outage_during_policy_surfaces_503_not_401(self):
"""A transient database outage during the live-policy gate surfaces a retryable 503, not a 401

View file

@ -448,7 +448,7 @@ async def test_handle_authentication_error_budget_exceeded():
)
assert exc_info.value.type == ProxyErrorTypes.budget_exceeded
assert int(exc_info.value.code) == status.HTTP_429_TOO_MANY_REQUESTS
assert int(exc_info.value.code) == status.HTTP_422_UNPROCESSABLE_CONTENT
@pytest.mark.asyncio
@ -687,7 +687,7 @@ def _http_request(client_host: str | None = "10.1.2.3", headers: dict[str, str]
{"allow_requests_on_db_unavailable": False},
{},
"10.1.2.3",
id="429_budget_exceeded",
id="422_budget_exceeded",
),
],
)
@ -697,7 +697,7 @@ async def test_auth_failure_logs_requester_ip_address(
request_kwargs: dict[str, dict[str, str]],
expected_ip: str,
) -> None:
"""401s and budget 429s are rejected before `add_litellm_data_to_request` stamps
"""401s and budget 422s are rejected before `add_litellm_data_to_request` stamps
the caller IP, so without this the failure logs (spend logs, prometheus client_ip)
had no IP, and a 401 rarely carries a key or user identity either."""
with (

View file

@ -75,7 +75,7 @@ async def test_over_first_window_raises():
await _virtual_key_multi_budget_check(valid_token=token)
err = exc_info.value
assert err.status_code == 429
assert err.status_code == 422
assert "24h" in str(err)
assert "Key over" in str(err)
@ -107,7 +107,7 @@ async def test_over_second_window_raises():
await _virtual_key_multi_budget_check(valid_token=token)
err = exc_info.value
assert err.status_code == 429
assert err.status_code == 422
assert "30d" in str(err)

View file

@ -8156,7 +8156,7 @@ async def test_reset_key_spend_resets_budget_windows(monkeypatch):
counter without also advancing reset_at is not durable either: the very
next request would re-sum the unchanged historical spend and put the
counter right back above the window's max_budget, so
_virtual_key_multi_budget_check kept raising BudgetExceededError (429) on
_virtual_key_multi_budget_check kept raising BudgetExceededError (422) on
every request even though the key's own reported spend read $0.
"""
mock_prisma_client = MagicMock()
@ -16593,7 +16593,7 @@ async def test_info_key_fn_reads_the_configured_budget_model_key(monkeypatch):
It used to probe a second, provider-stripped key because the counter was
written under the request model instead, which is what let a key report zero
usage while being blocked at 429.
usage while being blocked at 422.
"""
from unittest.mock import AsyncMock, MagicMock

View file

@ -2099,7 +2099,7 @@ class TestCursorVariantPerModelBudgetEnforcement:
response = _post_cursor_with_real_auth(valid_token, attrs, request_model="claude-opus-5-thinking-high")
assert response.status_code == 429, response.text
assert response.status_code == 422, response.text
error = response.json()["error"]
assert error["type"] == "budget_exceeded"
assert "exceeded budget for model=claude-opus-5" in error["message"]
@ -2110,8 +2110,8 @@ class TestCursorVariantPerModelBudgetEnforcement:
base_response = _post_cursor_with_real_auth(valid_token, attrs, request_model="claude-opus-5")
alias_response = _post_cursor_with_real_auth(valid_token, attrs, request_model="claude-opus-5-fast")
assert base_response.status_code == 429, base_response.text
assert alias_response.status_code == 429, alias_response.text
assert base_response.status_code == 422, base_response.text
assert alias_response.status_code == 422, alias_response.text
assert alias_response.json() == base_response.json()

View file

@ -495,7 +495,7 @@ class TestProxyBaseLLMRequestProcessing:
)
assert exc_info.value.type == ProxyErrorTypes.budget_exceeded
assert exc_info.value.code == "429"
assert exc_info.value.code == "422"
tag_budget_check.assert_awaited_once()
_, call_kwargs = tag_budget_check.call_args
assert call_kwargs["tags"] == ("guardrail-tag",)
@ -702,7 +702,7 @@ class TestProxyBaseLLMRequestProcessing:
)
assert exc_info.value.type == ProxyErrorTypes.budget_exceeded
assert exc_info.value.code == "429"
assert exc_info.value.code == "422"
assert "guardrail-tag" in exc_info.value.message
@pytest.mark.asyncio

View file

@ -10872,7 +10872,7 @@ async def test_realtime_session_rejected_in_pre_call_releases_the_budget_reserva
"""A rate-limit or guardrail rejection happens before route_request, so the
relay never runs and no success log can own the reservation. The endpoint
must release it on that exit too, or the key stays pinned at the reserved
amount and its next requests 429 with budget_exceeded while /key/info shows
amount and its next requests 422 with budget_exceeded while /key/info shows
spend 0 (reproduced live with rpm_limit=1). The client still gets the
pre-call error event and the 1011 close it got before."""
reservation: Final = {"reserved_cost": 0.55, "input_cost": 0.0, "finalized": False, "entries": []}

View file

@ -1248,6 +1248,16 @@ class TestFunctionCallTransformation:
assert "tool_choice" not in result
assert "tools" not in result
def test_safety_identifier_forwarded_to_chat_completion_request(self) -> None:
result: Final = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
model="bedrock/global.openai.gpt-5.6-luna",
input="hi",
responses_api_request={"safety_identifier": "user-7f3a"},
custom_llm_provider="bedrock",
)
assert result["safety_identifier"] == "user-7f3a"
def test_parallel_tool_calls_dropped_when_no_chat_tools_remain(self) -> None:
transform: Final = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request
codex_tool_search: Final = {

View file

@ -4230,3 +4230,30 @@ def test_completion_cost_prices_responses_websocket_turns_per_service_tier():
assert ws_cost == pytest.approx(_http_cost(100, 40, "default") + _http_cost(60, 10, "priority"))
assert ws_cost != pytest.approx(_http_cost(160, 50, "default"))
assert ws_cost != pytest.approx(_http_cost(160, 50, "priority"))
QWEN3_NEXT_REGIONS: Final = ("ap-northeast-1", "ap-south-1", "ap-southeast-2", "eu-west-1", "eu-west-2", "sa-east-1")
@pytest.mark.parametrize("region", QWEN3_NEXT_REGIONS)
def test_cost_per_token_bedrock_qwen3_next_uses_regional_entry_not_us_rate(
monkeypatch: pytest.MonkeyPatch, region: str
) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
regional: Final = litellm.model_cost[f"bedrock/{region}/qwen.qwen3-next-80b-a3b"]
us: Final = litellm.model_cost["qwen.qwen3-next-80b-a3b"]
assert regional["input_cost_per_token"] != us["input_cost_per_token"]
assert regional["output_cost_per_token"] != us["output_cost_per_token"]
prompt_tokens, completion_tokens = 1000, 500
prompt_usd, completion_usd = cost_per_token(
model=f"bedrock/{region}/qwen.qwen3-next-80b-a3b",
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
custom_llm_provider="bedrock",
)
assert prompt_usd == pytest.approx(prompt_tokens * regional["input_cost_per_token"])
assert completion_usd == pytest.approx(completion_tokens * regional["output_cost_per_token"])

View file

@ -1397,13 +1397,18 @@ class TestBudgetExceededErrorSurfacesUnifiedFields:
assert e.llm_provider == "anthropic"
def test_should_keep_existing_status_code_and_message(self):
# Backward-compat guard: existing callers depend on `status_code=429`
# Backward-compat guard: existing callers depend on `status_code=422`
# and the canonical message format.
e = litellm.BudgetExceededError(current_cost=0.000109, max_budget=0.0001)
assert e.status_code == 429
assert e.status_code == 422
assert "Current cost: 0.000109" in e.message
assert "Max budget: 0.0001" in e.message
def test_should_honor_budget_exceeded_status_code_override(self, monkeypatch: pytest.MonkeyPatch):
monkeypatch.setattr(litellm, "budget_exceeded_status_code", 429)
e = litellm.BudgetExceededError(current_cost=0.5, max_budget=0.1)
assert e.status_code == 429
def test_should_still_be_catchable_as_exception_not_rate_limit_error(self):
# Critical: we deliberately did NOT make BudgetExceededError a
# RateLimitError subclass. Existing `except BudgetExceededError:`
@ -1424,7 +1429,7 @@ class TestBudgetExceededErrorSurfacesUnifiedFields:
info = StandardLoggingPayloadSetup.get_error_information(e)
assert info["error_rate_limit_category"] == "litellm_rate_limit"
assert info["error_rate_limit_type"] == "budget"
assert info["error_code"] == "429"
assert info["error_code"] == "422"
assert info["error_class"] == "BudgetExceededError"
def test_should_propagate_llm_provider_to_standard_logging_payload(self):

View file

@ -652,6 +652,7 @@ def validate_model_cost_values(model_data, exceptions=None):
"cache_creation_input_audio_token_cost",
"cache_read_input_token_cost",
"cache_read_input_audio_token_cost",
"cache_read_input_image_token_cost",
"input_dbu_cost_per_token",
"output_db_cost_per_token",
"output_dbu_cost_per_token",
@ -740,6 +741,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"cache_read_input_token_cost_above_512k_tokens": {"type": "number"},
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": {"type": "number"},
"cache_read_input_audio_token_cost": {"type": "number"},
"cache_read_input_image_token_cost": {"type": "number"},
"audio_transcription_config": {"type": "string"},
"deprecation_date": {"type": "string"},
"input_cost_per_audio_per_second": {"type": "number"},

View file

@ -30908,6 +30908,8 @@ export interface components {
cache_creation_input_token_cost_ultrafast?: number | null;
/** Cache Read Input Audio Token Cost */
cache_read_input_audio_token_cost?: number | null;
/** Cache Read Input Image Token Cost */
cache_read_input_image_token_cost?: number | null;
/** Cache Read Input Token Cost */
cache_read_input_token_cost?: number | null;
/** Cache Read Input Token Cost Above 200K Tokens */
@ -41635,6 +41637,8 @@ export interface components {
cache_creation_input_token_cost_ultrafast?: number | null;
/** Cache Read Input Audio Token Cost */
cache_read_input_audio_token_cost?: number | null;
/** Cache Read Input Image Token Cost */
cache_read_input_image_token_cost?: number | null;
/** Cache Read Input Token Cost */
cache_read_input_token_cost?: number | null;
/** Cache Read Input Token Cost Above 200K Tokens */

View file

@ -44,6 +44,7 @@ bedrock/ap-northeast-1/minimax.minimax-m2.5
bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking
bedrock/ap-northeast-1/moonshotai.kimi-k2.5
bedrock/ap-northeast-1/qwen.qwen3-coder-next
bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b
bedrock/moonshotai.kimi-k2-thinking
bedrock/moonshotai.kimi-k2.5
bedrock/ap-south-1/meta.llama3-70b-instruct-v1:0
@ -54,6 +55,7 @@ bedrock/ap-south-1/minimax.minimax-m2.5
bedrock/ap-south-1/moonshotai.kimi-k2-thinking
bedrock/ap-south-1/moonshotai.kimi-k2.5
bedrock/ap-south-1/qwen.qwen3-coder-next
bedrock/ap-south-1/qwen.qwen3-next-80b-a3b
bedrock/ap-southeast-2/minimax.minimax-m2.5
bedrock/ap-southeast-3/deepseek.v3.2
bedrock/ap-southeast-3/minimax.minimax-m2.1
@ -83,11 +85,13 @@ bedrock/eu-west-1/meta.llama3-8b-instruct-v1:0
bedrock/eu-west-1/minimax.minimax-m2.1
bedrock/eu-west-1/minimax.minimax-m2.5
bedrock/eu-west-1/qwen.qwen3-coder-next
bedrock/eu-west-1/qwen.qwen3-next-80b-a3b
bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0
bedrock/eu-west-2/meta.llama3-8b-instruct-v1:0
bedrock/eu-west-2/minimax.minimax-m2.1
bedrock/eu-west-2/minimax.minimax-m2.5
bedrock/eu-west-2/qwen.qwen3-coder-next
bedrock/eu-west-2/qwen.qwen3-next-80b-a3b
bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2
bedrock/eu-west-3/mistral.mistral-large-2402-v1:0
bedrock/eu-west-3/mistral.mixtral-8x7b-instruct-v0:1
@ -103,6 +107,7 @@ bedrock/sa-east-1/minimax.minimax-m2.5
bedrock/sa-east-1/moonshotai.kimi-k2-thinking
bedrock/sa-east-1/moonshotai.kimi-k2.5
bedrock/sa-east-1/qwen.qwen3-coder-next
bedrock/sa-east-1/qwen.qwen3-next-80b-a3b
bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1
bedrock/us-east-1/1-month-commitment/anthropic.claude-v1
bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1
@ -240,3 +245,4 @@ bedrock/us-gov-east-1/anthropic.claude-sonnet-5
bedrock/us-gov-east-1/anthropic.claude-opus-4-8
bedrock/us-gov-east-1/anthropic.claude-opus-5
bedrock/us-gov-east-1/anthropic.claude-fable-5-1
bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b