mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(anthropic_wif): declare federation params as owned connection leaves and chart their metrics
Register the 18 Anthropic and 3 OpenAI federation params as frozen ConnectionSettings leaves so the owned-kwarg registry, the kwargs funnel and the request-body ban list read one declaration. Pass the deployment api_base through to the count-tokens handler instead of a pre-suffixed URL, which doubled the /count_tokens path on main's prompt-cache predictor. Add the five litellm_anthropic_wif_* families to the all-metrics Grafana dashboard.
This commit is contained in:
parent
9e97756239
commit
54de31e46c
8 changed files with 194 additions and 76 deletions
|
|
@ -5710,6 +5710,17 @@
|
|||
}
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_PROMETHEUS}"
|
||||
},
|
||||
"editorMode": "code",
|
||||
"expr": "histogram_quantile(0.95, sum(rate(litellm_anthropic_wif_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "anthropic_wif",
|
||||
"range": true,
|
||||
"refId": "A"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
|
|
@ -5719,7 +5730,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_auth_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "auth",
|
||||
"range": true,
|
||||
"refId": "A"
|
||||
"refId": "B"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5730,7 +5741,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_batch_write_to_db_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "batch_write_to_db",
|
||||
"range": true,
|
||||
"refId": "B"
|
||||
"refId": "C"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5741,7 +5752,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_postgres_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "postgres",
|
||||
"range": true,
|
||||
"refId": "C"
|
||||
"refId": "D"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5752,7 +5763,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_proxy_pre_call_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "proxy_pre_call",
|
||||
"range": true,
|
||||
"refId": "D"
|
||||
"refId": "E"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5763,7 +5774,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "redis",
|
||||
"range": true,
|
||||
"refId": "E"
|
||||
"refId": "F"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5774,7 +5785,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_org_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "redis_daily_org_spend_update_queue",
|
||||
"range": true,
|
||||
"refId": "F"
|
||||
"refId": "G"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5785,7 +5796,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_tag_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "redis_daily_tag_spend_update_queue",
|
||||
"range": true,
|
||||
"refId": "G"
|
||||
"refId": "H"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5796,7 +5807,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_daily_team_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "redis_daily_team_spend_update_queue",
|
||||
"range": true,
|
||||
"refId": "H"
|
||||
"refId": "I"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5807,7 +5818,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_redis_window_spend_update_queue_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "redis_window_spend_update_queue",
|
||||
"range": true,
|
||||
"refId": "I"
|
||||
"refId": "J"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5818,7 +5829,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_reset_budget_job_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "reset_budget_job",
|
||||
"range": true,
|
||||
"refId": "J"
|
||||
"refId": "K"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5829,7 +5840,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_router_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "router",
|
||||
"range": true,
|
||||
"refId": "K"
|
||||
"refId": "L"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5840,7 +5851,7 @@
|
|||
"expr": "histogram_quantile(0.95, sum(rate(litellm_self_latency_bucket[$__rate_interval])) by (le))",
|
||||
"legendFormat": "self",
|
||||
"range": true,
|
||||
"refId": "L"
|
||||
"refId": "M"
|
||||
}
|
||||
],
|
||||
"title": "Service latency p95 (litellm_<service>_latency)",
|
||||
|
|
@ -5888,6 +5899,28 @@
|
|||
}
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_PROMETHEUS}"
|
||||
},
|
||||
"editorMode": "code",
|
||||
"expr": "sum(rate(litellm_anthropic_wif_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "anthropic_wif",
|
||||
"range": true,
|
||||
"refId": "A"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_PROMETHEUS}"
|
||||
},
|
||||
"editorMode": "code",
|
||||
"expr": "sum(rate(litellm_anthropic_wif_cache_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "anthropic_wif_cache",
|
||||
"range": true,
|
||||
"refId": "B"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
|
|
@ -5897,7 +5930,7 @@
|
|||
"expr": "sum(rate(litellm_auth_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "auth",
|
||||
"range": true,
|
||||
"refId": "A"
|
||||
"refId": "C"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5908,7 +5941,7 @@
|
|||
"expr": "sum(rate(litellm_batch_write_to_db_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "batch_write_to_db",
|
||||
"range": true,
|
||||
"refId": "B"
|
||||
"refId": "D"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5919,7 +5952,7 @@
|
|||
"expr": "sum(rate(litellm_postgres_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "postgres",
|
||||
"range": true,
|
||||
"refId": "C"
|
||||
"refId": "E"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5930,7 +5963,7 @@
|
|||
"expr": "sum(rate(litellm_proxy_pre_call_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "proxy_pre_call",
|
||||
"range": true,
|
||||
"refId": "D"
|
||||
"refId": "F"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5941,7 +5974,7 @@
|
|||
"expr": "sum(rate(litellm_redis_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "redis",
|
||||
"range": true,
|
||||
"refId": "E"
|
||||
"refId": "G"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5952,7 +5985,7 @@
|
|||
"expr": "sum(rate(litellm_redis_daily_org_spend_update_queue_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "redis_daily_org_spend_update_queue",
|
||||
"range": true,
|
||||
"refId": "F"
|
||||
"refId": "H"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5963,7 +5996,7 @@
|
|||
"expr": "sum(rate(litellm_redis_daily_tag_spend_update_queue_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "redis_daily_tag_spend_update_queue",
|
||||
"range": true,
|
||||
"refId": "G"
|
||||
"refId": "I"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5974,7 +6007,7 @@
|
|||
"expr": "sum(rate(litellm_redis_daily_team_spend_update_queue_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "redis_daily_team_spend_update_queue",
|
||||
"range": true,
|
||||
"refId": "H"
|
||||
"refId": "J"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5985,7 +6018,7 @@
|
|||
"expr": "sum(rate(litellm_redis_window_spend_update_queue_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "redis_window_spend_update_queue",
|
||||
"range": true,
|
||||
"refId": "I"
|
||||
"refId": "K"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -5996,7 +6029,7 @@
|
|||
"expr": "sum(rate(litellm_reset_budget_job_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "reset_budget_job",
|
||||
"range": true,
|
||||
"refId": "J"
|
||||
"refId": "L"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6007,7 +6040,7 @@
|
|||
"expr": "sum(rate(litellm_router_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "router",
|
||||
"range": true,
|
||||
"refId": "K"
|
||||
"refId": "M"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6018,7 +6051,7 @@
|
|||
"expr": "sum(rate(litellm_self_total_requests_total[$__rate_interval]))",
|
||||
"legendFormat": "self",
|
||||
"range": true,
|
||||
"refId": "L"
|
||||
"refId": "N"
|
||||
}
|
||||
],
|
||||
"title": "Service request rate (litellm_<service>_total_requests)",
|
||||
|
|
@ -6066,6 +6099,28 @@
|
|||
}
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_PROMETHEUS}"
|
||||
},
|
||||
"editorMode": "code",
|
||||
"expr": "sum(rate(litellm_anthropic_wif_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "anthropic_wif / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "A"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "${DS_PROMETHEUS}"
|
||||
},
|
||||
"editorMode": "code",
|
||||
"expr": "sum(rate(litellm_anthropic_wif_cache_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "anthropic_wif_cache / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "B"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
|
|
@ -6075,7 +6130,7 @@
|
|||
"expr": "sum(rate(litellm_auth_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "auth / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "A"
|
||||
"refId": "C"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6086,7 +6141,7 @@
|
|||
"expr": "sum(rate(litellm_batch_write_to_db_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "batch_write_to_db / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "B"
|
||||
"refId": "D"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6097,7 +6152,7 @@
|
|||
"expr": "sum(rate(litellm_postgres_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "postgres / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "C"
|
||||
"refId": "E"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6108,7 +6163,7 @@
|
|||
"expr": "sum(rate(litellm_proxy_pre_call_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "proxy_pre_call / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "D"
|
||||
"refId": "F"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6119,7 +6174,7 @@
|
|||
"expr": "sum(rate(litellm_redis_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "redis / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "E"
|
||||
"refId": "G"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6130,7 +6185,7 @@
|
|||
"expr": "sum(rate(litellm_redis_daily_org_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "redis_daily_org_spend_update_queue / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "F"
|
||||
"refId": "H"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6141,7 +6196,7 @@
|
|||
"expr": "sum(rate(litellm_redis_daily_tag_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "redis_daily_tag_spend_update_queue / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "G"
|
||||
"refId": "I"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6152,7 +6207,7 @@
|
|||
"expr": "sum(rate(litellm_redis_daily_team_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "redis_daily_team_spend_update_queue / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "H"
|
||||
"refId": "J"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6163,7 +6218,7 @@
|
|||
"expr": "sum(rate(litellm_redis_window_spend_update_queue_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "redis_window_spend_update_queue / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "I"
|
||||
"refId": "K"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6174,7 +6229,7 @@
|
|||
"expr": "sum(rate(litellm_reset_budget_job_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "reset_budget_job / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "J"
|
||||
"refId": "L"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6185,7 +6240,7 @@
|
|||
"expr": "sum(rate(litellm_router_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "router / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "K"
|
||||
"refId": "M"
|
||||
},
|
||||
{
|
||||
"datasource": {
|
||||
|
|
@ -6196,7 +6251,7 @@
|
|||
"expr": "sum(rate(litellm_self_failed_requests_total[$__rate_interval])) by (error_class)",
|
||||
"legendFormat": "self / {{error_class}}",
|
||||
"range": true,
|
||||
"refId": "L"
|
||||
"refId": "N"
|
||||
}
|
||||
],
|
||||
"title": "Service failure rate (litellm_<service>_failed_requests)",
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
# LiteLLM All Prometheus Metrics dashboard
|
||||
|
||||
Every `litellm_*` metric family the proxy can expose on `/metrics` (136 families across 97 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
|
||||
Every `litellm_*` metric family the proxy can expose on `/metrics` (141 families across 97 panels), grouped into rows: proxy traffic, latency, spend and tokens, cache, LLM API deployments, key and team rate limits, budgets, guardrails, MCP, managed files and batches, users and teams, the Redis circuit breaker, the spend log cleanup job, and the `prometheus_system` service callback metrics (per-service latency, request and failure rates, spend update queue sizes). Panel titles are the metric names so you can grep the JSON for the metric you care about
|
||||
|
||||
Import `grafana_dashboard.json` from **Dashboards > New > Import** and pick your Prometheus data source when prompted (the `DS_PROMETHEUS` variable). Counters are plotted as `rate()` over `$__rate_interval`, histograms as p50 / p95 / p99, gauges as the raw value grouped by the most useful label. Every query names the metric exactly as the proxy emits it (counters carry the `_total` suffix the Prometheus client adds), and `tests/unit/integrations/test_prometheus_metric_name_consistency.py` fails if a metric is renamed without updating this dashboard
|
||||
|
||||
|
|
|
|||
|
|
@ -526,7 +526,6 @@ async def count_prompt_tokens(
|
|||
) -> int | None:
|
||||
try:
|
||||
native: Final = _CountBody.model_validate(body)
|
||||
count_url: Final = _messages_url(model, api_key, api_base) + "/count_tokens"
|
||||
result: Final = _CountResult.model_validate(
|
||||
await _counter.handle_count_tokens_request(
|
||||
model=model,
|
||||
|
|
@ -534,7 +533,7 @@ async def count_prompt_tokens(
|
|||
tools=_count_objects(native.tools) if native.tools is not None else None,
|
||||
system=_JSON_OBJECT.validate_python(MappingProxyType({"system": native.system}))["system"],
|
||||
api_key=api_key,
|
||||
api_base=count_url,
|
||||
api_base=api_base,
|
||||
optional_params=_JSON_OBJECT.validate_python(
|
||||
MappingProxyType({key: body[key] for key in COUNT_TOKEN_OPTION_NAMES if key in body})
|
||||
),
|
||||
|
|
|
|||
|
|
@ -104,10 +104,41 @@ class BedrockBatchConnection:
|
|||
bedrock_tags: Sequence[Mapping[str, str]] | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, kw_only=True)
|
||||
class AnthropicFederationConnection:
|
||||
anthropic_federation_rule_id: str | None = None
|
||||
anthropic_organization_id: str | None = None
|
||||
anthropic_service_account_id: str | None = None
|
||||
anthropic_federation_workspace_id: str | None = None
|
||||
anthropic_identity_token_file: str | None = None
|
||||
anthropic_identity_token: str | None = None
|
||||
anthropic_identity_source: str | None = None
|
||||
anthropic_issuer_url: str | None = None
|
||||
anthropic_issuer_subject: str | None = None
|
||||
anthropic_issuer_audience: str | None = None
|
||||
anthropic_issuer_ttl_seconds: int | None = None
|
||||
anthropic_issuer_signing_key_ref: str | None = None
|
||||
anthropic_keycloak_token_url: str | None = None
|
||||
anthropic_keycloak_client_id: str | None = None
|
||||
anthropic_keycloak_auth_method: str | None = None
|
||||
anthropic_keycloak_client_secret_ref: str | None = None
|
||||
anthropic_keycloak_scope: str | None = None
|
||||
anthropic_disable_workload_identity_federation: bool | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, kw_only=True)
|
||||
class OpenAIFederationConnection:
|
||||
openai_identity_provider_id: str | None = None
|
||||
openai_service_account_id: str | None = None
|
||||
openai_identity_token_file: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, kw_only=True)
|
||||
class ConnectionSettings:
|
||||
provider: ProviderConnection
|
||||
bedrock_batch: BedrockBatchConnection
|
||||
anthropic_federation: AnthropicFederationConnection
|
||||
openai_federation: OpenAIFederationConnection
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, kw_only=True)
|
||||
|
|
|
|||
|
|
@ -3965,7 +3965,6 @@ secret_bearing_wif_litellm_params: Final = tuple(sorted(WIF_SECRET_BEARING_KEYS)
|
|||
all_litellm_params = [ # rebind-ok: two star imports in litellm/__init__.py re-bind it
|
||||
*OWNED_KWARG_NAMES,
|
||||
*KWARG_ARTIFACTS,
|
||||
*server_owned_wif_litellm_params,
|
||||
*StandardCallbackDynamicParams.__annotations__,
|
||||
*CustomPricingLiteLLMParams.model_fields,
|
||||
]
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
"""litellm_params keys that configure workload identity federation.
|
||||
|
||||
The kwargs funnel (``litellm_core_utils.get_litellm_params``) and the request-body ban list
|
||||
(``types.utils.all_litellm_params``) both derive from these sets, so they live in a module with
|
||||
no litellm imports that either side can reach without a cycle. Every key here rides the funnel
|
||||
into ``litellm_params`` and is banned from request bodies, which also covers
|
||||
The key sets derive from the federation connection leaves in ``types.litellm_params``, so the
|
||||
kwargs funnel (``litellm_core_utils.get_litellm_params``) and the request-body ban list
|
||||
(``types.utils.all_litellm_params``) read one declaration. Every key rides the funnel into
|
||||
``litellm_params`` and is banned from request bodies, which also covers
|
||||
``anthropic_disable_workload_identity_federation``: the proxy sets it when a client redirects
|
||||
``api_base`` so a federated deployment stops minting for a base the caller chose, and a caller
|
||||
must not be able to set it in either direction.
|
||||
|
|
@ -11,36 +11,11 @@ must not be able to set it in either direction.
|
|||
|
||||
from typing import Final
|
||||
|
||||
ANTHROPIC_WIF_KWARGS_KEYS: Final = frozenset(
|
||||
{
|
||||
"anthropic_federation_rule_id",
|
||||
"anthropic_organization_id",
|
||||
"anthropic_service_account_id",
|
||||
"anthropic_federation_workspace_id",
|
||||
"anthropic_identity_token_file",
|
||||
"anthropic_identity_token",
|
||||
"anthropic_identity_source",
|
||||
"anthropic_issuer_url",
|
||||
"anthropic_issuer_subject",
|
||||
"anthropic_issuer_audience",
|
||||
"anthropic_issuer_ttl_seconds",
|
||||
"anthropic_issuer_signing_key_ref",
|
||||
"anthropic_keycloak_token_url",
|
||||
"anthropic_keycloak_client_id",
|
||||
"anthropic_keycloak_auth_method",
|
||||
"anthropic_keycloak_client_secret_ref",
|
||||
"anthropic_keycloak_scope",
|
||||
"anthropic_disable_workload_identity_federation",
|
||||
}
|
||||
)
|
||||
from litellm.types.litellm_params import AnthropicFederationConnection, OpenAIFederationConnection, wire_names
|
||||
|
||||
OPENAI_WIF_KWARGS_KEYS: Final = frozenset(
|
||||
{
|
||||
"openai_identity_provider_id",
|
||||
"openai_service_account_id",
|
||||
"openai_identity_token_file",
|
||||
}
|
||||
)
|
||||
ANTHROPIC_WIF_KWARGS_KEYS: Final = frozenset(wire_names(AnthropicFederationConnection))
|
||||
|
||||
OPENAI_WIF_KWARGS_KEYS: Final = frozenset(wire_names(OpenAIFederationConnection))
|
||||
|
||||
WIF_SECRET_BEARING_KEYS: Final = frozenset(
|
||||
{
|
||||
|
|
|
|||
|
|
@ -115,7 +115,16 @@ def test_prediction_header_eligibility(headers: Mapping[str, str], supported: bo
|
|||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_count_uses_same_version_and_preserves_native_input(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
@pytest.mark.parametrize(
|
||||
"api_base, count_url",
|
||||
[
|
||||
(None, "https://api.anthropic.com/v1/messages/count_tokens"),
|
||||
("https://gateway.example/v1/messages", "https://gateway.example/v1/messages/count_tokens"),
|
||||
],
|
||||
)
|
||||
async def test_provider_count_uses_same_version_and_preserves_native_input(
|
||||
monkeypatch: pytest.MonkeyPatch, api_base: str | None, count_url: str
|
||||
) -> None:
|
||||
body: Final = _body()
|
||||
requests: Final[list[httpx.Request]] = []
|
||||
|
||||
|
|
@ -128,12 +137,12 @@ async def test_provider_count_uses_same_version_and_preserves_native_input(monke
|
|||
client.client = httpx.AsyncClient(transport=httpx.MockTransport(provider))
|
||||
monkeypatch.setattr(count_handler, "get_async_httpx_client", lambda **kwargs: client)
|
||||
try:
|
||||
assert await count_prompt_tokens(_MODEL, _KEY, body) == 311
|
||||
assert await count_prompt_tokens(_MODEL, _KEY, body, api_base=api_base) == 311
|
||||
finally:
|
||||
await client.client.aclose()
|
||||
assert len(requests) == 1
|
||||
assert requests[0].headers["anthropic-version"] == DEFAULT_ANTHROPIC_API_VERSION
|
||||
assert requests[0].url == "https://api.anthropic.com/v1/messages/count_tokens"
|
||||
assert requests[0].url == count_url
|
||||
assert json.loads(requests[0].content) == body
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -95,6 +95,27 @@ CONNECTION_NAMES: Final = (
|
|||
"s3_secret_access_key",
|
||||
"s3_encryption_key_id",
|
||||
"bedrock_tags",
|
||||
"anthropic_federation_rule_id",
|
||||
"anthropic_organization_id",
|
||||
"anthropic_service_account_id",
|
||||
"anthropic_federation_workspace_id",
|
||||
"anthropic_identity_token_file",
|
||||
"anthropic_identity_token",
|
||||
"anthropic_identity_source",
|
||||
"anthropic_issuer_url",
|
||||
"anthropic_issuer_subject",
|
||||
"anthropic_issuer_audience",
|
||||
"anthropic_issuer_ttl_seconds",
|
||||
"anthropic_issuer_signing_key_ref",
|
||||
"anthropic_keycloak_token_url",
|
||||
"anthropic_keycloak_client_id",
|
||||
"anthropic_keycloak_auth_method",
|
||||
"anthropic_keycloak_client_secret_ref",
|
||||
"anthropic_keycloak_scope",
|
||||
"anthropic_disable_workload_identity_federation",
|
||||
"openai_identity_provider_id",
|
||||
"openai_service_account_id",
|
||||
"openai_identity_token_file",
|
||||
)
|
||||
|
||||
OPTION_NAMES: Final = (
|
||||
|
|
@ -491,6 +512,12 @@ TYPE_HINT_NAMESPACE: Final[Mapping[str, object]] = {
|
|||
LEAF_SAMPLES: Final[Mapping[type, Mapping[str, object]]] = {
|
||||
litellm_params.ProviderConnection: {"api_key": "k", "request_timeout": 1.5},
|
||||
litellm_params.BedrockBatchConnection: {"aws_batch_role_arn": "arn", "bedrock_tags": ({"k": "v"},)},
|
||||
litellm_params.AnthropicFederationConnection: {
|
||||
"anthropic_federation_rule_id": "fdrl_1",
|
||||
"anthropic_issuer_ttl_seconds": 300,
|
||||
"anthropic_disable_workload_identity_federation": True,
|
||||
},
|
||||
litellm_params.OpenAIFederationConnection: {"openai_identity_provider_id": "idp_1"},
|
||||
litellm_params.DispatchOptions: {"custom_llm_provider": "openai"},
|
||||
litellm_params.RoutingOptions: {
|
||||
"fallbacks": [{"model": "gpt-4o", "api_key": "k", "temperature": 0}],
|
||||
|
|
@ -526,6 +553,8 @@ LEAF_SAMPLES: Final[Mapping[type, Mapping[str, object]]] = {
|
|||
LEAF_BAD_SAMPLES: Final[Mapping[type, Mapping[str, object]]] = {
|
||||
litellm_params.ProviderConnection: {"api_key": 1},
|
||||
litellm_params.BedrockBatchConnection: {"aws_batch_role_arn": 1},
|
||||
litellm_params.AnthropicFederationConnection: {"anthropic_issuer_ttl_seconds": "300"},
|
||||
litellm_params.OpenAIFederationConnection: {"openai_identity_provider_id": 1},
|
||||
litellm_params.DispatchOptions: {"custom_llm_provider": 1},
|
||||
litellm_params.RoutingOptions: {"num_retries": "2"},
|
||||
litellm_params.DeploymentOptions: {"rpm": "2"},
|
||||
|
|
@ -620,6 +649,24 @@ def test_routing_options_accept_every_strategy_the_router_accepts(strategy: str)
|
|||
NAMES_SHARED_WITH_TYPED_MODELS: Final[Mapping[str, tuple[str, ...]]] = MappingProxyType(
|
||||
{
|
||||
"credentials": (
|
||||
"anthropic_disable_workload_identity_federation",
|
||||
"anthropic_federation_rule_id",
|
||||
"anthropic_federation_workspace_id",
|
||||
"anthropic_identity_source",
|
||||
"anthropic_identity_token",
|
||||
"anthropic_identity_token_file",
|
||||
"anthropic_issuer_audience",
|
||||
"anthropic_issuer_signing_key_ref",
|
||||
"anthropic_issuer_subject",
|
||||
"anthropic_issuer_ttl_seconds",
|
||||
"anthropic_issuer_url",
|
||||
"anthropic_keycloak_auth_method",
|
||||
"anthropic_keycloak_client_id",
|
||||
"anthropic_keycloak_client_secret_ref",
|
||||
"anthropic_keycloak_scope",
|
||||
"anthropic_keycloak_token_url",
|
||||
"anthropic_organization_id",
|
||||
"anthropic_service_account_id",
|
||||
"api_base",
|
||||
"api_key",
|
||||
"api_version",
|
||||
|
|
@ -630,6 +677,9 @@ NAMES_SHARED_WITH_TYPED_MODELS: Final[Mapping[str, tuple[str, ...]]] = MappingPr
|
|||
"bedrock_tags",
|
||||
"client_id",
|
||||
"client_secret",
|
||||
"openai_identity_provider_id",
|
||||
"openai_identity_token_file",
|
||||
"openai_service_account_id",
|
||||
"region_name",
|
||||
"s3_access_key_id",
|
||||
"s3_bucket_name",
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue