Litellm ishaan april2 (#25113)

* feat: add brave/search to model_prices_and_context_window.json (#25042)

Brave Search is supported by litellm as a search provider (documented at
docs.litellm.ai/docs/search/brave and listed in provider_endpoints_support.json)
but was missing from model_prices_and_context_window.json, making it invisible
to any code that discovers search providers from litellm.model_cost.

Cost: $0.005/query ($5 per 1,000 requests) per https://brave.com/search/api/

* feat(models): add NVIDIA Nemotron 3 Super 120B on Bedrock (#24588)

* feat(models): add NVIDIA Nemotron 3 Super 120B on Bedrock

Add model definition for nvidia.nemotron-3-super-120b-a12b-v1 via
Bedrock Converse API with pricing, context window (256k/32k), and
capability flags (function calling, tool choice, system messages).

* fix model ID to nvidia.nemotron-super-3-120b + add tests

Correct the Bedrock model ID from nvidia.nemotron-3-super-120b-a12b-v1
(NVIDIA's internal name) to nvidia.nemotron-super-3-120b (the actual
AWS Bedrock programmatic model ID). Add unit tests verifying model
resolution, pricing, and context window.

* fix(proxy): allow JWT auth for /v1/mcp/server sub-paths (#24698)

mcp_routes only contained "/v1/mcp/server" (exact match). Starlette's
compile_path produces an end-anchored regex, so sub-paths like
/register, /health, /submissions, /oauth/* all failed the JWT
allowed_routes_check. Add a {path:path} wildcard entry so all
sub-paths are covered.

---------

Co-authored-by: Daniel Yudelevich <4537920+yudelevi@users.noreply.github.com>
Co-authored-by: michelligabriele <gabriele.michelli@icloud.com>
This commit is contained in:
ishaan-berri 2026-04-04 12:31:49 -07:00 • committed by GitHub
parent 24886394b3
commit 4c06e4379b
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 106 additions and 0 deletions

View file

@ -432,6 +432,7 @@ class LiteLLMRoutes(enum.Enum):
"/mcp-rest/tools/list",
"/mcp-rest/tools/call",
"/v1/mcp/server",
"/v1/mcp/server/{path:path}",
]
agent_routes = [

View file

@ -0,0 +1,51 @@
"""
Test suite for NVIDIA Nemotron Super 3 120B on AWS Bedrock
Verifies model configuration, pricing, and regional availability.
"""
import os
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "true"
import pytest
from litellm import get_model_info
MODEL_NAME = "nvidia.nemotron-super-3-120b"
class TestNemotronSuper3120B:
"""Test model definition for nvidia.nemotron-super-3-120b"""
def test_model_info_primary_region(self):
"""Test model resolves in us-east-1"""
model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}")
assert model_info is not None, f"Model {MODEL_NAME} not found"
assert model_info["max_input_tokens"] == 256000
assert model_info["max_output_tokens"] == 32000
assert model_info["litellm_provider"] == "bedrock_converse"
assert model_info["mode"] == "chat"
assert model_info["supports_function_calling"] is True
def test_pricing_configured(self):
"""Verify pricing matches AWS Bedrock rates"""
model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}")
assert model_info["input_cost_per_token"] == 1.5e-07
assert model_info["output_cost_per_token"] == 6.5e-07
def test_context_window(self):
"""Nemotron Super 3 120B has 256K input, 32K output on Bedrock"""
model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}")
assert model_info["max_input_tokens"] == 256000
assert model_info["max_output_tokens"] == 32000
def test_resolves_without_region(self):
"""Test model resolves with just bedrock/ prefix"""
model_info = get_model_info(f"bedrock/{MODEL_NAME}")
assert model_info is not None, f"Model {MODEL_NAME} not found without region"
assert model_info["max_input_tokens"] == 256000

View file

@ -159,6 +159,32 @@ async def test_mcp_route_check_passes_for_team():
)
@pytest.mark.asyncio
async def test_mcp_route_check_passes_for_team_server_subpaths():
"""
Verify that allowed_routes_check returns True for /v1/mcp/server sub-paths with default settings.
Regression test for JWT users accessing /v1/mcp/server/register and similar endpoints.
"""
from litellm.proxy._types import LitellmUserRoles
from litellm.proxy.auth.auth_checks import allowed_routes_check
jwt_auth = LiteLLM_JWTAuth()
for route in [
"/v1/mcp/server/register",
"/v1/mcp/server/health",
"/v1/mcp/server/abc/approve",
]:
is_allowed = allowed_routes_check(
user_role=LitellmUserRoles.TEAM,
user_route=route,
litellm_proxy_roles=jwt_auth,
)
assert is_allowed is True, (
f"Route {route} should be allowed for TEAM role with default settings"
)
@pytest.mark.asyncio
async def test_e2e_jwt_team_mcp_permissions_enforced(monkeypatch):
"""

View file

@ -124,6 +124,34 @@ def test_virtual_key_mcp_routes_allows_v1_mcp_server():
assert result is True
@pytest.mark.parametrize(
"route",
[
"/v1/mcp/server/register",
"/v1/mcp/server/health",
"/v1/mcp/server/submissions",
"/v1/mcp/server/abc123",
"/v1/mcp/server/abc123/approve",
"/v1/mcp/server/oauth/session",
"/v1/mcp/server/oauth/abc123/authorize",
],
)
def test_virtual_key_mcp_routes_allows_v1_mcp_server_subpaths(route):
"""Regression test: mcp_routes must allow /v1/mcp/server sub-paths (register, health, oauth, etc.)."""
valid_token = UserAPIKeyAuth(
user_id="test_user",
allowed_routes=["mcp_routes"],
)
result = RouteChecks.is_virtual_key_allowed_to_call_route(
route=route,
valid_token=valid_token,
)
assert result is True
def test_virtual_key_allowed_routes_with_litellm_routes_member_name_denied():
"""Test that virtual key is denied when route is not in the allowed LiteLLMRoutes group"""