From 4c06e4379bca30b29b57e4a31e527893cd22f93d Mon Sep 17 00:00:00 2001 From: ishaan-berri <155045088+ishaan-berri@users.noreply.github.com> Date: Sat, 4 Apr 2026 12:31:49 -0700 Subject: [PATCH] Litellm ishaan april2 (#25113) * feat: add brave/search to model_prices_and_context_window.json (#25042) Brave Search is supported by litellm as a search provider (documented at docs.litellm.ai/docs/search/brave and listed in provider_endpoints_support.json) but was missing from model_prices_and_context_window.json, making it invisible to any code that discovers search providers from litellm.model_cost. Cost: $0.005/query ($5 per 1,000 requests) per https://brave.com/search/api/ * feat(models): add NVIDIA Nemotron 3 Super 120B on Bedrock (#24588) * feat(models): add NVIDIA Nemotron 3 Super 120B on Bedrock Add model definition for nvidia.nemotron-3-super-120b-a12b-v1 via Bedrock Converse API with pricing, context window (256k/32k), and capability flags (function calling, tool choice, system messages). * fix model ID to nvidia.nemotron-super-3-120b + add tests Correct the Bedrock model ID from nvidia.nemotron-3-super-120b-a12b-v1 (NVIDIA's internal name) to nvidia.nemotron-super-3-120b (the actual AWS Bedrock programmatic model ID). Add unit tests verifying model resolution, pricing, and context window. * fix(proxy): allow JWT auth for /v1/mcp/server sub-paths (#24698) mcp_routes only contained "/v1/mcp/server" (exact match). Starlette's compile_path produces an end-anchored regex, so sub-paths like /register, /health, /submissions, /oauth/* all failed the JWT allowed_routes_check. Add a {path:path} wildcard entry so all sub-paths are covered. --------- Co-authored-by: Daniel Yudelevich <4537920+yudelevi@users.noreply.github.com> Co-authored-by: michelligabriele --- litellm/proxy/_types.py | 1 + tests/litellm/test_bedrock_nemotron_super.py | 51 +++++++++++++++++++ .../mcp_server/test_jwt_mcp_enforcement.py | 26 ++++++++++ .../proxy/auth/test_route_checks.py | 28 ++++++++++ 4 files changed, 106 insertions(+) create mode 100644 tests/litellm/test_bedrock_nemotron_super.py diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 0fad0817702..07f3ee64d25 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -432,6 +432,7 @@ class LiteLLMRoutes(enum.Enum): "/mcp-rest/tools/list", "/mcp-rest/tools/call", "/v1/mcp/server", + "/v1/mcp/server/{path:path}", ] agent_routes = [ diff --git a/tests/litellm/test_bedrock_nemotron_super.py b/tests/litellm/test_bedrock_nemotron_super.py new file mode 100644 index 00000000000..8b081f10d1d --- /dev/null +++ b/tests/litellm/test_bedrock_nemotron_super.py @@ -0,0 +1,51 @@ +""" +Test suite for NVIDIA Nemotron Super 3 120B on AWS Bedrock +Verifies model configuration, pricing, and regional availability. +""" + +import os + +os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "true" + +import pytest + +from litellm import get_model_info + + +MODEL_NAME = "nvidia.nemotron-super-3-120b" + + +class TestNemotronSuper3120B: + """Test model definition for nvidia.nemotron-super-3-120b""" + + def test_model_info_primary_region(self): + """Test model resolves in us-east-1""" + model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}") + + assert model_info is not None, f"Model {MODEL_NAME} not found" + assert model_info["max_input_tokens"] == 256000 + assert model_info["max_output_tokens"] == 32000 + assert model_info["litellm_provider"] == "bedrock_converse" + assert model_info["mode"] == "chat" + assert model_info["supports_function_calling"] is True + + def test_pricing_configured(self): + """Verify pricing matches AWS Bedrock rates""" + model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}") + + assert model_info["input_cost_per_token"] == 1.5e-07 + assert model_info["output_cost_per_token"] == 6.5e-07 + + def test_context_window(self): + """Nemotron Super 3 120B has 256K input, 32K output on Bedrock""" + model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}") + + assert model_info["max_input_tokens"] == 256000 + assert model_info["max_output_tokens"] == 32000 + + def test_resolves_without_region(self): + """Test model resolves with just bedrock/ prefix""" + model_info = get_model_info(f"bedrock/{MODEL_NAME}") + + assert model_info is not None, f"Model {MODEL_NAME} not found without region" + assert model_info["max_input_tokens"] == 256000 diff --git a/tests/test_litellm/proxy/_experimental/mcp_server/test_jwt_mcp_enforcement.py b/tests/test_litellm/proxy/_experimental/mcp_server/test_jwt_mcp_enforcement.py index e93638df441..cc9d45c05b2 100644 --- a/tests/test_litellm/proxy/_experimental/mcp_server/test_jwt_mcp_enforcement.py +++ b/tests/test_litellm/proxy/_experimental/mcp_server/test_jwt_mcp_enforcement.py @@ -159,6 +159,32 @@ async def test_mcp_route_check_passes_for_team(): ) +@pytest.mark.asyncio +async def test_mcp_route_check_passes_for_team_server_subpaths(): + """ + Verify that allowed_routes_check returns True for /v1/mcp/server sub-paths with default settings. + Regression test for JWT users accessing /v1/mcp/server/register and similar endpoints. + """ + from litellm.proxy._types import LitellmUserRoles + from litellm.proxy.auth.auth_checks import allowed_routes_check + + jwt_auth = LiteLLM_JWTAuth() + + for route in [ + "/v1/mcp/server/register", + "/v1/mcp/server/health", + "/v1/mcp/server/abc/approve", + ]: + is_allowed = allowed_routes_check( + user_role=LitellmUserRoles.TEAM, + user_route=route, + litellm_proxy_roles=jwt_auth, + ) + assert is_allowed is True, ( + f"Route {route} should be allowed for TEAM role with default settings" + ) + + @pytest.mark.asyncio async def test_e2e_jwt_team_mcp_permissions_enforced(monkeypatch): """ diff --git a/tests/test_litellm/proxy/auth/test_route_checks.py b/tests/test_litellm/proxy/auth/test_route_checks.py index 83703cd4edd..f1344a302d7 100644 --- a/tests/test_litellm/proxy/auth/test_route_checks.py +++ b/tests/test_litellm/proxy/auth/test_route_checks.py @@ -124,6 +124,34 @@ def test_virtual_key_mcp_routes_allows_v1_mcp_server(): assert result is True +@pytest.mark.parametrize( + "route", + [ + "/v1/mcp/server/register", + "/v1/mcp/server/health", + "/v1/mcp/server/submissions", + "/v1/mcp/server/abc123", + "/v1/mcp/server/abc123/approve", + "/v1/mcp/server/oauth/session", + "/v1/mcp/server/oauth/abc123/authorize", + ], +) +def test_virtual_key_mcp_routes_allows_v1_mcp_server_subpaths(route): + """Regression test: mcp_routes must allow /v1/mcp/server sub-paths (register, health, oauth, etc.).""" + + valid_token = UserAPIKeyAuth( + user_id="test_user", + allowed_routes=["mcp_routes"], + ) + + result = RouteChecks.is_virtual_key_allowed_to_call_route( + route=route, + valid_token=valid_token, + ) + + assert result is True + + def test_virtual_key_allowed_routes_with_litellm_routes_member_name_denied(): """Test that virtual key is denied when route is not in the allowed LiteLLMRoutes group"""