mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Litellm ishaan april2 (#25113)
* feat: add brave/search to model_prices_and_context_window.json (#25042) Brave Search is supported by litellm as a search provider (documented at docs.litellm.ai/docs/search/brave and listed in provider_endpoints_support.json) but was missing from model_prices_and_context_window.json, making it invisible to any code that discovers search providers from litellm.model_cost. Cost: $0.005/query ($5 per 1,000 requests) per https://brave.com/search/api/ * feat(models): add NVIDIA Nemotron 3 Super 120B on Bedrock (#24588) * feat(models): add NVIDIA Nemotron 3 Super 120B on Bedrock Add model definition for nvidia.nemotron-3-super-120b-a12b-v1 via Bedrock Converse API with pricing, context window (256k/32k), and capability flags (function calling, tool choice, system messages). * fix model ID to nvidia.nemotron-super-3-120b + add tests Correct the Bedrock model ID from nvidia.nemotron-3-super-120b-a12b-v1 (NVIDIA's internal name) to nvidia.nemotron-super-3-120b (the actual AWS Bedrock programmatic model ID). Add unit tests verifying model resolution, pricing, and context window. * fix(proxy): allow JWT auth for /v1/mcp/server sub-paths (#24698) mcp_routes only contained "/v1/mcp/server" (exact match). Starlette's compile_path produces an end-anchored regex, so sub-paths like /register, /health, /submissions, /oauth/* all failed the JWT allowed_routes_check. Add a {path:path} wildcard entry so all sub-paths are covered. --------- Co-authored-by: Daniel Yudelevich <4537920+yudelevi@users.noreply.github.com> Co-authored-by: michelligabriele <gabriele.michelli@icloud.com>
This commit is contained in:
parent
24886394b3
commit
4c06e4379b
4 changed files with 106 additions and 0 deletions
|
|
@ -432,6 +432,7 @@ class LiteLLMRoutes(enum.Enum):
|
|||
"/mcp-rest/tools/list",
|
||||
"/mcp-rest/tools/call",
|
||||
"/v1/mcp/server",
|
||||
"/v1/mcp/server/{path:path}",
|
||||
]
|
||||
|
||||
agent_routes = [
|
||||
|
|
|
|||
51
tests/litellm/test_bedrock_nemotron_super.py
Normal file
51
tests/litellm/test_bedrock_nemotron_super.py
Normal file
|
|
@ -0,0 +1,51 @@
|
|||
"""
|
||||
Test suite for NVIDIA Nemotron Super 3 120B on AWS Bedrock
|
||||
Verifies model configuration, pricing, and regional availability.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "true"
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm import get_model_info
|
||||
|
||||
|
||||
MODEL_NAME = "nvidia.nemotron-super-3-120b"
|
||||
|
||||
|
||||
class TestNemotronSuper3120B:
|
||||
"""Test model definition for nvidia.nemotron-super-3-120b"""
|
||||
|
||||
def test_model_info_primary_region(self):
|
||||
"""Test model resolves in us-east-1"""
|
||||
model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}")
|
||||
|
||||
assert model_info is not None, f"Model {MODEL_NAME} not found"
|
||||
assert model_info["max_input_tokens"] == 256000
|
||||
assert model_info["max_output_tokens"] == 32000
|
||||
assert model_info["litellm_provider"] == "bedrock_converse"
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model_info["supports_function_calling"] is True
|
||||
|
||||
def test_pricing_configured(self):
|
||||
"""Verify pricing matches AWS Bedrock rates"""
|
||||
model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}")
|
||||
|
||||
assert model_info["input_cost_per_token"] == 1.5e-07
|
||||
assert model_info["output_cost_per_token"] == 6.5e-07
|
||||
|
||||
def test_context_window(self):
|
||||
"""Nemotron Super 3 120B has 256K input, 32K output on Bedrock"""
|
||||
model_info = get_model_info(f"bedrock/us-east-1/{MODEL_NAME}")
|
||||
|
||||
assert model_info["max_input_tokens"] == 256000
|
||||
assert model_info["max_output_tokens"] == 32000
|
||||
|
||||
def test_resolves_without_region(self):
|
||||
"""Test model resolves with just bedrock/ prefix"""
|
||||
model_info = get_model_info(f"bedrock/{MODEL_NAME}")
|
||||
|
||||
assert model_info is not None, f"Model {MODEL_NAME} not found without region"
|
||||
assert model_info["max_input_tokens"] == 256000
|
||||
|
|
@ -159,6 +159,32 @@ async def test_mcp_route_check_passes_for_team():
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_mcp_route_check_passes_for_team_server_subpaths():
|
||||
"""
|
||||
Verify that allowed_routes_check returns True for /v1/mcp/server sub-paths with default settings.
|
||||
Regression test for JWT users accessing /v1/mcp/server/register and similar endpoints.
|
||||
"""
|
||||
from litellm.proxy._types import LitellmUserRoles
|
||||
from litellm.proxy.auth.auth_checks import allowed_routes_check
|
||||
|
||||
jwt_auth = LiteLLM_JWTAuth()
|
||||
|
||||
for route in [
|
||||
"/v1/mcp/server/register",
|
||||
"/v1/mcp/server/health",
|
||||
"/v1/mcp/server/abc/approve",
|
||||
]:
|
||||
is_allowed = allowed_routes_check(
|
||||
user_role=LitellmUserRoles.TEAM,
|
||||
user_route=route,
|
||||
litellm_proxy_roles=jwt_auth,
|
||||
)
|
||||
assert is_allowed is True, (
|
||||
f"Route {route} should be allowed for TEAM role with default settings"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_e2e_jwt_team_mcp_permissions_enforced(monkeypatch):
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -124,6 +124,34 @@ def test_virtual_key_mcp_routes_allows_v1_mcp_server():
|
|||
assert result is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"route",
|
||||
[
|
||||
"/v1/mcp/server/register",
|
||||
"/v1/mcp/server/health",
|
||||
"/v1/mcp/server/submissions",
|
||||
"/v1/mcp/server/abc123",
|
||||
"/v1/mcp/server/abc123/approve",
|
||||
"/v1/mcp/server/oauth/session",
|
||||
"/v1/mcp/server/oauth/abc123/authorize",
|
||||
],
|
||||
)
|
||||
def test_virtual_key_mcp_routes_allows_v1_mcp_server_subpaths(route):
|
||||
"""Regression test: mcp_routes must allow /v1/mcp/server sub-paths (register, health, oauth, etc.)."""
|
||||
|
||||
valid_token = UserAPIKeyAuth(
|
||||
user_id="test_user",
|
||||
allowed_routes=["mcp_routes"],
|
||||
)
|
||||
|
||||
result = RouteChecks.is_virtual_key_allowed_to_call_route(
|
||||
route=route,
|
||||
valid_token=valid_token,
|
||||
)
|
||||
|
||||
assert result is True
|
||||
|
||||
|
||||
def test_virtual_key_allowed_routes_with_litellm_routes_member_name_denied():
|
||||
"""Test that virtual key is denied when route is not in the allowed LiteLLMRoutes group"""
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue