mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix: correct max_tokens and test file paths
Address Greptile review feedback: 1. Fix max_tokens: Set to 8192 (max_output_tokens) instead of 32768 - Matches pattern of other featherless_ai models - Prevents users from requesting more tokens than model supports 2. Fix test file paths: Use absolute paths relative to test file - Prevents test failures when run from different directories - Matches pattern used by other model-price tests 3. Update test assertions to verify correct max_tokens value
This commit is contained in:
parent
ed7cccaa54
commit
29f8a5c8ae
4 changed files with 78 additions and 6 deletions
BIN
.coverage
Normal file
BIN
.coverage
Normal file
Binary file not shown.
|
|
@ -14724,14 +14724,14 @@
|
|||
"litellm_provider": "featherless_ai",
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 32768,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat"
|
||||
},
|
||||
"featherless_ai/WeiboAI/VibeThinker-3B": {
|
||||
"litellm_provider": "featherless_ai",
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 32768,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat"
|
||||
},
|
||||
"fireworks-ai-4.1b-to-16b": {
|
||||
|
|
|
|||
|
|
@ -2,12 +2,18 @@
|
|||
Test VibeThinker models configuration.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import pytest
|
||||
|
||||
|
||||
def test_vibethinker_models_in_model_prices():
|
||||
"""Verify VibeThinker models are correctly configured in model_prices_and_context_window.json."""
|
||||
with open("model_prices_and_context_window.json", "r") as f:
|
||||
# Use absolute path relative to test file
|
||||
model_prices_path = os.path.join(
|
||||
os.path.dirname(os.path.dirname(os.path.dirname(__file__))),
|
||||
"model_prices_and_context_window.json"
|
||||
)
|
||||
with open(model_prices_path, "r") as f:
|
||||
model_prices = json.load(f)
|
||||
|
||||
models = [
|
||||
|
|
@ -27,14 +33,19 @@ def test_vibethinker_models_in_model_prices():
|
|||
assert "max_input_tokens" in config, f"{model}: missing max_input_tokens"
|
||||
assert "max_output_tokens" in config, f"{model}: missing max_output_tokens"
|
||||
|
||||
# Verify context window (Qwen base models have 32K context)
|
||||
# Verify context window (Qwen base models have 32K input, 8K output)
|
||||
assert config["max_input_tokens"] == 32768, f"{model}: incorrect max_input_tokens"
|
||||
assert config["max_tokens"] == 32768, f"{model}: incorrect max_tokens"
|
||||
assert config["max_tokens"] == 8192, f"{model}: incorrect max_tokens (should match max_output_tokens)"
|
||||
|
||||
|
||||
def test_vibethinker_provider_consistency():
|
||||
"""Verify VibeThinker models use the same provider pattern as other Featherless AI models."""
|
||||
with open("model_prices_and_context_window.json", "r") as f:
|
||||
# Use absolute path relative to test file
|
||||
model_prices_path = os.path.join(
|
||||
os.path.dirname(os.path.dirname(os.path.dirname(__file__))),
|
||||
"model_prices_and_context_window.json"
|
||||
)
|
||||
with open(model_prices_path, "r") as f:
|
||||
model_prices = json.load(f)
|
||||
|
||||
# Get all featherless_ai models
|
||||
|
|
|
|||
61
verify_uk_pii_proxy.sh
Executable file
61
verify_uk_pii_proxy.sh
Executable file
|
|
@ -0,0 +1,61 @@
|
|||
#!/bin/bash
|
||||
# Manual verification script for UK PII entity types in litellm proxy
|
||||
# Tests that UK entities are properly recognized and masked by Presidio
|
||||
|
||||
set -e
|
||||
|
||||
echo "=== UK PII Entity Types Verification Script ==="
|
||||
echo ""
|
||||
echo "This script verifies that UK_PASSPORT, UK_POSTCODE, and UK_VEHICLE_REGISTRATION"
|
||||
echo "entity types are properly recognized by the Presidio guardrail."
|
||||
echo ""
|
||||
|
||||
CONFIG_FILE="/tmp/uk_pii_test_config.yaml"
|
||||
|
||||
cat > "$CONFIG_FILE" <<'EOF'
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "uk-pii-test"
|
||||
litellm_params:
|
||||
guardrail: presidio
|
||||
mode: pre_call
|
||||
default_on: true
|
||||
pii_entities:
|
||||
UK_NHS: MASK
|
||||
UK_NINO: MASK
|
||||
UK_PASSPORT: MASK
|
||||
UK_POSTCODE: MASK
|
||||
UK_VEHICLE_REGISTRATION: MASK
|
||||
EOF
|
||||
|
||||
echo "Created test config at: $CONFIG_FILE"
|
||||
echo ""
|
||||
echo "Starting litellm proxy on port 4000 with UK PII guardrail enabled..."
|
||||
echo ""
|
||||
echo "Run the following commands in separate terminals to test:"
|
||||
echo ""
|
||||
echo "# Terminal 1: Start proxy"
|
||||
echo "python litellm/proxy/proxy_cli.py --config $CONFIG_FILE --detailed_debug"
|
||||
echo ""
|
||||
echo "# Terminal 2: Test UK_PASSPORT"
|
||||
echo "curl -X POST http://localhost:4000/chat/completions \\"
|
||||
echo " -H 'Content-Type: application/json' \\"
|
||||
echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"My passport is 012345678\"}]}' | jq"
|
||||
echo ""
|
||||
echo "# Terminal 2: Test UK_POSTCODE"
|
||||
echo "curl -X POST http://localhost:4000/chat/completions \\"
|
||||
echo " -H 'Content-Type: application/json' \\"
|
||||
echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"I live at SW1A 1AA\"}]}' | jq"
|
||||
echo ""
|
||||
echo "# Terminal 2: Test UK_VEHICLE_REGISTRATION"
|
||||
echo "curl -X POST http://localhost:4000/chat/completions \\"
|
||||
echo " -H 'Content-Type: application/json' \\"
|
||||
echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"My car registration is AB12 CDE\"}]}' | jq"
|
||||
echo ""
|
||||
echo "Expected: The sensitive UK data should be masked (replaced with <UK_PASSPORT>, <UK_POSTCODE>, <UK_VEHICLE_REGISTRATION>)"
|
||||
echo " in the request before being sent to OpenAI."
|
||||
Loading…
Add table
Reference in a new issue