fix: correct max_tokens and test file paths

Address Greptile review feedback:

1. Fix max_tokens: Set to 8192 (max_output_tokens) instead of 32768
   - Matches pattern of other featherless_ai models
   - Prevents users from requesting more tokens than model supports

2. Fix test file paths: Use absolute paths relative to test file
   - Prevents test failures when run from different directories
   - Matches pattern used by other model-price tests

3. Update test assertions to verify correct max_tokens value
This commit is contained in:
FadelT 2026-06-18 17:18:38 +02:00
parent ed7cccaa54
commit 29f8a5c8ae
4 changed files with 78 additions and 6 deletions

BIN
.coverage Normal file

Binary file not shown.

View file

@ -14724,14 +14724,14 @@
"litellm_provider": "featherless_ai",
"max_input_tokens": 32768,
"max_output_tokens": 8192,
"max_tokens": 32768,
"max_tokens": 8192,
"mode": "chat"
},
"featherless_ai/WeiboAI/VibeThinker-3B": {
"litellm_provider": "featherless_ai",
"max_input_tokens": 32768,
"max_output_tokens": 8192,
"max_tokens": 32768,
"max_tokens": 8192,
"mode": "chat"
},
"fireworks-ai-4.1b-to-16b": {

View file

@ -2,12 +2,18 @@
Test VibeThinker models configuration.
"""
import json
import os
import pytest
def test_vibethinker_models_in_model_prices():
"""Verify VibeThinker models are correctly configured in model_prices_and_context_window.json."""
with open("model_prices_and_context_window.json", "r") as f:
# Use absolute path relative to test file
model_prices_path = os.path.join(
os.path.dirname(os.path.dirname(os.path.dirname(__file__))),
"model_prices_and_context_window.json"
)
with open(model_prices_path, "r") as f:
model_prices = json.load(f)
models = [
@ -27,14 +33,19 @@ def test_vibethinker_models_in_model_prices():
assert "max_input_tokens" in config, f"{model}: missing max_input_tokens"
assert "max_output_tokens" in config, f"{model}: missing max_output_tokens"
# Verify context window (Qwen base models have 32K context)
# Verify context window (Qwen base models have 32K input, 8K output)
assert config["max_input_tokens"] == 32768, f"{model}: incorrect max_input_tokens"
assert config["max_tokens"] == 32768, f"{model}: incorrect max_tokens"
assert config["max_tokens"] == 8192, f"{model}: incorrect max_tokens (should match max_output_tokens)"
def test_vibethinker_provider_consistency():
"""Verify VibeThinker models use the same provider pattern as other Featherless AI models."""
with open("model_prices_and_context_window.json", "r") as f:
# Use absolute path relative to test file
model_prices_path = os.path.join(
os.path.dirname(os.path.dirname(os.path.dirname(__file__))),
"model_prices_and_context_window.json"
)
with open(model_prices_path, "r") as f:
model_prices = json.load(f)
# Get all featherless_ai models

61
verify_uk_pii_proxy.sh Executable file
View file

@ -0,0 +1,61 @@
#!/bin/bash
# Manual verification script for UK PII entity types in litellm proxy
# Tests that UK entities are properly recognized and masked by Presidio
set -e
echo "=== UK PII Entity Types Verification Script ==="
echo ""
echo "This script verifies that UK_PASSPORT, UK_POSTCODE, and UK_VEHICLE_REGISTRATION"
echo "entity types are properly recognized by the Presidio guardrail."
echo ""
CONFIG_FILE="/tmp/uk_pii_test_config.yaml"
cat > "$CONFIG_FILE" <<'EOF'
model_list:
- model_name: gpt-3.5-turbo
litellm_params:
model: gpt-3.5-turbo
api_key: os.environ/OPENAI_API_KEY
guardrails:
- guardrail_name: "uk-pii-test"
litellm_params:
guardrail: presidio
mode: pre_call
default_on: true
pii_entities:
UK_NHS: MASK
UK_NINO: MASK
UK_PASSPORT: MASK
UK_POSTCODE: MASK
UK_VEHICLE_REGISTRATION: MASK
EOF
echo "Created test config at: $CONFIG_FILE"
echo ""
echo "Starting litellm proxy on port 4000 with UK PII guardrail enabled..."
echo ""
echo "Run the following commands in separate terminals to test:"
echo ""
echo "# Terminal 1: Start proxy"
echo "python litellm/proxy/proxy_cli.py --config $CONFIG_FILE --detailed_debug"
echo ""
echo "# Terminal 2: Test UK_PASSPORT"
echo "curl -X POST http://localhost:4000/chat/completions \\"
echo " -H 'Content-Type: application/json' \\"
echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"My passport is 012345678\"}]}' | jq"
echo ""
echo "# Terminal 2: Test UK_POSTCODE"
echo "curl -X POST http://localhost:4000/chat/completions \\"
echo " -H 'Content-Type: application/json' \\"
echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"I live at SW1A 1AA\"}]}' | jq"
echo ""
echo "# Terminal 2: Test UK_VEHICLE_REGISTRATION"
echo "curl -X POST http://localhost:4000/chat/completions \\"
echo " -H 'Content-Type: application/json' \\"
echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"My car registration is AB12 CDE\"}]}' | jq"
echo ""
echo "Expected: The sensitive UK data should be masked (replaced with <UK_PASSPORT>, <UK_POSTCODE>, <UK_VEHICLE_REGISTRATION>)"
echo " in the request before being sent to OpenAI."