From 29f8a5c8ae0c13dc3a0e6ba7369387fefebd9383 Mon Sep 17 00:00:00 2001 From: FadelT Date: Thu, 18 Jun 2026 17:18:38 +0200 Subject: [PATCH] fix: correct max_tokens and test file paths Address Greptile review feedback: 1. Fix max_tokens: Set to 8192 (max_output_tokens) instead of 32768 - Matches pattern of other featherless_ai models - Prevents users from requesting more tokens than model supports 2. Fix test file paths: Use absolute paths relative to test file - Prevents test failures when run from different directories - Matches pattern used by other model-price tests 3. Update test assertions to verify correct max_tokens value --- .coverage | Bin 0 -> 53248 bytes model_prices_and_context_window.json | 4 +- tests/test_litellm/test_vibethinker_models.py | 19 ++++-- verify_uk_pii_proxy.sh | 61 ++++++++++++++++++ 4 files changed, 78 insertions(+), 6 deletions(-) create mode 100644 .coverage create mode 100755 verify_uk_pii_proxy.sh diff --git a/.coverage b/.coverage new file mode 100644 index 0000000000000000000000000000000000000000..ad8a392e831d16d7b2b55d0322675841f7722546 GIT binary patch literal 53248 zcmeI)%WoS+90%~-y7k76BL{`Z3OQsh$g%3USRnB_K!8Y8RDwc4oN&kX*k17NI{V1u zfRIb2BK`*c#!4JGapCuy+4b5$EE!x_}J=5P$##3L`LoTyq+Gd-l^`W7(goIFthwn)j{GzdboVJrSqJpFKJe z=A1aJ3fhj2#IXp23o%on7`am=Tz}{eWbFE55l>W>I*HVf*DN|ms}?8h=={9bbBB~G zR%0q+7P_7cuf#9vs;v{GcZ=0!Y)()i${qWhLL8>qTOw2=6)JzAB2$Y)ci5`7e*dlN zG?^m;Ep3^{yV9H)Ik%|EmH6&iq85$c z7J1IL4#N-Hf`gzQ@b+f8@!^hbQLj{&(b<$fI`w1{2l}{2jo<6iTkY8!8&2ckzAlH@<0!nHFAmN20FN zt&|%l2kW}dq6_r8i{0vcuk1AT_wA<@-Hp=LN`E;kT_|hGTc*+MlBZ;pgxN&$vPm+_ zk$Ght)&Q<7j`MvyX`;M;iA0@5t(WK9>(n^*2OkU-$)YZ|soYHJE zlWs^umgZoy0tikxl$2QZgCy4dmk)LK~wODCt{{Y?->E-kCPY(*bs zXGL=&Bc6q#rsxO3R7u~g4V2?yME7(fq3dvSRr(|^Bf8=;#^+64r)+WIY5tT3T{gaG z$Md8=OxG)3hE1t7_w(T7`bFQQKZp3GuR5AUq>tZNtZPWCm;()1B&xpOi2 zk|7_OJMfHXn!Ru6vQ+5@q-u-5;M?_oE&9O*0SG_<0uX=z1Rwwb2tWV=5P(4O1WNWD z+cDq&Us&}Q#S0J-K>z{}fB*y_009U<00Izz00jP@0^4?Je=q$z0PlCY?{(g(5`5P$##AOHafKmY;|fB*y_009UTNuXNVucUtpp#OpYc9DWZED(SI1Rwwb2tWV= z5P$##AOHafTo>Ty|MZ{!VS@k!AOHafKmY;|fB*y_009Unc>Z7fHb#UHfB*y_009U< z00Izz00bZa0X+XlAAkS^AOHafKmY;|fB*y_009UTUjWbli{HkG5CRZ@00bZa0SG_< L0uX=z1R(GqsZJkp literal 0 HcmV?d00001 diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 53a4149b9c9..b69310f26eb 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -14724,14 +14724,14 @@ "litellm_provider": "featherless_ai", "max_input_tokens": 32768, "max_output_tokens": 8192, - "max_tokens": 32768, + "max_tokens": 8192, "mode": "chat" }, "featherless_ai/WeiboAI/VibeThinker-3B": { "litellm_provider": "featherless_ai", "max_input_tokens": 32768, "max_output_tokens": 8192, - "max_tokens": 32768, + "max_tokens": 8192, "mode": "chat" }, "fireworks-ai-4.1b-to-16b": { diff --git a/tests/test_litellm/test_vibethinker_models.py b/tests/test_litellm/test_vibethinker_models.py index 2f5873d41c1..ba0e6c0ac98 100644 --- a/tests/test_litellm/test_vibethinker_models.py +++ b/tests/test_litellm/test_vibethinker_models.py @@ -2,12 +2,18 @@ Test VibeThinker models configuration. """ import json +import os import pytest def test_vibethinker_models_in_model_prices(): """Verify VibeThinker models are correctly configured in model_prices_and_context_window.json.""" - with open("model_prices_and_context_window.json", "r") as f: + # Use absolute path relative to test file + model_prices_path = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(__file__))), + "model_prices_and_context_window.json" + ) + with open(model_prices_path, "r") as f: model_prices = json.load(f) models = [ @@ -27,14 +33,19 @@ def test_vibethinker_models_in_model_prices(): assert "max_input_tokens" in config, f"{model}: missing max_input_tokens" assert "max_output_tokens" in config, f"{model}: missing max_output_tokens" - # Verify context window (Qwen base models have 32K context) + # Verify context window (Qwen base models have 32K input, 8K output) assert config["max_input_tokens"] == 32768, f"{model}: incorrect max_input_tokens" - assert config["max_tokens"] == 32768, f"{model}: incorrect max_tokens" + assert config["max_tokens"] == 8192, f"{model}: incorrect max_tokens (should match max_output_tokens)" def test_vibethinker_provider_consistency(): """Verify VibeThinker models use the same provider pattern as other Featherless AI models.""" - with open("model_prices_and_context_window.json", "r") as f: + # Use absolute path relative to test file + model_prices_path = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(__file__))), + "model_prices_and_context_window.json" + ) + with open(model_prices_path, "r") as f: model_prices = json.load(f) # Get all featherless_ai models diff --git a/verify_uk_pii_proxy.sh b/verify_uk_pii_proxy.sh new file mode 100755 index 00000000000..72d239ab418 --- /dev/null +++ b/verify_uk_pii_proxy.sh @@ -0,0 +1,61 @@ +#!/bin/bash +# Manual verification script for UK PII entity types in litellm proxy +# Tests that UK entities are properly recognized and masked by Presidio + +set -e + +echo "=== UK PII Entity Types Verification Script ===" +echo "" +echo "This script verifies that UK_PASSPORT, UK_POSTCODE, and UK_VEHICLE_REGISTRATION" +echo "entity types are properly recognized by the Presidio guardrail." +echo "" + +CONFIG_FILE="/tmp/uk_pii_test_config.yaml" + +cat > "$CONFIG_FILE" <<'EOF' +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +guardrails: + - guardrail_name: "uk-pii-test" + litellm_params: + guardrail: presidio + mode: pre_call + default_on: true + pii_entities: + UK_NHS: MASK + UK_NINO: MASK + UK_PASSPORT: MASK + UK_POSTCODE: MASK + UK_VEHICLE_REGISTRATION: MASK +EOF + +echo "Created test config at: $CONFIG_FILE" +echo "" +echo "Starting litellm proxy on port 4000 with UK PII guardrail enabled..." +echo "" +echo "Run the following commands in separate terminals to test:" +echo "" +echo "# Terminal 1: Start proxy" +echo "python litellm/proxy/proxy_cli.py --config $CONFIG_FILE --detailed_debug" +echo "" +echo "# Terminal 2: Test UK_PASSPORT" +echo "curl -X POST http://localhost:4000/chat/completions \\" +echo " -H 'Content-Type: application/json' \\" +echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"My passport is 012345678\"}]}' | jq" +echo "" +echo "# Terminal 2: Test UK_POSTCODE" +echo "curl -X POST http://localhost:4000/chat/completions \\" +echo " -H 'Content-Type: application/json' \\" +echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"I live at SW1A 1AA\"}]}' | jq" +echo "" +echo "# Terminal 2: Test UK_VEHICLE_REGISTRATION" +echo "curl -X POST http://localhost:4000/chat/completions \\" +echo " -H 'Content-Type: application/json' \\" +echo " -d '{\"model\": \"gpt-3.5-turbo\", \"messages\": [{\"role\": \"user\", \"content\": \"My car registration is AB12 CDE\"}]}' | jq" +echo "" +echo "Expected: The sensitive UK data should be masked (replaced with , , )" +echo " in the request before being sent to OpenAI."