diff --git a/litellm/constants.py b/litellm/constants.py index 78d5e5760d1..34a4f37ad6f 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -157,6 +157,7 @@ NON_LLM_CONNECTION_TIMEOUT = int( os.getenv("NON_LLM_CONNECTION_TIMEOUT", 15) ) # timeout for adjacent services (e.g. jwt auth) MAX_EXCEPTION_MESSAGE_LENGTH = int(os.getenv("MAX_EXCEPTION_MESSAGE_LENGTH", 2000)) +MAX_STRING_LENGTH_PROMPT_IN_DB = int(os.getenv("MAX_STRING_LENGTH_PROMPT_IN_DB", 1000)) BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75)) REPLICATE_POLLING_DELAY_SECONDS = float( os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5) @@ -486,227 +487,247 @@ _openai_like_providers: List = [ "watsonx", ] # private helper. similar to openai but require some custom auth / endpoint handling, so can't use the openai sdk # well supported replicate llms -replicate_models: set = set([ - # llama replicate supported LLMs - "replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf", - "a16z-infra/llama-2-13b-chat:2a7f981751ec7fdf87b5b91ad4db53683a98082e9ff7bfd12c8cd5ea85980a52", - "meta/codellama-13b:1c914d844307b0588599b8393480a3ba917b660c7e9dfae681542b5325f228db", - # Vicuna - "replicate/vicuna-13b:6282abe6a492de4145d7bb601023762212f9ddbbe78278bd6771c8b3b2f2a13b", - "joehoover/instructblip-vicuna13b:c4c54e3c8c97cd50c2d2fec9be3b6065563ccf7d43787fb99f84151b867178fe", - # Flan T-5 - "daanelson/flan-t5-large:ce962b3f6792a57074a601d3979db5839697add2e4e02696b3ced4c022d4767f", - # Others - "replicate/dolly-v2-12b:ef0e1aefc61f8e096ebe4db6b2bacc297daf2ef6899f0f7e001ec445893500e5", - "replit/replit-code-v1-3b:b84f4c074b807211cd75e3e8b1589b6399052125b4c27106e43d47189e8415ad", -]) +replicate_models: set = set( + [ + # llama replicate supported LLMs + "replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf", + "a16z-infra/llama-2-13b-chat:2a7f981751ec7fdf87b5b91ad4db53683a98082e9ff7bfd12c8cd5ea85980a52", + "meta/codellama-13b:1c914d844307b0588599b8393480a3ba917b660c7e9dfae681542b5325f228db", + # Vicuna + "replicate/vicuna-13b:6282abe6a492de4145d7bb601023762212f9ddbbe78278bd6771c8b3b2f2a13b", + "joehoover/instructblip-vicuna13b:c4c54e3c8c97cd50c2d2fec9be3b6065563ccf7d43787fb99f84151b867178fe", + # Flan T-5 + "daanelson/flan-t5-large:ce962b3f6792a57074a601d3979db5839697add2e4e02696b3ced4c022d4767f", + # Others + "replicate/dolly-v2-12b:ef0e1aefc61f8e096ebe4db6b2bacc297daf2ef6899f0f7e001ec445893500e5", + "replit/replit-code-v1-3b:b84f4c074b807211cd75e3e8b1589b6399052125b4c27106e43d47189e8415ad", + ] +) -clarifai_models: set = set([ - "clarifai/meta.Llama-3.Llama-3-8B-Instruct", - "clarifai/gcp.generate.gemma-1_1-7b-it", - "clarifai/mistralai.completion.mixtral-8x22B", - "clarifai/cohere.generate.command-r-plus", - "clarifai/databricks.drbx.dbrx-instruct", - "clarifai/mistralai.completion.mistral-large", - "clarifai/mistralai.completion.mistral-medium", - "clarifai/mistralai.completion.mistral-small", - "clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1", - "clarifai/gcp.generate.gemma-2b-it", - "clarifai/gcp.generate.gemma-7b-it", - "clarifai/deci.decilm.deciLM-7B-instruct", - "clarifai/mistralai.completion.mistral-7B-Instruct", - "clarifai/gcp.generate.gemini-pro", - "clarifai/anthropic.completion.claude-v1", - "clarifai/anthropic.completion.claude-instant-1_2", - "clarifai/anthropic.completion.claude-instant", - "clarifai/anthropic.completion.claude-v2", - "clarifai/anthropic.completion.claude-2_1", - "clarifai/meta.Llama-2.codeLlama-70b-Python", - "clarifai/meta.Llama-2.codeLlama-70b-Instruct", - "clarifai/openai.completion.gpt-3_5-turbo-instruct", - "clarifai/meta.Llama-2.llama2-7b-chat", - "clarifai/meta.Llama-2.llama2-13b-chat", - "clarifai/meta.Llama-2.llama2-70b-chat", - "clarifai/openai.chat-completion.gpt-4-turbo", - "clarifai/microsoft.text-generation.phi-2", - "clarifai/meta.Llama-2.llama2-7b-chat-vllm", - "clarifai/upstage.solar.solar-10_7b-instruct", - "clarifai/openchat.openchat.openchat-3_5-1210", - "clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B", - "clarifai/gcp.generate.text-bison", - "clarifai/meta.Llama-2.llamaGuard-7b", - "clarifai/fblgit.una-cybertron.una-cybertron-7b-v2", - "clarifai/openai.chat-completion.GPT-4", - "clarifai/openai.chat-completion.GPT-3_5-turbo", - "clarifai/ai21.complete.Jurassic2-Grande", - "clarifai/ai21.complete.Jurassic2-Grande-Instruct", - "clarifai/ai21.complete.Jurassic2-Jumbo-Instruct", - "clarifai/ai21.complete.Jurassic2-Jumbo", - "clarifai/ai21.complete.Jurassic2-Large", - "clarifai/cohere.generate.cohere-generate-command", - "clarifai/wizardlm.generate.wizardCoder-Python-34B", - "clarifai/wizardlm.generate.wizardLM-70B", - "clarifai/tiiuae.falcon.falcon-40b-instruct", - "clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat", - "clarifai/gcp.generate.code-gecko", - "clarifai/gcp.generate.code-bison", - "clarifai/mistralai.completion.mistral-7B-OpenOrca", - "clarifai/mistralai.completion.openHermes-2-mistral-7B", - "clarifai/wizardlm.generate.wizardLM-13B", - "clarifai/huggingface-research.zephyr.zephyr-7B-alpha", - "clarifai/wizardlm.generate.wizardCoder-15B", - "clarifai/microsoft.text-generation.phi-1_5", - "clarifai/databricks.Dolly-v2.dolly-v2-12b", - "clarifai/bigcode.code.StarCoder", - "clarifai/salesforce.xgen.xgen-7b-8k-instruct", - "clarifai/mosaicml.mpt.mpt-7b-instruct", - "clarifai/anthropic.completion.claude-3-opus", - "clarifai/anthropic.completion.claude-3-sonnet", - "clarifai/gcp.generate.gemini-1_5-pro", - "clarifai/gcp.generate.imagen-2", - "clarifai/salesforce.blip.general-english-image-caption-blip-2", -]) +clarifai_models: set = set( + [ + "clarifai/meta.Llama-3.Llama-3-8B-Instruct", + "clarifai/gcp.generate.gemma-1_1-7b-it", + "clarifai/mistralai.completion.mixtral-8x22B", + "clarifai/cohere.generate.command-r-plus", + "clarifai/databricks.drbx.dbrx-instruct", + "clarifai/mistralai.completion.mistral-large", + "clarifai/mistralai.completion.mistral-medium", + "clarifai/mistralai.completion.mistral-small", + "clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1", + "clarifai/gcp.generate.gemma-2b-it", + "clarifai/gcp.generate.gemma-7b-it", + "clarifai/deci.decilm.deciLM-7B-instruct", + "clarifai/mistralai.completion.mistral-7B-Instruct", + "clarifai/gcp.generate.gemini-pro", + "clarifai/anthropic.completion.claude-v1", + "clarifai/anthropic.completion.claude-instant-1_2", + "clarifai/anthropic.completion.claude-instant", + "clarifai/anthropic.completion.claude-v2", + "clarifai/anthropic.completion.claude-2_1", + "clarifai/meta.Llama-2.codeLlama-70b-Python", + "clarifai/meta.Llama-2.codeLlama-70b-Instruct", + "clarifai/openai.completion.gpt-3_5-turbo-instruct", + "clarifai/meta.Llama-2.llama2-7b-chat", + "clarifai/meta.Llama-2.llama2-13b-chat", + "clarifai/meta.Llama-2.llama2-70b-chat", + "clarifai/openai.chat-completion.gpt-4-turbo", + "clarifai/microsoft.text-generation.phi-2", + "clarifai/meta.Llama-2.llama2-7b-chat-vllm", + "clarifai/upstage.solar.solar-10_7b-instruct", + "clarifai/openchat.openchat.openchat-3_5-1210", + "clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B", + "clarifai/gcp.generate.text-bison", + "clarifai/meta.Llama-2.llamaGuard-7b", + "clarifai/fblgit.una-cybertron.una-cybertron-7b-v2", + "clarifai/openai.chat-completion.GPT-4", + "clarifai/openai.chat-completion.GPT-3_5-turbo", + "clarifai/ai21.complete.Jurassic2-Grande", + "clarifai/ai21.complete.Jurassic2-Grande-Instruct", + "clarifai/ai21.complete.Jurassic2-Jumbo-Instruct", + "clarifai/ai21.complete.Jurassic2-Jumbo", + "clarifai/ai21.complete.Jurassic2-Large", + "clarifai/cohere.generate.cohere-generate-command", + "clarifai/wizardlm.generate.wizardCoder-Python-34B", + "clarifai/wizardlm.generate.wizardLM-70B", + "clarifai/tiiuae.falcon.falcon-40b-instruct", + "clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat", + "clarifai/gcp.generate.code-gecko", + "clarifai/gcp.generate.code-bison", + "clarifai/mistralai.completion.mistral-7B-OpenOrca", + "clarifai/mistralai.completion.openHermes-2-mistral-7B", + "clarifai/wizardlm.generate.wizardLM-13B", + "clarifai/huggingface-research.zephyr.zephyr-7B-alpha", + "clarifai/wizardlm.generate.wizardCoder-15B", + "clarifai/microsoft.text-generation.phi-1_5", + "clarifai/databricks.Dolly-v2.dolly-v2-12b", + "clarifai/bigcode.code.StarCoder", + "clarifai/salesforce.xgen.xgen-7b-8k-instruct", + "clarifai/mosaicml.mpt.mpt-7b-instruct", + "clarifai/anthropic.completion.claude-3-opus", + "clarifai/anthropic.completion.claude-3-sonnet", + "clarifai/gcp.generate.gemini-1_5-pro", + "clarifai/gcp.generate.imagen-2", + "clarifai/salesforce.blip.general-english-image-caption-blip-2", + ] +) -huggingface_models: set = set([ - "meta-llama/Llama-2-7b-hf", - "meta-llama/Llama-2-7b-chat-hf", - "meta-llama/Llama-2-13b-hf", - "meta-llama/Llama-2-13b-chat-hf", - "meta-llama/Llama-2-70b-hf", - "meta-llama/Llama-2-70b-chat-hf", - "meta-llama/Llama-2-7b", - "meta-llama/Llama-2-7b-chat", - "meta-llama/Llama-2-13b", - "meta-llama/Llama-2-13b-chat", - "meta-llama/Llama-2-70b", - "meta-llama/Llama-2-70b-chat", -]) # these have been tested on extensively. But by default all text2text-generation and text-generation models are supported by liteLLM. - https://docs.litellm.ai/docs/providers -empower_models = set([ - "empower/empower-functions", - "empower/empower-functions-small", -]) +huggingface_models: set = set( + [ + "meta-llama/Llama-2-7b-hf", + "meta-llama/Llama-2-7b-chat-hf", + "meta-llama/Llama-2-13b-hf", + "meta-llama/Llama-2-13b-chat-hf", + "meta-llama/Llama-2-70b-hf", + "meta-llama/Llama-2-70b-chat-hf", + "meta-llama/Llama-2-7b", + "meta-llama/Llama-2-7b-chat", + "meta-llama/Llama-2-13b", + "meta-llama/Llama-2-13b-chat", + "meta-llama/Llama-2-70b", + "meta-llama/Llama-2-70b-chat", + ] +) # these have been tested on extensively. But by default all text2text-generation and text-generation models are supported by liteLLM. - https://docs.litellm.ai/docs/providers +empower_models = set( + [ + "empower/empower-functions", + "empower/empower-functions-small", + ] +) -together_ai_models: set = set([ - # llama llms - chat - "togethercomputer/llama-2-70b-chat", - # llama llms - language / instruct - "togethercomputer/llama-2-70b", - "togethercomputer/LLaMA-2-7B-32K", - "togethercomputer/Llama-2-7B-32K-Instruct", - "togethercomputer/llama-2-7b", - # falcon llms - "togethercomputer/falcon-40b-instruct", - "togethercomputer/falcon-7b-instruct", - # alpaca - "togethercomputer/alpaca-7b", - # chat llms - "HuggingFaceH4/starchat-alpha", - # code llms - "togethercomputer/CodeLlama-34b", - "togethercomputer/CodeLlama-34b-Instruct", - "togethercomputer/CodeLlama-34b-Python", - "defog/sqlcoder", - "NumbersStation/nsql-llama-2-7B", - "WizardLM/WizardCoder-15B-V1.0", - "WizardLM/WizardCoder-Python-34B-V1.0", - # language llms - "NousResearch/Nous-Hermes-Llama2-13b", - "Austism/chronos-hermes-13b", - "upstage/SOLAR-0-70b-16bit", - "WizardLM/WizardLM-70B-V1.0", -]) - # supports all together ai models, just pass in the model id e.g. completion(model="together_computer/replit_code_3b",...) +together_ai_models: set = set( + [ + # llama llms - chat + "togethercomputer/llama-2-70b-chat", + # llama llms - language / instruct + "togethercomputer/llama-2-70b", + "togethercomputer/LLaMA-2-7B-32K", + "togethercomputer/Llama-2-7B-32K-Instruct", + "togethercomputer/llama-2-7b", + # falcon llms + "togethercomputer/falcon-40b-instruct", + "togethercomputer/falcon-7b-instruct", + # alpaca + "togethercomputer/alpaca-7b", + # chat llms + "HuggingFaceH4/starchat-alpha", + # code llms + "togethercomputer/CodeLlama-34b", + "togethercomputer/CodeLlama-34b-Instruct", + "togethercomputer/CodeLlama-34b-Python", + "defog/sqlcoder", + "NumbersStation/nsql-llama-2-7B", + "WizardLM/WizardCoder-15B-V1.0", + "WizardLM/WizardCoder-Python-34B-V1.0", + # language llms + "NousResearch/Nous-Hermes-Llama2-13b", + "Austism/chronos-hermes-13b", + "upstage/SOLAR-0-70b-16bit", + "WizardLM/WizardLM-70B-V1.0", + ] +) +# supports all together ai models, just pass in the model id e.g. completion(model="together_computer/replit_code_3b",...) -baseten_models: set = set([ - "qvv0xeq", - "q841o8w", - "31dxrj3", -]) # FALCON 7B # WizardLM # Mosaic ML +baseten_models: set = set( + [ + "qvv0xeq", + "q841o8w", + "31dxrj3", + ] +) # FALCON 7B # WizardLM # Mosaic ML -featherless_ai_models: set = set([ - "featherless-ai/Qwerky-72B", - "featherless-ai/Qwerky-QwQ-32B", - "Qwen/Qwen2.5-72B-Instruct", - "all-hands/openhands-lm-32b-v0.1", - "Qwen/Qwen2.5-Coder-32B-Instruct", - "deepseek-ai/DeepSeek-V3-0324", - "mistralai/Mistral-Small-24B-Instruct-2501", - "mistralai/Mistral-Nemo-Instruct-2407", - "ProdeusUnity/Stellar-Odyssey-12b-v0.0", -]) +featherless_ai_models: set = set( + [ + "featherless-ai/Qwerky-72B", + "featherless-ai/Qwerky-QwQ-32B", + "Qwen/Qwen2.5-72B-Instruct", + "all-hands/openhands-lm-32b-v0.1", + "Qwen/Qwen2.5-Coder-32B-Instruct", + "deepseek-ai/DeepSeek-V3-0324", + "mistralai/Mistral-Small-24B-Instruct-2501", + "mistralai/Mistral-Nemo-Instruct-2407", + "ProdeusUnity/Stellar-Odyssey-12b-v0.0", + ] +) -nebius_models: set = set([ - # deepseek models - "deepseek-ai/DeepSeek-R1-0528", - "deepseek-ai/DeepSeek-V3-0324", - "deepseek-ai/DeepSeek-V3", - "deepseek-ai/DeepSeek-R1", - "deepseek-ai/DeepSeek-R1-Distill-Llama-70B", - # google models - "google/gemma-2-2b-it", - "google/gemma-2-9b-it-fast", - # llama models - "meta-llama/Llama-3.3-70B-Instruct", - "meta-llama/Meta-Llama-3.1-70B-Instruct", - "meta-llama/Meta-Llama-3.1-8B-Instruct", - "meta-llama/Meta-Llama-3.1-405B-Instruct", - "NousResearch/Hermes-3-Llama-405B", - # microsoft models - "microsoft/phi-4", - # mistral models - "mistralai/Mistral-Nemo-Instruct-2407", - "mistralai/Devstral-Small-2505", - # moonshot models - "moonshotai/Kimi-K2-Instruct", - # nvidia models - "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1", - "nvidia/Llama-3_3-Nemotron-Super-49B-v1", - # openai models - "openai/gpt-oss-120b", - "openai/gpt-oss-20b", - # qwen models - "Qwen/Qwen3-Coder-480B-A35B-Instruct", - "Qwen/Qwen3-235B-A22B-Instruct-2507", - "Qwen/Qwen3-235B-A22B", - "Qwen/Qwen3-30B-A3B", - "Qwen/Qwen3-32B", - "Qwen/Qwen3-14B", - "Qwen/Qwen3-4B-fast", - "Qwen/Qwen2.5-Coder-7B", - "Qwen/Qwen2.5-Coder-32B-Instruct", - "Qwen/Qwen2.5-72B-Instruct", - "Qwen/QwQ-32B", - "Qwen/Qwen3-30B-A3B-Thinking-2507", - "Qwen/Qwen3-30B-A3B-Instruct-2507", - # zai models - "zai-org/GLM-4.5", - "zai-org/GLM-4.5-Air", - # other models - "aaditya/Llama3-OpenBioLLM-70B", - "ProdeusUnity/Stellar-Odyssey-12b-v0.0", - "all-hands/openhands-lm-32b-v0.1", -]) +nebius_models: set = set( + [ + # deepseek models + "deepseek-ai/DeepSeek-R1-0528", + "deepseek-ai/DeepSeek-V3-0324", + "deepseek-ai/DeepSeek-V3", + "deepseek-ai/DeepSeek-R1", + "deepseek-ai/DeepSeek-R1-Distill-Llama-70B", + # google models + "google/gemma-2-2b-it", + "google/gemma-2-9b-it-fast", + # llama models + "meta-llama/Llama-3.3-70B-Instruct", + "meta-llama/Meta-Llama-3.1-70B-Instruct", + "meta-llama/Meta-Llama-3.1-8B-Instruct", + "meta-llama/Meta-Llama-3.1-405B-Instruct", + "NousResearch/Hermes-3-Llama-405B", + # microsoft models + "microsoft/phi-4", + # mistral models + "mistralai/Mistral-Nemo-Instruct-2407", + "mistralai/Devstral-Small-2505", + # moonshot models + "moonshotai/Kimi-K2-Instruct", + # nvidia models + "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1", + "nvidia/Llama-3_3-Nemotron-Super-49B-v1", + # openai models + "openai/gpt-oss-120b", + "openai/gpt-oss-20b", + # qwen models + "Qwen/Qwen3-Coder-480B-A35B-Instruct", + "Qwen/Qwen3-235B-A22B-Instruct-2507", + "Qwen/Qwen3-235B-A22B", + "Qwen/Qwen3-30B-A3B", + "Qwen/Qwen3-32B", + "Qwen/Qwen3-14B", + "Qwen/Qwen3-4B-fast", + "Qwen/Qwen2.5-Coder-7B", + "Qwen/Qwen2.5-Coder-32B-Instruct", + "Qwen/Qwen2.5-72B-Instruct", + "Qwen/QwQ-32B", + "Qwen/Qwen3-30B-A3B-Thinking-2507", + "Qwen/Qwen3-30B-A3B-Instruct-2507", + # zai models + "zai-org/GLM-4.5", + "zai-org/GLM-4.5-Air", + # other models + "aaditya/Llama3-OpenBioLLM-70B", + "ProdeusUnity/Stellar-Odyssey-12b-v0.0", + "all-hands/openhands-lm-32b-v0.1", + ] +) -dashscope_models: set = set([ - "qwen-turbo", - "qwen-plus", - "qwen-max", - "qwen-turbo-latest", - "qwen-plus-latest", - "qwen-max-latest", - "qwq-32b", - "qwen3-235b-a22b", - "qwen3-32b", - "qwen3-30b-a3b", -]) +dashscope_models: set = set( + [ + "qwen-turbo", + "qwen-plus", + "qwen-max", + "qwen-turbo-latest", + "qwen-plus-latest", + "qwen-max-latest", + "qwq-32b", + "qwen3-235b-a22b", + "qwen3-32b", + "qwen3-30b-a3b", + ] +) -nebius_embedding_models: set = set([ - "BAAI/bge-en-icl", - "BAAI/bge-multilingual-gemma2", - "intfloat/e5-mistral-7b-instruct", -]) +nebius_embedding_models: set = set( + [ + "BAAI/bge-en-icl", + "BAAI/bge-multilingual-gemma2", + "intfloat/e5-mistral-7b-instruct", + ] +) BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[ "cohere", @@ -721,20 +742,24 @@ BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[ ] open_ai_embedding_models: set = set(["text-embedding-ada-002"]) -cohere_embedding_models: set = set([ - "embed-v4.0", - "embed-english-v3.0", - "embed-english-light-v3.0", - "embed-multilingual-v3.0", - "embed-english-v2.0", - "embed-english-light-v2.0", - "embed-multilingual-v2.0", -]) -bedrock_embedding_models: set = set([ - "amazon.titan-embed-text-v1", - "cohere.embed-english-v3", - "cohere.embed-multilingual-v3", -]) +cohere_embedding_models: set = set( + [ + "embed-v4.0", + "embed-english-v3.0", + "embed-english-light-v3.0", + "embed-multilingual-v3.0", + "embed-english-v2.0", + "embed-english-light-v2.0", + "embed-multilingual-v2.0", + ] +) +bedrock_embedding_models: set = set( + [ + "amazon.titan-embed-text-v1", + "cohere.embed-english-v3", + "cohere.embed-multilingual-v3", + ] +) known_tokenizer_config = { "mistralai/Mistral-7B-Instruct-v0.1": { diff --git a/litellm/proxy/spend_tracking/spend_tracking_utils.py b/litellm/proxy/spend_tracking/spend_tracking_utils.py index 653426e8bd2..4a2bd796c47 100644 --- a/litellm/proxy/spend_tracking/spend_tracking_utils.py +++ b/litellm/proxy/spend_tracking/spend_tracking_utils.py @@ -10,7 +10,7 @@ from pydantic import BaseModel import litellm from litellm._logging import verbose_proxy_logger -from litellm.constants import REDACTED_BY_LITELM_STRING +from litellm.constants import REDACTED_BY_LITELM_STRING, MAX_STRING_LENGTH_PROMPT_IN_DB from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.proxy._types import SpendLogsMetadata, SpendLogsPayload @@ -53,7 +53,7 @@ def _get_spend_logs_metadata( guardrail_information: Optional[StandardLoggingGuardrailInformation] = None, usage_object: Optional[dict] = None, model_map_information: Optional[StandardLoggingModelInformation] = None, - cold_storage_object_key: Optional[str] = None + cold_storage_object_key: Optional[str] = None, ) -> SpendLogsMetadata: if metadata is None: return SpendLogsMetadata( @@ -101,7 +101,7 @@ def _get_spend_logs_metadata( clean_metadata["usage_object"] = usage_object clean_metadata["model_map_information"] = model_map_information clean_metadata["cold_storage_object_key"] = cold_storage_object_key - + return clean_metadata @@ -481,10 +481,9 @@ def _sanitize_request_body_for_spend_logs_payload( ) -> dict: """ Recursively sanitize request body to prevent logging large base64 strings or other large values. - Truncates strings longer than 1000 characters and handles nested dictionaries. + Truncates strings longer than MAX_STRING_LENGTH_PROMPT_IN_DB characters and handles nested dictionaries. """ from litellm.constants import LITELLM_TRUNCATED_PAYLOAD_FIELD - MAX_STRING_LENGTH = 1000 if visited is None: visited = set() @@ -501,8 +500,8 @@ def _sanitize_request_body_for_spend_logs_payload( elif isinstance(value, list): return [_sanitize_value(item) for item in value] elif isinstance(value, str): - if len(value) > MAX_STRING_LENGTH: - return f"{value[:MAX_STRING_LENGTH]}... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} {len(value) - MAX_STRING_LENGTH} chars)" + if len(value) > MAX_STRING_LENGTH_PROMPT_IN_DB: + return f"{value[:MAX_STRING_LENGTH_PROMPT_IN_DB]}... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} {len(value) - MAX_STRING_LENGTH_PROMPT_IN_DB} chars)" return value return value diff --git a/litellm/router_strategy/simple_shuffle.py b/litellm/router_strategy/simple_shuffle.py index 3bfd1fdc51b..ca82ddc6aa1 100644 --- a/litellm/router_strategy/simple_shuffle.py +++ b/litellm/router_strategy/simple_shuffle.py @@ -9,7 +9,6 @@ import random from typing import TYPE_CHECKING, Any, Dict, List, Union from litellm._logging import verbose_router_logger -from litellm.litellm_core_utils.core_helpers import safe_divide if TYPE_CHECKING: from litellm.router import Router as _Router diff --git a/tests/logging_callback_tests/test_spend_logs.py b/tests/logging_callback_tests/test_spend_logs.py index 5eed5971605..46e4e2cfcc4 100644 --- a/tests/logging_callback_tests/test_spend_logs.py +++ b/tests/logging_callback_tests/test_spend_logs.py @@ -25,7 +25,7 @@ from typing import Optional import pytest import litellm -from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload +from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload, _sanitize_request_body_for_spend_logs_payload from litellm.proxy._types import SpendLogsMetadata, SpendLogsPayload @@ -396,3 +396,91 @@ def test_spend_logs_payload_with_prompts_enabled(monkeypatch): payload_disabled: SpendLogsPayload = get_logging_payload(**input_args) assert payload_disabled["messages"] == "{}" assert payload_disabled["response"] == "{}" + + +def test_large_request_no_truncation_threshold(): + """ + Test that MAX_STRING_LENGTH_PROMPT_IN_DB constant is used for request body sanitization + """ + from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB, LITELLM_TRUNCATED_PAYLOAD_FIELD + + # Create a large string that exceeds the threshold + large_content = "x" * (MAX_STRING_LENGTH_PROMPT_IN_DB + 500) + + request_body = { + "messages": [ + {"role": "user", "content": large_content} + ], + "model": "gpt-4" + } + + sanitized = _sanitize_request_body_for_spend_logs_payload(request_body) + + # Verify the content was truncated + truncated_content = sanitized["messages"][0]["content"] + assert len(truncated_content) > MAX_STRING_LENGTH_PROMPT_IN_DB # includes truncation message + assert truncated_content.startswith("x" * MAX_STRING_LENGTH_PROMPT_IN_DB) + assert LITELLM_TRUNCATED_PAYLOAD_FIELD in truncated_content + assert "500 chars" in truncated_content + + +def test_small_request_no_truncation(): + """ + Test that small strings are not truncated by MAX_STRING_LENGTH_PROMPT_IN_DB + """ + from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB + + # Create a small string that's under the threshold + small_content = "x" * (MAX_STRING_LENGTH_PROMPT_IN_DB - 100) + + request_body = { + "messages": [ + {"role": "user", "content": small_content} + ], + "model": "gpt-4" + } + + sanitized = _sanitize_request_body_for_spend_logs_payload(request_body) + + # Verify the content was NOT truncated + assert sanitized["messages"][0]["content"] == small_content + assert len(sanitized["messages"][0]["content"]) == MAX_STRING_LENGTH_PROMPT_IN_DB - 100 + + +def test_configurable_string_length_env_var(monkeypatch): + """ + Test that MAX_STRING_LENGTH_PROMPT_IN_DB can be configured via environment variable + """ + # Set environment variable to a custom value + monkeypatch.setenv("MAX_STRING_LENGTH_PROMPT_IN_DB", "500") + + # Import after setting env var to ensure it picks up the new value + import importlib + import litellm.constants + import litellm.proxy.spend_tracking.spend_tracking_utils + importlib.reload(litellm.constants) + importlib.reload(litellm.proxy.spend_tracking.spend_tracking_utils) + + from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB, LITELLM_TRUNCATED_PAYLOAD_FIELD + from litellm.proxy.spend_tracking.spend_tracking_utils import _sanitize_request_body_for_spend_logs_payload + + # Verify the constant was set to the env var value + assert MAX_STRING_LENGTH_PROMPT_IN_DB == 500 + + # Test truncation with the custom value + large_content = "y" * 750 # 250 chars over the custom limit + + request_body = { + "messages": [ + {"role": "user", "content": large_content} + ], + "model": "gpt-4" + } + + sanitized = _sanitize_request_body_for_spend_logs_payload(request_body) + + # Verify truncation occurred at the custom threshold + truncated_content = sanitized["messages"][0]["content"] + assert truncated_content.startswith("y" * 500) + assert LITELLM_TRUNCATED_PAYLOAD_FIELD in truncated_content + assert "250 chars" in truncated_content diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py index 422bbdc9c8b..8fa6324cdd8 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py @@ -757,7 +757,7 @@ def test_optional_combine_thinking_block_with_none_content( # Second chunk with reasoning_content and None content second_chunk = { - "id": "chunk2", + "id": "chunk2", "object": "chat.completion.chunk", "created": 1741037891, "model": "deepseek-reasoner", @@ -776,16 +776,13 @@ def test_optional_combine_thinking_block_with_none_content( # Final chunk with actual content - should add tag final_chunk = { "id": "chunk3", - "object": "chat.completion.chunk", + "object": "chat.completion.chunk", "created": 1741037892, "model": "deepseek-reasoner", "choices": [ { "index": 0, - "delta": { - "content": "The answer is 42", - "reasoning_content": None - }, + "delta": {"content": "The answer is 42", "reasoning_content": None}, "finish_reason": None, } ], @@ -796,12 +793,15 @@ def test_optional_combine_thinking_block_with_none_content( initialized_custom_stream_wrapper._optional_combine_thinking_block_in_choices( first_response ) - assert first_response.choices[0].delta.content == "Let me think about this problem" + assert ( + first_response.choices[0].delta.content + == "Let me think about this problem" + ) assert not hasattr(first_response.choices[0].delta, "reasoning_content") assert initialized_custom_stream_wrapper.sent_first_thinking_block is True # Process second chunk - should work with continued reasoning - second_response = ModelResponseStream(**second_chunk) + second_response = ModelResponseStream(**second_chunk) initialized_custom_stream_wrapper._optional_combine_thinking_block_in_choices( second_response ) @@ -822,76 +822,99 @@ def test_has_special_delta_content( initialized_custom_stream_wrapper: CustomStreamWrapper, ): """Test the _has_special_delta_content helper method""" - + # Test empty choices empty_response = ModelResponseStream( id="test", created=1742056047, model=None, choices=[] ) - assert not initialized_custom_stream_wrapper._has_special_delta_content(empty_response) - + assert not initialized_custom_stream_wrapper._has_special_delta_content( + empty_response + ) + # Test with tool_calls (simulate with mock object) tool_call_response = ModelResponseStream( - id="test", created=1742056047, model=None, + id="test", + created=1742056047, + model=None, choices=[ StreamingChoices( - finish_reason=None, index=0, - delta=Delta(content=None, tool_calls=[{"id": "test"}]) + finish_reason=None, + index=0, + delta=Delta( + content=None, + tool_calls=[ + { + "id": "test", + "function": {"arguments": "{}", "name": "test_func"}, + } + ], + ), ) - ] + ], ) - assert initialized_custom_stream_wrapper._has_special_delta_content(tool_call_response) - + assert initialized_custom_stream_wrapper._has_special_delta_content( + tool_call_response + ) + # Test with function_call (simulate with mock object) function_call_response = ModelResponseStream( - id="test", created=1742056047, model=None, + id="test", + created=1742056047, + model=None, choices=[ StreamingChoices( - finish_reason=None, index=0, - delta=Delta(content=None, function_call={"name": "test_func"}) + finish_reason=None, + index=0, + delta=Delta( + content=None, function_call={"name": "test_func", "arguments": "{}"} + ), ) - ] + ], ) - assert initialized_custom_stream_wrapper._has_special_delta_content(function_call_response) - + assert initialized_custom_stream_wrapper._has_special_delta_content( + function_call_response + ) + # Test with audio (simulate by adding audio attribute) audio_response = ModelResponseStream( - id="test", created=1742056047, model=None, + id="test", + created=1742056047, + model=None, choices=[ - StreamingChoices( - finish_reason=None, index=0, - delta=Delta(content=None) - ) - ] + StreamingChoices(finish_reason=None, index=0, delta=Delta(content=None)) + ], ) # Manually add audio attribute to delta audio_response.choices[0].delta.audio = {"transcript": "test"} assert initialized_custom_stream_wrapper._has_special_delta_content(audio_response) - + # Test with image (simulate by adding image attribute) image_response = ModelResponseStream( - id="test", created=1742056047, model=None, + id="test", + created=1742056047, + model=None, choices=[ - StreamingChoices( - finish_reason=None, index=0, - delta=Delta(content=None) - ) - ] + StreamingChoices(finish_reason=None, index=0, delta=Delta(content=None)) + ], ) # Manually add image attribute to delta image_response.choices[0].delta.image = {"url": "test.jpg"} assert initialized_custom_stream_wrapper._has_special_delta_content(image_response) - + # Test with regular content (should return False) regular_response = ModelResponseStream( - id="test", created=1742056047, model=None, + id="test", + created=1742056047, + model=None, choices=[ StreamingChoices( - finish_reason=None, index=0, - delta=Delta(content="Hello world") + finish_reason=None, index=0, delta=Delta(content="Hello world") ) - ] + ], + ) + assert not initialized_custom_stream_wrapper._has_special_delta_content( + regular_response ) - assert not initialized_custom_stream_wrapper._has_special_delta_content(regular_response) def test_handle_special_delta_content( @@ -899,21 +922,26 @@ def test_handle_special_delta_content( ): """Test the _handle_special_delta_content helper method""" test_response = ModelResponseStream( - id="test", created=1742056047, model=None, + id="test", + created=1742056047, + model=None, choices=[ StreamingChoices( - finish_reason=None, index=0, - delta=Delta(content="test", role="assistant") + finish_reason=None, + index=0, + delta=Delta(content="test", role="assistant"), ) - ] + ], ) - + # The method should call strip_role_from_delta - result = initialized_custom_stream_wrapper._handle_special_delta_content(test_response) - + result = initialized_custom_stream_wrapper._handle_special_delta_content( + test_response + ) + # Should return the same response object (modified) assert result is test_response - + # Should have set sent_first_chunk to True assert initialized_custom_stream_wrapper.sent_first_chunk is True @@ -922,32 +950,38 @@ def test_has_any_special_delta_attributes( initialized_custom_stream_wrapper: CustomStreamWrapper, ): """Test the _has_any_special_delta_attributes helper method""" - + # Test with delta that has audio attribute class MockDelta: def __init__(self): self.audio = {"transcript": "Hello world"} - + audio_delta = MockDelta() - result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(audio_delta) + result = initialized_custom_stream_wrapper._has_any_special_delta_attributes( + audio_delta + ) assert result is True - + # Test with delta that has image attribute class MockDeltaImage: def __init__(self): self.image = {"url": "test.jpg"} - + image_delta = MockDeltaImage() - result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(image_delta) + result = initialized_custom_stream_wrapper._has_any_special_delta_attributes( + image_delta + ) assert result is True - + # Test with delta that has no special attributes class MockDeltaRegular: def __init__(self): self.content = "regular content" - + regular_delta = MockDeltaRegular() - result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(regular_delta) + result = initialized_custom_stream_wrapper._has_any_special_delta_attributes( + regular_delta + ) assert result is False @@ -955,48 +989,50 @@ def test_handle_special_delta_attributes( initialized_custom_stream_wrapper: CustomStreamWrapper, ): """Test the _handle_special_delta_attributes helper method""" - + # Create a model response model_response = ModelResponseStream( - id="test", created=1742056047, model=None, + id="test", + created=1742056047, + model=None, choices=[ - StreamingChoices( - finish_reason=None, index=0, - delta=Delta(content="test") - ) - ] + StreamingChoices(finish_reason=None, index=0, delta=Delta(content="test")) + ], ) - + # Test with delta that has audio attribute class MockDelta: def __init__(self): self.audio = {"transcript": "Hello world"} - + audio_delta = MockDelta() - initialized_custom_stream_wrapper._handle_special_delta_attributes(audio_delta, model_response) - + initialized_custom_stream_wrapper._handle_special_delta_attributes( + audio_delta, model_response + ) + # Should copy the audio attribute assert hasattr(model_response.choices[0].delta, "audio") assert model_response.choices[0].delta.audio == {"transcript": "Hello world"} - + # Test with delta that has image attribute class MockDeltaImage: def __init__(self): self.image = {"url": "test.jpg"} - + image_delta = MockDeltaImage() model_response2 = ModelResponseStream( - id="test", created=1742056047, model=None, + id="test", + created=1742056047, + model=None, choices=[ - StreamingChoices( - finish_reason=None, index=0, - delta=Delta(content="test") - ) - ] + StreamingChoices(finish_reason=None, index=0, delta=Delta(content="test")) + ], ) - - initialized_custom_stream_wrapper._handle_special_delta_attributes(image_delta, model_response2) - + + initialized_custom_stream_wrapper._handle_special_delta_attributes( + image_delta, model_response2 + ) + # Should copy the image attribute assert hasattr(model_response2.choices[0].delta, "image") assert model_response2.choices[0].delta.image == {"url": "test.jpg"} @@ -1006,30 +1042,38 @@ def test_has_special_delta_attribute( initialized_custom_stream_wrapper: CustomStreamWrapper, ): """Test the _has_special_delta_attribute helper method""" - + # Test with None delta - assert not initialized_custom_stream_wrapper._has_special_delta_attribute(None, "audio") - + assert not initialized_custom_stream_wrapper._has_special_delta_attribute( + None, "audio" + ) + # Test with delta that has the attribute class MockDelta: def __init__(self): self.audio = {"transcript": "test"} - + delta_with_audio = MockDelta() - assert initialized_custom_stream_wrapper._has_special_delta_attribute(delta_with_audio, "audio") - + assert initialized_custom_stream_wrapper._has_special_delta_attribute( + delta_with_audio, "audio" + ) + # Test with delta that doesn't have the attribute class MockDeltaNoAudio: def __init__(self): self.content = "test" - + delta_without_audio = MockDeltaNoAudio() - assert not initialized_custom_stream_wrapper._has_special_delta_attribute(delta_without_audio, "audio") - + assert not initialized_custom_stream_wrapper._has_special_delta_attribute( + delta_without_audio, "audio" + ) + # Test with delta that has the attribute but it's None class MockDeltaNone: def __init__(self): self.audio = None - + delta_with_none = MockDeltaNone() - assert not initialized_custom_stream_wrapper._has_special_delta_attribute(delta_with_none, "audio") + assert not initialized_custom_stream_wrapper._has_special_delta_attribute( + delta_with_none, "audio" + )