Merge pull request #14042 from WilsonSunBritten/10988-set-truncation-threshold

Allow configuration to set threshold before request entry in spend log gets truncated
This commit is contained in:
Krish Dholakia 2025-08-29 06:24:28 -07:00 • committed by GitHub
commit fffccc0675
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 481 additions and 326 deletions

View file

@ -157,6 +157,7 @@ NON_LLM_CONNECTION_TIMEOUT = int(
os.getenv("NON_LLM_CONNECTION_TIMEOUT", 15)
) # timeout for adjacent services (e.g. jwt auth)
MAX_EXCEPTION_MESSAGE_LENGTH = int(os.getenv("MAX_EXCEPTION_MESSAGE_LENGTH", 2000))
MAX_STRING_LENGTH_PROMPT_IN_DB = int(os.getenv("MAX_STRING_LENGTH_PROMPT_IN_DB", 1000))
BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75))
REPLICATE_POLLING_DELAY_SECONDS = float(
os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5)
@ -486,227 +487,247 @@ _openai_like_providers: List = [
"watsonx",
] # private helper. similar to openai but require some custom auth / endpoint handling, so can't use the openai sdk
# well supported replicate llms
replicate_models: set = set([
# llama replicate supported LLMs
"replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf",
"a16z-infra/llama-2-13b-chat:2a7f981751ec7fdf87b5b91ad4db53683a98082e9ff7bfd12c8cd5ea85980a52",
"meta/codellama-13b:1c914d844307b0588599b8393480a3ba917b660c7e9dfae681542b5325f228db",
# Vicuna
"replicate/vicuna-13b:6282abe6a492de4145d7bb601023762212f9ddbbe78278bd6771c8b3b2f2a13b",
"joehoover/instructblip-vicuna13b:c4c54e3c8c97cd50c2d2fec9be3b6065563ccf7d43787fb99f84151b867178fe",
# Flan T-5
"daanelson/flan-t5-large:ce962b3f6792a57074a601d3979db5839697add2e4e02696b3ced4c022d4767f",
# Others
"replicate/dolly-v2-12b:ef0e1aefc61f8e096ebe4db6b2bacc297daf2ef6899f0f7e001ec445893500e5",
"replit/replit-code-v1-3b:b84f4c074b807211cd75e3e8b1589b6399052125b4c27106e43d47189e8415ad",
])
replicate_models: set = set(
[
# llama replicate supported LLMs
"replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf",
"a16z-infra/llama-2-13b-chat:2a7f981751ec7fdf87b5b91ad4db53683a98082e9ff7bfd12c8cd5ea85980a52",
"meta/codellama-13b:1c914d844307b0588599b8393480a3ba917b660c7e9dfae681542b5325f228db",
# Vicuna
"replicate/vicuna-13b:6282abe6a492de4145d7bb601023762212f9ddbbe78278bd6771c8b3b2f2a13b",
"joehoover/instructblip-vicuna13b:c4c54e3c8c97cd50c2d2fec9be3b6065563ccf7d43787fb99f84151b867178fe",
# Flan T-5
"daanelson/flan-t5-large:ce962b3f6792a57074a601d3979db5839697add2e4e02696b3ced4c022d4767f",
# Others
"replicate/dolly-v2-12b:ef0e1aefc61f8e096ebe4db6b2bacc297daf2ef6899f0f7e001ec445893500e5",
"replit/replit-code-v1-3b:b84f4c074b807211cd75e3e8b1589b6399052125b4c27106e43d47189e8415ad",
]
)
clarifai_models: set = set([
"clarifai/meta.Llama-3.Llama-3-8B-Instruct",
"clarifai/gcp.generate.gemma-1_1-7b-it",
"clarifai/mistralai.completion.mixtral-8x22B",
"clarifai/cohere.generate.command-r-plus",
"clarifai/databricks.drbx.dbrx-instruct",
"clarifai/mistralai.completion.mistral-large",
"clarifai/mistralai.completion.mistral-medium",
"clarifai/mistralai.completion.mistral-small",
"clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1",
"clarifai/gcp.generate.gemma-2b-it",
"clarifai/gcp.generate.gemma-7b-it",
"clarifai/deci.decilm.deciLM-7B-instruct",
"clarifai/mistralai.completion.mistral-7B-Instruct",
"clarifai/gcp.generate.gemini-pro",
"clarifai/anthropic.completion.claude-v1",
"clarifai/anthropic.completion.claude-instant-1_2",
"clarifai/anthropic.completion.claude-instant",
"clarifai/anthropic.completion.claude-v2",
"clarifai/anthropic.completion.claude-2_1",
"clarifai/meta.Llama-2.codeLlama-70b-Python",
"clarifai/meta.Llama-2.codeLlama-70b-Instruct",
"clarifai/openai.completion.gpt-3_5-turbo-instruct",
"clarifai/meta.Llama-2.llama2-7b-chat",
"clarifai/meta.Llama-2.llama2-13b-chat",
"clarifai/meta.Llama-2.llama2-70b-chat",
"clarifai/openai.chat-completion.gpt-4-turbo",
"clarifai/microsoft.text-generation.phi-2",
"clarifai/meta.Llama-2.llama2-7b-chat-vllm",
"clarifai/upstage.solar.solar-10_7b-instruct",
"clarifai/openchat.openchat.openchat-3_5-1210",
"clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B",
"clarifai/gcp.generate.text-bison",
"clarifai/meta.Llama-2.llamaGuard-7b",
"clarifai/fblgit.una-cybertron.una-cybertron-7b-v2",
"clarifai/openai.chat-completion.GPT-4",
"clarifai/openai.chat-completion.GPT-3_5-turbo",
"clarifai/ai21.complete.Jurassic2-Grande",
"clarifai/ai21.complete.Jurassic2-Grande-Instruct",
"clarifai/ai21.complete.Jurassic2-Jumbo-Instruct",
"clarifai/ai21.complete.Jurassic2-Jumbo",
"clarifai/ai21.complete.Jurassic2-Large",
"clarifai/cohere.generate.cohere-generate-command",
"clarifai/wizardlm.generate.wizardCoder-Python-34B",
"clarifai/wizardlm.generate.wizardLM-70B",
"clarifai/tiiuae.falcon.falcon-40b-instruct",
"clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat",
"clarifai/gcp.generate.code-gecko",
"clarifai/gcp.generate.code-bison",
"clarifai/mistralai.completion.mistral-7B-OpenOrca",
"clarifai/mistralai.completion.openHermes-2-mistral-7B",
"clarifai/wizardlm.generate.wizardLM-13B",
"clarifai/huggingface-research.zephyr.zephyr-7B-alpha",
"clarifai/wizardlm.generate.wizardCoder-15B",
"clarifai/microsoft.text-generation.phi-1_5",
"clarifai/databricks.Dolly-v2.dolly-v2-12b",
"clarifai/bigcode.code.StarCoder",
"clarifai/salesforce.xgen.xgen-7b-8k-instruct",
"clarifai/mosaicml.mpt.mpt-7b-instruct",
"clarifai/anthropic.completion.claude-3-opus",
"clarifai/anthropic.completion.claude-3-sonnet",
"clarifai/gcp.generate.gemini-1_5-pro",
"clarifai/gcp.generate.imagen-2",
"clarifai/salesforce.blip.general-english-image-caption-blip-2",
])
clarifai_models: set = set(
[
"clarifai/meta.Llama-3.Llama-3-8B-Instruct",
"clarifai/gcp.generate.gemma-1_1-7b-it",
"clarifai/mistralai.completion.mixtral-8x22B",
"clarifai/cohere.generate.command-r-plus",
"clarifai/databricks.drbx.dbrx-instruct",
"clarifai/mistralai.completion.mistral-large",
"clarifai/mistralai.completion.mistral-medium",
"clarifai/mistralai.completion.mistral-small",
"clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1",
"clarifai/gcp.generate.gemma-2b-it",
"clarifai/gcp.generate.gemma-7b-it",
"clarifai/deci.decilm.deciLM-7B-instruct",
"clarifai/mistralai.completion.mistral-7B-Instruct",
"clarifai/gcp.generate.gemini-pro",
"clarifai/anthropic.completion.claude-v1",
"clarifai/anthropic.completion.claude-instant-1_2",
"clarifai/anthropic.completion.claude-instant",
"clarifai/anthropic.completion.claude-v2",
"clarifai/anthropic.completion.claude-2_1",
"clarifai/meta.Llama-2.codeLlama-70b-Python",
"clarifai/meta.Llama-2.codeLlama-70b-Instruct",
"clarifai/openai.completion.gpt-3_5-turbo-instruct",
"clarifai/meta.Llama-2.llama2-7b-chat",
"clarifai/meta.Llama-2.llama2-13b-chat",
"clarifai/meta.Llama-2.llama2-70b-chat",
"clarifai/openai.chat-completion.gpt-4-turbo",
"clarifai/microsoft.text-generation.phi-2",
"clarifai/meta.Llama-2.llama2-7b-chat-vllm",
"clarifai/upstage.solar.solar-10_7b-instruct",
"clarifai/openchat.openchat.openchat-3_5-1210",
"clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B",
"clarifai/gcp.generate.text-bison",
"clarifai/meta.Llama-2.llamaGuard-7b",
"clarifai/fblgit.una-cybertron.una-cybertron-7b-v2",
"clarifai/openai.chat-completion.GPT-4",
"clarifai/openai.chat-completion.GPT-3_5-turbo",
"clarifai/ai21.complete.Jurassic2-Grande",
"clarifai/ai21.complete.Jurassic2-Grande-Instruct",
"clarifai/ai21.complete.Jurassic2-Jumbo-Instruct",
"clarifai/ai21.complete.Jurassic2-Jumbo",
"clarifai/ai21.complete.Jurassic2-Large",
"clarifai/cohere.generate.cohere-generate-command",
"clarifai/wizardlm.generate.wizardCoder-Python-34B",
"clarifai/wizardlm.generate.wizardLM-70B",
"clarifai/tiiuae.falcon.falcon-40b-instruct",
"clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat",
"clarifai/gcp.generate.code-gecko",
"clarifai/gcp.generate.code-bison",
"clarifai/mistralai.completion.mistral-7B-OpenOrca",
"clarifai/mistralai.completion.openHermes-2-mistral-7B",
"clarifai/wizardlm.generate.wizardLM-13B",
"clarifai/huggingface-research.zephyr.zephyr-7B-alpha",
"clarifai/wizardlm.generate.wizardCoder-15B",
"clarifai/microsoft.text-generation.phi-1_5",
"clarifai/databricks.Dolly-v2.dolly-v2-12b",
"clarifai/bigcode.code.StarCoder",
"clarifai/salesforce.xgen.xgen-7b-8k-instruct",
"clarifai/mosaicml.mpt.mpt-7b-instruct",
"clarifai/anthropic.completion.claude-3-opus",
"clarifai/anthropic.completion.claude-3-sonnet",
"clarifai/gcp.generate.gemini-1_5-pro",
"clarifai/gcp.generate.imagen-2",
"clarifai/salesforce.blip.general-english-image-caption-blip-2",
]
)
huggingface_models: set = set([
"meta-llama/Llama-2-7b-hf",
"meta-llama/Llama-2-7b-chat-hf",
"meta-llama/Llama-2-13b-hf",
"meta-llama/Llama-2-13b-chat-hf",
"meta-llama/Llama-2-70b-hf",
"meta-llama/Llama-2-70b-chat-hf",
"meta-llama/Llama-2-7b",
"meta-llama/Llama-2-7b-chat",
"meta-llama/Llama-2-13b",
"meta-llama/Llama-2-13b-chat",
"meta-llama/Llama-2-70b",
"meta-llama/Llama-2-70b-chat",
]) # these have been tested on extensively. But by default all text2text-generation and text-generation models are supported by liteLLM. - https://docs.litellm.ai/docs/providers
empower_models = set([
"empower/empower-functions",
"empower/empower-functions-small",
])
huggingface_models: set = set(
[
"meta-llama/Llama-2-7b-hf",
"meta-llama/Llama-2-7b-chat-hf",
"meta-llama/Llama-2-13b-hf",
"meta-llama/Llama-2-13b-chat-hf",
"meta-llama/Llama-2-70b-hf",
"meta-llama/Llama-2-70b-chat-hf",
"meta-llama/Llama-2-7b",
"meta-llama/Llama-2-7b-chat",
"meta-llama/Llama-2-13b",
"meta-llama/Llama-2-13b-chat",
"meta-llama/Llama-2-70b",
"meta-llama/Llama-2-70b-chat",
]
) # these have been tested on extensively. But by default all text2text-generation and text-generation models are supported by liteLLM. - https://docs.litellm.ai/docs/providers
empower_models = set(
[
"empower/empower-functions",
"empower/empower-functions-small",
]
)
together_ai_models: set = set([
# llama llms - chat
"togethercomputer/llama-2-70b-chat",
# llama llms - language / instruct
"togethercomputer/llama-2-70b",
"togethercomputer/LLaMA-2-7B-32K",
"togethercomputer/Llama-2-7B-32K-Instruct",
"togethercomputer/llama-2-7b",
# falcon llms
"togethercomputer/falcon-40b-instruct",
"togethercomputer/falcon-7b-instruct",
# alpaca
"togethercomputer/alpaca-7b",
# chat llms
"HuggingFaceH4/starchat-alpha",
# code llms
"togethercomputer/CodeLlama-34b",
"togethercomputer/CodeLlama-34b-Instruct",
"togethercomputer/CodeLlama-34b-Python",
"defog/sqlcoder",
"NumbersStation/nsql-llama-2-7B",
"WizardLM/WizardCoder-15B-V1.0",
"WizardLM/WizardCoder-Python-34B-V1.0",
# language llms
"NousResearch/Nous-Hermes-Llama2-13b",
"Austism/chronos-hermes-13b",
"upstage/SOLAR-0-70b-16bit",
"WizardLM/WizardLM-70B-V1.0",
])
# supports all together ai models, just pass in the model id e.g. completion(model="together_computer/replit_code_3b",...)
together_ai_models: set = set(
[
# llama llms - chat
"togethercomputer/llama-2-70b-chat",
# llama llms - language / instruct
"togethercomputer/llama-2-70b",
"togethercomputer/LLaMA-2-7B-32K",
"togethercomputer/Llama-2-7B-32K-Instruct",
"togethercomputer/llama-2-7b",
# falcon llms
"togethercomputer/falcon-40b-instruct",
"togethercomputer/falcon-7b-instruct",
# alpaca
"togethercomputer/alpaca-7b",
# chat llms
"HuggingFaceH4/starchat-alpha",
# code llms
"togethercomputer/CodeLlama-34b",
"togethercomputer/CodeLlama-34b-Instruct",
"togethercomputer/CodeLlama-34b-Python",
"defog/sqlcoder",
"NumbersStation/nsql-llama-2-7B",
"WizardLM/WizardCoder-15B-V1.0",
"WizardLM/WizardCoder-Python-34B-V1.0",
# language llms
"NousResearch/Nous-Hermes-Llama2-13b",
"Austism/chronos-hermes-13b",
"upstage/SOLAR-0-70b-16bit",
"WizardLM/WizardLM-70B-V1.0",
]
)
# supports all together ai models, just pass in the model id e.g. completion(model="together_computer/replit_code_3b",...)
baseten_models: set = set([
"qvv0xeq",
"q841o8w",
"31dxrj3",
]) # FALCON 7B # WizardLM # Mosaic ML
baseten_models: set = set(
[
"qvv0xeq",
"q841o8w",
"31dxrj3",
]
) # FALCON 7B # WizardLM # Mosaic ML
featherless_ai_models: set = set([
"featherless-ai/Qwerky-72B",
"featherless-ai/Qwerky-QwQ-32B",
"Qwen/Qwen2.5-72B-Instruct",
"all-hands/openhands-lm-32b-v0.1",
"Qwen/Qwen2.5-Coder-32B-Instruct",
"deepseek-ai/DeepSeek-V3-0324",
"mistralai/Mistral-Small-24B-Instruct-2501",
"mistralai/Mistral-Nemo-Instruct-2407",
"ProdeusUnity/Stellar-Odyssey-12b-v0.0",
])
featherless_ai_models: set = set(
[
"featherless-ai/Qwerky-72B",
"featherless-ai/Qwerky-QwQ-32B",
"Qwen/Qwen2.5-72B-Instruct",
"all-hands/openhands-lm-32b-v0.1",
"Qwen/Qwen2.5-Coder-32B-Instruct",
"deepseek-ai/DeepSeek-V3-0324",
"mistralai/Mistral-Small-24B-Instruct-2501",
"mistralai/Mistral-Nemo-Instruct-2407",
"ProdeusUnity/Stellar-Odyssey-12b-v0.0",
]
)
nebius_models: set = set([
# deepseek models
"deepseek-ai/DeepSeek-R1-0528",
"deepseek-ai/DeepSeek-V3-0324",
"deepseek-ai/DeepSeek-V3",
"deepseek-ai/DeepSeek-R1",
"deepseek-ai/DeepSeek-R1-Distill-Llama-70B",
# google models
"google/gemma-2-2b-it",
"google/gemma-2-9b-it-fast",
# llama models
"meta-llama/Llama-3.3-70B-Instruct",
"meta-llama/Meta-Llama-3.1-70B-Instruct",
"meta-llama/Meta-Llama-3.1-8B-Instruct",
"meta-llama/Meta-Llama-3.1-405B-Instruct",
"NousResearch/Hermes-3-Llama-405B",
# microsoft models
"microsoft/phi-4",
# mistral models
"mistralai/Mistral-Nemo-Instruct-2407",
"mistralai/Devstral-Small-2505",
# moonshot models
"moonshotai/Kimi-K2-Instruct",
# nvidia models
"nvidia/Llama-3_1-Nemotron-Ultra-253B-v1",
"nvidia/Llama-3_3-Nemotron-Super-49B-v1",
# openai models
"openai/gpt-oss-120b",
"openai/gpt-oss-20b",
# qwen models
"Qwen/Qwen3-Coder-480B-A35B-Instruct",
"Qwen/Qwen3-235B-A22B-Instruct-2507",
"Qwen/Qwen3-235B-A22B",
"Qwen/Qwen3-30B-A3B",
"Qwen/Qwen3-32B",
"Qwen/Qwen3-14B",
"Qwen/Qwen3-4B-fast",
"Qwen/Qwen2.5-Coder-7B",
"Qwen/Qwen2.5-Coder-32B-Instruct",
"Qwen/Qwen2.5-72B-Instruct",
"Qwen/QwQ-32B",
"Qwen/Qwen3-30B-A3B-Thinking-2507",
"Qwen/Qwen3-30B-A3B-Instruct-2507",
# zai models
"zai-org/GLM-4.5",
"zai-org/GLM-4.5-Air",
# other models
"aaditya/Llama3-OpenBioLLM-70B",
"ProdeusUnity/Stellar-Odyssey-12b-v0.0",
"all-hands/openhands-lm-32b-v0.1",
])
nebius_models: set = set(
[
# deepseek models
"deepseek-ai/DeepSeek-R1-0528",
"deepseek-ai/DeepSeek-V3-0324",
"deepseek-ai/DeepSeek-V3",
"deepseek-ai/DeepSeek-R1",
"deepseek-ai/DeepSeek-R1-Distill-Llama-70B",
# google models
"google/gemma-2-2b-it",
"google/gemma-2-9b-it-fast",
# llama models
"meta-llama/Llama-3.3-70B-Instruct",
"meta-llama/Meta-Llama-3.1-70B-Instruct",
"meta-llama/Meta-Llama-3.1-8B-Instruct",
"meta-llama/Meta-Llama-3.1-405B-Instruct",
"NousResearch/Hermes-3-Llama-405B",
# microsoft models
"microsoft/phi-4",
# mistral models
"mistralai/Mistral-Nemo-Instruct-2407",
"mistralai/Devstral-Small-2505",
# moonshot models
"moonshotai/Kimi-K2-Instruct",
# nvidia models
"nvidia/Llama-3_1-Nemotron-Ultra-253B-v1",
"nvidia/Llama-3_3-Nemotron-Super-49B-v1",
# openai models
"openai/gpt-oss-120b",
"openai/gpt-oss-20b",
# qwen models
"Qwen/Qwen3-Coder-480B-A35B-Instruct",
"Qwen/Qwen3-235B-A22B-Instruct-2507",
"Qwen/Qwen3-235B-A22B",
"Qwen/Qwen3-30B-A3B",
"Qwen/Qwen3-32B",
"Qwen/Qwen3-14B",
"Qwen/Qwen3-4B-fast",
"Qwen/Qwen2.5-Coder-7B",
"Qwen/Qwen2.5-Coder-32B-Instruct",
"Qwen/Qwen2.5-72B-Instruct",
"Qwen/QwQ-32B",
"Qwen/Qwen3-30B-A3B-Thinking-2507",
"Qwen/Qwen3-30B-A3B-Instruct-2507",
# zai models
"zai-org/GLM-4.5",
"zai-org/GLM-4.5-Air",
# other models
"aaditya/Llama3-OpenBioLLM-70B",
"ProdeusUnity/Stellar-Odyssey-12b-v0.0",
"all-hands/openhands-lm-32b-v0.1",
]
)
dashscope_models: set = set([
"qwen-turbo",
"qwen-plus",
"qwen-max",
"qwen-turbo-latest",
"qwen-plus-latest",
"qwen-max-latest",
"qwq-32b",
"qwen3-235b-a22b",
"qwen3-32b",
"qwen3-30b-a3b",
])
dashscope_models: set = set(
[
"qwen-turbo",
"qwen-plus",
"qwen-max",
"qwen-turbo-latest",
"qwen-plus-latest",
"qwen-max-latest",
"qwq-32b",
"qwen3-235b-a22b",
"qwen3-32b",
"qwen3-30b-a3b",
]
)
nebius_embedding_models: set = set([
"BAAI/bge-en-icl",
"BAAI/bge-multilingual-gemma2",
"intfloat/e5-mistral-7b-instruct",
])
nebius_embedding_models: set = set(
[
"BAAI/bge-en-icl",
"BAAI/bge-multilingual-gemma2",
"intfloat/e5-mistral-7b-instruct",
]
)
BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[
"cohere",
@ -721,20 +742,24 @@ BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[
]
open_ai_embedding_models: set = set(["text-embedding-ada-002"])
cohere_embedding_models: set = set([
"embed-v4.0",
"embed-english-v3.0",
"embed-english-light-v3.0",
"embed-multilingual-v3.0",
"embed-english-v2.0",
"embed-english-light-v2.0",
"embed-multilingual-v2.0",
])
bedrock_embedding_models: set = set([
"amazon.titan-embed-text-v1",
"cohere.embed-english-v3",
"cohere.embed-multilingual-v3",
])
cohere_embedding_models: set = set(
[
"embed-v4.0",
"embed-english-v3.0",
"embed-english-light-v3.0",
"embed-multilingual-v3.0",
"embed-english-v2.0",
"embed-english-light-v2.0",
"embed-multilingual-v2.0",
]
)
bedrock_embedding_models: set = set(
[
"amazon.titan-embed-text-v1",
"cohere.embed-english-v3",
"cohere.embed-multilingual-v3",
]
)
known_tokenizer_config = {
"mistralai/Mistral-7B-Instruct-v0.1": {

View file

@ -10,7 +10,7 @@ from pydantic import BaseModel
import litellm
from litellm._logging import verbose_proxy_logger
from litellm.constants import REDACTED_BY_LITELM_STRING
from litellm.constants import REDACTED_BY_LITELM_STRING, MAX_STRING_LENGTH_PROMPT_IN_DB
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
from litellm.proxy._types import SpendLogsMetadata, SpendLogsPayload
@ -53,7 +53,7 @@ def _get_spend_logs_metadata(
guardrail_information: Optional[StandardLoggingGuardrailInformation] = None,
usage_object: Optional[dict] = None,
model_map_information: Optional[StandardLoggingModelInformation] = None,
cold_storage_object_key: Optional[str] = None
cold_storage_object_key: Optional[str] = None,
) -> SpendLogsMetadata:
if metadata is None:
return SpendLogsMetadata(
@ -101,7 +101,7 @@ def _get_spend_logs_metadata(
clean_metadata["usage_object"] = usage_object
clean_metadata["model_map_information"] = model_map_information
clean_metadata["cold_storage_object_key"] = cold_storage_object_key
return clean_metadata
@ -481,10 +481,9 @@ def _sanitize_request_body_for_spend_logs_payload(
) -> dict:
"""
Recursively sanitize request body to prevent logging large base64 strings or other large values.
Truncates strings longer than 1000 characters and handles nested dictionaries.
Truncates strings longer than MAX_STRING_LENGTH_PROMPT_IN_DB characters and handles nested dictionaries.
"""
from litellm.constants import LITELLM_TRUNCATED_PAYLOAD_FIELD
MAX_STRING_LENGTH = 1000
if visited is None:
visited = set()
@ -501,8 +500,8 @@ def _sanitize_request_body_for_spend_logs_payload(
elif isinstance(value, list):
return [_sanitize_value(item) for item in value]
elif isinstance(value, str):
if len(value) > MAX_STRING_LENGTH:
return f"{value[:MAX_STRING_LENGTH]}... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} {len(value) - MAX_STRING_LENGTH} chars)"
if len(value) > MAX_STRING_LENGTH_PROMPT_IN_DB:
return f"{value[:MAX_STRING_LENGTH_PROMPT_IN_DB]}... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} {len(value) - MAX_STRING_LENGTH_PROMPT_IN_DB} chars)"
return value
return value

View file

@ -9,7 +9,6 @@ import random
from typing import TYPE_CHECKING, Any, Dict, List, Union
from litellm._logging import verbose_router_logger
from litellm.litellm_core_utils.core_helpers import safe_divide
if TYPE_CHECKING:
from litellm.router import Router as _Router

View file

@ -25,7 +25,7 @@ from typing import Optional
import pytest
import litellm
from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload
from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload, _sanitize_request_body_for_spend_logs_payload
from litellm.proxy._types import SpendLogsMetadata, SpendLogsPayload
@ -396,3 +396,91 @@ def test_spend_logs_payload_with_prompts_enabled(monkeypatch):
payload_disabled: SpendLogsPayload = get_logging_payload(**input_args)
assert payload_disabled["messages"] == "{}"
assert payload_disabled["response"] == "{}"
def test_large_request_no_truncation_threshold():
"""
Test that MAX_STRING_LENGTH_PROMPT_IN_DB constant is used for request body sanitization
"""
from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB, LITELLM_TRUNCATED_PAYLOAD_FIELD
# Create a large string that exceeds the threshold
large_content = "x" * (MAX_STRING_LENGTH_PROMPT_IN_DB + 500)
request_body = {
"messages": [
{"role": "user", "content": large_content}
],
"model": "gpt-4"
}
sanitized = _sanitize_request_body_for_spend_logs_payload(request_body)
# Verify the content was truncated
truncated_content = sanitized["messages"][0]["content"]
assert len(truncated_content) > MAX_STRING_LENGTH_PROMPT_IN_DB # includes truncation message
assert truncated_content.startswith("x" * MAX_STRING_LENGTH_PROMPT_IN_DB)
assert LITELLM_TRUNCATED_PAYLOAD_FIELD in truncated_content
assert "500 chars" in truncated_content
def test_small_request_no_truncation():
"""
Test that small strings are not truncated by MAX_STRING_LENGTH_PROMPT_IN_DB
"""
from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB
# Create a small string that's under the threshold
small_content = "x" * (MAX_STRING_LENGTH_PROMPT_IN_DB - 100)
request_body = {
"messages": [
{"role": "user", "content": small_content}
],
"model": "gpt-4"
}
sanitized = _sanitize_request_body_for_spend_logs_payload(request_body)
# Verify the content was NOT truncated
assert sanitized["messages"][0]["content"] == small_content
assert len(sanitized["messages"][0]["content"]) == MAX_STRING_LENGTH_PROMPT_IN_DB - 100
def test_configurable_string_length_env_var(monkeypatch):
"""
Test that MAX_STRING_LENGTH_PROMPT_IN_DB can be configured via environment variable
"""
# Set environment variable to a custom value
monkeypatch.setenv("MAX_STRING_LENGTH_PROMPT_IN_DB", "500")
# Import after setting env var to ensure it picks up the new value
import importlib
import litellm.constants
import litellm.proxy.spend_tracking.spend_tracking_utils
importlib.reload(litellm.constants)
importlib.reload(litellm.proxy.spend_tracking.spend_tracking_utils)
from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB, LITELLM_TRUNCATED_PAYLOAD_FIELD
from litellm.proxy.spend_tracking.spend_tracking_utils import _sanitize_request_body_for_spend_logs_payload
# Verify the constant was set to the env var value
assert MAX_STRING_LENGTH_PROMPT_IN_DB == 500
# Test truncation with the custom value
large_content = "y" * 750 # 250 chars over the custom limit
request_body = {
"messages": [
{"role": "user", "content": large_content}
],
"model": "gpt-4"
}
sanitized = _sanitize_request_body_for_spend_logs_payload(request_body)
# Verify truncation occurred at the custom threshold
truncated_content = sanitized["messages"][0]["content"]
assert truncated_content.startswith("y" * 500)
assert LITELLM_TRUNCATED_PAYLOAD_FIELD in truncated_content
assert "250 chars" in truncated_content

View file

@ -757,7 +757,7 @@ def test_optional_combine_thinking_block_with_none_content(
# Second chunk with reasoning_content and None content
second_chunk = {
"id": "chunk2",
"id": "chunk2",
"object": "chat.completion.chunk",
"created": 1741037891,
"model": "deepseek-reasoner",
@ -776,16 +776,13 @@ def test_optional_combine_thinking_block_with_none_content(
# Final chunk with actual content - should add </think> tag
final_chunk = {
"id": "chunk3",
"object": "chat.completion.chunk",
"object": "chat.completion.chunk",
"created": 1741037892,
"model": "deepseek-reasoner",
"choices": [
{
"index": 0,
"delta": {
"content": "The answer is 42",
"reasoning_content": None
},
"delta": {"content": "The answer is 42", "reasoning_content": None},
"finish_reason": None,
}
],
@ -796,12 +793,15 @@ def test_optional_combine_thinking_block_with_none_content(
initialized_custom_stream_wrapper._optional_combine_thinking_block_in_choices(
first_response
)
assert first_response.choices[0].delta.content == "<think>Let me think about this problem"
assert (
first_response.choices[0].delta.content
== "<think>Let me think about this problem"
)
assert not hasattr(first_response.choices[0].delta, "reasoning_content")
assert initialized_custom_stream_wrapper.sent_first_thinking_block is True
# Process second chunk - should work with continued reasoning
second_response = ModelResponseStream(**second_chunk)
second_response = ModelResponseStream(**second_chunk)
initialized_custom_stream_wrapper._optional_combine_thinking_block_in_choices(
second_response
)
@ -822,76 +822,99 @@ def test_has_special_delta_content(
initialized_custom_stream_wrapper: CustomStreamWrapper,
):
"""Test the _has_special_delta_content helper method"""
# Test empty choices
empty_response = ModelResponseStream(
id="test", created=1742056047, model=None, choices=[]
)
assert not initialized_custom_stream_wrapper._has_special_delta_content(empty_response)
assert not initialized_custom_stream_wrapper._has_special_delta_content(
empty_response
)
# Test with tool_calls (simulate with mock object)
tool_call_response = ModelResponseStream(
id="test", created=1742056047, model=None,
id="test",
created=1742056047,
model=None,
choices=[
StreamingChoices(
finish_reason=None, index=0,
delta=Delta(content=None, tool_calls=[{"id": "test"}])
finish_reason=None,
index=0,
delta=Delta(
content=None,
tool_calls=[
{
"id": "test",
"function": {"arguments": "{}", "name": "test_func"},
}
],
),
)
]
],
)
assert initialized_custom_stream_wrapper._has_special_delta_content(tool_call_response)
assert initialized_custom_stream_wrapper._has_special_delta_content(
tool_call_response
)
# Test with function_call (simulate with mock object)
function_call_response = ModelResponseStream(
id="test", created=1742056047, model=None,
id="test",
created=1742056047,
model=None,
choices=[
StreamingChoices(
finish_reason=None, index=0,
delta=Delta(content=None, function_call={"name": "test_func"})
finish_reason=None,
index=0,
delta=Delta(
content=None, function_call={"name": "test_func", "arguments": "{}"}
),
)
]
],
)
assert initialized_custom_stream_wrapper._has_special_delta_content(function_call_response)
assert initialized_custom_stream_wrapper._has_special_delta_content(
function_call_response
)
# Test with audio (simulate by adding audio attribute)
audio_response = ModelResponseStream(
id="test", created=1742056047, model=None,
id="test",
created=1742056047,
model=None,
choices=[
StreamingChoices(
finish_reason=None, index=0,
delta=Delta(content=None)
)
]
StreamingChoices(finish_reason=None, index=0, delta=Delta(content=None))
],
)
# Manually add audio attribute to delta
audio_response.choices[0].delta.audio = {"transcript": "test"}
assert initialized_custom_stream_wrapper._has_special_delta_content(audio_response)
# Test with image (simulate by adding image attribute)
image_response = ModelResponseStream(
id="test", created=1742056047, model=None,
id="test",
created=1742056047,
model=None,
choices=[
StreamingChoices(
finish_reason=None, index=0,
delta=Delta(content=None)
)
]
StreamingChoices(finish_reason=None, index=0, delta=Delta(content=None))
],
)
# Manually add image attribute to delta
image_response.choices[0].delta.image = {"url": "test.jpg"}
assert initialized_custom_stream_wrapper._has_special_delta_content(image_response)
# Test with regular content (should return False)
regular_response = ModelResponseStream(
id="test", created=1742056047, model=None,
id="test",
created=1742056047,
model=None,
choices=[
StreamingChoices(
finish_reason=None, index=0,
delta=Delta(content="Hello world")
finish_reason=None, index=0, delta=Delta(content="Hello world")
)
]
],
)
assert not initialized_custom_stream_wrapper._has_special_delta_content(
regular_response
)
assert not initialized_custom_stream_wrapper._has_special_delta_content(regular_response)
def test_handle_special_delta_content(
@ -899,21 +922,26 @@ def test_handle_special_delta_content(
):
"""Test the _handle_special_delta_content helper method"""
test_response = ModelResponseStream(
id="test", created=1742056047, model=None,
id="test",
created=1742056047,
model=None,
choices=[
StreamingChoices(
finish_reason=None, index=0,
delta=Delta(content="test", role="assistant")
finish_reason=None,
index=0,
delta=Delta(content="test", role="assistant"),
)
]
],
)
# The method should call strip_role_from_delta
result = initialized_custom_stream_wrapper._handle_special_delta_content(test_response)
result = initialized_custom_stream_wrapper._handle_special_delta_content(
test_response
)
# Should return the same response object (modified)
assert result is test_response
# Should have set sent_first_chunk to True
assert initialized_custom_stream_wrapper.sent_first_chunk is True
@ -922,32 +950,38 @@ def test_has_any_special_delta_attributes(
initialized_custom_stream_wrapper: CustomStreamWrapper,
):
"""Test the _has_any_special_delta_attributes helper method"""
# Test with delta that has audio attribute
class MockDelta:
def __init__(self):
self.audio = {"transcript": "Hello world"}
audio_delta = MockDelta()
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(audio_delta)
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(
audio_delta
)
assert result is True
# Test with delta that has image attribute
class MockDeltaImage:
def __init__(self):
self.image = {"url": "test.jpg"}
image_delta = MockDeltaImage()
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(image_delta)
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(
image_delta
)
assert result is True
# Test with delta that has no special attributes
class MockDeltaRegular:
def __init__(self):
self.content = "regular content"
regular_delta = MockDeltaRegular()
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(regular_delta)
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(
regular_delta
)
assert result is False
@ -955,48 +989,50 @@ def test_handle_special_delta_attributes(
initialized_custom_stream_wrapper: CustomStreamWrapper,
):
"""Test the _handle_special_delta_attributes helper method"""
# Create a model response
model_response = ModelResponseStream(
id="test", created=1742056047, model=None,
id="test",
created=1742056047,
model=None,
choices=[
StreamingChoices(
finish_reason=None, index=0,
delta=Delta(content="test")
)
]
StreamingChoices(finish_reason=None, index=0, delta=Delta(content="test"))
],
)
# Test with delta that has audio attribute
class MockDelta:
def __init__(self):
self.audio = {"transcript": "Hello world"}
audio_delta = MockDelta()
initialized_custom_stream_wrapper._handle_special_delta_attributes(audio_delta, model_response)
initialized_custom_stream_wrapper._handle_special_delta_attributes(
audio_delta, model_response
)
# Should copy the audio attribute
assert hasattr(model_response.choices[0].delta, "audio")
assert model_response.choices[0].delta.audio == {"transcript": "Hello world"}
# Test with delta that has image attribute
class MockDeltaImage:
def __init__(self):
self.image = {"url": "test.jpg"}
image_delta = MockDeltaImage()
model_response2 = ModelResponseStream(
id="test", created=1742056047, model=None,
id="test",
created=1742056047,
model=None,
choices=[
StreamingChoices(
finish_reason=None, index=0,
delta=Delta(content="test")
)
]
StreamingChoices(finish_reason=None, index=0, delta=Delta(content="test"))
],
)
initialized_custom_stream_wrapper._handle_special_delta_attributes(image_delta, model_response2)
initialized_custom_stream_wrapper._handle_special_delta_attributes(
image_delta, model_response2
)
# Should copy the image attribute
assert hasattr(model_response2.choices[0].delta, "image")
assert model_response2.choices[0].delta.image == {"url": "test.jpg"}
@ -1006,30 +1042,38 @@ def test_has_special_delta_attribute(
initialized_custom_stream_wrapper: CustomStreamWrapper,
):
"""Test the _has_special_delta_attribute helper method"""
# Test with None delta
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(None, "audio")
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(
None, "audio"
)
# Test with delta that has the attribute
class MockDelta:
def __init__(self):
self.audio = {"transcript": "test"}
delta_with_audio = MockDelta()
assert initialized_custom_stream_wrapper._has_special_delta_attribute(delta_with_audio, "audio")
assert initialized_custom_stream_wrapper._has_special_delta_attribute(
delta_with_audio, "audio"
)
# Test with delta that doesn't have the attribute
class MockDeltaNoAudio:
def __init__(self):
self.content = "test"
delta_without_audio = MockDeltaNoAudio()
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(delta_without_audio, "audio")
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(
delta_without_audio, "audio"
)
# Test with delta that has the attribute but it's None
class MockDeltaNone:
def __init__(self):
self.audio = None
delta_with_none = MockDeltaNone()
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(delta_with_none, "audio")
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(
delta_with_none, "audio"
)