mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge pull request #14042 from WilsonSunBritten/10988-set-truncation-threshold
Allow configuration to set threshold before request entry in spend log gets truncated
This commit is contained in:
commit
fffccc0675
5 changed files with 481 additions and 326 deletions
|
|
@ -157,6 +157,7 @@ NON_LLM_CONNECTION_TIMEOUT = int(
|
|||
os.getenv("NON_LLM_CONNECTION_TIMEOUT", 15)
|
||||
) # timeout for adjacent services (e.g. jwt auth)
|
||||
MAX_EXCEPTION_MESSAGE_LENGTH = int(os.getenv("MAX_EXCEPTION_MESSAGE_LENGTH", 2000))
|
||||
MAX_STRING_LENGTH_PROMPT_IN_DB = int(os.getenv("MAX_STRING_LENGTH_PROMPT_IN_DB", 1000))
|
||||
BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75))
|
||||
REPLICATE_POLLING_DELAY_SECONDS = float(
|
||||
os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5)
|
||||
|
|
@ -486,227 +487,247 @@ _openai_like_providers: List = [
|
|||
"watsonx",
|
||||
] # private helper. similar to openai but require some custom auth / endpoint handling, so can't use the openai sdk
|
||||
# well supported replicate llms
|
||||
replicate_models: set = set([
|
||||
# llama replicate supported LLMs
|
||||
"replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf",
|
||||
"a16z-infra/llama-2-13b-chat:2a7f981751ec7fdf87b5b91ad4db53683a98082e9ff7bfd12c8cd5ea85980a52",
|
||||
"meta/codellama-13b:1c914d844307b0588599b8393480a3ba917b660c7e9dfae681542b5325f228db",
|
||||
# Vicuna
|
||||
"replicate/vicuna-13b:6282abe6a492de4145d7bb601023762212f9ddbbe78278bd6771c8b3b2f2a13b",
|
||||
"joehoover/instructblip-vicuna13b:c4c54e3c8c97cd50c2d2fec9be3b6065563ccf7d43787fb99f84151b867178fe",
|
||||
# Flan T-5
|
||||
"daanelson/flan-t5-large:ce962b3f6792a57074a601d3979db5839697add2e4e02696b3ced4c022d4767f",
|
||||
# Others
|
||||
"replicate/dolly-v2-12b:ef0e1aefc61f8e096ebe4db6b2bacc297daf2ef6899f0f7e001ec445893500e5",
|
||||
"replit/replit-code-v1-3b:b84f4c074b807211cd75e3e8b1589b6399052125b4c27106e43d47189e8415ad",
|
||||
])
|
||||
replicate_models: set = set(
|
||||
[
|
||||
# llama replicate supported LLMs
|
||||
"replicate/llama-2-70b-chat:2796ee9483c3fd7aa2e171d38f4ca12251a30609463dcfd4cd76703f22e96cdf",
|
||||
"a16z-infra/llama-2-13b-chat:2a7f981751ec7fdf87b5b91ad4db53683a98082e9ff7bfd12c8cd5ea85980a52",
|
||||
"meta/codellama-13b:1c914d844307b0588599b8393480a3ba917b660c7e9dfae681542b5325f228db",
|
||||
# Vicuna
|
||||
"replicate/vicuna-13b:6282abe6a492de4145d7bb601023762212f9ddbbe78278bd6771c8b3b2f2a13b",
|
||||
"joehoover/instructblip-vicuna13b:c4c54e3c8c97cd50c2d2fec9be3b6065563ccf7d43787fb99f84151b867178fe",
|
||||
# Flan T-5
|
||||
"daanelson/flan-t5-large:ce962b3f6792a57074a601d3979db5839697add2e4e02696b3ced4c022d4767f",
|
||||
# Others
|
||||
"replicate/dolly-v2-12b:ef0e1aefc61f8e096ebe4db6b2bacc297daf2ef6899f0f7e001ec445893500e5",
|
||||
"replit/replit-code-v1-3b:b84f4c074b807211cd75e3e8b1589b6399052125b4c27106e43d47189e8415ad",
|
||||
]
|
||||
)
|
||||
|
||||
clarifai_models: set = set([
|
||||
"clarifai/meta.Llama-3.Llama-3-8B-Instruct",
|
||||
"clarifai/gcp.generate.gemma-1_1-7b-it",
|
||||
"clarifai/mistralai.completion.mixtral-8x22B",
|
||||
"clarifai/cohere.generate.command-r-plus",
|
||||
"clarifai/databricks.drbx.dbrx-instruct",
|
||||
"clarifai/mistralai.completion.mistral-large",
|
||||
"clarifai/mistralai.completion.mistral-medium",
|
||||
"clarifai/mistralai.completion.mistral-small",
|
||||
"clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1",
|
||||
"clarifai/gcp.generate.gemma-2b-it",
|
||||
"clarifai/gcp.generate.gemma-7b-it",
|
||||
"clarifai/deci.decilm.deciLM-7B-instruct",
|
||||
"clarifai/mistralai.completion.mistral-7B-Instruct",
|
||||
"clarifai/gcp.generate.gemini-pro",
|
||||
"clarifai/anthropic.completion.claude-v1",
|
||||
"clarifai/anthropic.completion.claude-instant-1_2",
|
||||
"clarifai/anthropic.completion.claude-instant",
|
||||
"clarifai/anthropic.completion.claude-v2",
|
||||
"clarifai/anthropic.completion.claude-2_1",
|
||||
"clarifai/meta.Llama-2.codeLlama-70b-Python",
|
||||
"clarifai/meta.Llama-2.codeLlama-70b-Instruct",
|
||||
"clarifai/openai.completion.gpt-3_5-turbo-instruct",
|
||||
"clarifai/meta.Llama-2.llama2-7b-chat",
|
||||
"clarifai/meta.Llama-2.llama2-13b-chat",
|
||||
"clarifai/meta.Llama-2.llama2-70b-chat",
|
||||
"clarifai/openai.chat-completion.gpt-4-turbo",
|
||||
"clarifai/microsoft.text-generation.phi-2",
|
||||
"clarifai/meta.Llama-2.llama2-7b-chat-vllm",
|
||||
"clarifai/upstage.solar.solar-10_7b-instruct",
|
||||
"clarifai/openchat.openchat.openchat-3_5-1210",
|
||||
"clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B",
|
||||
"clarifai/gcp.generate.text-bison",
|
||||
"clarifai/meta.Llama-2.llamaGuard-7b",
|
||||
"clarifai/fblgit.una-cybertron.una-cybertron-7b-v2",
|
||||
"clarifai/openai.chat-completion.GPT-4",
|
||||
"clarifai/openai.chat-completion.GPT-3_5-turbo",
|
||||
"clarifai/ai21.complete.Jurassic2-Grande",
|
||||
"clarifai/ai21.complete.Jurassic2-Grande-Instruct",
|
||||
"clarifai/ai21.complete.Jurassic2-Jumbo-Instruct",
|
||||
"clarifai/ai21.complete.Jurassic2-Jumbo",
|
||||
"clarifai/ai21.complete.Jurassic2-Large",
|
||||
"clarifai/cohere.generate.cohere-generate-command",
|
||||
"clarifai/wizardlm.generate.wizardCoder-Python-34B",
|
||||
"clarifai/wizardlm.generate.wizardLM-70B",
|
||||
"clarifai/tiiuae.falcon.falcon-40b-instruct",
|
||||
"clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat",
|
||||
"clarifai/gcp.generate.code-gecko",
|
||||
"clarifai/gcp.generate.code-bison",
|
||||
"clarifai/mistralai.completion.mistral-7B-OpenOrca",
|
||||
"clarifai/mistralai.completion.openHermes-2-mistral-7B",
|
||||
"clarifai/wizardlm.generate.wizardLM-13B",
|
||||
"clarifai/huggingface-research.zephyr.zephyr-7B-alpha",
|
||||
"clarifai/wizardlm.generate.wizardCoder-15B",
|
||||
"clarifai/microsoft.text-generation.phi-1_5",
|
||||
"clarifai/databricks.Dolly-v2.dolly-v2-12b",
|
||||
"clarifai/bigcode.code.StarCoder",
|
||||
"clarifai/salesforce.xgen.xgen-7b-8k-instruct",
|
||||
"clarifai/mosaicml.mpt.mpt-7b-instruct",
|
||||
"clarifai/anthropic.completion.claude-3-opus",
|
||||
"clarifai/anthropic.completion.claude-3-sonnet",
|
||||
"clarifai/gcp.generate.gemini-1_5-pro",
|
||||
"clarifai/gcp.generate.imagen-2",
|
||||
"clarifai/salesforce.blip.general-english-image-caption-blip-2",
|
||||
])
|
||||
clarifai_models: set = set(
|
||||
[
|
||||
"clarifai/meta.Llama-3.Llama-3-8B-Instruct",
|
||||
"clarifai/gcp.generate.gemma-1_1-7b-it",
|
||||
"clarifai/mistralai.completion.mixtral-8x22B",
|
||||
"clarifai/cohere.generate.command-r-plus",
|
||||
"clarifai/databricks.drbx.dbrx-instruct",
|
||||
"clarifai/mistralai.completion.mistral-large",
|
||||
"clarifai/mistralai.completion.mistral-medium",
|
||||
"clarifai/mistralai.completion.mistral-small",
|
||||
"clarifai/mistralai.completion.mixtral-8x7B-Instruct-v0_1",
|
||||
"clarifai/gcp.generate.gemma-2b-it",
|
||||
"clarifai/gcp.generate.gemma-7b-it",
|
||||
"clarifai/deci.decilm.deciLM-7B-instruct",
|
||||
"clarifai/mistralai.completion.mistral-7B-Instruct",
|
||||
"clarifai/gcp.generate.gemini-pro",
|
||||
"clarifai/anthropic.completion.claude-v1",
|
||||
"clarifai/anthropic.completion.claude-instant-1_2",
|
||||
"clarifai/anthropic.completion.claude-instant",
|
||||
"clarifai/anthropic.completion.claude-v2",
|
||||
"clarifai/anthropic.completion.claude-2_1",
|
||||
"clarifai/meta.Llama-2.codeLlama-70b-Python",
|
||||
"clarifai/meta.Llama-2.codeLlama-70b-Instruct",
|
||||
"clarifai/openai.completion.gpt-3_5-turbo-instruct",
|
||||
"clarifai/meta.Llama-2.llama2-7b-chat",
|
||||
"clarifai/meta.Llama-2.llama2-13b-chat",
|
||||
"clarifai/meta.Llama-2.llama2-70b-chat",
|
||||
"clarifai/openai.chat-completion.gpt-4-turbo",
|
||||
"clarifai/microsoft.text-generation.phi-2",
|
||||
"clarifai/meta.Llama-2.llama2-7b-chat-vllm",
|
||||
"clarifai/upstage.solar.solar-10_7b-instruct",
|
||||
"clarifai/openchat.openchat.openchat-3_5-1210",
|
||||
"clarifai/togethercomputer.stripedHyena.stripedHyena-Nous-7B",
|
||||
"clarifai/gcp.generate.text-bison",
|
||||
"clarifai/meta.Llama-2.llamaGuard-7b",
|
||||
"clarifai/fblgit.una-cybertron.una-cybertron-7b-v2",
|
||||
"clarifai/openai.chat-completion.GPT-4",
|
||||
"clarifai/openai.chat-completion.GPT-3_5-turbo",
|
||||
"clarifai/ai21.complete.Jurassic2-Grande",
|
||||
"clarifai/ai21.complete.Jurassic2-Grande-Instruct",
|
||||
"clarifai/ai21.complete.Jurassic2-Jumbo-Instruct",
|
||||
"clarifai/ai21.complete.Jurassic2-Jumbo",
|
||||
"clarifai/ai21.complete.Jurassic2-Large",
|
||||
"clarifai/cohere.generate.cohere-generate-command",
|
||||
"clarifai/wizardlm.generate.wizardCoder-Python-34B",
|
||||
"clarifai/wizardlm.generate.wizardLM-70B",
|
||||
"clarifai/tiiuae.falcon.falcon-40b-instruct",
|
||||
"clarifai/togethercomputer.RedPajama.RedPajama-INCITE-7B-Chat",
|
||||
"clarifai/gcp.generate.code-gecko",
|
||||
"clarifai/gcp.generate.code-bison",
|
||||
"clarifai/mistralai.completion.mistral-7B-OpenOrca",
|
||||
"clarifai/mistralai.completion.openHermes-2-mistral-7B",
|
||||
"clarifai/wizardlm.generate.wizardLM-13B",
|
||||
"clarifai/huggingface-research.zephyr.zephyr-7B-alpha",
|
||||
"clarifai/wizardlm.generate.wizardCoder-15B",
|
||||
"clarifai/microsoft.text-generation.phi-1_5",
|
||||
"clarifai/databricks.Dolly-v2.dolly-v2-12b",
|
||||
"clarifai/bigcode.code.StarCoder",
|
||||
"clarifai/salesforce.xgen.xgen-7b-8k-instruct",
|
||||
"clarifai/mosaicml.mpt.mpt-7b-instruct",
|
||||
"clarifai/anthropic.completion.claude-3-opus",
|
||||
"clarifai/anthropic.completion.claude-3-sonnet",
|
||||
"clarifai/gcp.generate.gemini-1_5-pro",
|
||||
"clarifai/gcp.generate.imagen-2",
|
||||
"clarifai/salesforce.blip.general-english-image-caption-blip-2",
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
huggingface_models: set = set([
|
||||
"meta-llama/Llama-2-7b-hf",
|
||||
"meta-llama/Llama-2-7b-chat-hf",
|
||||
"meta-llama/Llama-2-13b-hf",
|
||||
"meta-llama/Llama-2-13b-chat-hf",
|
||||
"meta-llama/Llama-2-70b-hf",
|
||||
"meta-llama/Llama-2-70b-chat-hf",
|
||||
"meta-llama/Llama-2-7b",
|
||||
"meta-llama/Llama-2-7b-chat",
|
||||
"meta-llama/Llama-2-13b",
|
||||
"meta-llama/Llama-2-13b-chat",
|
||||
"meta-llama/Llama-2-70b",
|
||||
"meta-llama/Llama-2-70b-chat",
|
||||
]) # these have been tested on extensively. But by default all text2text-generation and text-generation models are supported by liteLLM. - https://docs.litellm.ai/docs/providers
|
||||
empower_models = set([
|
||||
"empower/empower-functions",
|
||||
"empower/empower-functions-small",
|
||||
])
|
||||
huggingface_models: set = set(
|
||||
[
|
||||
"meta-llama/Llama-2-7b-hf",
|
||||
"meta-llama/Llama-2-7b-chat-hf",
|
||||
"meta-llama/Llama-2-13b-hf",
|
||||
"meta-llama/Llama-2-13b-chat-hf",
|
||||
"meta-llama/Llama-2-70b-hf",
|
||||
"meta-llama/Llama-2-70b-chat-hf",
|
||||
"meta-llama/Llama-2-7b",
|
||||
"meta-llama/Llama-2-7b-chat",
|
||||
"meta-llama/Llama-2-13b",
|
||||
"meta-llama/Llama-2-13b-chat",
|
||||
"meta-llama/Llama-2-70b",
|
||||
"meta-llama/Llama-2-70b-chat",
|
||||
]
|
||||
) # these have been tested on extensively. But by default all text2text-generation and text-generation models are supported by liteLLM. - https://docs.litellm.ai/docs/providers
|
||||
empower_models = set(
|
||||
[
|
||||
"empower/empower-functions",
|
||||
"empower/empower-functions-small",
|
||||
]
|
||||
)
|
||||
|
||||
together_ai_models: set = set([
|
||||
# llama llms - chat
|
||||
"togethercomputer/llama-2-70b-chat",
|
||||
# llama llms - language / instruct
|
||||
"togethercomputer/llama-2-70b",
|
||||
"togethercomputer/LLaMA-2-7B-32K",
|
||||
"togethercomputer/Llama-2-7B-32K-Instruct",
|
||||
"togethercomputer/llama-2-7b",
|
||||
# falcon llms
|
||||
"togethercomputer/falcon-40b-instruct",
|
||||
"togethercomputer/falcon-7b-instruct",
|
||||
# alpaca
|
||||
"togethercomputer/alpaca-7b",
|
||||
# chat llms
|
||||
"HuggingFaceH4/starchat-alpha",
|
||||
# code llms
|
||||
"togethercomputer/CodeLlama-34b",
|
||||
"togethercomputer/CodeLlama-34b-Instruct",
|
||||
"togethercomputer/CodeLlama-34b-Python",
|
||||
"defog/sqlcoder",
|
||||
"NumbersStation/nsql-llama-2-7B",
|
||||
"WizardLM/WizardCoder-15B-V1.0",
|
||||
"WizardLM/WizardCoder-Python-34B-V1.0",
|
||||
# language llms
|
||||
"NousResearch/Nous-Hermes-Llama2-13b",
|
||||
"Austism/chronos-hermes-13b",
|
||||
"upstage/SOLAR-0-70b-16bit",
|
||||
"WizardLM/WizardLM-70B-V1.0",
|
||||
])
|
||||
# supports all together ai models, just pass in the model id e.g. completion(model="together_computer/replit_code_3b",...)
|
||||
together_ai_models: set = set(
|
||||
[
|
||||
# llama llms - chat
|
||||
"togethercomputer/llama-2-70b-chat",
|
||||
# llama llms - language / instruct
|
||||
"togethercomputer/llama-2-70b",
|
||||
"togethercomputer/LLaMA-2-7B-32K",
|
||||
"togethercomputer/Llama-2-7B-32K-Instruct",
|
||||
"togethercomputer/llama-2-7b",
|
||||
# falcon llms
|
||||
"togethercomputer/falcon-40b-instruct",
|
||||
"togethercomputer/falcon-7b-instruct",
|
||||
# alpaca
|
||||
"togethercomputer/alpaca-7b",
|
||||
# chat llms
|
||||
"HuggingFaceH4/starchat-alpha",
|
||||
# code llms
|
||||
"togethercomputer/CodeLlama-34b",
|
||||
"togethercomputer/CodeLlama-34b-Instruct",
|
||||
"togethercomputer/CodeLlama-34b-Python",
|
||||
"defog/sqlcoder",
|
||||
"NumbersStation/nsql-llama-2-7B",
|
||||
"WizardLM/WizardCoder-15B-V1.0",
|
||||
"WizardLM/WizardCoder-Python-34B-V1.0",
|
||||
# language llms
|
||||
"NousResearch/Nous-Hermes-Llama2-13b",
|
||||
"Austism/chronos-hermes-13b",
|
||||
"upstage/SOLAR-0-70b-16bit",
|
||||
"WizardLM/WizardLM-70B-V1.0",
|
||||
]
|
||||
)
|
||||
# supports all together ai models, just pass in the model id e.g. completion(model="together_computer/replit_code_3b",...)
|
||||
|
||||
|
||||
baseten_models: set = set([
|
||||
"qvv0xeq",
|
||||
"q841o8w",
|
||||
"31dxrj3",
|
||||
]) # FALCON 7B # WizardLM # Mosaic ML
|
||||
baseten_models: set = set(
|
||||
[
|
||||
"qvv0xeq",
|
||||
"q841o8w",
|
||||
"31dxrj3",
|
||||
]
|
||||
) # FALCON 7B # WizardLM # Mosaic ML
|
||||
|
||||
featherless_ai_models: set = set([
|
||||
"featherless-ai/Qwerky-72B",
|
||||
"featherless-ai/Qwerky-QwQ-32B",
|
||||
"Qwen/Qwen2.5-72B-Instruct",
|
||||
"all-hands/openhands-lm-32b-v0.1",
|
||||
"Qwen/Qwen2.5-Coder-32B-Instruct",
|
||||
"deepseek-ai/DeepSeek-V3-0324",
|
||||
"mistralai/Mistral-Small-24B-Instruct-2501",
|
||||
"mistralai/Mistral-Nemo-Instruct-2407",
|
||||
"ProdeusUnity/Stellar-Odyssey-12b-v0.0",
|
||||
])
|
||||
featherless_ai_models: set = set(
|
||||
[
|
||||
"featherless-ai/Qwerky-72B",
|
||||
"featherless-ai/Qwerky-QwQ-32B",
|
||||
"Qwen/Qwen2.5-72B-Instruct",
|
||||
"all-hands/openhands-lm-32b-v0.1",
|
||||
"Qwen/Qwen2.5-Coder-32B-Instruct",
|
||||
"deepseek-ai/DeepSeek-V3-0324",
|
||||
"mistralai/Mistral-Small-24B-Instruct-2501",
|
||||
"mistralai/Mistral-Nemo-Instruct-2407",
|
||||
"ProdeusUnity/Stellar-Odyssey-12b-v0.0",
|
||||
]
|
||||
)
|
||||
|
||||
nebius_models: set = set([
|
||||
# deepseek models
|
||||
"deepseek-ai/DeepSeek-R1-0528",
|
||||
"deepseek-ai/DeepSeek-V3-0324",
|
||||
"deepseek-ai/DeepSeek-V3",
|
||||
"deepseek-ai/DeepSeek-R1",
|
||||
"deepseek-ai/DeepSeek-R1-Distill-Llama-70B",
|
||||
# google models
|
||||
"google/gemma-2-2b-it",
|
||||
"google/gemma-2-9b-it-fast",
|
||||
# llama models
|
||||
"meta-llama/Llama-3.3-70B-Instruct",
|
||||
"meta-llama/Meta-Llama-3.1-70B-Instruct",
|
||||
"meta-llama/Meta-Llama-3.1-8B-Instruct",
|
||||
"meta-llama/Meta-Llama-3.1-405B-Instruct",
|
||||
"NousResearch/Hermes-3-Llama-405B",
|
||||
# microsoft models
|
||||
"microsoft/phi-4",
|
||||
# mistral models
|
||||
"mistralai/Mistral-Nemo-Instruct-2407",
|
||||
"mistralai/Devstral-Small-2505",
|
||||
# moonshot models
|
||||
"moonshotai/Kimi-K2-Instruct",
|
||||
# nvidia models
|
||||
"nvidia/Llama-3_1-Nemotron-Ultra-253B-v1",
|
||||
"nvidia/Llama-3_3-Nemotron-Super-49B-v1",
|
||||
# openai models
|
||||
"openai/gpt-oss-120b",
|
||||
"openai/gpt-oss-20b",
|
||||
# qwen models
|
||||
"Qwen/Qwen3-Coder-480B-A35B-Instruct",
|
||||
"Qwen/Qwen3-235B-A22B-Instruct-2507",
|
||||
"Qwen/Qwen3-235B-A22B",
|
||||
"Qwen/Qwen3-30B-A3B",
|
||||
"Qwen/Qwen3-32B",
|
||||
"Qwen/Qwen3-14B",
|
||||
"Qwen/Qwen3-4B-fast",
|
||||
"Qwen/Qwen2.5-Coder-7B",
|
||||
"Qwen/Qwen2.5-Coder-32B-Instruct",
|
||||
"Qwen/Qwen2.5-72B-Instruct",
|
||||
"Qwen/QwQ-32B",
|
||||
"Qwen/Qwen3-30B-A3B-Thinking-2507",
|
||||
"Qwen/Qwen3-30B-A3B-Instruct-2507",
|
||||
# zai models
|
||||
"zai-org/GLM-4.5",
|
||||
"zai-org/GLM-4.5-Air",
|
||||
# other models
|
||||
"aaditya/Llama3-OpenBioLLM-70B",
|
||||
"ProdeusUnity/Stellar-Odyssey-12b-v0.0",
|
||||
"all-hands/openhands-lm-32b-v0.1",
|
||||
])
|
||||
nebius_models: set = set(
|
||||
[
|
||||
# deepseek models
|
||||
"deepseek-ai/DeepSeek-R1-0528",
|
||||
"deepseek-ai/DeepSeek-V3-0324",
|
||||
"deepseek-ai/DeepSeek-V3",
|
||||
"deepseek-ai/DeepSeek-R1",
|
||||
"deepseek-ai/DeepSeek-R1-Distill-Llama-70B",
|
||||
# google models
|
||||
"google/gemma-2-2b-it",
|
||||
"google/gemma-2-9b-it-fast",
|
||||
# llama models
|
||||
"meta-llama/Llama-3.3-70B-Instruct",
|
||||
"meta-llama/Meta-Llama-3.1-70B-Instruct",
|
||||
"meta-llama/Meta-Llama-3.1-8B-Instruct",
|
||||
"meta-llama/Meta-Llama-3.1-405B-Instruct",
|
||||
"NousResearch/Hermes-3-Llama-405B",
|
||||
# microsoft models
|
||||
"microsoft/phi-4",
|
||||
# mistral models
|
||||
"mistralai/Mistral-Nemo-Instruct-2407",
|
||||
"mistralai/Devstral-Small-2505",
|
||||
# moonshot models
|
||||
"moonshotai/Kimi-K2-Instruct",
|
||||
# nvidia models
|
||||
"nvidia/Llama-3_1-Nemotron-Ultra-253B-v1",
|
||||
"nvidia/Llama-3_3-Nemotron-Super-49B-v1",
|
||||
# openai models
|
||||
"openai/gpt-oss-120b",
|
||||
"openai/gpt-oss-20b",
|
||||
# qwen models
|
||||
"Qwen/Qwen3-Coder-480B-A35B-Instruct",
|
||||
"Qwen/Qwen3-235B-A22B-Instruct-2507",
|
||||
"Qwen/Qwen3-235B-A22B",
|
||||
"Qwen/Qwen3-30B-A3B",
|
||||
"Qwen/Qwen3-32B",
|
||||
"Qwen/Qwen3-14B",
|
||||
"Qwen/Qwen3-4B-fast",
|
||||
"Qwen/Qwen2.5-Coder-7B",
|
||||
"Qwen/Qwen2.5-Coder-32B-Instruct",
|
||||
"Qwen/Qwen2.5-72B-Instruct",
|
||||
"Qwen/QwQ-32B",
|
||||
"Qwen/Qwen3-30B-A3B-Thinking-2507",
|
||||
"Qwen/Qwen3-30B-A3B-Instruct-2507",
|
||||
# zai models
|
||||
"zai-org/GLM-4.5",
|
||||
"zai-org/GLM-4.5-Air",
|
||||
# other models
|
||||
"aaditya/Llama3-OpenBioLLM-70B",
|
||||
"ProdeusUnity/Stellar-Odyssey-12b-v0.0",
|
||||
"all-hands/openhands-lm-32b-v0.1",
|
||||
]
|
||||
)
|
||||
|
||||
dashscope_models: set = set([
|
||||
"qwen-turbo",
|
||||
"qwen-plus",
|
||||
"qwen-max",
|
||||
"qwen-turbo-latest",
|
||||
"qwen-plus-latest",
|
||||
"qwen-max-latest",
|
||||
"qwq-32b",
|
||||
"qwen3-235b-a22b",
|
||||
"qwen3-32b",
|
||||
"qwen3-30b-a3b",
|
||||
])
|
||||
dashscope_models: set = set(
|
||||
[
|
||||
"qwen-turbo",
|
||||
"qwen-plus",
|
||||
"qwen-max",
|
||||
"qwen-turbo-latest",
|
||||
"qwen-plus-latest",
|
||||
"qwen-max-latest",
|
||||
"qwq-32b",
|
||||
"qwen3-235b-a22b",
|
||||
"qwen3-32b",
|
||||
"qwen3-30b-a3b",
|
||||
]
|
||||
)
|
||||
|
||||
nebius_embedding_models: set = set([
|
||||
"BAAI/bge-en-icl",
|
||||
"BAAI/bge-multilingual-gemma2",
|
||||
"intfloat/e5-mistral-7b-instruct",
|
||||
])
|
||||
nebius_embedding_models: set = set(
|
||||
[
|
||||
"BAAI/bge-en-icl",
|
||||
"BAAI/bge-multilingual-gemma2",
|
||||
"intfloat/e5-mistral-7b-instruct",
|
||||
]
|
||||
)
|
||||
|
||||
BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[
|
||||
"cohere",
|
||||
|
|
@ -721,20 +742,24 @@ BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[
|
|||
]
|
||||
|
||||
open_ai_embedding_models: set = set(["text-embedding-ada-002"])
|
||||
cohere_embedding_models: set = set([
|
||||
"embed-v4.0",
|
||||
"embed-english-v3.0",
|
||||
"embed-english-light-v3.0",
|
||||
"embed-multilingual-v3.0",
|
||||
"embed-english-v2.0",
|
||||
"embed-english-light-v2.0",
|
||||
"embed-multilingual-v2.0",
|
||||
])
|
||||
bedrock_embedding_models: set = set([
|
||||
"amazon.titan-embed-text-v1",
|
||||
"cohere.embed-english-v3",
|
||||
"cohere.embed-multilingual-v3",
|
||||
])
|
||||
cohere_embedding_models: set = set(
|
||||
[
|
||||
"embed-v4.0",
|
||||
"embed-english-v3.0",
|
||||
"embed-english-light-v3.0",
|
||||
"embed-multilingual-v3.0",
|
||||
"embed-english-v2.0",
|
||||
"embed-english-light-v2.0",
|
||||
"embed-multilingual-v2.0",
|
||||
]
|
||||
)
|
||||
bedrock_embedding_models: set = set(
|
||||
[
|
||||
"amazon.titan-embed-text-v1",
|
||||
"cohere.embed-english-v3",
|
||||
"cohere.embed-multilingual-v3",
|
||||
]
|
||||
)
|
||||
|
||||
known_tokenizer_config = {
|
||||
"mistralai/Mistral-7B-Instruct-v0.1": {
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ from pydantic import BaseModel
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.constants import REDACTED_BY_LITELM_STRING
|
||||
from litellm.constants import REDACTED_BY_LITELM_STRING, MAX_STRING_LENGTH_PROMPT_IN_DB
|
||||
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs
|
||||
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
|
||||
from litellm.proxy._types import SpendLogsMetadata, SpendLogsPayload
|
||||
|
|
@ -53,7 +53,7 @@ def _get_spend_logs_metadata(
|
|||
guardrail_information: Optional[StandardLoggingGuardrailInformation] = None,
|
||||
usage_object: Optional[dict] = None,
|
||||
model_map_information: Optional[StandardLoggingModelInformation] = None,
|
||||
cold_storage_object_key: Optional[str] = None
|
||||
cold_storage_object_key: Optional[str] = None,
|
||||
) -> SpendLogsMetadata:
|
||||
if metadata is None:
|
||||
return SpendLogsMetadata(
|
||||
|
|
@ -101,7 +101,7 @@ def _get_spend_logs_metadata(
|
|||
clean_metadata["usage_object"] = usage_object
|
||||
clean_metadata["model_map_information"] = model_map_information
|
||||
clean_metadata["cold_storage_object_key"] = cold_storage_object_key
|
||||
|
||||
|
||||
return clean_metadata
|
||||
|
||||
|
||||
|
|
@ -481,10 +481,9 @@ def _sanitize_request_body_for_spend_logs_payload(
|
|||
) -> dict:
|
||||
"""
|
||||
Recursively sanitize request body to prevent logging large base64 strings or other large values.
|
||||
Truncates strings longer than 1000 characters and handles nested dictionaries.
|
||||
Truncates strings longer than MAX_STRING_LENGTH_PROMPT_IN_DB characters and handles nested dictionaries.
|
||||
"""
|
||||
from litellm.constants import LITELLM_TRUNCATED_PAYLOAD_FIELD
|
||||
MAX_STRING_LENGTH = 1000
|
||||
|
||||
if visited is None:
|
||||
visited = set()
|
||||
|
|
@ -501,8 +500,8 @@ def _sanitize_request_body_for_spend_logs_payload(
|
|||
elif isinstance(value, list):
|
||||
return [_sanitize_value(item) for item in value]
|
||||
elif isinstance(value, str):
|
||||
if len(value) > MAX_STRING_LENGTH:
|
||||
return f"{value[:MAX_STRING_LENGTH]}... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} {len(value) - MAX_STRING_LENGTH} chars)"
|
||||
if len(value) > MAX_STRING_LENGTH_PROMPT_IN_DB:
|
||||
return f"{value[:MAX_STRING_LENGTH_PROMPT_IN_DB]}... ({LITELLM_TRUNCATED_PAYLOAD_FIELD} {len(value) - MAX_STRING_LENGTH_PROMPT_IN_DB} chars)"
|
||||
return value
|
||||
return value
|
||||
|
||||
|
|
|
|||
|
|
@ -9,7 +9,6 @@ import random
|
|||
from typing import TYPE_CHECKING, Any, Dict, List, Union
|
||||
|
||||
from litellm._logging import verbose_router_logger
|
||||
from litellm.litellm_core_utils.core_helpers import safe_divide
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.router import Router as _Router
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ from typing import Optional
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload
|
||||
from litellm.proxy.spend_tracking.spend_tracking_utils import get_logging_payload, _sanitize_request_body_for_spend_logs_payload
|
||||
from litellm.proxy._types import SpendLogsMetadata, SpendLogsPayload
|
||||
|
||||
|
||||
|
|
@ -396,3 +396,91 @@ def test_spend_logs_payload_with_prompts_enabled(monkeypatch):
|
|||
payload_disabled: SpendLogsPayload = get_logging_payload(**input_args)
|
||||
assert payload_disabled["messages"] == "{}"
|
||||
assert payload_disabled["response"] == "{}"
|
||||
|
||||
|
||||
def test_large_request_no_truncation_threshold():
|
||||
"""
|
||||
Test that MAX_STRING_LENGTH_PROMPT_IN_DB constant is used for request body sanitization
|
||||
"""
|
||||
from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB, LITELLM_TRUNCATED_PAYLOAD_FIELD
|
||||
|
||||
# Create a large string that exceeds the threshold
|
||||
large_content = "x" * (MAX_STRING_LENGTH_PROMPT_IN_DB + 500)
|
||||
|
||||
request_body = {
|
||||
"messages": [
|
||||
{"role": "user", "content": large_content}
|
||||
],
|
||||
"model": "gpt-4"
|
||||
}
|
||||
|
||||
sanitized = _sanitize_request_body_for_spend_logs_payload(request_body)
|
||||
|
||||
# Verify the content was truncated
|
||||
truncated_content = sanitized["messages"][0]["content"]
|
||||
assert len(truncated_content) > MAX_STRING_LENGTH_PROMPT_IN_DB # includes truncation message
|
||||
assert truncated_content.startswith("x" * MAX_STRING_LENGTH_PROMPT_IN_DB)
|
||||
assert LITELLM_TRUNCATED_PAYLOAD_FIELD in truncated_content
|
||||
assert "500 chars" in truncated_content
|
||||
|
||||
|
||||
def test_small_request_no_truncation():
|
||||
"""
|
||||
Test that small strings are not truncated by MAX_STRING_LENGTH_PROMPT_IN_DB
|
||||
"""
|
||||
from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB
|
||||
|
||||
# Create a small string that's under the threshold
|
||||
small_content = "x" * (MAX_STRING_LENGTH_PROMPT_IN_DB - 100)
|
||||
|
||||
request_body = {
|
||||
"messages": [
|
||||
{"role": "user", "content": small_content}
|
||||
],
|
||||
"model": "gpt-4"
|
||||
}
|
||||
|
||||
sanitized = _sanitize_request_body_for_spend_logs_payload(request_body)
|
||||
|
||||
# Verify the content was NOT truncated
|
||||
assert sanitized["messages"][0]["content"] == small_content
|
||||
assert len(sanitized["messages"][0]["content"]) == MAX_STRING_LENGTH_PROMPT_IN_DB - 100
|
||||
|
||||
|
||||
def test_configurable_string_length_env_var(monkeypatch):
|
||||
"""
|
||||
Test that MAX_STRING_LENGTH_PROMPT_IN_DB can be configured via environment variable
|
||||
"""
|
||||
# Set environment variable to a custom value
|
||||
monkeypatch.setenv("MAX_STRING_LENGTH_PROMPT_IN_DB", "500")
|
||||
|
||||
# Import after setting env var to ensure it picks up the new value
|
||||
import importlib
|
||||
import litellm.constants
|
||||
import litellm.proxy.spend_tracking.spend_tracking_utils
|
||||
importlib.reload(litellm.constants)
|
||||
importlib.reload(litellm.proxy.spend_tracking.spend_tracking_utils)
|
||||
|
||||
from litellm.constants import MAX_STRING_LENGTH_PROMPT_IN_DB, LITELLM_TRUNCATED_PAYLOAD_FIELD
|
||||
from litellm.proxy.spend_tracking.spend_tracking_utils import _sanitize_request_body_for_spend_logs_payload
|
||||
|
||||
# Verify the constant was set to the env var value
|
||||
assert MAX_STRING_LENGTH_PROMPT_IN_DB == 500
|
||||
|
||||
# Test truncation with the custom value
|
||||
large_content = "y" * 750 # 250 chars over the custom limit
|
||||
|
||||
request_body = {
|
||||
"messages": [
|
||||
{"role": "user", "content": large_content}
|
||||
],
|
||||
"model": "gpt-4"
|
||||
}
|
||||
|
||||
sanitized = _sanitize_request_body_for_spend_logs_payload(request_body)
|
||||
|
||||
# Verify truncation occurred at the custom threshold
|
||||
truncated_content = sanitized["messages"][0]["content"]
|
||||
assert truncated_content.startswith("y" * 500)
|
||||
assert LITELLM_TRUNCATED_PAYLOAD_FIELD in truncated_content
|
||||
assert "250 chars" in truncated_content
|
||||
|
|
|
|||
|
|
@ -757,7 +757,7 @@ def test_optional_combine_thinking_block_with_none_content(
|
|||
|
||||
# Second chunk with reasoning_content and None content
|
||||
second_chunk = {
|
||||
"id": "chunk2",
|
||||
"id": "chunk2",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1741037891,
|
||||
"model": "deepseek-reasoner",
|
||||
|
|
@ -776,16 +776,13 @@ def test_optional_combine_thinking_block_with_none_content(
|
|||
# Final chunk with actual content - should add </think> tag
|
||||
final_chunk = {
|
||||
"id": "chunk3",
|
||||
"object": "chat.completion.chunk",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1741037892,
|
||||
"model": "deepseek-reasoner",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {
|
||||
"content": "The answer is 42",
|
||||
"reasoning_content": None
|
||||
},
|
||||
"delta": {"content": "The answer is 42", "reasoning_content": None},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
|
|
@ -796,12 +793,15 @@ def test_optional_combine_thinking_block_with_none_content(
|
|||
initialized_custom_stream_wrapper._optional_combine_thinking_block_in_choices(
|
||||
first_response
|
||||
)
|
||||
assert first_response.choices[0].delta.content == "<think>Let me think about this problem"
|
||||
assert (
|
||||
first_response.choices[0].delta.content
|
||||
== "<think>Let me think about this problem"
|
||||
)
|
||||
assert not hasattr(first_response.choices[0].delta, "reasoning_content")
|
||||
assert initialized_custom_stream_wrapper.sent_first_thinking_block is True
|
||||
|
||||
# Process second chunk - should work with continued reasoning
|
||||
second_response = ModelResponseStream(**second_chunk)
|
||||
second_response = ModelResponseStream(**second_chunk)
|
||||
initialized_custom_stream_wrapper._optional_combine_thinking_block_in_choices(
|
||||
second_response
|
||||
)
|
||||
|
|
@ -822,76 +822,99 @@ def test_has_special_delta_content(
|
|||
initialized_custom_stream_wrapper: CustomStreamWrapper,
|
||||
):
|
||||
"""Test the _has_special_delta_content helper method"""
|
||||
|
||||
|
||||
# Test empty choices
|
||||
empty_response = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None, choices=[]
|
||||
)
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_content(empty_response)
|
||||
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_content(
|
||||
empty_response
|
||||
)
|
||||
|
||||
# Test with tool_calls (simulate with mock object)
|
||||
tool_call_response = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None,
|
||||
id="test",
|
||||
created=1742056047,
|
||||
model=None,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=None, index=0,
|
||||
delta=Delta(content=None, tool_calls=[{"id": "test"}])
|
||||
finish_reason=None,
|
||||
index=0,
|
||||
delta=Delta(
|
||||
content=None,
|
||||
tool_calls=[
|
||||
{
|
||||
"id": "test",
|
||||
"function": {"arguments": "{}", "name": "test_func"},
|
||||
}
|
||||
],
|
||||
),
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
assert initialized_custom_stream_wrapper._has_special_delta_content(tool_call_response)
|
||||
|
||||
assert initialized_custom_stream_wrapper._has_special_delta_content(
|
||||
tool_call_response
|
||||
)
|
||||
|
||||
# Test with function_call (simulate with mock object)
|
||||
function_call_response = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None,
|
||||
id="test",
|
||||
created=1742056047,
|
||||
model=None,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=None, index=0,
|
||||
delta=Delta(content=None, function_call={"name": "test_func"})
|
||||
finish_reason=None,
|
||||
index=0,
|
||||
delta=Delta(
|
||||
content=None, function_call={"name": "test_func", "arguments": "{}"}
|
||||
),
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
assert initialized_custom_stream_wrapper._has_special_delta_content(function_call_response)
|
||||
|
||||
assert initialized_custom_stream_wrapper._has_special_delta_content(
|
||||
function_call_response
|
||||
)
|
||||
|
||||
# Test with audio (simulate by adding audio attribute)
|
||||
audio_response = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None,
|
||||
id="test",
|
||||
created=1742056047,
|
||||
model=None,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=None, index=0,
|
||||
delta=Delta(content=None)
|
||||
)
|
||||
]
|
||||
StreamingChoices(finish_reason=None, index=0, delta=Delta(content=None))
|
||||
],
|
||||
)
|
||||
# Manually add audio attribute to delta
|
||||
audio_response.choices[0].delta.audio = {"transcript": "test"}
|
||||
assert initialized_custom_stream_wrapper._has_special_delta_content(audio_response)
|
||||
|
||||
|
||||
# Test with image (simulate by adding image attribute)
|
||||
image_response = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None,
|
||||
id="test",
|
||||
created=1742056047,
|
||||
model=None,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=None, index=0,
|
||||
delta=Delta(content=None)
|
||||
)
|
||||
]
|
||||
StreamingChoices(finish_reason=None, index=0, delta=Delta(content=None))
|
||||
],
|
||||
)
|
||||
# Manually add image attribute to delta
|
||||
image_response.choices[0].delta.image = {"url": "test.jpg"}
|
||||
assert initialized_custom_stream_wrapper._has_special_delta_content(image_response)
|
||||
|
||||
|
||||
# Test with regular content (should return False)
|
||||
regular_response = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None,
|
||||
id="test",
|
||||
created=1742056047,
|
||||
model=None,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=None, index=0,
|
||||
delta=Delta(content="Hello world")
|
||||
finish_reason=None, index=0, delta=Delta(content="Hello world")
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_content(
|
||||
regular_response
|
||||
)
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_content(regular_response)
|
||||
|
||||
|
||||
def test_handle_special_delta_content(
|
||||
|
|
@ -899,21 +922,26 @@ def test_handle_special_delta_content(
|
|||
):
|
||||
"""Test the _handle_special_delta_content helper method"""
|
||||
test_response = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None,
|
||||
id="test",
|
||||
created=1742056047,
|
||||
model=None,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=None, index=0,
|
||||
delta=Delta(content="test", role="assistant")
|
||||
finish_reason=None,
|
||||
index=0,
|
||||
delta=Delta(content="test", role="assistant"),
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
# The method should call strip_role_from_delta
|
||||
result = initialized_custom_stream_wrapper._handle_special_delta_content(test_response)
|
||||
|
||||
result = initialized_custom_stream_wrapper._handle_special_delta_content(
|
||||
test_response
|
||||
)
|
||||
|
||||
# Should return the same response object (modified)
|
||||
assert result is test_response
|
||||
|
||||
|
||||
# Should have set sent_first_chunk to True
|
||||
assert initialized_custom_stream_wrapper.sent_first_chunk is True
|
||||
|
||||
|
|
@ -922,32 +950,38 @@ def test_has_any_special_delta_attributes(
|
|||
initialized_custom_stream_wrapper: CustomStreamWrapper,
|
||||
):
|
||||
"""Test the _has_any_special_delta_attributes helper method"""
|
||||
|
||||
|
||||
# Test with delta that has audio attribute
|
||||
class MockDelta:
|
||||
def __init__(self):
|
||||
self.audio = {"transcript": "Hello world"}
|
||||
|
||||
|
||||
audio_delta = MockDelta()
|
||||
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(audio_delta)
|
||||
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(
|
||||
audio_delta
|
||||
)
|
||||
assert result is True
|
||||
|
||||
|
||||
# Test with delta that has image attribute
|
||||
class MockDeltaImage:
|
||||
def __init__(self):
|
||||
self.image = {"url": "test.jpg"}
|
||||
|
||||
|
||||
image_delta = MockDeltaImage()
|
||||
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(image_delta)
|
||||
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(
|
||||
image_delta
|
||||
)
|
||||
assert result is True
|
||||
|
||||
|
||||
# Test with delta that has no special attributes
|
||||
class MockDeltaRegular:
|
||||
def __init__(self):
|
||||
self.content = "regular content"
|
||||
|
||||
|
||||
regular_delta = MockDeltaRegular()
|
||||
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(regular_delta)
|
||||
result = initialized_custom_stream_wrapper._has_any_special_delta_attributes(
|
||||
regular_delta
|
||||
)
|
||||
assert result is False
|
||||
|
||||
|
||||
|
|
@ -955,48 +989,50 @@ def test_handle_special_delta_attributes(
|
|||
initialized_custom_stream_wrapper: CustomStreamWrapper,
|
||||
):
|
||||
"""Test the _handle_special_delta_attributes helper method"""
|
||||
|
||||
|
||||
# Create a model response
|
||||
model_response = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None,
|
||||
id="test",
|
||||
created=1742056047,
|
||||
model=None,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=None, index=0,
|
||||
delta=Delta(content="test")
|
||||
)
|
||||
]
|
||||
StreamingChoices(finish_reason=None, index=0, delta=Delta(content="test"))
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
# Test with delta that has audio attribute
|
||||
class MockDelta:
|
||||
def __init__(self):
|
||||
self.audio = {"transcript": "Hello world"}
|
||||
|
||||
|
||||
audio_delta = MockDelta()
|
||||
initialized_custom_stream_wrapper._handle_special_delta_attributes(audio_delta, model_response)
|
||||
|
||||
initialized_custom_stream_wrapper._handle_special_delta_attributes(
|
||||
audio_delta, model_response
|
||||
)
|
||||
|
||||
# Should copy the audio attribute
|
||||
assert hasattr(model_response.choices[0].delta, "audio")
|
||||
assert model_response.choices[0].delta.audio == {"transcript": "Hello world"}
|
||||
|
||||
|
||||
# Test with delta that has image attribute
|
||||
class MockDeltaImage:
|
||||
def __init__(self):
|
||||
self.image = {"url": "test.jpg"}
|
||||
|
||||
|
||||
image_delta = MockDeltaImage()
|
||||
model_response2 = ModelResponseStream(
|
||||
id="test", created=1742056047, model=None,
|
||||
id="test",
|
||||
created=1742056047,
|
||||
model=None,
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
finish_reason=None, index=0,
|
||||
delta=Delta(content="test")
|
||||
)
|
||||
]
|
||||
StreamingChoices(finish_reason=None, index=0, delta=Delta(content="test"))
|
||||
],
|
||||
)
|
||||
|
||||
initialized_custom_stream_wrapper._handle_special_delta_attributes(image_delta, model_response2)
|
||||
|
||||
|
||||
initialized_custom_stream_wrapper._handle_special_delta_attributes(
|
||||
image_delta, model_response2
|
||||
)
|
||||
|
||||
# Should copy the image attribute
|
||||
assert hasattr(model_response2.choices[0].delta, "image")
|
||||
assert model_response2.choices[0].delta.image == {"url": "test.jpg"}
|
||||
|
|
@ -1006,30 +1042,38 @@ def test_has_special_delta_attribute(
|
|||
initialized_custom_stream_wrapper: CustomStreamWrapper,
|
||||
):
|
||||
"""Test the _has_special_delta_attribute helper method"""
|
||||
|
||||
|
||||
# Test with None delta
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(None, "audio")
|
||||
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(
|
||||
None, "audio"
|
||||
)
|
||||
|
||||
# Test with delta that has the attribute
|
||||
class MockDelta:
|
||||
def __init__(self):
|
||||
self.audio = {"transcript": "test"}
|
||||
|
||||
|
||||
delta_with_audio = MockDelta()
|
||||
assert initialized_custom_stream_wrapper._has_special_delta_attribute(delta_with_audio, "audio")
|
||||
|
||||
assert initialized_custom_stream_wrapper._has_special_delta_attribute(
|
||||
delta_with_audio, "audio"
|
||||
)
|
||||
|
||||
# Test with delta that doesn't have the attribute
|
||||
class MockDeltaNoAudio:
|
||||
def __init__(self):
|
||||
self.content = "test"
|
||||
|
||||
|
||||
delta_without_audio = MockDeltaNoAudio()
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(delta_without_audio, "audio")
|
||||
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(
|
||||
delta_without_audio, "audio"
|
||||
)
|
||||
|
||||
# Test with delta that has the attribute but it's None
|
||||
class MockDeltaNone:
|
||||
def __init__(self):
|
||||
self.audio = None
|
||||
|
||||
|
||||
delta_with_none = MockDeltaNone()
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(delta_with_none, "audio")
|
||||
assert not initialized_custom_stream_wrapper._has_special_delta_attribute(
|
||||
delta_with_none, "audio"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue