From 122f037990da9c008130cde0ec5bd54259b3d0b0 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Wed, 7 May 2025 14:19:35 -0700 Subject: [PATCH] fix(ollama/completion): add 'max_completion_token' support for ollama --- .../llms/ollama/completion/transformation.py | 11 ++++++++--- .../proxy/_experimental/out/onboarding.html | 1 - litellm/proxy/_new_secret_config.yaml | 19 ++++++------------- 3 files changed, 14 insertions(+), 17 deletions(-) delete mode 100644 litellm/proxy/_experimental/out/onboarding.html diff --git a/litellm/llms/ollama/completion/transformation.py b/litellm/llms/ollama/completion/transformation.py index c619fd8cfb7..b2557a53e95 100644 --- a/litellm/llms/ollama/completion/transformation.py +++ b/litellm/llms/ollama/completion/transformation.py @@ -150,6 +150,7 @@ class OllamaConfig(BaseConfig): "frequency_penalty", "stop", "response_format", + "max_completion_tokens", ] def map_openai_params( @@ -160,7 +161,7 @@ class OllamaConfig(BaseConfig): drop_params: bool, ) -> dict: for param, value in non_default_params.items(): - if param == "max_tokens": + if param == "max_tokens" or param == "max_completion_tokens": optional_params["num_predict"] = value if param == "stream": optional_params["stream"] = value @@ -257,9 +258,13 @@ class OllamaConfig(BaseConfig): model_response.choices[0].finish_reason = "stop" if request_data.get("format", "") == "json": response_content = json.loads(response_json["response"]) - + # Check if this is a function call format with name/arguments structure - if isinstance(response_content, dict) and "name" in response_content and "arguments" in response_content: + if ( + isinstance(response_content, dict) + and "name" in response_content + and "arguments" in response_content + ): # Handle as function call (original behavior) function_call = response_content message = litellm.Message( diff --git a/litellm/proxy/_experimental/out/onboarding.html b/litellm/proxy/_experimental/out/onboarding.html deleted file mode 100644 index 755880844d4..00000000000 --- a/litellm/proxy/_experimental/out/onboarding.html +++ /dev/null @@ -1 +0,0 @@ -LiteLLM Dashboard \ No newline at end of file diff --git a/litellm/proxy/_new_secret_config.yaml b/litellm/proxy/_new_secret_config.yaml index e1a444a6d9c..11069452097 100644 --- a/litellm/proxy/_new_secret_config.yaml +++ b/litellm/proxy/_new_secret_config.yaml @@ -60,16 +60,9 @@ model_list: - model_name: gemini/gemini-2.0-flash litellm_params: model: gemini/gemini-2.0-flash - -litellm_settings: - num_retries: 0 - check_provider_endpoint: true - cache: true - callbacks: ["otel"] - -files_settings: - - custom_llm_provider: gemini - api_key: os.environ/GEMINI_API_KEY - -general_settings: - store_prompts_in_spend_logs: true \ No newline at end of file + - model_name: llama-qwen + litellm_params: + model: ollama/qwen2:0.5b + model_info: + input_cost_per_token: 0.75 + output_cost_per_token: 3