fix(ollama/completion): add 'max_completion_token' support for ollama

This commit is contained in:
Krrish Dholakia 2025-05-07 14:19:35 -07:00
parent 314b68cfd4
commit 122f037990
3 changed files with 14 additions and 17 deletions

View file

@ -150,6 +150,7 @@ class OllamaConfig(BaseConfig):
"frequency_penalty",
"stop",
"response_format",
"max_completion_tokens",
]
def map_openai_params(
@ -160,7 +161,7 @@ class OllamaConfig(BaseConfig):
drop_params: bool,
) -> dict:
for param, value in non_default_params.items():
if param == "max_tokens":
if param == "max_tokens" or param == "max_completion_tokens":
optional_params["num_predict"] = value
if param == "stream":
optional_params["stream"] = value
@ -257,9 +258,13 @@ class OllamaConfig(BaseConfig):
model_response.choices[0].finish_reason = "stop"
if request_data.get("format", "") == "json":
response_content = json.loads(response_json["response"])
# Check if this is a function call format with name/arguments structure
if isinstance(response_content, dict) and "name" in response_content and "arguments" in response_content:
if (
isinstance(response_content, dict)
and "name" in response_content
and "arguments" in response_content
):
# Handle as function call (original behavior)
function_call = response_content
message = litellm.Message(

File diff suppressed because one or more lines are too long

View file

@ -60,16 +60,9 @@ model_list:
- model_name: gemini/gemini-2.0-flash
litellm_params:
model: gemini/gemini-2.0-flash
litellm_settings:
num_retries: 0
check_provider_endpoint: true
cache: true
callbacks: ["otel"]
files_settings:
- custom_llm_provider: gemini
api_key: os.environ/GEMINI_API_KEY
general_settings:
store_prompts_in_spend_logs: true
- model_name: llama-qwen
litellm_params:
model: ollama/qwen2:0.5b
model_info:
input_cost_per_token: 0.75
output_cost_per_token: 3