diff --git a/docs/my-website/docs/integrations/index.md b/docs/my-website/docs/integrations/index.md index 95c922cce89..e75530ba381 100644 --- a/docs/my-website/docs/integrations/index.md +++ b/docs/my-website/docs/integrations/index.md @@ -10,6 +10,7 @@ This section covers integrations with various tools and services that can be use ## Observability & Monitoring - **[Langfuse](../observability/langfuse_integration.md)** - LLM observability and analytics +- **[MLflow](../observability/mlflow.md)** - LLM observability, evaluation, and prompt management - **[Prometheus](../proxy/prometheus.md)** - Metrics collection and monitoring - **[PagerDuty](../proxy/pagerduty.md)** - Incident response and alerting - **[Datadog](../observability/datadog.md)** diff --git a/docs/my-website/docs/observability/mlflow.md b/docs/my-website/docs/observability/mlflow.md index 5fa46bdfdac..b9a2d34ee10 100644 --- a/docs/my-website/docs/observability/mlflow.md +++ b/docs/my-website/docs/observability/mlflow.md @@ -4,12 +4,11 @@ import Image from '@theme/IdealImage'; ## What is MLflow? -**MLflow** is an end-to-end open source MLOps platform for [experiment tracking](https://www.mlflow.org/docs/latest/tracking.html), [model management](https://www.mlflow.org/docs/latest/models.html), [evaluation](https://www.mlflow.org/docs/latest/llms/llm-evaluate/index.html), [observability (tracing)](https://www.mlflow.org/docs/latest/llms/tracing/index.html), and [deployment](https://www.mlflow.org/docs/latest/deployment/index.html). MLflow empowers teams to collaboratively develop and refine LLM applications efficiently. +**MLflow** is an open source developer platform for [observability (tracing)](https://www.mlflow.org/docs/latest/llms/tracing/index.html), [evaluation](https://www.mlflow.org/docs/latest/llms/llm-evaluate/index.html), and [prompt management](https://www.mlflow.org/docs/latest/llms/prompt-management/index.html). MLflow empowers teams to collaboratively develop and refine LLM applications and agents efficiently. MLflow’s integration with LiteLLM supports advanced observability compatible with OpenTelemetry. - - + ## Getting Started @@ -49,61 +48,97 @@ response = litellm.completion( ) ``` -Open the MLflow UI and go to the `Traces` tab to view logged traces: +In a terminal, start the MLflow server and open the UI by visiting the URL (e.g. `http://localhost:5000`) in your browser. Select the "Default" experiment to find the logged traces: ```bash -mlflow ui +mlflow server --port 5000 ``` + + + ## Tracing Tool Calls MLflow integration with LiteLLM support tracking tool calls in addition to the messages. ```python +import json +import litellm import mlflow +from mlflow.entities import SpanType # Enable MLflow auto-tracing for LiteLLM mlflow.litellm.autolog() -# Define the tool function. -def get_weather(location: str) -> str: - if location == "Tokyo": +# Define the tool function. Decorate it with `@mlflow.trace` to create a span for its execution. +@mlflow.trace(span_type=SpanType.TOOL) +def get_weather(city: str) -> str: + if city == "Tokyo": return "sunny" - elif location == "Paris": + elif city == "Paris": return "rainy" return "unknown" -# Define function spec -get_weather_tool = { - "type": "function", - "function": { - "name": "get_weather", - "description": "Get the current weather in a given location", - "parameters": { - "properties": { - "location": { - "description": "The city and state, e.g., San Francisco, CA", - "type": "string", - }, + +tools = [ + { + "type": "function", + "function": { + "name": "get_weather", + "parameters": { + "type": "object", + "properties": {"city": {"type": "string"}}, }, - "required": ["location"], - "type": "object", }, - }, -} + } +] -# Call LiteLLM as usual -response = litellm.completion( - model="gpt-4o-mini", - messages=[ - {"role": "user", "content": "What's the weather like in Paris today?"} - ], - tools=[get_weather_tool] -) +_tool_functions = {"get_weather": get_weather} + + +# Define a simple tool calling agent +@mlflow.trace(span_type=SpanType.AGENT) +def run_tool_agent(question: str): + messages = [{"role": "user", "content": question}] + + # Invoke the model with the given question and available tools + response = litellm.completion( + model="gpt-4o-mini", + messages=messages, + tools=tools, + ) + ai_msg = response.choices[0].message + messages.append(ai_msg) + + # If the model request tool call(s), invoke the function with the specified arguments + if tool_calls := ai_msg.tool_calls: + for tool_call in tool_calls: + function_name = tool_call.function.name + if tool_func := _tool_functions.get(function_name): + args = json.loads(tool_call.function.arguments) + tool_result = tool_func(**args) + else: + raise RuntimeError("An invalid tool is returned from the assistant!") + + messages.append( + { + "role": "tool", + "tool_call_id": tool_call.id, + "content": tool_result, + } + ) + + # Sent the tool results to the model and get a new response + response = litellm.completion(model="gpt-4o-mini", messages=messages) + + return response.choices[0].message.content + + +# Run the tool calling agent +question = "What's the weather like in Paris today?" +answer = run_tool_agent(question) ``` - - ## Evaluation diff --git a/docs/my-website/docs/tutorials/eval_suites.md b/docs/my-website/docs/tutorials/eval_suites.md index b533da99367..126d86ebdd2 100644 --- a/docs/my-website/docs/tutorials/eval_suites.md +++ b/docs/my-website/docs/tutorials/eval_suites.md @@ -5,26 +5,65 @@ import TabItem from '@theme/TabItem'; # Evaluate LLMs - MLflow Evals, Auto Eval ## Using LiteLLM with MLflow -MLflow provides an API `mlflow.evaluate()` to help evaluate your LLMs https://mlflow.org/docs/latest/llms/llm-evaluate/index.html + +[MLflow](https://mlflow.org/docs/latest/genai/eval-monitor.html) provides a powerful capability for evaluating your LLM applications and agents with 50+ built-in quality metrics and custom scoring criteria. This tutorial shows how to use MLflow to evaluate LLM applications and agents powered by LiteLLM. + + ### Pre Requisites ```shell -pip install litellm +pip install litellm mlflow>=3.3 ``` + +### Step 1: Configure MLflow + +In a terminal, start the MLflow server. + ```shell -pip install mlflow +mlflow server --port 5000 ``` +Then create a new notebook or a Python script to run the evaluation. Import MLflow and set the tracking URI and experiment name. -### Step 1: Start LiteLLM Proxy on the CLI -LiteLLM allows you to create an OpenAI compatible server for all supported LLMs. [More information on litellm proxy here](https://docs.litellm.ai/docs/simple_proxy) +- **Tracking URI**: The URL of the MLflow server. MLflow will determine where to send evaluation result based on this URI. +- **Experiment**: Experiment is a container in MLflow that groups evaluation runs, metrics, traces, etc. You can think of it as sort of a folder or a project. + +```python evaluation.py + +import mlflow + +mlflow.set_tracking_uri("http://localhost:5000") # <- The MLflow server URL +mlflow.set_experiment("LiteLLM Evaluation") # <- Specify any name you want for your experiment and MLflow will create it if it doesn't exist. +``` + +### Step 2: Define your inference logic with LiteLLM + +First, define a simple function that generates responses by invoking LLM API through LiteLLM. + +```python +def predict_fn(question: str) -> str: + response = litellm.completion( + model="gpt-5.1-mini", + messages=[ + {"role": "system", "content": "Answer the following question in two sentences."}, + {"role": "user", "content": question}, + ], + ) + return response.choices[0].message.content +``` + +:::info + +During evaluation, MLflow will automatically **[trace](https://mlflow.org/docs/latest/llms/tracing/index.html)** the LiteLLM calls and store them in the evaluation run. These traces are useful for debugging the root cause of low-quality responses and improve the model performance. + +::: + +Alternatively, you can use the LiteLLM proxy to create an OpenAI compatible server for all supported LLMs. [More information on litellm proxy here](https://docs.litellm.ai/docs/simple_proxy) ```shell $ litellm --model huggingface/bigcode/starcoder - #INFO: Proxy running on http://0.0.0.0:8000 ``` - **Here's how you can create the proxy for other supported llms** @@ -152,73 +191,83 @@ $ litellm --model command-nightly - -### Step 2: Run MLflow -Before running the eval we will set `openai.api_base` to the litellm proxy from Step 1 - -```python -openai.api_base = "http://0.0.0.0:8000" -``` - ```python import openai -import pandas as pd -openai.api_key = "anything" # this can be anything, we set the key on the proxy -openai.api_base = "http://0.0.0.0:8000" # set api base to the proxy from step 1 - -import mlflow -eval_data = pd.DataFrame( - { - "inputs": [ - "What is the largest country", - "What is the weather in sf?", - ], - "ground_truth": [ - "India is a large country", - "It's cold in SF today" - ], - } +client = openai.OpenAI( + api_key="anything", # this can be anything, we set the key on the proxy + base_url="http://0.0.0.0:8000" # your proxy url ) -with mlflow.start_run() as run: - system_prompt = "Answer the following question in two sentences" - logged_model_info = mlflow.openai.log_model( - model="gpt-3.5", - task=openai.ChatCompletion, - artifact_path="model", +def predict_fn(question: str) -> str: + response = client.chat.completions.create( + model="", messages=[ - {"role": "system", "content": system_prompt}, - {"role": "user", "content": "{question}"}, - ], - ) - - # Use predefined question-answering metrics to evaluate our model. - results = mlflow.evaluate( - logged_model_info.model_uri, - eval_data, - targets="ground_truth", - model_type="question-answering", - ) - print(f"See aggregated evaluation results below: \n{results.metrics}") - - # Evaluation result for each data record is available in `results.tables`. - eval_table = results.tables["eval_results_table"] - print(f"See evaluation table below: \n{eval_table}") - - + {"role": "system", "content": "Answer the following question in two sentences."}, + {"role": "user", "content": question}, + ]) + return response.choices[0].message.content ``` -### MLflow Output +### Step 3: Prepare the evaluation dataset + +Define the evaluation dataset with input questions, and optional expectations (= ground truth answers). + +```python +eval_data = [ + { + "inputs": {"question": "What is the largest country?"}, + "expectations": {"expected_response": "Russia is the largest country by area."}, + }, + { + "inputs": {"question": "What is the weather in SF?"}, + "expectations": {"expected_response": "I cannot provide real-time weather information."}, + }, +] ``` -{'toxicity/v1/mean': 0.00014476531214313582, 'toxicity/v1/variance': 2.5759661361262862e-12, 'toxicity/v1/p90': 0.00014604929747292773, 'toxicity/v1/ratio': 0.0, 'exact_match/v1': 0.0} -Downloading artifacts: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:00<00:00, 1890.18it/s] -See evaluation table below: - inputs ground_truth outputs token_count toxicity/v1/score -0 What is the largest country India is a large country Russia is the largest country in the world in... 14 0.000146 -1 What is the weather in sf? It's cold in SF today I'm sorry, I cannot provide the current weath... 36 0.000143 + + +### Step 4: Define evaluation metrics + +MLflow provides 50+ built-in evaluation metrics and a flexible API for defining custom ones. + +In MLflow, a **scorer** is a class or a function that generates evaluation metrics for a given data record. In this example, we will use two built-in scorers: + +- `Correctness`: LLM-as-a-Judge metric to check if the response is correct according to the expectation. +- `Guidelines`: Flexible built-in metric that allow you to define custom LLM-as-a-Judge with a simple natural language guidelines. + +```python +from mlflow.genai.scorers import Correctness, Guidelines + +scorers = [ + Correctness(), + Guidelines(name="is_concise", guidelines="The answer must be concise and no longer than two sentences."), +] ``` +See [MLflow documentation](https://mlflow.org/docs/latest/genai/eval-monitor/scorers/) for more details about supported scorers. + +### Step 5: Run the evaluation + +Now we are ready to run the evaluation. Pass the evaluation dataset, prediction function, and scorers to the `mlflow.genai.evaluate` function. + +```python +results = mlflow.genai.evaluate( + data=eval_data, + predict_fn=predict_fn, + scorers=scorers, +) +``` + +### Review the Results + +When the evaluation is complete, MLflow will show the link to the evaluation run in the terminal. Open the link in your browser to see the evaluation run and detailed results for each data record. + +### Next Steps + +- Check out [MLflow LiteLLM Integration](../observability/mlflow.md) for more details about MLflow LiteLLM integration. +- Visit [Scorers Documentation](https://mlflow.org/docs/latest/genai/eval-monitor/scorers/) for the full list of supported scorers and find the one that fits your needs. + ## Using LiteLLM with AutoEval AutoEvals is a tool for quickly and easily evaluating AI model outputs using best practices. diff --git a/docs/my-website/img/mlflow_evaluation_results.png b/docs/my-website/img/mlflow_evaluation_results.png new file mode 100644 index 00000000000..2372b790786 Binary files /dev/null and b/docs/my-website/img/mlflow_evaluation_results.png differ diff --git a/docs/my-website/img/mlflow_tool_calling_tracing.png b/docs/my-website/img/mlflow_tool_calling_tracing.png index 4d4a0e8fc50..a4ee24df60b 100644 Binary files a/docs/my-website/img/mlflow_tool_calling_tracing.png and b/docs/my-website/img/mlflow_tool_calling_tracing.png differ diff --git a/docs/my-website/img/mlflow_tracing.png b/docs/my-website/img/mlflow_tracing.png index aee1fb79ea1..ab8e5ebdcf2 100644 Binary files a/docs/my-website/img/mlflow_tracing.png and b/docs/my-website/img/mlflow_tracing.png differ