From da1451e49300c30d66af426fc265994d5f5a8daa Mon Sep 17 00:00:00 2001 From: ishaan-jaff Date: Wed, 8 Nov 2023 17:25:59 -0800 Subject: [PATCH] (docs) use FLASK benchmarks with proxy --- .../docs/tutorials/lm_evaluation_harness.md | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/docs/my-website/docs/tutorials/lm_evaluation_harness.md b/docs/my-website/docs/tutorials/lm_evaluation_harness.md index a4488076e4b..a3d0c082e22 100644 --- a/docs/my-website/docs/tutorials/lm_evaluation_harness.md +++ b/docs/my-website/docs/tutorials/lm_evaluation_harness.md @@ -1,8 +1,9 @@ import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# LM-Evaluation Harness with TGI +# Benchmark LLMs - LM Harness, Flask +## LM Harness Benchmarks Evaluate LLMs 20x faster with TGI via litellm proxy's `/completions` endpoint. This tutorial assumes you're using the `big-refactor` branch of [lm-evaluation-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/big-refactor) @@ -41,7 +42,7 @@ python3 -m lm_eval \ ``` -## FLASK - Fine-grained Language Model Evaluation Based on Alignment Skill Sets +## FLASK - Fine-grained Language Model Evaluation Use litellm to evaluate any LLM on FLASK https://github.com/kaistAI/FLASK **Step 1: Start the local proxy** @@ -57,12 +58,14 @@ $ export OPENAI_API_BASE=http://0.0.0.0:8000 **Step 3 Run with FLASK** ```shell -cd FLASK -cd gpt_review +git clone https://github.com/kaistAI/FLASK +``` +```shell +cd FLASK/gpt_review ``` Run the eval -``shell +```shell python gpt4_eval.py -q '../evaluation_set/flask_evaluation.jsonl' ```