diff --git a/litellm/llms/prompt_templates/factory.py b/litellm/llms/prompt_templates/factory.py index a6d1d64386f..1a576f43a33 100644 --- a/litellm/llms/prompt_templates/factory.py +++ b/litellm/llms/prompt_templates/factory.py @@ -1359,11 +1359,8 @@ def prompt_factory( "meta-llama/llama-3" in model or "meta-llama-3" in model ) and "instruct" in model: return hf_chat_template( - model=model, + model="meta-llama/Meta-Llama-3-8B-Instruct", messages=messages, - chat_template=known_tokenizer_config[ # type: ignore - "meta-llama/Meta-Llama-3-8B-Instruct" - ]["tokenizer"]["chat_template"], ) elif ( "tiiuae/falcon" in model diff --git a/litellm/tests/test_router_fallbacks.py b/litellm/tests/test_router_fallbacks.py index 98a2449f06b..51d9451a87e 100644 --- a/litellm/tests/test_router_fallbacks.py +++ b/litellm/tests/test_router_fallbacks.py @@ -258,6 +258,7 @@ def test_sync_fallbacks_embeddings(): model_list=model_list, fallbacks=[{"bad-azure-embedding-model": ["good-azure-embedding-model"]}], set_verbose=False, + num_retries=0, ) customHandler = MyCustomHandler() litellm.callbacks = [customHandler] @@ -393,7 +394,7 @@ def test_dynamic_fallbacks_sync(): }, ] - router = Router(model_list=model_list, set_verbose=True) + router = Router(model_list=model_list, set_verbose=True, num_retries=0) kwargs = {} kwargs["model"] = "azure/gpt-3.5-turbo" kwargs["messages"] = [{"role": "user", "content": "Hey, how's it going?"}]