diff --git a/litellm/__pycache__/main.cpython-311.pyc b/litellm/__pycache__/main.cpython-311.pyc index 27a715c21d9..0d2b93327bc 100644 Binary files a/litellm/__pycache__/main.cpython-311.pyc and b/litellm/__pycache__/main.cpython-311.pyc differ diff --git a/litellm/__pycache__/utils.cpython-311.pyc b/litellm/__pycache__/utils.cpython-311.pyc index 1013f790171..fdb4b7b7bd9 100644 Binary files a/litellm/__pycache__/utils.cpython-311.pyc and b/litellm/__pycache__/utils.cpython-311.pyc differ diff --git a/litellm/llms/huggingface_restapi.py b/litellm/llms/huggingface_restapi.py index de1781ee0d5..0d61a8ad01f 100644 --- a/litellm/llms/huggingface_restapi.py +++ b/litellm/llms/huggingface_restapi.py @@ -56,6 +56,7 @@ def completion( if task == "conversational": inference_params = copy.deepcopy(optional_params) inference_params.pop("details") + inference_params.pop("return_full_text") past_user_inputs = [] generated_responses = [] text = "" diff --git a/litellm/tests/test_completion.py b/litellm/tests/test_completion.py index 4dce3c9ec62..be11ffdc540 100644 --- a/litellm/tests/test_completion.py +++ b/litellm/tests/test_completion.py @@ -420,6 +420,16 @@ def test_completion_azure_deployment_id(): pytest.fail(f"Error occurred: {e}") # test_completion_azure_deployment_id() +# def test_hf_conversational_task(): +# try: +# messages = [{ "content": "There's a llama in my garden 😱 What should I do?","role": "user"}] +# # e.g. Call 'facebook/blenderbot-400M-distill' hosted on HF Inference endpoints +# response = completion(model="huggingface/facebook/blenderbot-400M-distill", messages=messages, task="conversational") +# print(f"response: {response}") +# except Exception as e: +# pytest.fail(f"Error occurred: {e}") + +# test_hf_conversational_task() # Replicate API endpoints are unstable -> throw random CUDA errors -> this means our tests can fail even if our tests weren't incorrect. # def test_completion_replicate_llama_2():