mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
add example using litellm proxy with gemini context caching
This commit is contained in:
parent
3178724579
commit
e2e5e46979
2 changed files with 47 additions and 34 deletions
|
|
@ -3,20 +3,14 @@ model_list:
|
|||
litellm_params:
|
||||
model: openai/fake
|
||||
api_key: fake-key
|
||||
api_base: https://exampleopenaiendpoint-production.up.railwaz.app/
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
- model_name: fireworks-llama-v3-70b-instruct
|
||||
litellm_params:
|
||||
model: fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct
|
||||
api_key: "os.environ/FIREWORKS"
|
||||
# provider specific wildcard routing
|
||||
- model_name: "anthropic/*"
|
||||
- model_name: "*"
|
||||
litellm_params:
|
||||
model: "anthropic/*"
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
- model_name: "groq/*"
|
||||
litellm_params:
|
||||
model: "groq/*"
|
||||
api_key: os.environ/GROQ_API_KEY
|
||||
model: "*"
|
||||
- model_name: "*"
|
||||
litellm_params:
|
||||
model: openai/*
|
||||
|
|
@ -25,37 +19,18 @@ model_list:
|
|||
litellm_params:
|
||||
model: mistral/mistral-small-latest
|
||||
api_key: "os.environ/MISTRAL_API_KEY"
|
||||
- model_name: tts
|
||||
- model_name: gemini-1.5-pro-001
|
||||
litellm_params:
|
||||
model: openai/tts-1
|
||||
api_key: "os.environ/OPENAI_API_KEY"
|
||||
model_info:
|
||||
mode: audio_speech
|
||||
|
||||
|
||||
# for /files endpoints
|
||||
files_settings:
|
||||
- custom_llm_provider: azure
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app
|
||||
api_key: fake-key
|
||||
api_version: "2023-03-15-preview"
|
||||
- custom_llm_provider: openai
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
model: vertex_ai_beta/gemini-1.5-pro-001
|
||||
vertex_project: "adroit-crow-413218"
|
||||
vertex_location: "us-central1"
|
||||
vertex_credentials: "adroit-crow-413218-a956eef1a2a8.json"
|
||||
# Add path to service account.json
|
||||
|
||||
|
||||
|
||||
general_settings:
|
||||
master_key: sk-1234
|
||||
pass_through_endpoints:
|
||||
- path: "/v1/rerank" # route you want to add to LiteLLM Proxy Server
|
||||
target: "https://api.cohere.com/v1/rerank" # URL this route should forward requests to
|
||||
headers: # headers to forward to this URL
|
||||
content-type: application/json # (Optional) Extra Headers to pass to this endpoint
|
||||
accept: application/json
|
||||
forward_headers: True
|
||||
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["otel"] # 👈 KEY CHANGE
|
||||
success_callback: ["prometheus"]
|
||||
failure_callback: ["prometheus"]
|
||||
38
litellm/proxy/tests/test_gemini_context_caching.py
Normal file
38
litellm/proxy/tests/test_gemini_context_caching.py
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
import datetime
|
||||
|
||||
import openai
|
||||
import vertexai
|
||||
from vertexai.generative_models import Content, Part
|
||||
from vertexai.preview import caching
|
||||
from vertexai.preview.generative_models import GenerativeModel
|
||||
|
||||
client = openai.OpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000")
|
||||
vertexai.init(project="adroit-crow-413218", location="us-central1")
|
||||
print("creating cached content")
|
||||
contents_here: list[Content] = [
|
||||
Content(role="user", parts=[Part.from_text("huge string of text here" * 10000)])
|
||||
]
|
||||
cached_content = caching.CachedContent.create(
|
||||
model_name="gemini-1.5-pro-001",
|
||||
contents=contents_here,
|
||||
expire_time=datetime.datetime(2024, 8, 10),
|
||||
)
|
||||
|
||||
created_Caches = caching.CachedContent.list()
|
||||
|
||||
print("created_Caches contents=", created_Caches)
|
||||
|
||||
response = client.chat.completions.create( # type: ignore
|
||||
model="gemini-1.5-pro-001",
|
||||
max_tokens=8192,
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "quote all everything above this message",
|
||||
},
|
||||
],
|
||||
temperature="0.7",
|
||||
extra_body={"cached_content": cached_content.resource_name},
|
||||
)
|
||||
|
||||
print("response from proxy", response)
|
||||
Loading…
Add table
Reference in a new issue