mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
docs - 1.74.0.rc (#12347)
* doc updates * doc fixes * doc fixes * docs fix * doc fix * doc clean up * doc fixes * cleanup
This commit is contained in:
parent
912e5084e5
commit
303c4bd628
3 changed files with 251 additions and 12 deletions
236
docs/my-website/docs/generateContent.md
Normal file
236
docs/my-website/docs/generateContent.md
Normal file
|
|
@ -0,0 +1,236 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Google AI generateContent
|
||||
|
||||
Use LiteLLM to call Google AI's generateContent endpoints for text generation, multimodal interactions, and streaming responses.
|
||||
|
||||
## Overview
|
||||
|
||||
| Feature | Supported | Notes |
|
||||
|-------|-------|-------|
|
||||
| Cost Tracking | ✅ | |
|
||||
| Logging | ✅ | works across all integrations |
|
||||
| End-user Tracking | ✅ | |
|
||||
| Streaming | ✅ | |
|
||||
| Fallbacks | ✅ | between supported models |
|
||||
| Loadbalancing | ✅ | between supported models |
|
||||
|
||||
## Usage
|
||||
---
|
||||
|
||||
### LiteLLM Python SDK
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="basic" label="Basic Usage">
|
||||
|
||||
#### Non-streaming example
|
||||
```python showLineNumbers title="Basic Text Generation"
|
||||
from litellm.google_genai import agenerate_content
|
||||
from google.genai.types import ContentDict, PartDict
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
contents = ContentDict(
|
||||
parts=[
|
||||
PartDict(text="Hello, can you tell me a short joke?")
|
||||
],
|
||||
role="user",
|
||||
)
|
||||
|
||||
response = await agenerate_content(
|
||||
contents=contents,
|
||||
model="gemini/gemini-2.0-flash",
|
||||
max_tokens=100,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Streaming example
|
||||
```python showLineNumbers title="Streaming Text Generation"
|
||||
from litellm.google_genai import agenerate_content_stream
|
||||
from google.genai.types import ContentDict, PartDict
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
contents = ContentDict(
|
||||
parts=[
|
||||
PartDict(text="Write a long story about space exploration")
|
||||
],
|
||||
role="user",
|
||||
)
|
||||
|
||||
response = await agenerate_content_stream(
|
||||
contents=contents,
|
||||
model="gemini/gemini-2.0-flash",
|
||||
max_tokens=500,
|
||||
)
|
||||
|
||||
async for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="sync" label="Sync Usage">
|
||||
|
||||
#### Sync non-streaming example
|
||||
```python showLineNumbers title="Sync Text Generation"
|
||||
from litellm.google_genai import generate_content
|
||||
from google.genai.types import ContentDict, PartDict
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
contents = ContentDict(
|
||||
parts=[
|
||||
PartDict(text="Hello, can you tell me a short joke?")
|
||||
],
|
||||
role="user",
|
||||
)
|
||||
|
||||
response = generate_content(
|
||||
contents=contents,
|
||||
model="gemini/gemini-2.0-flash",
|
||||
max_tokens=100,
|
||||
)
|
||||
print(response)
|
||||
```
|
||||
|
||||
#### Sync streaming example
|
||||
```python showLineNumbers title="Sync Streaming Text Generation"
|
||||
from litellm.google_genai import generate_content_stream
|
||||
from google.genai.types import ContentDict, PartDict
|
||||
import os
|
||||
|
||||
# Set API key
|
||||
os.environ["GEMINI_API_KEY"] = "your-gemini-api-key"
|
||||
|
||||
contents = ContentDict(
|
||||
parts=[
|
||||
PartDict(text="Write a long story about space exploration")
|
||||
],
|
||||
role="user",
|
||||
)
|
||||
|
||||
response = generate_content_stream(
|
||||
contents=contents,
|
||||
model="gemini/gemini-2.0-flash",
|
||||
max_tokens=500,
|
||||
)
|
||||
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### LiteLLM Proxy Server
|
||||
|
||||
1. Setup config.yaml
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: gemini-flash
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash
|
||||
api_key: os.environ/GEMINI_API_KEY
|
||||
```
|
||||
|
||||
2. Start proxy
|
||||
|
||||
```bash
|
||||
litellm --config /path/to/config.yaml
|
||||
```
|
||||
|
||||
3. Test it!
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="gemini-proxy" label="Google GenAI SDK">
|
||||
|
||||
```python showLineNumbers title="Google GenAI SDK with LiteLLM Proxy"
|
||||
from google.genai import Client
|
||||
import os
|
||||
|
||||
# Configure Google GenAI SDK to use LiteLLM proxy
|
||||
os.environ["GOOGLE_GEMINI_BASE_URL"] = "http://localhost:4000"
|
||||
os.environ["GEMINI_API_KEY"] = "sk-1234"
|
||||
|
||||
client = Client()
|
||||
|
||||
response = client.models.generate_content(
|
||||
model="gemini-flash",
|
||||
contents=[
|
||||
{
|
||||
"parts": [{"text": "Write a short story about AI"}],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
config={"max_output_tokens": 100}
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem value="curl-proxy" label="curl">
|
||||
|
||||
#### Generate Content
|
||||
|
||||
```bash showLineNumbers title="generateContent via LiteLLM Proxy"
|
||||
curl -L -X POST 'http://localhost:4000/v1beta/models/gemini-flash:generateContent' \
|
||||
-H 'content-type: application/json' \
|
||||
-H 'authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"contents": [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "Write a short story about AI"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"generationConfig": {
|
||||
"maxOutputTokens": 100
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
#### Stream Generate Content
|
||||
|
||||
```bash showLineNumbers title="streamGenerateContent via LiteLLM Proxy"
|
||||
curl -L -X POST 'http://localhost:4000/v1beta/models/gemini-flash:streamGenerateContent' \
|
||||
-H 'content-type: application/json' \
|
||||
-H 'authorization: Bearer sk-1234' \
|
||||
-d '{
|
||||
"contents": [
|
||||
{
|
||||
"parts": [
|
||||
{
|
||||
"text": "Write a long story about space exploration"
|
||||
}
|
||||
],
|
||||
"role": "user"
|
||||
}
|
||||
],
|
||||
"generationConfig": {
|
||||
"maxOutputTokens": 500
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
|
||||
## Related
|
||||
|
||||
- [Use LiteLLM with gemini-cli](../docs/tutorials/litellm_gemini_cli)
|
||||
|
|
@ -53,7 +53,7 @@ pip install litellm==1.74.0.post1
|
|||
|
||||
### MCP Gateway: Segregate MCP tools
|
||||
|
||||
### Python SDK: 2.3 Second Faster Python SDK Import Times
|
||||
### Python SDK: 2.3 Second Faster Import Times
|
||||
|
||||
This release brings significant performance improvements to the Python SDK with 2.3 seconds faster import times. We've refactored the initialization process to reduce startup overhead, making LiteLLM more efficient for applications that need quick initialization. This is a major improvement for applications that need to initialize LiteLLM quickly.
|
||||
|
||||
|
|
@ -62,6 +62,13 @@ This release brings significant performance improvements to the Python SDK with
|
|||
|
||||
## New Models / Updated Models
|
||||
|
||||
#### Pricing / Context Window Updates
|
||||
|
||||
| Provider | Model | Context Window | Input ($/1M tokens) | Output ($/1M tokens) | Type |
|
||||
| ----------- | -------------------------------------- | -------------- | ------------------- | -------------------- | ---- |
|
||||
| Watsonx | `watsonx/mistralai/mistral-large` | 131k | $3.00 | $10.00 | New |
|
||||
| Azure AI | `azure_ai/cohere-rerank-v3.5` | 4k | $2.00/1k queries | - | New (Rerank) |
|
||||
|
||||
|
||||
#### Features
|
||||
- **[🆕 GitHub Copilot](../../docs/providers/github_copilot)** - Use GitHub Copilot API with LiteLLM - [PR](https://github.com/BerriAI/litellm/pull/12325), [Get Started](../../docs/providers/github_copilot)
|
||||
|
|
@ -88,19 +95,13 @@ This release brings significant performance improvements to the Python SDK with
|
|||
- Fix default parameters for ollama-chat - [PR](https://github.com/BerriAI/litellm/pull/12201)
|
||||
- **[VLLM](../../docs/providers/vllm)**
|
||||
- Add 'audio_url' message type support - [PR](https://github.com/BerriAI/litellm/pull/12270)
|
||||
- **[Hugging Face](../../docs/providers/huggingface)**
|
||||
- Fix Hugging Face tests - [PR](https://github.com/BerriAI/litellm/pull/12286)
|
||||
|
||||
|
||||
---
|
||||
|
||||
## LLM API Endpoints
|
||||
|
||||
#### Features
|
||||
- **[/generateContent](../../docs/generate_content)**
|
||||
- Allow passing litellm_params - [PR](https://github.com/BerriAI/litellm/pull/12177)
|
||||
- Only pass supported params when using OpenAI models - [PR](https://github.com/BerriAI/litellm/pull/12297)
|
||||
- Fix using gemini-cli with Vertex Anthropic Models - [PR](https://github.com/BerriAI/litellm/pull/12246)
|
||||
|
||||
- **[/batches](../../docs/batches)**
|
||||
- Support batch retrieve with target model Query Param - [PR](https://github.com/BerriAI/litellm/pull/12228)
|
||||
- Anthropic completion bridge improvements - [PR](https://github.com/BerriAI/litellm/pull/12228)
|
||||
|
|
@ -122,6 +123,10 @@ This release brings significant performance improvements to the Python SDK with
|
|||
- Non-anthropic models token usage returned - [PR](https://github.com/BerriAI/litellm/pull/12184)
|
||||
- **[/chat/completions](../../docs/providers/anthropic_unified)**
|
||||
- Support Cursor IDE tool_choice format `{"type": "auto"}` - [PR](https://github.com/BerriAI/litellm/pull/12168)
|
||||
- **[/generateContent](../../docs/generate_content)**
|
||||
- Allow passing litellm_params - [PR](https://github.com/BerriAI/litellm/pull/12177)
|
||||
- Only pass supported params when using OpenAI models - [PR](https://github.com/BerriAI/litellm/pull/12297)
|
||||
- Fix using gemini-cli with Vertex Anthropic Models - [PR](https://github.com/BerriAI/litellm/pull/12246)
|
||||
- **Streaming**
|
||||
- Fix Error code: 307 for LlamaAPI Streaming Chat - [PR](https://github.com/BerriAI/litellm/pull/11946)
|
||||
- Store finish reason even if is_finished - [PR](https://github.com/BerriAI/litellm/pull/12250)
|
||||
|
|
@ -130,12 +135,9 @@ This release brings significant performance improvements to the Python SDK with
|
|||
|
||||
## Spend Tracking / Budget Improvements
|
||||
|
||||
#### Features
|
||||
- **Cost Tracking**
|
||||
- VertexAI Anthropic streaming cost tracking with prompt caching fixes - [PR](https://github.com/BerriAI/litellm/pull/12188)
|
||||
|
||||
#### Bugs
|
||||
- Fix allow strings in calculate cost - [PR](https://github.com/BerriAI/litellm/pull/12200)
|
||||
- VertexAI Anthropic streaming cost tracking with prompt caching fixes - [PR](https://github.com/BerriAI/litellm/pull/12188)
|
||||
|
||||
---
|
||||
|
||||
|
|
|
|||
|
|
@ -256,6 +256,7 @@ const sidebars = {
|
|||
"embedding/supported_embedding",
|
||||
"anthropic_unified",
|
||||
"mcp",
|
||||
"generateContent",
|
||||
{
|
||||
type: "category",
|
||||
label: "/images",
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue