[Fix] Fix web search model info regression, deprecated prompt caching model, undocumented env keys

- Revert test_anthropic_web_search_in_model_info to use claude-3-5-haiku-latest
  (model info test doesn't make API calls, so the -latest alias is fine here)
- Replace claude-3-7-sonnet-20250219 with claude-sonnet-4-5-20250929 in
  test_anthropic_prompt_caching.py (10 instances)
- Include pending doc updates for COMPETITOR_LLM_TEMPERATURE and
  MAX_COMPETITOR_NAMES env vars in config_settings.md

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
yuneng-jiang 2026-02-19 14:22:47 -08:00
parent f459808c3d
commit c67120efd5
3 changed files with 13 additions and 11 deletions

View file

@ -485,6 +485,7 @@ router_settings:
| CUSTOM_TIKTOKEN_CACHE_DIR | Custom directory for Tiktoken cache
| CONFIDENT_API_KEY | API key for Confident AI (Deepeval) Logging service
| COHERE_API_BASE | Base URL for Cohere API. Default is https://api.cohere.com
| COMPETITOR_LLM_TEMPERATURE | Temperature setting for the LLM used in competitor discovery. Default is 0.3
| DATABASE_HOST | Hostname for the database server
| DATABASE_NAME | Name of the database
| DATABASE_PASSWORD | Password for the database user
@ -806,6 +807,7 @@ router_settings:
| LOGGING_WORKER_MAX_QUEUE_SIZE | Maximum size of the logging worker queue. When the queue is full, the worker aggressively clears tasks to make room instead of dropping logs. Default is 50,000
| LOGGING_WORKER_MAX_TIME_PER_COROUTINE | Maximum time in seconds allowed for each coroutine in the logging worker before timing out. Default is 20.0
| LOGGING_WORKER_CLEAR_PERCENTAGE | Percentage of the queue to extract when clearing. Default is 50%
| MAX_COMPETITOR_NAMES | Maximum number of competitor names allowed in policy template enrichment. Default is 100
| MAX_EXCEPTION_MESSAGE_LENGTH | Maximum length for exception messages. Default is 2000
| MAX_ITERATIONS_TO_CLEAR_QUEUE | Maximum number of iterations to attempt when clearing the logging worker queue during shutdown. Default is 200
| MAX_TIME_TO_CLEAR_QUEUE | Maximum time in seconds to spend clearing the logging worker queue during shutdown. Default is 5.0

View file

@ -57,7 +57,7 @@ async def test_litellm_anthropic_prompt_caching_tools():
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello!"}],
"model": "claude-3-7-sonnet-20250219",
"model": "claude-sonnet-4-5-20250929",
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 6},
@ -74,7 +74,7 @@ async def test_litellm_anthropic_prompt_caching_tools():
# Act: Call the litellm.acompletion function
response = await litellm.acompletion(
api_key="mock_api_key",
model="anthropic/claude-3-7-sonnet-20250219",
model="anthropic/claude-sonnet-4-5-20250929",
messages=[
{"role": "user", "content": "What's the weather like in Boston today?"}
],
@ -154,7 +154,7 @@ async def test_litellm_anthropic_prompt_caching_tools():
}
],
"max_tokens": 64000,
"model": "claude-3-7-sonnet-20250219",
"model": "claude-sonnet-4-5-20250929",
}
mock_post.assert_called_once_with(
@ -240,7 +240,7 @@ async def test_anthropic_vertex_ai_prompt_caching(anthropic_messages, sync_mode)
async def test_anthropic_api_prompt_caching_basic():
litellm.set_verbose = True
response = await litellm.acompletion(
model="anthropic/claude-3-7-sonnet-20250219",
model="anthropic/claude-sonnet-4-5-20250929",
messages=[
# System Message
{
@ -308,7 +308,7 @@ async def test_anthropic_api_prompt_caching_basic_with_cache_creation():
litellm.set_verbose = True
response = await litellm.acompletion(
model="anthropic/claude-3-7-sonnet-20250219",
model="anthropic/claude-sonnet-4-5-20250929",
messages=[
# System Message
{
@ -460,7 +460,7 @@ async def test_anthropic_api_prompt_caching_with_content_str():
async def test_anthropic_api_prompt_caching_no_headers():
litellm.set_verbose = True
response = await litellm.acompletion(
model="anthropic/claude-3-7-sonnet-20250219",
model="anthropic/claude-sonnet-4-5-20250929",
messages=[
# System Message
{
@ -520,7 +520,7 @@ async def test_anthropic_api_prompt_caching_no_headers():
@pytest.mark.asyncio()
async def test_anthropic_api_prompt_caching_streaming():
response = await litellm.acompletion(
model="anthropic/claude-3-7-sonnet-20250219",
model="anthropic/claude-sonnet-4-5-20250929",
messages=[
# System Message
{
@ -603,7 +603,7 @@ async def test_litellm_anthropic_prompt_caching_system():
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello!"}],
"model": "claude-3-7-sonnet-20250219",
"model": "claude-sonnet-4-5-20250929",
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 6},
@ -620,7 +620,7 @@ async def test_litellm_anthropic_prompt_caching_system():
# Act: Call the litellm.acompletion function
response = await litellm.acompletion(
api_key="mock_api_key",
model="anthropic/claude-3-7-sonnet-20250219",
model="anthropic/claude-sonnet-4-5-20250929",
messages=[
{
"role": "system",
@ -681,7 +681,7 @@ async def test_litellm_anthropic_prompt_caching_system():
}
],
"max_tokens": 64000,
"model": "claude-3-7-sonnet-20250219",
"model": "claude-sonnet-4-5-20250929",
}
mock_post.assert_called_once_with(

View file

@ -433,7 +433,7 @@ def test_anthropic_web_search_in_model_info():
"anthropic/claude-sonnet-4-5-20250929",
"anthropic/claude-3-5-sonnet-20241022",
"anthropic/claude-3-5-haiku-20241022",
"anthropic/claude-haiku-4-5-20251001",
"anthropic/claude-3-5-haiku-latest",
]
for model in supported_models:
from litellm.utils import get_model_info