mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
test(reasoning-effort-grid): cover Claude Opus 4.8 across provider routes (#29327)
* test(logging): align DB metrics event_metadata assertions with safe redaction PR #28909 hardened log_db_metrics to emit a minimal, non-sensitive event_metadata (only table_name when present, otherwise None) instead of dumping function_name, function_kwargs, and function_args onto the span. The test in test_log_db_redis_services was not updated and still asserted "function_name" in event_metadata, which raised TypeError (argument of type 'NoneType' is not iterable) and turned the logging_testing CI job red on litellm_internal_staging. Update test_log_db_metrics_success to assert event_metadata is None when no table_name is passed, and add test_log_db_metrics_event_metadata_is_safe as a regression guard verifying that only the table name surfaces and that sensitive kwargs (tokens, prisma client) are never dumped. * test(bedrock): self-heal opus-4-7 grid cells when unentitled on CI The bedrock-claude-opus-4-7 converse cells are unentitled on the Bedrock CI account, so they were marked xfail. xfail keeps reporting them as expected failures even after access is granted, so the wire translation never gets verified again. Now the cell makes the call and skips only when Bedrock replies "is not available for this account"; the moment the model is entitled the same cells run their full assertions with no edit. A focused unit test pins the tolerance predicate so any other failure still surfaces loudly and the available path still runs the assertions. * test(reasoning-effort-grid): add claude-opus-4-8 across provider routes Adds claude-opus-4-8 to the anthropic, azure, vertex and bedrock-converse routes (275 cells total) so the reasoning-effort wire translation is covered for the new model. The bedrock opus-4-8 and opus-4-7 cells reuse the self-heal path: they run the call and skip only on Bedrock's "is not available for this account" reply, then assert in full once the model is entitled. The azure and vertex opus-4-8 cells stay xfail until a Foundry deployment exists and Vertex availability is confirmed. The shared xhigh+max capability set is renamed to _CAPS_XHIGH_MAX now that more than one model uses it.
This commit is contained in:
parent
94a043efb2
commit
152b1177e5
2 changed files with 56 additions and 7 deletions
|
|
@ -22,6 +22,7 @@ class ModelEntry:
|
|||
required_env: FrozenSet[str] = field(default_factory=frozenset)
|
||||
caps: FrozenSet[str] = field(default_factory=frozenset)
|
||||
unavailable_error: Optional[str] = None
|
||||
fail_reason: Optional[str] = None
|
||||
bedrock_effort_ceiling: Optional[str] = None
|
||||
|
||||
def params(self) -> Dict[str, str]:
|
||||
|
|
@ -126,7 +127,7 @@ _VERTEX_REQ = frozenset({"VERTEX_PROJECT"})
|
|||
_BEDROCK_REQ = frozenset({"AWS_ACCESS_KEY_ID", "AWS_SECRET_ACCESS_KEY"})
|
||||
|
||||
|
||||
_CAPS_OPUS_4_7: FrozenSet[str] = frozenset(
|
||||
_CAPS_XHIGH_MAX: FrozenSet[str] = frozenset(
|
||||
{"supports_xhigh_reasoning_effort", "supports_max_reasoning_effort"}
|
||||
)
|
||||
_CAPS_4_6: FrozenSet[str] = frozenset({"supports_max_reasoning_effort"})
|
||||
|
|
@ -134,12 +135,19 @@ _CAPS_NONE: FrozenSet[str] = frozenset()
|
|||
|
||||
|
||||
ANTHROPIC_DIRECT_MODELS: Tuple[ModelEntry, ...] = (
|
||||
ModelEntry(
|
||||
alias="claude-opus-4-8",
|
||||
model="anthropic/claude-opus-4-8",
|
||||
mode="adaptive",
|
||||
required_env=_ANTHROPIC_REQ,
|
||||
caps=_CAPS_XHIGH_MAX,
|
||||
),
|
||||
ModelEntry(
|
||||
alias="claude-opus-4-7",
|
||||
model="anthropic/claude-opus-4-7",
|
||||
mode="adaptive",
|
||||
required_env=_ANTHROPIC_REQ,
|
||||
caps=_CAPS_OPUS_4_7,
|
||||
caps=_CAPS_XHIGH_MAX,
|
||||
),
|
||||
ModelEntry(
|
||||
alias="claude-sonnet-4-6",
|
||||
|
|
@ -159,12 +167,25 @@ ANTHROPIC_DIRECT_MODELS: Tuple[ModelEntry, ...] = (
|
|||
|
||||
|
||||
AZURE_AI_MODELS: Tuple[ModelEntry, ...] = (
|
||||
ModelEntry(
|
||||
alias="azure-claude-opus-4-8",
|
||||
model="azure_ai/claude-opus-4-8",
|
||||
mode="adaptive",
|
||||
required_env=_AZURE_FOUNDRY_REQ,
|
||||
caps=_CAPS_XHIGH_MAX,
|
||||
fail_reason=(
|
||||
"claude-opus-4-8 has no deployment on the CI Microsoft Foundry "
|
||||
"resource yet; Foundry returns DeploymentNotFound until someone "
|
||||
"creates the opus-4-8 deployment, so this cell stays loud in CI. "
|
||||
"Remove this fail_reason once the deployment exists."
|
||||
),
|
||||
),
|
||||
ModelEntry(
|
||||
alias="azure-claude-opus-4-7",
|
||||
model="azure_ai/claude-opus-4-7",
|
||||
mode="adaptive",
|
||||
required_env=_AZURE_FOUNDRY_REQ,
|
||||
caps=_CAPS_OPUS_4_7,
|
||||
caps=_CAPS_XHIGH_MAX,
|
||||
),
|
||||
ModelEntry(
|
||||
alias="azure-claude-opus-4-6",
|
||||
|
|
@ -191,13 +212,27 @@ AZURE_AI_MODELS: Tuple[ModelEntry, ...] = (
|
|||
|
||||
|
||||
VERTEX_AI_MODELS: Tuple[ModelEntry, ...] = (
|
||||
ModelEntry(
|
||||
alias="vertex-claude-opus-4-8",
|
||||
model="vertex_ai/claude-opus-4-8",
|
||||
mode="adaptive",
|
||||
extra_params=(("vertex_location", "global"),),
|
||||
required_env=_VERTEX_REQ,
|
||||
caps=_CAPS_XHIGH_MAX,
|
||||
fail_reason=(
|
||||
"claude-opus-4-8 availability on the CI Vertex project is not yet "
|
||||
"confirmed for this brand-new release, so this cell stays loud in "
|
||||
"CI until verified. Remove this fail_reason once the model is "
|
||||
"confirmed available on the global Vertex endpoint."
|
||||
),
|
||||
),
|
||||
ModelEntry(
|
||||
alias="vertex-claude-opus-4-7",
|
||||
model="vertex_ai/claude-opus-4-7",
|
||||
mode="adaptive",
|
||||
extra_params=(("vertex_location", "global"),),
|
||||
required_env=_VERTEX_REQ,
|
||||
caps=_CAPS_OPUS_4_7,
|
||||
caps=_CAPS_XHIGH_MAX,
|
||||
),
|
||||
ModelEntry(
|
||||
alias="vertex-claude-opus-4-6",
|
||||
|
|
@ -227,13 +262,24 @@ VERTEX_AI_MODELS: Tuple[ModelEntry, ...] = (
|
|||
|
||||
|
||||
BEDROCK_CONVERSE_MODELS: Tuple[ModelEntry, ...] = (
|
||||
ModelEntry(
|
||||
alias="bedrock-claude-opus-4-8",
|
||||
model="bedrock/converse/us.anthropic.claude-opus-4-8",
|
||||
mode="adaptive",
|
||||
extra_params=(("aws_region_name", "us-east-1"),),
|
||||
required_env=_BEDROCK_REQ,
|
||||
caps=_CAPS_XHIGH_MAX,
|
||||
bedrock_effort_ceiling="xhigh",
|
||||
unavailable_error="is not available for this account",
|
||||
),
|
||||
ModelEntry(
|
||||
alias="bedrock-claude-opus-4-7",
|
||||
model="bedrock/converse/us.anthropic.claude-opus-4-7",
|
||||
mode="adaptive",
|
||||
extra_params=(("aws_region_name", "us-east-1"),),
|
||||
required_env=_BEDROCK_REQ,
|
||||
caps=_CAPS_OPUS_4_7,
|
||||
caps=_CAPS_XHIGH_MAX,
|
||||
bedrock_effort_ceiling="xhigh",
|
||||
unavailable_error="is not available for this account",
|
||||
),
|
||||
ModelEntry(
|
||||
|
|
|
|||
|
|
@ -173,6 +173,9 @@ async def test_reasoning_effort_grid(
|
|||
if skip_reason:
|
||||
pytest.skip(skip_reason)
|
||||
|
||||
if model.fail_reason:
|
||||
pytest.xfail(model.fail_reason)
|
||||
|
||||
if route_name == "bedrock_invoke_messages":
|
||||
status, exc = await _call_messages(model, effort)
|
||||
else:
|
||||
|
|
@ -197,8 +200,8 @@ async def test_reasoning_effort_grid(
|
|||
|
||||
|
||||
def test_grid_cell_count() -> None:
|
||||
assert len(_PARAMS) == 21 * 11, (
|
||||
f"expected 231 cells (21 provider x model combos x 11 efforts), "
|
||||
assert len(_PARAMS) == 25 * 11, (
|
||||
f"expected 275 cells (25 provider x model combos x 11 efforts), "
|
||||
f"got {len(_PARAMS)}"
|
||||
)
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue