From 4ee161a152a9b27ff10f87ba50d3173b1eb30903 Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Thu, 16 Jul 2026 18:15:31 -0700 Subject: [PATCH] fix(e2e): long_context sonnet-4-6, complexity router assert, UI create modal long_context_1m used claude-sonnet-4-5 which is 200k-capped; switch those cells to sonnet-4-6 (1M) and register the aliases. Harden complexity router registration wait and accept alias or provider-prefixed spend models. Soften create-key modal open against ALB redirect races --- .../_pr_gate_unit_tests/test_compat_models.py | 9 ++-- .../long_context_1m/test_anthropic.py | 7 ++- .../claude_code/long_context_1m/test_azure.py | 2 +- .../long_context_1m/test_bedrock_converse.py | 2 +- .../long_context_1m/test_bedrock_invoke.py | 2 +- .../long_context_1m/test_vertex_ai.py | 2 +- tests/e2e/claude_code/test_config.yaml | 44 ++++++++++--------- .../test_key_models_dropdown_e2e.py | 12 ++++- tests/e2e/router/conftest.py | 14 +++++- .../e2e/router/test_complexity_router_e2e.py | 16 ++++--- 10 files changed, 68 insertions(+), 42 deletions(-) diff --git a/tests/e2e/claude_code/_pr_gate_unit_tests/test_compat_models.py b/tests/e2e/claude_code/_pr_gate_unit_tests/test_compat_models.py index 6614e75e2f4..23760842bb6 100644 --- a/tests/e2e/claude_code/_pr_gate_unit_tests/test_compat_models.py +++ b/tests/e2e/claude_code/_pr_gate_unit_tests/test_compat_models.py @@ -82,10 +82,11 @@ def test_yaml_has_no_unused_declarations() -> None: ) -def test_load_returns_fifteen_deployments() -> None: - """The compat matrix is 3 tiers x 5 provider surfaces = 15. Pin the - count so a future edit to the yaml can't silently drop a tier.""" - assert len(load_all_deployments()) == 15 +def test_load_returns_expected_deployment_count() -> None: + """3 tiers x 5 surfaces = 15 base deployments, plus sonnet-4-6 on each + surface for long_context_1m (sonnet-4-5 is 200k-capped). Pin the count + so a future edit to the yaml can't silently drop a tier.""" + assert len(load_all_deployments()) == 20 def test_deployments_are_hashable_and_frozen() -> None: diff --git a/tests/e2e/claude_code/long_context_1m/test_anthropic.py b/tests/e2e/claude_code/long_context_1m/test_anthropic.py index b9bbd1c2fe7..0317ce2a452 100644 --- a/tests/e2e/claude_code/long_context_1m/test_anthropic.py +++ b/tests/e2e/claude_code/long_context_1m/test_anthropic.py @@ -66,11 +66,10 @@ from claude_code.cli_driver import ( ) -# Haiku 4.5 is excluded -- only Sonnet 4.6 and Opus 4.7 support the -# 1M-context beta. See module docstring for the per-cell-aggregator -# rationale. +# Haiku 4.5 is excluded (200k window). Sonnet 4.5 is also 200k-capped on +# the Anthropic API; long_context uses sonnet-4-6 (1M) + opus-4-7. ANTHROPIC_MODELS: Sequence[str] = ( - "claude-sonnet-4-5", + "claude-sonnet-4-6", "claude-opus-4-7", ) diff --git a/tests/e2e/claude_code/long_context_1m/test_azure.py b/tests/e2e/claude_code/long_context_1m/test_azure.py index d62214d2758..1c5bdd5df7e 100644 --- a/tests/e2e/claude_code/long_context_1m/test_azure.py +++ b/tests/e2e/claude_code/long_context_1m/test_azure.py @@ -70,7 +70,7 @@ from claude_code.cli_driver import ( # 1M-context beta. See module docstring for the per-cell-aggregator # rationale. AZURE_MODELS: Sequence[str] = ( - "claude-sonnet-4-5-azure", + "claude-sonnet-4-6-azure", "claude-opus-4-7-azure", ) diff --git a/tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py b/tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py index 3c2fd4f02cc..a9ffbbd09d7 100644 --- a/tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py +++ b/tests/e2e/claude_code/long_context_1m/test_bedrock_converse.py @@ -70,7 +70,7 @@ from claude_code.cli_driver import ( # 1M-context beta. See module docstring for the per-cell-aggregator # rationale. BEDROCK_CONVERSE_MODELS: Sequence[str] = ( - "claude-sonnet-4-5-bedrock-converse", + "claude-sonnet-4-6-bedrock-converse", "claude-opus-4-7-bedrock-converse", ) diff --git a/tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py b/tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py index 4801d405760..a1164f79ee8 100644 --- a/tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py +++ b/tests/e2e/claude_code/long_context_1m/test_bedrock_invoke.py @@ -70,7 +70,7 @@ from claude_code.cli_driver import ( # 1M-context beta. See module docstring for the per-cell-aggregator # rationale. BEDROCK_INVOKE_MODELS: Sequence[str] = ( - "claude-sonnet-4-5-bedrock-invoke", + "claude-sonnet-4-6-bedrock-invoke", "claude-opus-4-7-bedrock-invoke", ) diff --git a/tests/e2e/claude_code/long_context_1m/test_vertex_ai.py b/tests/e2e/claude_code/long_context_1m/test_vertex_ai.py index efa96bf076d..6eb75940542 100644 --- a/tests/e2e/claude_code/long_context_1m/test_vertex_ai.py +++ b/tests/e2e/claude_code/long_context_1m/test_vertex_ai.py @@ -70,7 +70,7 @@ from claude_code.cli_driver import ( # 1M-context beta. See module docstring for the per-cell-aggregator # rationale. VERTEX_AI_MODELS: Sequence[str] = ( - "claude-sonnet-4-5-vertex", + "claude-sonnet-4-6-vertex", "claude-opus-4-7-vertex", ) diff --git a/tests/e2e/claude_code/test_config.yaml b/tests/e2e/claude_code/test_config.yaml index ba57ca4ccb7..c5f2121c42c 100644 --- a/tests/e2e/claude_code/test_config.yaml +++ b/tests/e2e/claude_code/test_config.yaml @@ -25,14 +25,15 @@ model_list: litellm_params: model: anthropic/claude-sonnet-4-5 api_key: os.environ/ANTHROPIC_API_KEY - extra_headers: - anthropic-beta: "context-1m-2025-08-07" + # 1M-context sonnet for long_context_1m only (sonnet-4-5 is capped at 200k). + - model_name: claude-sonnet-4-6 + litellm_params: + model: anthropic/claude-sonnet-4-6 + api_key: os.environ/ANTHROPIC_API_KEY - model_name: claude-opus-4-7 litellm_params: model: anthropic/claude-opus-4-7 api_key: os.environ/ANTHROPIC_API_KEY - extra_headers: - anthropic-beta: "context-1m-2025-08-07" # ---- Bedrock (InvokeModel) ---- - model_name: claude-haiku-4-5-bedrock-invoke @@ -43,14 +44,14 @@ model_list: litellm_params: model: bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0 aws_region_name: us-east-1 - extra_headers: - anthropic-beta: "context-1m-2025-08-07" + - model_name: claude-sonnet-4-6-bedrock-invoke + litellm_params: + model: bedrock/us.anthropic.claude-sonnet-4-6 + aws_region_name: us-east-1 - model_name: claude-opus-4-7-bedrock-invoke litellm_params: model: bedrock/us.anthropic.claude-opus-4-7 aws_region_name: us-east-1 - extra_headers: - anthropic-beta: "context-1m-2025-08-07" # ---- Bedrock (Converse) ---- - model_name: claude-haiku-4-5-bedrock-converse @@ -61,14 +62,14 @@ model_list: litellm_params: model: bedrock/converse/us.anthropic.claude-sonnet-4-5-20250929-v1:0 aws_region_name: us-east-1 - extra_headers: - anthropic-beta: "context-1m-2025-08-07" + - model_name: claude-sonnet-4-6-bedrock-converse + litellm_params: + model: bedrock/converse/us.anthropic.claude-sonnet-4-6 + aws_region_name: us-east-1 - model_name: claude-opus-4-7-bedrock-converse litellm_params: model: bedrock/converse/us.anthropic.claude-opus-4-7 aws_region_name: us-east-1 - extra_headers: - anthropic-beta: "context-1m-2025-08-07" # ---- Vertex AI ---- # `use_in_pass_through: true` registers each deployment's @@ -90,16 +91,18 @@ model_list: vertex_project: os.environ/VERTEXAI_PROJECT vertex_location: global use_in_pass_through: true - extra_headers: - anthropic-beta: "context-1m-2025-08-07" + - model_name: claude-sonnet-4-6-vertex + litellm_params: + model: vertex_ai/claude-sonnet-4-6 + vertex_project: os.environ/VERTEXAI_PROJECT + vertex_location: global + use_in_pass_through: true - model_name: claude-opus-4-7-vertex litellm_params: model: vertex_ai/claude-opus-4-7 vertex_project: os.environ/VERTEXAI_PROJECT vertex_location: global use_in_pass_through: true - extra_headers: - anthropic-beta: "context-1m-2025-08-07" # ---- Microsoft Foundry (Anthropic deployments on Azure) ---- - model_name: claude-haiku-4-5-azure @@ -112,15 +115,16 @@ model_list: model: azure_ai/claude-sonnet-4-5 api_base: os.environ/AZURE_AI_API_BASE api_key: os.environ/AZURE_AI_API_KEY - extra_headers: - anthropic-beta: "context-1m-2025-08-07" + - model_name: claude-sonnet-4-6-azure + litellm_params: + model: azure_ai/claude-sonnet-4-6 + api_base: os.environ/AZURE_AI_API_BASE + api_key: os.environ/AZURE_AI_API_KEY - model_name: claude-opus-4-7-azure litellm_params: model: azure_ai/claude-opus-4-7 api_base: os.environ/AZURE_AI_API_BASE api_key: os.environ/AZURE_AI_API_KEY - extra_headers: - anthropic-beta: "context-1m-2025-08-07" general_settings: # Claude Code sends provider-specific headers (e.g. anthropic-beta) we diff --git a/tests/e2e/management/test_key_models_dropdown_e2e.py b/tests/e2e/management/test_key_models_dropdown_e2e.py index 36b3d606d51..b1d97d6a196 100644 --- a/tests/e2e/management/test_key_models_dropdown_e2e.py +++ b/tests/e2e/management/test_key_models_dropdown_e2e.py @@ -46,8 +46,16 @@ def _models_dropdown_texts(page: Page, must_contain: str) -> list[str]: def _open_create_key_modal(page: Page) -> None: - page.goto(f"{UI_BASE_URL}/ui/api-keys/?create=true") - expect(page.locator(".ant-modal").first).to_be_visible() + # Stage ALB sometimes races an auth redirect that aborts the first goto + # to /ui/api-keys/?create=true. Land on the list first, then open create. + page.goto(f"{UI_BASE_URL}/ui/api-keys/", wait_until="domcontentloaded") + page.goto(f"{UI_BASE_URL}/ui/api-keys/?create=true", wait_until="domcontentloaded") + modal = page.locator(".ant-modal").first + try: + expect(modal).to_be_visible(timeout=15_000) + except AssertionError: + page.get_by_role("button", name="+ Create New Key").click() + expect(modal).to_be_visible(timeout=15_000) def _select_team(page: Page, alias: str) -> None: diff --git a/tests/e2e/router/conftest.py b/tests/e2e/router/conftest.py index 32868594777..f5a13f084c9 100644 --- a/tests/e2e/router/conftest.py +++ b/tests/e2e/router/conftest.py @@ -98,12 +98,18 @@ def _ensure_complexity_smart_router( # pyright: ignore[reportUnusedFunction] # Compose already declares it in docker-compose.yml; stage does not. Register via /model/new when missing and tear down only what we created. + + Always re-check servability after registration: a bare /model/new 200 is not + enough if control→data propagation lags (stage) or the data plane filters + the virtual name. Without this the chat cell fails with a vague + ``Invalid model name`` 400 instead of a registration error. """ gateway = client.gateway if _model_is_servable(gateway, ROUTER_MODEL): yield return + model_id: str | None = None try: model_id = _register_router_model(gateway) except (AssertionError, RequestException) as exc: @@ -117,6 +123,12 @@ def _ensure_complexity_smart_router( # pyright: ignore[reportUnusedFunction] # try: _await_router_model_servable(gateway) + if not _model_is_servable(gateway, ROUTER_MODEL): + raise AssertionError( + f"{ROUTER_MODEL!r} registered as {model_id!r} but still missing " + f"from /v1/models after wait; data-plane sync failed" + ) yield finally: - gateway.delete_model(model_id) + if model_id is not None: + gateway.delete_model(model_id) diff --git a/tests/e2e/router/test_complexity_router_e2e.py b/tests/e2e/router/test_complexity_router_e2e.py index 88d79a9cac0..a1f009e6eee 100644 --- a/tests/e2e/router/test_complexity_router_e2e.py +++ b/tests/e2e/router/test_complexity_router_e2e.py @@ -30,9 +30,11 @@ ROUTER_MODEL = "complexity-smart-router" # Lexically simple (heuristic -> SIMPLE) but a hard reasoning question (LLM -> above SIMPLE). LEXICALLY_SIMPLE_HARD_PROMPT = "Is P equal to NP?" # SIMPLE tier backend; served only when the classifier silently falls back to heuristic. -HEURISTIC_TIER_MODEL = "openai/gpt-5.5" +# Spend logs may store the alias (gpt-5.5) or the provider-prefixed form depending on +# how the deployment is registered (compose vs /model/new). +HEURISTIC_TIER_MODELS = frozenset({"openai/gpt-5.5", "gpt-5.5"}) # MEDIUM/COMPLEX/REASONING tier backend; served only when the LLM classifier runs. -LLM_TIER_MODEL = "anthropic/claude-haiku-4-5" +LLM_TIER_MODELS = frozenset({"anthropic/claude-haiku-4-5", "claude-haiku-4-5"}) class TestComplexityRouterLlmClassifier: @@ -54,9 +56,9 @@ class TestComplexityRouterLlmClassifier: rows = client.gateway.poll_logs_for_key(scoped_key, min_rows=1) served = [row.model for row in rows] - assert served == [LLM_TIER_MODEL], ( - f"expected the request to be served by {LLM_TIER_MODEL!r} (the higher-tier " - f"backend the LLM classifier picks for a hard prompt), but the spend log shows " - f"{served!r}. {HEURISTIC_TIER_MODEL!r} means the LLM classifier silently failed " - f"and the router fell back to heuristic scoring (SIMPLE) - the pre-fix regression" + assert served and all(model in LLM_TIER_MODELS for model in served), ( + f"expected the request to be served by one of {sorted(LLM_TIER_MODELS)!r} " + f"(higher-tier backend the LLM classifier picks for a hard prompt), but the " + f"spend log shows {served!r}. One of {sorted(HEURISTIC_TIER_MODELS)!r} means " + f"the LLM classifier silently failed or scored SIMPLE (heuristic/fallback path)" )