mirror of
https://github.com/usestrix/strix.git
synced 2026-10-01 02:03:55 +00:00
chore(models): remove the model quality warning and its allowlists
This commit is contained in:
parent
9b72488c92
commit
ef272b8e0d
9 changed files with 3 additions and 250 deletions
|
|
@ -632,61 +632,6 @@ DEFAULT_MODEL_RETRY = ModelRetrySettings(
|
|||
),
|
||||
)
|
||||
|
||||
RECOMMENDED_MODEL_NAMES = (
|
||||
"zai/glm-5.3",
|
||||
"zai/glm-5.3-flash",
|
||||
"openai/gpt-5.6-sol",
|
||||
"openai/gpt-5.6-terra",
|
||||
"openai/gpt-5.6-luna",
|
||||
"openai/gpt-5.6",
|
||||
"openai/gpt-5.5-pro",
|
||||
"openai/gpt-5.5",
|
||||
"openai/gpt-5.4",
|
||||
"openai/gpt-5.3-codex",
|
||||
"anthropic/claude-fable-5-1",
|
||||
"anthropic/claude-fable-5",
|
||||
"anthropic/claude-opus-5",
|
||||
"anthropic/claude-opus-4-8",
|
||||
"anthropic/claude-sonnet-5",
|
||||
"anthropic/claude-sonnet-4-6",
|
||||
"vertex_ai/gemini-3.1-pro-preview",
|
||||
"gemini/gemini-3.1-pro-preview",
|
||||
"vertex_ai/gemini-3.7-flash",
|
||||
"gemini/gemini-3.7-flash",
|
||||
"gemini/gemini-3.6-flash",
|
||||
"deepseek/deepseek-v4-pro",
|
||||
"deepseek/deepseek-v4-flash",
|
||||
"dashscope/qwen3.8-max",
|
||||
"dashscope/qwen3.7-max-2026-06-08",
|
||||
"moonshot/kimi-k3",
|
||||
"moonshot/kimi-k2.7-code",
|
||||
)
|
||||
|
||||
_RECOMMENDED_MODEL_NAME_SET = frozenset(name.lower() for name in RECOMMENDED_MODEL_NAMES)
|
||||
|
||||
# Matched against the bare model name only: the route (``openai/``, ``openrouter/``,
|
||||
# a local gateway, ...) says nothing about the model's quality.
|
||||
FRONTIER_MODEL_PREFIXES = (
|
||||
"gpt-5",
|
||||
"claude-fable-5",
|
||||
"claude-opus-5",
|
||||
"claude-opus-4",
|
||||
"claude-sonnet-5",
|
||||
"claude-sonnet-4",
|
||||
"gemini-3",
|
||||
"deepseek-v4",
|
||||
"deepseek-r1",
|
||||
"deepseek-reasoner",
|
||||
"qwen3.8",
|
||||
"qwen3.7",
|
||||
"qwen3-max",
|
||||
"kimi-k3",
|
||||
"kimi-k2.7",
|
||||
"kimi-k2.6",
|
||||
"glm-5.3",
|
||||
"glm-5.2",
|
||||
)
|
||||
|
||||
|
||||
def configure_sdk_model_defaults(settings: Settings) -> None:
|
||||
"""Apply Strix config to SDK-native defaults."""
|
||||
|
|
@ -947,43 +892,6 @@ def model_supports_reasoning(model_name: str) -> bool:
|
|||
return bool(entry and entry.get("supports_reasoning"))
|
||||
|
||||
|
||||
def is_recommended_or_frontier_model(model_name: str) -> bool:
|
||||
"""Return whether a model is recommended or in a frontier model family."""
|
||||
name = _normalized_model_name(model_name)
|
||||
if not name:
|
||||
return False
|
||||
if name in _RECOMMENDED_MODEL_NAME_SET:
|
||||
return True
|
||||
bare_model_name = name.rsplit("/", 1)[-1]
|
||||
return _matches_model_prefix(bare_model_name, FRONTIER_MODEL_PREFIXES)
|
||||
|
||||
|
||||
def _normalized_model_name(model_name: str) -> str:
|
||||
name = model_name.strip().lower()
|
||||
for prefix in ("litellm/", "any-llm/"):
|
||||
if name.startswith(prefix):
|
||||
name = name[len(prefix) :]
|
||||
break
|
||||
return name
|
||||
|
||||
|
||||
def _matches_model_prefix(model_name: str, model_prefixes: tuple[str, ...]) -> bool:
|
||||
return any(
|
||||
candidate.startswith(prefix)
|
||||
for candidate in _model_name_candidates(model_name)
|
||||
for prefix in model_prefixes
|
||||
)
|
||||
|
||||
|
||||
def _model_name_candidates(model_name: str) -> tuple[str, ...]:
|
||||
if "." not in model_name:
|
||||
return (model_name,)
|
||||
suffixes = tuple(
|
||||
model_name.split(".", index)[-1] for index in range(1, model_name.count(".") + 1)
|
||||
)
|
||||
return (model_name, *suffixes)
|
||||
|
||||
|
||||
def is_known_openai_bare_model(model_name: str) -> bool:
|
||||
import litellm
|
||||
|
||||
|
|
|
|||
|
|
@ -135,14 +135,12 @@ def _subscription_error_hint(exc: BaseException) -> str | None:
|
|||
return None
|
||||
|
||||
|
||||
async def warm_up_llm(show_model_warning: bool = True) -> None:
|
||||
async def warm_up_llm() -> None:
|
||||
from agents.models.interface import ModelTracing
|
||||
|
||||
from strix.config.models import (
|
||||
RECOMMENDED_MODEL_NAMES,
|
||||
configure_sdk_model_defaults,
|
||||
is_known_openai_bare_model,
|
||||
is_recommended_or_frontier_model,
|
||||
)
|
||||
from strix.core.inputs import make_model_settings
|
||||
|
||||
|
|
@ -186,32 +184,6 @@ async def warm_up_llm(show_model_warning: bool = True) -> None:
|
|||
)
|
||||
sys.exit(1)
|
||||
|
||||
if show_model_warning and raw_model and not is_recommended_or_frontier_model(raw_model):
|
||||
warn_text = Text()
|
||||
warn_text.append("MODEL QUALITY WARNING", style="bold yellow")
|
||||
warn_text.append("\n\n", style="white")
|
||||
warn_text.append(f"'{raw_model}'", style="bold cyan")
|
||||
warn_text.append(
|
||||
" is not a recommended frontier model for Strix.\nSecurity scans work best with:\n",
|
||||
style="white",
|
||||
)
|
||||
for recommended_model in RECOMMENDED_MODEL_NAMES:
|
||||
warn_text.append(f"• {recommended_model}\n", style="bold cyan")
|
||||
warn_text.append(
|
||||
"\nYou can continue, but weaker models may miss vulnerabilities "
|
||||
"or produce lower-quality findings.",
|
||||
style="white",
|
||||
)
|
||||
console.print(
|
||||
Panel(
|
||||
warn_text,
|
||||
title="[bold white]STRIX",
|
||||
title_align="left",
|
||||
border_style="yellow",
|
||||
padding=(1, 2),
|
||||
),
|
||||
)
|
||||
|
||||
await preflight_model_connection(raw_model, settings=settings)
|
||||
logger.info("LLM warm-up succeeded for model %s", (llm.model or "").strip())
|
||||
|
||||
|
|
@ -403,7 +375,7 @@ def _bootstrap_scan(args: argparse.Namespace) -> None:
|
|||
"""
|
||||
set_scan_phase("preflight")
|
||||
try:
|
||||
asyncio.run(warm_up_llm(show_model_warning=True))
|
||||
asyncio.run(warm_up_llm())
|
||||
except ModelConnectionError as exc:
|
||||
report_error("model_connection_failed", exc)
|
||||
_print_model_connection_error(exc, exc.model_name)
|
||||
|
|
|
|||
|
|
@ -11,7 +11,6 @@ from pathlib import Path
|
|||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from strix.config import load_settings
|
||||
from strix.config.models import is_recommended_or_frontier_model
|
||||
from strix.config.settings import DEFAULT_MAX_TURNS
|
||||
from strix.interface.tui.backend.live_view import TuiLiveView
|
||||
from strix.interface.tui.backend.projection import (
|
||||
|
|
@ -189,11 +188,6 @@ class TuiController:
|
|||
subscription = False
|
||||
with contextlib.suppress(Exception):
|
||||
subscription = is_subscription_run(self.report_state)
|
||||
model_warning = ""
|
||||
if model and not is_recommended_or_frontier_model(model):
|
||||
model_warning = (
|
||||
f"{model} is not a recommended frontier model. Pentest quality could be degraded."
|
||||
)
|
||||
state = {
|
||||
"setup_mode": self.setup_mode,
|
||||
"scan_started": self.scan_started,
|
||||
|
|
@ -211,7 +205,6 @@ class TuiController:
|
|||
"scope_mode": self.scope_mode,
|
||||
"diff_base": terminal_projection(self.diff_base, max_string=256),
|
||||
"model": terminal_projection(model, max_string=256),
|
||||
"model_warning": terminal_projection(model_warning, max_string=512),
|
||||
"caido_url": terminal_projection(
|
||||
getattr(self.report_state, "caido_url", None), max_string=1024
|
||||
),
|
||||
|
|
|
|||
|
|
@ -150,7 +150,6 @@ def bounded_state_projection(state: dict[str, Any]) -> dict[str, Any]:
|
|||
key: state["usage"][key] for key in ("total_tokens", "cost") if key in state["usage"]
|
||||
}
|
||||
state["error"] = terminal_projection(state["error"], max_string=512)
|
||||
state["model_warning"] = terminal_projection(state["model_warning"], max_string=256)
|
||||
state["caido_url"] = terminal_projection(state["caido_url"], max_string=256)
|
||||
state["viewer_url"] = terminal_projection(state["viewer_url"], max_string=256)
|
||||
if encoded_size(state) <= STATE_TARGET_BYTES:
|
||||
|
|
@ -173,7 +172,6 @@ def bounded_state_projection(state: dict[str, Any]) -> dict[str, Any]:
|
|||
"scope_mode": state["scope_mode"],
|
||||
"diff_base": state["diff_base"],
|
||||
"model": state["model"],
|
||||
"model_warning": "",
|
||||
"caido_url": None,
|
||||
"messages": [],
|
||||
"usage": state["usage"],
|
||||
|
|
|
|||
|
|
@ -370,17 +370,6 @@ func TestStartedSnapshotTransitionsToLiveView(t *testing.T) {
|
|||
}
|
||||
}
|
||||
|
||||
func TestSplashModelWarningRendersTheBackendSentenceOnce(t *testing.T) {
|
||||
warning := "openai/glm-5.3 is not a recommended frontier model. Pentest quality could be degraded."
|
||||
got := ansi.Strip(splashModelWarning("openai/glm-5.3", warning))
|
||||
if got != "⚠ "+warning {
|
||||
t.Fatalf("splash warning = %q, want %q", got, "⚠ "+warning)
|
||||
}
|
||||
if got := ansi.Strip(splashModelWarning("other/model", warning)); got != "⚠ "+warning {
|
||||
t.Fatalf("splash warning with unrelated model = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSetupStartScreenFitsNarrowTerminal(t *testing.T) {
|
||||
model := New(nil)
|
||||
model.width, model.height = 40, 18
|
||||
|
|
|
|||
|
|
@ -423,27 +423,12 @@ func (m Model) splashView() string {
|
|||
content := wordmark() + "\n\n" +
|
||||
welcome + "\n" + version + "\n" + tagline + "\n\n" +
|
||||
start.String() + "\n\n" + url
|
||||
if warn := m.snapshot.ModelWarning; warn != "" {
|
||||
content += "\n\n" + splashModelWarning(m.snapshot.Model, warn)
|
||||
}
|
||||
panel := lipgloss.NewStyle().Border(lipgloss.RoundedBorder()).BorderForeground(green).Padding(1, 6).Align(lipgloss.Center).Render(content)
|
||||
// #splash_screen background is solid black.
|
||||
return lipgloss.Place(m.width, m.height, lipgloss.Center, lipgloss.Center, panel,
|
||||
lipgloss.WithWhitespaceBackground(black))
|
||||
}
|
||||
|
||||
// splashModelWarning renders the backend's full warning sentence, with the
|
||||
// model name highlighted when the sentence leads with it.
|
||||
func splashModelWarning(model, warning string) string {
|
||||
yellow := lipgloss.Color("#eab308")
|
||||
out := lipgloss.NewStyle().Bold(true).Foreground(yellow).Render("⚠ ")
|
||||
if model != "" && strings.HasPrefix(warning, model) {
|
||||
out += lipgloss.NewStyle().Bold(true).Foreground(render.Cyan).Render(model)
|
||||
warning = strings.TrimPrefix(warning, model)
|
||||
}
|
||||
return out + lipgloss.NewStyle().Foreground(yellow).Render(warning)
|
||||
}
|
||||
|
||||
// chatPaneKey identifies everything the bordered trace depends on.
|
||||
type chatPaneKey struct {
|
||||
offset int
|
||||
|
|
|
|||
|
|
@ -71,7 +71,6 @@ type Snapshot struct {
|
|||
ScopeMode string `json:"scope_mode"`
|
||||
DiffBase string `json:"diff_base"`
|
||||
Model string `json:"model"`
|
||||
ModelWarning string `json:"model_warning"`
|
||||
CaidoURL string `json:"caido_url"`
|
||||
Messages []Message `json:"messages"`
|
||||
Agents []Agent `json:"-"`
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
"""Tests for LLM model recommendation helpers."""
|
||||
"""Tests for LLM model configuration helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
|
@ -11,12 +11,10 @@ from agents.models.openai_chatcompletions import OpenAIChatCompletionsModel
|
|||
from agents.models.openai_responses import OpenAIResponsesModel
|
||||
|
||||
from strix.config.models import (
|
||||
RECOMMENDED_MODEL_NAMES,
|
||||
StrixProvider,
|
||||
_NonStreamingModel,
|
||||
_TurnGuardModel,
|
||||
configure_sdk_model_defaults,
|
||||
is_recommended_or_frontier_model,
|
||||
request_timeout_extra_args,
|
||||
routes_through_litellm,
|
||||
supports_strict_tool_schemas,
|
||||
|
|
@ -26,11 +24,6 @@ from strix.config.settings import Settings
|
|||
from strix.llm.request_log import RequestLoggingModel
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model_name", RECOMMENDED_MODEL_NAMES)
|
||||
def test_recommended_models_are_accepted(model_name: str) -> None:
|
||||
assert is_recommended_or_frontier_model(model_name)
|
||||
|
||||
|
||||
def test_request_timeout_extra_args_positive() -> None:
|
||||
assert request_timeout_extra_args(300) == {"timeout": 300}
|
||||
assert request_timeout_extra_args(10) == {"timeout": 10}
|
||||
|
|
@ -48,80 +41,6 @@ def test_request_timeout_extra_args_disabled(value: float | None) -> None:
|
|||
assert request_timeout_extra_args(value) is None
|
||||
|
||||
|
||||
def test_recommended_models_are_matched_case_insensitively() -> None:
|
||||
assert is_recommended_or_frontier_model("Vertex_AI/Gemini-3-Pro-Preview")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[
|
||||
"gpt-5.5",
|
||||
"chatgpt/gpt-5.4",
|
||||
"litellm/openai/gpt-5.4-pro",
|
||||
"azure_ai/gpt-5.5-pro",
|
||||
"bedrock_mantle/openai.gpt-5.5",
|
||||
"anthropic/claude-opus-5",
|
||||
"anthropic/claude-opus-4-8",
|
||||
"anthropic.claude-opus-4-8",
|
||||
"anthropic/claude-opus-4-7",
|
||||
"anthropic/claude-fable-5",
|
||||
"anthropic/claude-sonnet-5",
|
||||
"vertex_ai/claude-sonnet-5@default",
|
||||
"vertex_ai/claude-sonnet-4-6@default",
|
||||
"any-llm/anthropic/claude-sonnet-4-6",
|
||||
"vertex_ai/gemini-3.1-pro-preview",
|
||||
"openrouter/google/gemini-3.1-pro-preview",
|
||||
"deepseek/deepseek-v4-pro",
|
||||
"deepseek/deepseek-r1-0528",
|
||||
"deepseek/deepseek-reasoner",
|
||||
"dashscope/qwen3-max-2026-01-23",
|
||||
"qwen3.7-max",
|
||||
"dashscope/qwen3.8-max",
|
||||
"moonshot/kimi-k2.6",
|
||||
"kimi-k2.7-code",
|
||||
"moonshot/kimi-k3",
|
||||
"anthropic/claude-fable-5-1",
|
||||
"vertex_ai/claude-fable-5-1@default",
|
||||
"gemini/gemini-3.7-flash",
|
||||
"glm-5.3",
|
||||
"zai/glm-5.3-flash",
|
||||
"openrouter/z-ai/glm-5.3",
|
||||
"novita/zai-org/glm-5.2",
|
||||
"openai/glm-5.3",
|
||||
"openai/zai-org/glm-5.3",
|
||||
"hosted_vllm/glm-5.3",
|
||||
"openai/claude-opus-4-8",
|
||||
"openai/deepseek-v4-pro",
|
||||
"custom-ollama/gpt-5-mini-local",
|
||||
"custom-provider/claude-opus-4-local",
|
||||
"custom-provider/glm-5.3-local",
|
||||
],
|
||||
)
|
||||
def test_frontier_model_families_are_accepted(model_name: str) -> None:
|
||||
assert is_recommended_or_frontier_model(model_name)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[
|
||||
"",
|
||||
"openai/gpt-4.1",
|
||||
"anthropic/claude-3-5-sonnet-latest",
|
||||
"ollama/llama3.1",
|
||||
"deepseek/deepseek-chat",
|
||||
"xai/grok-4.5",
|
||||
"openrouter/x-ai/grok-4",
|
||||
"mistral/mistral-medium-3-5",
|
||||
"mistral/magistral-medium-latest",
|
||||
"zai/glm-4.7",
|
||||
"openai/glm-4.7",
|
||||
"openrouter/z-ai/glm-5",
|
||||
],
|
||||
)
|
||||
def test_non_frontier_models_are_rejected(model_name: str) -> None:
|
||||
assert not is_recommended_or_frontier_model(model_name)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[
|
||||
|
|
|
|||
|
|
@ -129,16 +129,6 @@ async def test_large_target_list_reports_truncated_snapshot_count() -> None:
|
|||
assert len(snapshot["targets"]) == 16
|
||||
|
||||
|
||||
def test_state_populates_model_warning_for_non_frontier_model() -> None:
|
||||
os.environ["STRIX_LLM"] = "openai/gpt-3.5-turbo"
|
||||
loader._cached = None
|
||||
|
||||
warning = TuiController(args()).snapshot()["model_warning"]
|
||||
|
||||
assert "openai/gpt-3.5-turbo" in warning
|
||||
assert "not a recommended frontier model" in warning
|
||||
|
||||
|
||||
def test_setup_restores_prepared_cli_targets() -> None:
|
||||
setup_args = args()
|
||||
setup_args.targets_info = [
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue