diff --git a/strix/config/models.py b/strix/config/models.py
index 8fe43e21..709ef267 100644
--- a/strix/config/models.py
+++ b/strix/config/models.py
@@ -537,12 +537,28 @@ class StrixProvider(MultiProvider):
codex.get_subscription_client(),
reasoning_effort=llm.reasoning_effort,
)
- elif oc and oc.uses_responses:
+ elif oc and oc.protocol == opencode.PROTOCOL_RESPONSES:
model = _CodexResponsesModel(
oc.slug,
opencode.get_subscription_client(oc.base_url),
reasoning_effort=llm.reasoning_effort,
)
+ elif oc and oc.protocol == opencode.PROTOCOL_MESSAGES:
+ # Claude models are served on Anthropic's ``/messages``, which the
+ # OpenAI SDK cannot speak: it has no Messages method and sends the
+ # key as a bearer token rather than ``x-api-key``. LiteLLM's
+ # Anthropic route handles both, so the gateway becomes an Anthropic
+ # base URL with the subscription key.
+ from agents.extensions.models.litellm_model import LitellmModel
+
+ model = LitellmModel(
+ model=f"anthropic/{oc.slug}",
+ base_url=oc.messages_url,
+ api_key=opencode.get_api_key(),
+ )
+ if llm.disable_streaming:
+ model = _NonStreamingModel(model)
+ idle_timeout = 0.0
elif oc:
model = OpenAIChatCompletionsModel(
oc.slug, opencode.get_subscription_client(oc.base_url)
@@ -634,7 +650,14 @@ def configure_sdk_model_defaults(settings: Settings) -> None:
"""Apply Strix config to SDK-native defaults."""
llm = settings.llm
set_tracing_disabled(True)
- if codex.subscription_model(llm.model) or opencode.subscription_model(llm.model):
+ oc = opencode.subscription_model(llm.model)
+ if codex.subscription_model(llm.model) or oc:
+ # A subscription run carries its own client and credentials, so none of
+ # the api_key/api_base defaults below apply. The Anthropic route is the
+ # exception: it goes through LiteLLM, which still needs the
+ # compatibility flags and the cost callback.
+ if oc is not None and oc.protocol == opencode.PROTOCOL_MESSAGES:
+ _configure_litellm_compatibility()
return
_configure_litellm_compatibility()
_configure_openrouter_attribution(llm.model)
@@ -821,7 +844,9 @@ def uses_chat_completions_tool_schema(model_name: str, settings: Settings) -> bo
return False
oc = opencode.subscription_model(model_name)
if oc:
- return not oc.uses_responses
+ # Chat Completions takes JSON function tools; so does the LiteLLM
+ # Anthropic route, which translates them to Anthropic tool blocks.
+ return oc.protocol != opencode.PROTOCOL_RESPONSES
model = model_name.strip().lower()
if "/" in model and not model.startswith("openai/"):
return True
diff --git a/strix/config/opencode.py b/strix/config/opencode.py
index f7de0fe4..980d3c93 100644
--- a/strix/config/opencode.py
+++ b/strix/config/opencode.py
@@ -1,15 +1,18 @@
-"""OpenCode subscription auth: API-key sign-in and the OpenAI clients that
-route inference through the OpenCode gateway.
+"""OpenCode subscription auth: API-key sign-in and the clients that route
+inference through the OpenCode gateway.
-Covers both OpenCode offerings — Zen (pay-as-you-go credits) and Go (the
-monthly subscription) — which share one account and API key but live behind
+Covers both OpenCode offerings, Zen (pay-as-you-go credits) and Go (the
+monthly subscription), which share one account and API key but live behind
different gateway base URLs. Unlike the ChatGPT subscription there is no
OAuth: the user copies a plain API key from https://opencode.ai/auth, and
using the gateway from other agents is officially supported.
-Model routing follows the endpoint each model is served on (see
-https://opencode.ai/docs/zen/): GPT models use the Responses API, everything
-else the OpenAI-compatible Chat Completions API.
+The gateway speaks three protocols and serves each model family on exactly
+one of them (see https://opencode.ai/docs/zen/), answering a request sent to
+the wrong one with an unhandled 500 rather than a 404. ``_protocol()`` holds
+the mapping; ``SubscriptionModel.protocol`` carries the result. Claude runs on
+Anthropic's ``/messages``, which the OpenAI SDK cannot speak, so that route
+goes through LiteLLM instead of the clients built here.
"""
from __future__ import annotations
@@ -45,31 +48,85 @@ class OpencodeAuthError(Exception):
super().__init__(message or code)
+PROTOCOL_CHAT = "chat"
+PROTOCOL_RESPONSES = "responses"
+PROTOCOL_MESSAGES = "messages"
+
+PLAN_ZEN = "zen"
+PLAN_GO = "go"
+
+_PLAN_LABELS = {PLAN_ZEN: "OpenCode Zen", PLAN_GO: "OpenCode Go"}
+
+
@dataclass(frozen=True)
class SubscriptionModel:
slug: str
base_url: str
- uses_responses: bool
+ protocol: str
+ plan: str
+
+ @property
+ def uses_responses(self) -> bool:
+ return self.protocol == PROTOCOL_RESPONSES
+
+ @property
+ def messages_url(self) -> str:
+ """Anthropic-protocol endpoint for this gateway, e.g. ``.../zen/v1/messages``."""
+ return f"{self.base_url}/messages"
+
+ @property
+ def label(self) -> str:
+ return _PLAN_LABELS[self.plan]
+
+ @property
+ def metered(self) -> bool:
+ """Whether a run spends money per request.
+
+ Zen bills prepaid credits per request, so its runs cost real money and
+ must not be reported as free. Go is a flat monthly fee, where a run's
+ marginal cost genuinely is zero.
+ """
+ return self.plan == PLAN_ZEN
-def _uses_responses(slug: str, base_url: str) -> bool:
+def _protocol(slug: str, base_url: str) -> str:
+ """Which wire protocol the gateway serves *slug* on.
+
+ The gateway routes by model family and answers a request sent to the wrong
+ protocol with an unhandled 500 rather than a 404, so the mapping has to be
+ right. Probed against both gateways per family:
+
+ * Claude on Anthropic's ``/messages``
+ * GPT, Grok (Zen) and Muse on OpenAI's ``/responses``
+ * DeepSeek, MiniMax, Kimi, GLM and Qwen on Chat Completions
+
+ Grok is absent from the Go catalog, so its Zen-only Responses route costs
+ nothing there. Kimi and Qwen also answer on ``/messages``, but Chat
+ Completions works for them on both plans and stays the single mapping.
+ """
lowered = slug.lower()
- if lowered.startswith("gpt-"):
- return True
- # Grok is served via Responses on Zen but Chat Completions on Go.
- return lowered.startswith("grok") and base_url == ZEN_BASE_URL
+ if lowered.startswith("claude-"):
+ return PROTOCOL_MESSAGES
+ if lowered.startswith(("gpt-", "muse-")):
+ return PROTOCOL_RESPONSES
+ if lowered.startswith("grok") and base_url == ZEN_BASE_URL:
+ return PROTOCOL_RESPONSES
+ return PROTOCOL_CHAT
def subscription_model(model_name: str | None) -> SubscriptionModel | None:
"""The gateway model behind an ``opencode/`` or ``opencode-go/`` STRIX_LLM."""
name = (model_name or "").strip()
lowered = name.lower()
- for prefix, base_url in ((GO_PREFIX, GO_BASE_URL), (ZEN_PREFIX, ZEN_BASE_URL)):
+ for prefix, base_url, plan in (
+ (GO_PREFIX, GO_BASE_URL, PLAN_GO),
+ (ZEN_PREFIX, ZEN_BASE_URL, PLAN_ZEN),
+ ):
if lowered.startswith(prefix):
slug = name[len(prefix) :]
if not slug:
return None
- return SubscriptionModel(slug, base_url, _uses_responses(slug, base_url))
+ return SubscriptionModel(slug, base_url, _protocol(slug, base_url), plan)
return None
@@ -149,6 +206,16 @@ def auth_mode(model_name: str | None) -> str:
return "api_key"
+def subscription_plan(model_name: str | None) -> str | None:
+ """Which OpenCode plan STRIX_LLM runs on: "zen", "go", or None.
+
+ Recorded alongside ``subscription_provider`` rather than folded into it, so
+ consumers that compare the provider against "opencode" keep working.
+ """
+ oc = subscription_model(model_name)
+ return oc.plan if oc else None
+
+
def subscription_provider(model_name: str | None) -> str | None:
"""The subscription behind STRIX_LLM: "opencode", "chatgpt", or None."""
if subscription_model(model_name):
diff --git a/strix/core/inputs.py b/strix/core/inputs.py
index 1123be43..96676e39 100644
--- a/strix/core/inputs.py
+++ b/strix/core/inputs.py
@@ -326,9 +326,12 @@ def _prompt_cache_extra_args(model_name: str) -> dict[str, Any] | None:
"""
if not is_claude_model(model_name) or not routes_through_litellm(model_name):
return None
- # OpenCode routes use the raw OpenAI SDK, which rejects this LiteLLM-only
- # argument; the gateway applies Anthropic prompt caching itself.
- if opencode.subscription_model(model_name):
+ # OpenCode's Chat Completions and Responses routes use the raw OpenAI SDK,
+ # which rejects this LiteLLM-only argument. Its Anthropic route does go
+ # through LiteLLM, so the injection points apply there as they would for a
+ # direct Anthropic key.
+ oc = opencode.subscription_model(model_name)
+ if oc is not None and oc.protocol != opencode.PROTOCOL_MESSAGES:
return None
if is_bedrock_route(model_name) and not bedrock_route_supports_prompt_caching(model_name):
return None
diff --git a/strix/interface/environment.py b/strix/interface/environment.py
index fb2d2d2b..ed01dc82 100644
--- a/strix/interface/environment.py
+++ b/strix/interface/environment.py
@@ -37,14 +37,15 @@ def validate_environment() -> None:
logger.info("Environment OK (ChatGPT subscription)")
return
- if opencode.subscription_model(settings.llm.model):
+ oc = opencode.subscription_model(settings.llm.model)
+ if oc:
if not opencode.is_authenticated():
console.print(
- f"[red]STRIX_LLM={settings.llm.model} uses your OpenCode subscription, "
+ f"[red]STRIX_LLM={settings.llm.model} runs on {oc.label}, "
"but you're not signed in.[/] Run [cyan]strix auth login opencode[/] first."
)
sys.exit(1)
- logger.info("Environment OK (OpenCode subscription)")
+ logger.info("Environment OK (%s)", oc.label)
return
if not settings.llm.model:
diff --git a/strix/interface/utils.py b/strix/interface/utils.py
index a4fb55f8..f2d784d7 100644
--- a/strix/interface/utils.py
+++ b/strix/interface/utils.py
@@ -290,12 +290,20 @@ def subscription_label() -> str:
"""Display name of the subscription behind the configured model."""
from strix.config import opencode
- model = load_settings().llm.model
- if opencode.subscription_model(model):
- return "OpenCode subscription"
+ oc = opencode.subscription_model(load_settings().llm.model)
+ if oc:
+ return oc.label
return "ChatGPT subscription"
+def subscription_is_metered() -> bool:
+ """Whether the run spends per-request credits rather than a flat plan."""
+ from strix.config import opencode
+
+ oc = opencode.subscription_model(load_settings().llm.model)
+ return oc is not None and oc.metered
+
+
def _int_stat(usage: dict[str, Any], key: str) -> int:
try:
return max(0, int(usage.get(key) or 0))
@@ -336,7 +344,9 @@ def _build_llm_usage_stats(
if not usage or _int_stat(usage, "requests") <= 0:
stats_text.append("\n")
stats_text.append("Cost ", style="dim")
- if subscription:
+ if subscription and subscription_is_metered():
+ stats_text.append("credits ", style="#22c55e")
+ elif subscription:
stats_text.append("$0.00 ", style="#22c55e")
stats_text.append("(subscription) ", style="dim")
else:
@@ -365,7 +375,19 @@ def _build_llm_usage_stats(
stats_text.append("Output Tokens ", style="dim")
stats_text.append(format_token_count(output_tokens), style="white")
- if subscription:
+ if subscription and subscription_is_metered():
+ # Zen spends prepaid credits per request, so a run is not free. Its
+ # Anthropic route runs through LiteLLM and yields a real charge; the
+ # OpenAI-SDK routes report none, and an unpriced run says so rather
+ # than claiming $0.00.
+ stats_text.append(" · ", style="dim white")
+ stats_text.append("Cost ", style="dim")
+ if cost > 0:
+ stats_text.append(f"${cost:.4f}", style="#22c55e")
+ stats_text.append(" (credits)", style="dim")
+ else:
+ stats_text.append("credits", style="#22c55e")
+ elif subscription:
stats_text.append(" · ", style="dim white")
stats_text.append("Cost ", style="dim")
stats_text.append("$0.00", style="#22c55e")
diff --git a/strix/interface/viewer/frontend/src/components/RunDetails.tsx b/strix/interface/viewer/frontend/src/components/RunDetails.tsx
index c169d0de..33daf4a0 100644
--- a/strix/interface/viewer/frontend/src/components/RunDetails.tsx
+++ b/strix/interface/viewer/frontend/src/components/RunDetails.tsx
@@ -104,8 +104,20 @@ export function RunDetails({
const subscriptionProvider =
str(raw.subscription_provider) ??
(models.some((m) => m.toLowerCase().startsWith("opencode")) ? "opencode" : "chatgpt");
+ // Runs recorded before subscription_plan existed still carry the model string,
+ // whose prefix names the plan.
+ const subscriptionPlan =
+ str(raw.subscription_plan) ??
+ (models.some((m) => m.toLowerCase().startsWith("opencode-go/")) ? "go" : "zen");
const subscriptionLabel =
- subscriptionProvider === "opencode" ? "OpenCode subscription" : "ChatGPT subscription";
+ subscriptionProvider === "opencode"
+ ? subscriptionPlan === "go"
+ ? "OpenCode Go"
+ : "OpenCode Zen"
+ : "ChatGPT subscription";
+ // Zen bills prepaid credits per request, so its runs are not free and there is
+ // no price table to estimate them from. Go is a flat monthly plan.
+ const metered = subscriptionProvider === "opencode" && subscriptionPlan === "zen";
const sub = (n: number, word: string) => (
({formatNumber(n)} {word})
@@ -205,7 +217,21 @@ export function RunDetails({
)}
{totalTokens != null && {formatNumber(totalTokens)}}
- {subscription ? (
+ {subscription && metered ? (
+
+ {cost != null && cost > 0 ? (
+ <>
+ ${cost.toFixed(2)}
+ (Zen credits)
+ >
+ ) : (
+ <>
+ credits
+ (not priced locally)
+ >
+ )}
+
+ ) : subscription ? (
$0.00
(subscription)
diff --git a/strix/report/state.py b/strix/report/state.py
index 3ed2154c..a526fdec 100644
--- a/strix/report/state.py
+++ b/strix/report/state.py
@@ -153,7 +153,12 @@ class ReportState:
self._llm_usage = LLMUsageLedger()
self._telemetry_llm_usage_baseline: dict[str, Any] = {}
auth_mode = opencode.auth_mode(load_settings().llm.model)
- self._llm_usage.zero_cost = auth_mode == "subscription"
+ oc = opencode.subscription_model(load_settings().llm.model)
+ # A flat subscription has no per-run charge to report. Zen bills prepaid
+ # credits per request, so its cost is real and stays tracked.
+ self._llm_usage.zero_cost = auth_mode == "subscription" and not (
+ oc is not None and oc.metered
+ )
self.run_record: dict[str, Any] = {
"run_id": self.run_id,
"run_name": self.run_name,
@@ -162,6 +167,7 @@ class ReportState:
"status": "running",
"auth_mode": auth_mode,
"subscription_provider": opencode.subscription_provider(load_settings().llm.model),
+ "subscription_plan": opencode.subscription_plan(load_settings().llm.model),
"targets_info": [],
"llm_usage": self._build_llm_usage_record(),
}
diff --git a/tests/test_cli_target_list.py b/tests/test_cli_target_list.py
index 6ba4ca22..d20230b1 100644
--- a/tests/test_cli_target_list.py
+++ b/tests/test_cli_target_list.py
@@ -228,7 +228,10 @@ def test_resume_still_requires_targets_or_a_workspace(
assert "has no targets_info" in capsys.readouterr().err
-def test_resume_non_object_run_json_exits(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]) -> None:
+
+def test_resume_non_object_run_json_exits(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
+) -> None:
monkeypatch.chdir(tmp_path)
run_dir = tmp_path / "strix_runs" / "pentest_abcd"
run_dir.mkdir(parents=True)
diff --git a/tests/test_inputs.py b/tests/test_inputs.py
index a99a3a47..365a87d7 100644
--- a/tests/test_inputs.py
+++ b/tests/test_inputs.py
@@ -123,10 +123,20 @@ def test_make_model_settings_no_prompt_cache_for_non_claude(model_name: str) ->
@pytest.mark.parametrize("model_name", ["opencode/claude-sonnet-5", "opencode-go/claude-sonnet-5"])
-def test_no_prompt_cache_for_opencode_claude(model_name: str) -> None:
- # The OpenCode route uses the raw OpenAI SDK, whose create() rejects the
- # LiteLLM-only cache_control_injection_points argument.
- assert _cache_points(model_name) is None
+def test_prompt_cache_for_opencode_claude(model_name: str) -> None:
+ # Claude on OpenCode runs through LiteLLM's Anthropic route, which consumes
+ # cache_control_injection_points. The gateway's other two routes use the raw
+ # OpenAI SDK, whose create() rejects this LiteLLM-only argument.
+ assert _cache_points(model_name) == [
+ {"location": "message", "role": "system"},
+ {"location": "message", "index": -1},
+ ]
+
+
+def test_no_prompt_cache_for_opencode_openai_routes() -> None:
+ # A "claude" substring cannot smuggle the LiteLLM-only argument onto a route
+ # that is served by the raw OpenAI SDK.
+ assert _cache_points("opencode/gpt-5.4-claude-tuned") is None
def test_no_prompt_cache_for_unmapped_bedrock_claude_model(monkeypatch: Any) -> None:
diff --git a/tests/test_opencode_auth.py b/tests/test_opencode_auth.py
index 3c120cd4..d05c12e8 100644
--- a/tests/test_opencode_auth.py
+++ b/tests/test_opencode_auth.py
@@ -23,25 +23,104 @@ def _tmp_store(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path:
@pytest.mark.parametrize(
- ("model", "slug", "base_url", "uses_responses"),
+ ("model", "slug", "base_url", "protocol"),
[
- ("opencode/claude-sonnet-5", "claude-sonnet-5", opencode.ZEN_BASE_URL, False),
- ("opencode/gpt-5.4", "gpt-5.4", opencode.ZEN_BASE_URL, True),
- ("opencode/grok-4.5", "grok-4.5", opencode.ZEN_BASE_URL, True),
- ("OpenCode/Kimi-K3", "Kimi-K3", opencode.ZEN_BASE_URL, False),
- ("opencode-go/kimi-k3", "kimi-k3", opencode.GO_BASE_URL, False),
- ("opencode-go/gpt-5.6-luna", "gpt-5.6-luna", opencode.GO_BASE_URL, True),
- ("opencode-go/grok-4.5", "grok-4.5", opencode.GO_BASE_URL, False),
+ (
+ "opencode/claude-sonnet-5",
+ "claude-sonnet-5",
+ opencode.ZEN_BASE_URL,
+ opencode.PROTOCOL_MESSAGES,
+ ),
+ ("opencode/gpt-5.4", "gpt-5.4", opencode.ZEN_BASE_URL, opencode.PROTOCOL_RESPONSES),
+ ("opencode/grok-4.5", "grok-4.5", opencode.ZEN_BASE_URL, opencode.PROTOCOL_RESPONSES),
+ ("OpenCode/Kimi-K3", "Kimi-K3", opencode.ZEN_BASE_URL, opencode.PROTOCOL_CHAT),
+ (
+ "OpenCode/Claude-Opus-5",
+ "Claude-Opus-5",
+ opencode.ZEN_BASE_URL,
+ opencode.PROTOCOL_MESSAGES,
+ ),
+ ("opencode-go/kimi-k3", "kimi-k3", opencode.GO_BASE_URL, opencode.PROTOCOL_CHAT),
+ (
+ "opencode-go/gpt-5.6-luna",
+ "gpt-5.6-luna",
+ opencode.GO_BASE_URL,
+ opencode.PROTOCOL_RESPONSES,
+ ),
+ ("opencode-go/grok-4.5", "grok-4.5", opencode.GO_BASE_URL, opencode.PROTOCOL_CHAT),
+ # Probed per family against both gateways; a wrong protocol 500s.
+ (
+ "opencode/muse-spark-1.2",
+ "muse-spark-1.2",
+ opencode.ZEN_BASE_URL,
+ opencode.PROTOCOL_RESPONSES,
+ ),
+ (
+ "opencode/deepseek-v4-pro",
+ "deepseek-v4-pro",
+ opencode.ZEN_BASE_URL,
+ opencode.PROTOCOL_CHAT,
+ ),
+ ("opencode/minimax-m3", "minimax-m3", opencode.ZEN_BASE_URL, opencode.PROTOCOL_CHAT),
+ ("opencode/qwen3.6-plus", "qwen3.6-plus", opencode.ZEN_BASE_URL, opencode.PROTOCOL_CHAT),
+ ("opencode/glm-5.2", "glm-5.2", opencode.ZEN_BASE_URL, opencode.PROTOCOL_CHAT),
+ ("opencode/grok-4.6", "grok-4.6", opencode.ZEN_BASE_URL, opencode.PROTOCOL_RESPONSES),
+ (
+ "opencode/gpt-5.6-luna",
+ "gpt-5.6-luna",
+ opencode.ZEN_BASE_URL,
+ opencode.PROTOCOL_RESPONSES,
+ ),
+ (
+ "opencode/claude-opus-5",
+ "claude-opus-5",
+ opencode.ZEN_BASE_URL,
+ opencode.PROTOCOL_MESSAGES,
+ ),
],
)
def test_subscription_model_parses_prefixes(
- model: str, slug: str, base_url: str, uses_responses: bool
+ model: str, slug: str, base_url: str, protocol: str
) -> None:
parsed = opencode.subscription_model(model)
assert parsed is not None
assert parsed.slug == slug
assert parsed.base_url == base_url
- assert parsed.uses_responses == uses_responses
+ assert parsed.protocol == protocol
+ assert parsed.uses_responses is (protocol == opencode.PROTOCOL_RESPONSES)
+
+
+def test_claude_route_targets_the_anthropic_endpoint() -> None:
+ parsed = opencode.subscription_model("opencode/claude-sonnet-5")
+ assert parsed is not None
+ assert parsed.messages_url == "https://opencode.ai/zen/v1/messages"
+
+
+@pytest.mark.parametrize(
+ ("model", "plan", "label", "metered"),
+ [
+ ("opencode/claude-sonnet-5", opencode.PLAN_ZEN, "OpenCode Zen", True),
+ ("opencode/kimi-k3", opencode.PLAN_ZEN, "OpenCode Zen", True),
+ ("opencode-go/kimi-k3", opencode.PLAN_GO, "OpenCode Go", False),
+ ("OpenCode-Go/GPT-5.6-Luna", opencode.PLAN_GO, "OpenCode Go", False),
+ ],
+)
+def test_plan_is_labelled_and_metered_per_prefix(
+ model: str, plan: str, label: str, metered: bool
+) -> None:
+ parsed = opencode.subscription_model(model)
+ assert parsed is not None
+ assert parsed.plan == plan
+ assert parsed.label == label
+ # Zen bills prepaid credits per request; Go is a flat monthly plan.
+ assert parsed.metered is metered
+ assert opencode.subscription_plan(model) == plan
+
+
+def test_subscription_plan_is_none_off_opencode() -> None:
+ assert opencode.subscription_plan("chatgpt/gpt-5.4") is None
+ assert opencode.subscription_plan("anthropic/claude-sonnet-5") is None
+ assert opencode.subscription_plan(None) is None
@pytest.mark.parametrize(