diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml
index b6d816ad3..6bdb56028 100644
--- a/docs/public/api-reference/fabro-api.yaml
+++ b/docs/public/api-reference/fabro-api.yaml
@@ -7590,10 +7590,15 @@ components:
example: 128000
ReasoningEffortFeature:
- description: Whether the model endpoint supports a native reasoning-effort parameter.
+ description: >-
+ Whether the model endpoint supports a native reasoning-effort
+ parameter. `levels` accepts discrete effort levels; `always_adaptive`
+ accepts effort levels with natively always-on adaptive thinking;
+ `none` has no native effort parameter.
type: string
enum:
- levels
+ - always_adaptive
- none
ReasoningEffort:
@@ -7615,6 +7620,7 @@ components:
- reasoning
- reasoning_effort
- prompt_cache
+ - sampling_params
properties:
tools:
type: boolean
@@ -7630,6 +7636,9 @@ components:
prompt_cache:
type: boolean
description: Whether the model endpoint supports prompt caching.
+ sampling_params:
+ type: boolean
+ description: Whether the model accepts classic sampling parameters (temperature, top_p).
ModelCosts:
description: Pricing per million tokens in USD.
diff --git a/docs/public/changelog/2026-06-10.mdx b/docs/public/changelog/2026-06-10.mdx
new file mode 100644
index 000000000..e34908250
--- /dev/null
+++ b/docs/public/changelog/2026-06-10.mdx
@@ -0,0 +1,24 @@
+---
+title: "Claude Fable 5 support"
+date: "2026-06-10"
+---
+
+## Claude Fable 5 support
+
+Fabro now includes Anthropic's `claude-fable-5` in the built-in model catalog, with `fable` and `claude-fable` aliases. It supports the model's 1M context window, 128K max output, vision, tools, prompt caching, and native `reasoning_effort` levels without changing the default Anthropic model from `claude-sonnet-4-6`.
+
+Fable refusals are now handled through Fabro's provider-neutral fallback system. When Anthropic returns `stop_reason: "refusal"`, Fabro surfaces a content-filter LLM error with the refusal detail preserved and retries through `run.model.fallbacks` when configured.
+
+## More
+
+
+- Added `claude-fable-5` to the Anthropic catalog at $10 / $50 per MTok input/output pricing
+- Added aliases `fable` and `claude-fable`
+- Kept `claude-sonnet-4-6` as the Anthropic default
+- Omitted deprecated Anthropic sampling parameters for Fable requests
+
+
+
+- Fable refusal responses now participate in configured model fallback chains
+- Refusals without a configured fallback surface as clear LLM content-filter errors
+
diff --git a/docs/public/core-concepts/models.mdx b/docs/public/core-concepts/models.mdx
index e36fcd7fd..722cf62ca 100644
--- a/docs/public/core-concepts/models.mdx
+++ b/docs/public/core-concepts/models.mdx
@@ -13,6 +13,7 @@ No single model is best at everything. Fabro lets you assign the right model to
| Model | Provider | Aliases | Context | Cost (in/out per Mtok) | Speed |
|---|---|---|---|---|---|
+| `claude-fable-5` | anthropic | `fable`, `claude-fable` | 1M | $10.00 / $50.00 | n/a |
| `claude-opus-4-8` | anthropic | `opus`, `claude-opus` | 1M | $5.00 / $25.00 | 25 tok/s |
| `claude-opus-4-7` | anthropic | | 1M | $5.00 / $25.00 | 25 tok/s |
| `claude-opus-4-6` | anthropic | | 1M | $5.00 / $25.00 | 25 tok/s |
@@ -36,6 +37,8 @@ No single model is best at everything. Fabro lets you assign the right model to
Each provider requires its own API key. Server-backed workflows read provider credentials from the server vault (for example `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, or `GEMINI_API_KEY` set with `fabro secret set` or `fabro provider login`). Standalone SDK/CLI flows can opt into env-backed credential sources explicitly. See the [Quick Start](/getting-started/quick-start) for setup.
+Claude Fable 5 is available as an explicit model but is not the default Anthropic model. If Fable refuses a request, Fabro reports the refusal as a content-filter LLM error and applies the configured `run.model.fallbacks` chain when one is present.
+
## Configuring providers and models
Fabro's catalog starts with the built-in providers and models, then merges any `[llm]` entries from settings. Provider and model IDs are strings, so a server or project can add an OpenAI-compatible provider without a Fabro release.
diff --git a/docs/public/docs.json b/docs/public/docs.json
index c60ddb57b..a4efd3df3 100644
--- a/docs/public/docs.json
+++ b/docs/public/docs.json
@@ -251,6 +251,13 @@
"tab": "Changelog",
"icon": "clock-rotate-left",
"groups": [
+ {
+ "group": "June 2026",
+ "icon": "clock-rotate-left",
+ "pages": [
+ "changelog/2026-06-10"
+ ]
+ },
{
"group": "May 2026",
"icon": "clock-rotate-left",
diff --git a/docs/public/reference/user-configuration.mdx b/docs/public/reference/user-configuration.mdx
index bdad7e3c5..d14c91631 100644
--- a/docs/public/reference/user-configuration.mdx
+++ b/docs/public/reference/user-configuration.mdx
@@ -279,14 +279,15 @@ cache_input_cost_per_mtok = 0.60
| `tools` | boolean | `false` | Whether the model supports tool calls. |
| `vision` | boolean | `false` | Whether the model accepts image inputs. |
| `reasoning` | boolean | `false` | Whether the model has reasoning behavior. |
-| `reasoning_effort` | `"levels"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. |
+| `reasoning_effort` | `"levels"` \| `"always_adaptive"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter. |
| `prompt_cache` | boolean | `false` | Whether prompt cache pricing/usage applies. |
+| `sampling_params` | boolean | `true` | Whether the model accepts classic sampling parameters (`temperature`, `top_p`). |
## `[llm.models..controls]`
| Key | Type / values | Default | Description |
|---|---|---|---|
-| `reasoning_effort` | array | all standard levels when feature is `"levels"` | User-facing reasoning effort values Fabro may send for this model. Can be set explicitly for reasoning models whose provider adapter maps effort to a non-native API shape. |
+| `reasoning_effort` | array | all standard levels when feature is `"levels"` or `"always_adaptive"` | User-facing reasoning effort values Fabro may send for this model. Can be set explicitly for reasoning models whose provider adapter maps effort to a non-native API shape. |
| `speed` | array | `[]` | Additional speeds beyond implicit `standard`; do not list `standard`. |
## `[llm.models..costs]`
diff --git a/docs/superpowers/plans/2026-06-10-catalog-driven-model-behaviors.md b/docs/superpowers/plans/2026-06-10-catalog-driven-model-behaviors.md
new file mode 100644
index 000000000..c99c59000
--- /dev/null
+++ b/docs/superpowers/plans/2026-06-10-catalog-driven-model-behaviors.md
@@ -0,0 +1,622 @@
+# Catalog-Driven Model Behaviors Implementation Plan
+
+> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
+
+**Goal:** Remove all model-specific leakage (`CLAUDE_FABLE_5_MODEL`, `is_fable`, hard-coded `"Claude Fable 5"` strings) from production Rust code in the Anthropic adapter, expressing the underlying model capabilities as catalog TOML data instead.
+
+**Architecture:** Two new catalog mechanisms replace the `is_fable` special cases: (1) a `ReasoningEffortFeature::Adaptive` variant ("effort levels supported; thinking is natively always-on; manual thinking toggle rejected"), (2) a neutral `ModelFeatures.sampling_params: bool` flag ("accepts `temperature`/`top_p`"). The third special case — the `context-1m-2025-08-07` beta header heuristic — is **deleted outright**: per current Anthropic docs (verified 2026-06-10 against platform.claude.com/docs/en/build-with-claude/context-windows), 1M context is GA on Opus 4.6/4.7/4.8 and Sonnet 4.6 with no beta header, and is Fable 5's default. No catalog model needs the header, so no replacement mechanism is built (request-level `provider_options.anthropic.beta_headers` remains as the escape hatch for custom needs). The refusal error message becomes model-derived instead of hard-coding "Claude Fable 5". Tasks 1–2 build pure mechanism (zero behavior change); Task 3 flips the TOML data and adapter code together, with the existing Fable wire tests as the oracle.
+
+**Tech Stack:** Rust (serde, strum, insta snapshots, httpmock wire tests), OpenAPI spec (`docs/public/api-reference/fabro-api.yaml`) regenerated via progenitor (`cargo build -p fabro-api`), TypeScript client via openapi-generator.
+
+**Branch:** work on `feature/claude-fable-5-support` (the Fable PR branch). New commits on top; never amend.
+
+**Decisions confirmed with user (2026-06-10):** variant name `Adaptive`; feature name `sampling_params`; 1M beta header deleted (GA per docs — deliberate wire change: opus requests stop sending `context-1m-2025-08-07`, a no-op server-side); refusal message includes the model ID from the response.
+
+---
+
+## Background for the implementer
+
+The PR `feat(llm): add Claude Fable 5 support` introduced these production-code leaks in `lib/crates/fabro-llm/src/providers/anthropic.rs`:
+
+| Site | Behavior encoded | Replacement |
+|---|---|---|
+| `anthropic.rs:591` `const CLAUDE_FABLE_5_MODEL` | (lookup key) | deleted |
+| `anthropic.rs:1360` `!is_fable` in auto-adaptive thinking injection | thinking is native/always-on | `ReasoningEffortFeature::Adaptive` |
+| `anthropic.rs:1683` `validate_request` rejects manual thinking | same fact as above | `ReasoningEffortFeature::Adaptive` |
+| `anthropic.rs:1379` temperature/top_p forced to `None` | model rejects sampling params | `features.sampling_params = false` |
+| `anthropic.rs:1419,1465` `!is_fable` in `include_1m_context` | 1M is GA, not beta opt-in | heuristic + `CONTEXT_1M_BETA_HEADER` deleted (1M is GA on every catalog model that has it) |
+| `anthropic.rs` `refusal_error()` message `"Claude Fable 5 refused..."` | (nothing — refusals are wire-protocol-generic) | message uses the response's model ID |
+
+Key constraint: `Model` and `ModelFeatures` are reused by `fabro-api` via `with_replacement` (`lib/crates/fabro-api/build.rs`) and appear in the OpenAPI spec, with JSON-parity tests in `lib/crates/fabro-api/tests/model_features_round_trip.rs`. Any serialized-shape change to those types MUST update the spec in the same task.
+
+---
+
+### Task 1: Add `ReasoningEffortFeature::Adaptive`
+
+**Files:**
+- Modify: `lib/crates/fabro-model/src/types.rs` (enum at :22, `ModelFeatures` at :35, `Model::supports_reasoning_effort` at :117)
+- Modify: `lib/crates/fabro-model/src/catalog.rs:1430` (validation), `:1499-1500` (controls)
+- Modify: `docs/public/api-reference/fabro-api.yaml:7592` (enum schema)
+- Test: `lib/crates/fabro-model/src/types.rs` tests, `lib/crates/fabro-model/src/catalog.rs` tests
+
+- [ ] **Step 1: Write the failing tests**
+
+In `lib/crates/fabro-model/src/types.rs` test module:
+
+```rust
+#[test]
+fn reasoning_effort_feature_adaptive_round_trips() {
+ let parsed: ReasoningEffortFeature =
+ serde_json::from_value(serde_json::json!("adaptive")).unwrap();
+ assert_eq!(parsed, ReasoningEffortFeature::Adaptive);
+ assert_eq!(
+ serde_json::to_value(parsed).unwrap(),
+ serde_json::json!("adaptive")
+ );
+ assert_eq!(parsed.to_string(), "adaptive");
+ assert_eq!("adaptive".parse::().unwrap(), parsed);
+}
+```
+
+(Check `fabro-model/Cargo.toml` for a `serde_json` dev-dependency first; mirror existing test patterns if a different parse path is conventional.)
+
+In `lib/crates/fabro-model/src/catalog.rs` test module (pattern: copy `catalog_from_settings_accepts_reasoning_effort_feature_levels` at :3404):
+
+```rust
+#[test]
+fn catalog_from_settings_accepts_reasoning_effort_feature_adaptive() {
+ let settings = minimal_settings(
+ r#"
+[providers.test]
+display_name = "Test"
+adapter = "openai"
+agent_profile = "openai"
+
+[models.model]
+provider = "test"
+display_name = "Model"
+family = "test"
+default = true
+
+[models.model.limits]
+context_window = 1000
+
+[models.model.features]
+tools = true
+vision = false
+reasoning = true
+reasoning_effort = "adaptive"
+prompt_cache = true
+"#,
+ );
+
+ let catalog = Catalog::from_settings(&settings).unwrap();
+ let model = catalog.get("model").unwrap();
+ assert_eq!(
+ model.features.reasoning_effort,
+ crate::ReasoningEffortFeature::Adaptive
+ );
+ assert!(model.supports_reasoning_effort());
+ // Adaptive models get the full default effort controls, same as Levels.
+ assert_eq!(
+ catalog
+ .model_settings("model")
+ .unwrap()
+ .controls
+ .reasoning_effort,
+ ReasoningEffort::VARIANTS.to_vec()
+ );
+}
+
+#[test]
+fn catalog_from_settings_rejects_adaptive_effort_without_reasoning() {
+ let settings = minimal_settings(
+ r#"
+[providers.test]
+display_name = "Test"
+adapter = "openai"
+agent_profile = "openai"
+
+[models.model]
+provider = "test"
+display_name = "Model"
+family = "test"
+default = true
+
+[models.model.limits]
+context_window = 1000
+
+[models.model.features]
+tools = true
+vision = false
+reasoning = false
+reasoning_effort = "adaptive"
+"#,
+ );
+
+ assert!(matches!(
+ Catalog::from_settings(&settings),
+ Err(CatalogBuildError::ReasoningEffortWithoutReasoning { .. })
+ ));
+}
+```
+
+- [ ] **Step 2: Run tests to verify they fail**
+
+Run: `cargo nextest run -p fabro-model`
+Expected: FAIL — `adaptive` does not parse (unknown variant).
+
+- [ ] **Step 3: Implement**
+
+`lib/crates/fabro-model/src/types.rs` — add variant (keep `None` as `#[default]`):
+
+```rust
+pub enum ReasoningEffortFeature {
+ Levels,
+ /// Effort levels are supported and thinking is natively always-on /
+ /// adaptive; a manual thinking on/off toggle is not accepted.
+ Adaptive,
+ #[default]
+ None,
+}
+```
+
+Add a method on `ModelFeatures` (single source of truth for "has a native effort param"):
+
+```rust
+impl ModelFeatures {
+ /// Whether the model endpoint accepts a native reasoning-effort level.
+ #[must_use]
+ pub fn supports_reasoning_effort(&self) -> bool {
+ matches!(
+ self.reasoning_effort,
+ ReasoningEffortFeature::Levels | ReasoningEffortFeature::Adaptive
+ )
+ }
+}
+```
+
+Change `Model::supports_reasoning_effort` (types.rs:116-118) to delegate:
+
+```rust
+pub fn supports_reasoning_effort(&self) -> bool {
+ self.features.supports_reasoning_effort()
+}
+```
+
+`lib/crates/fabro-model/src/catalog.rs:1430` — any effort feature requires `reasoning`:
+
+```rust
+if !reasoning && reasoning_effort != ReasoningEffortFeature::None {
+```
+
+`lib/crates/fabro-model/src/catalog.rs:1499-1500`:
+
+```rust
+let supports_native_reasoning_effort = features.supports_reasoning_effort();
+```
+
+`docs/public/api-reference/fabro-api.yaml:7592` — update the enum schema:
+
+```yaml
+ ReasoningEffortFeature:
+ description: >-
+ Whether the model endpoint supports a native reasoning-effort
+ parameter. `levels` accepts discrete effort levels; `adaptive`
+ accepts effort levels with natively always-on adaptive thinking;
+ `none` has no native effort parameter.
+ type: string
+ enum:
+ - levels
+ - adaptive
+ - none
+```
+
+- [ ] **Step 4: Run tests to verify they pass**
+
+Run: `cargo nextest run -p fabro-model && cargo build -p fabro-api && cargo nextest run -p fabro-api`
+Expected: PASS (fabro-api build regenerates types from the spec; parity tests still pass since the enum reuses the canonical Rust type).
+
+- [ ] **Step 5: Verify zero behavior change**
+
+Run: `cargo nextest run -p fabro-llm -p fabro-config`
+Expected: PASS — nothing references `Adaptive` yet; no TOML uses it.
+
+- [ ] **Step 6: Commit**
+
+```bash
+git add lib/crates/fabro-model docs/public/api-reference/fabro-api.yaml lib/crates/fabro-api
+git commit -m "feat(model): add adaptive reasoning-effort feature variant"
+```
+
+---
+
+### Task 2: Add `ModelFeatures.sampling_params`
+
+**Files:**
+- Modify: `lib/crates/fabro-model/src/types.rs` (`ModelFeatures` :35, test fixture :179)
+- Modify: `lib/crates/fabro-model/src/catalog.rs` (`SettingsModelFeatures` :115, `merge_model_features_settings` :1182, `build_model_features` :1419)
+- Modify: `lib/crates/fabro-config/src/layers/llm.rs` (layer `ModelFeatures`, ~:152)
+- Modify: `lib/crates/fabro-config/src/builders.rs:393` (`model_features_to_catalog`)
+- Modify: `docs/public/api-reference/fabro-api.yaml` (`ModelFeatures` schema, ~:7609)
+- Modify: `lib/crates/fabro-api/tests/model_features_round_trip.rs`
+- Modify (test fixtures that construct `ModelFeatures` literally — run `rg -n 'ModelFeatures \{' lib/crates` to confirm the full list): `lib/crates/fabro-cli/src/commands/model.rs:495,528`, `lib/crates/fabro-llm/src/model_test.rs:227`
+
+- [ ] **Step 1: Write the failing test**
+
+In `lib/crates/fabro-model/src/catalog.rs` test module:
+
+```rust
+#[test]
+fn catalog_from_settings_sampling_params_defaults_true_and_accepts_false() {
+ let settings = minimal_settings(
+ r#"
+[providers.test]
+display_name = "Test"
+adapter = "openai"
+agent_profile = "openai"
+
+[models.with-sampling]
+provider = "test"
+display_name = "With"
+family = "test"
+default = true
+
+[models.with-sampling.limits]
+context_window = 1000
+
+[models.with-sampling.features]
+tools = true
+vision = false
+reasoning = false
+
+[models.no-sampling]
+provider = "test"
+display_name = "Without"
+family = "test"
+
+[models.no-sampling.limits]
+context_window = 1000
+
+[models.no-sampling.features]
+tools = true
+vision = false
+reasoning = false
+sampling_params = false
+"#,
+ );
+
+ let catalog = Catalog::from_settings(&settings).unwrap();
+ assert!(catalog.get("with-sampling").unwrap().features.sampling_params);
+ assert!(!catalog.get("no-sampling").unwrap().features.sampling_params);
+}
+```
+
+- [ ] **Step 2: Run test to verify it fails**
+
+Run: `cargo nextest run -p fabro-model catalog_from_settings_sampling_params`
+Expected: FAIL to compile (no `sampling_params` field) — compile error is the failing state.
+
+- [ ] **Step 3: Implement**
+
+`lib/crates/fabro-model/src/types.rs` — add field to `ModelFeatures` (last position) and a serde default helper:
+
+```rust
+fn default_true() -> bool {
+ true
+}
+
+pub struct ModelFeatures {
+ // ... existing fields unchanged ...
+ /// Whether the model endpoint accepts classic sampling parameters
+ /// (`temperature`, `top_p`). Models with always-on adaptive behavior
+ /// reject them.
+ #[serde(default = "default_true")]
+ pub sampling_params: bool,
+}
+```
+
+Add a `Model` accessor next to `supports_prompt_cache` (types.rs:120):
+
+```rust
+pub fn supports_sampling_params(&self) -> bool {
+ self.features.sampling_params
+}
+```
+
+`lib/crates/fabro-model/src/catalog.rs`:
+
+```rust
+// SettingsModelFeatures (:115) — add:
+ #[serde(default)]
+ pub sampling_params: Option,
+
+// merge_model_features_settings (:1186) — add:
+ sampling_params: higher.sampling_params.or(fallback.sampling_params),
+
+// build_model_features Ok(ModelFeatures { ... }) (:1436) — add:
+ sampling_params: features.sampling_params.unwrap_or(true),
+```
+
+`lib/crates/fabro-config/src/layers/llm.rs` — add to the layer `ModelFeatures` struct:
+
+```rust
+ #[serde(default, skip_serializing_if = "Option::is_none")]
+ pub sampling_params: Option,
+```
+
+(`Option` already has a `Combine` impl via `impl_combine_or_option!` in `layers/combine.rs` — no macro change needed.)
+
+`lib/crates/fabro-config/src/builders.rs:393` `model_features_to_catalog` — add:
+
+```rust
+ sampling_params: features.sampling_params,
+```
+
+`docs/public/api-reference/fabro-api.yaml` `ModelFeatures` schema — add to `required` and `properties`:
+
+```yaml
+ required:
+ - tools
+ - vision
+ - reasoning
+ - reasoning_effort
+ - prompt_cache
+ - sampling_params
+ properties:
+ # ... existing ...
+ sampling_params:
+ type: boolean
+ description: Whether the model accepts classic sampling parameters (temperature, top_p).
+```
+
+`lib/crates/fabro-api/tests/model_features_round_trip.rs` — add `sampling_params: true` to the fixture and `assert_eq!(json["sampling_params"], true);`.
+
+Fix all remaining `ModelFeatures { ... }` struct literals the compiler flags (test fixtures in `fabro-cli/src/commands/model.rs`, `fabro-llm/src/model_test.rs`, `fabro-model/src/types.rs` test) by adding `sampling_params: true`.
+
+- [ ] **Step 4: Run tests to verify they pass**
+
+Run: `cargo build --workspace && cargo nextest run -p fabro-model -p fabro-config -p fabro-api -p fabro-cli -p fabro-llm`
+Expected: PASS. (fabro-api regenerates from the spec during build; parity test passes with the new field.)
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add lib/crates docs/public/api-reference/fabro-api.yaml
+git commit -m "feat(model): add sampling_params model feature flag"
+```
+
+---
+
+### Task 3: Refactor the Anthropic adapter onto catalog data
+
+This is the behavior task: the TOML flip and the adapter changes must land together (e.g. once Fable's TOML says `adaptive`, the old `supports_effort` check `== Levels` would wrongly convert effort to `budget_tokens` — so the existing Fable wire tests only pass with both halves in place).
+
+**Deliberate wire change:** Opus requests stop sending the `context-1m-2025-08-07` beta header. Per Anthropic's context-windows doc (verified 2026-06-10), 1M context is GA on Opus 4.6/4.7/4.8 (and Sonnet 4.6) on the Claude API — the header is a no-op. Users who genuinely need a beta header on a custom setup can still pass request-level `provider_options.anthropic.beta_headers`.
+
+**Files:**
+- Modify: `lib/crates/fabro-model/src/catalog/providers/anthropic.toml` (Fable features only)
+- Modify: `lib/crates/fabro-llm/src/providers/anthropic.rs`
+- Modify: `lib/crates/fabro-llm/tests/it/wire/anthropic.rs` (one new test)
+- Modify: `lib/crates/fabro-workflow/src/handler/llm/api.rs` (test fixture message only)
+
+- [ ] **Step 1: Write the failing wire test (opus no longer gets the 1M beta header)**
+
+In `lib/crates/fabro-llm/tests/it/wire/anthropic.rs`, next to `encode_fable_uses_api_id_effort_and_omits_1m_beta`:
+
+```rust
+#[tokio::test]
+async fn encode_opus_omits_1m_beta_header() {
+ let capture = encode_capture(
+ adapter().with_catalog(builtin_catalog()),
+ &base_request("claude-opus-4-8"),
+ )
+ .await;
+
+ assert!(
+ !header_value(&capture, "anthropic-beta")
+ .unwrap_or("")
+ .contains("context-1m-2025-08-07"),
+ "1M context is GA on opus; the legacy beta opt-in must not be sent"
+ );
+}
+```
+
+- [ ] **Step 2: Run the new test to verify it fails**
+
+Run: `cargo nextest run -p fabro-llm encode_opus_omits_1m_beta_header`
+Expected: FAIL — the current heuristic still adds the header for 1M-context catalog models.
+
+- [ ] **Step 3: Flip the TOML data**
+
+`lib/crates/fabro-model/src/catalog/providers/anthropic.toml` — Fable features block (lines 23-28) becomes:
+
+```toml
+[models."claude-fable-5".features]
+tools = true
+vision = true
+reasoning = true
+reasoning_effort = "adaptive"
+prompt_cache = true
+sampling_params = false
+```
+
+No changes to the opus entries.
+
+- [ ] **Step 4: Refactor `lib/crates/fabro-llm/src/providers/anthropic.rs`**
+
+Delete the model constant (line 591):
+
+```rust
+// DELETE: const CLAUDE_FABLE_5_MODEL: &str = "claude-fable-5";
+```
+
+Delete the 1M beta constant and heuristic:
+- Remove `const CONTEXT_1M_BETA_HEADER` (line 688).
+- Remove the `include_1m_context: bool` parameter from `build_beta_header` (line 690) and the block at :723 that appends the header.
+- At both call sites (`build_api_request` :1419-1426 and `count_input_tokens` :1465-1481): delete the `include_1m_context` computation and the argument.
+- In the test module, drop the final `false` argument from every `build_beta_header(...)` call (lines 2040, 2046, 2057, 2073, 2085, 2631, 2645, 3025).
+
+In `build_api_request` (around :1289):
+
+```rust
+ let model_info = common::catalog_model(adapter.catalog.as_deref(), &request.model);
+ let api_model = common::api_model_id(adapter.catalog.as_deref(), &request.model);
+ // DELETE: let is_fable = api_model == CLAUDE_FABLE_5_MODEL;
+```
+
+`supports_effort` (:1327) — Adaptive also takes the effort parameter:
+
+```rust
+ let supports_effort = model_info.is_none_or(Model::supports_reasoning_effort);
+```
+
+(Import: change `use fabro_model::{Catalog, ReasoningEffortFeature};` to also bring in `Model`.)
+
+Auto-adaptive thinking injection (:1359) — `Levels` only; `Adaptive` models are natively adaptive and must not receive a `thinking` param:
+
+```rust
+ let thinking = explicit_thinking.or_else(|| {
+ if model_info
+ .is_some_and(|m| m.features.reasoning_effort == ReasoningEffortFeature::Levels)
+ {
+ Some(serde_json::json!({"type": "adaptive"}))
+ } else {
+ None
+ }
+ });
+```
+
+Sampling gate (:1378):
+
+```rust
+ // Models with always-on adaptive behavior reject classic sampling knobs.
+ let (temperature, top_p) = if model_info.is_none_or(|m| m.features.sampling_params) {
+ (request.temperature, request.top_p)
+ } else {
+ (None, None)
+ };
+```
+
+`validate_request` (:1676):
+
+```rust
+ fn validate_request(&self, request: &Request) -> Result<(), Error> {
+ if let Some(tool_choice) = &request.tool_choice {
+ crate::provider::validate_tool_choice(self, tool_choice)?;
+ }
+
+ let model_info = common::catalog_model(self.catalog.as_deref(), &request.model);
+ if let Some(model) = model_info
+ .filter(|m| m.features.reasoning_effort == ReasoningEffortFeature::Adaptive)
+ {
+ if let Some(kind @ ("enabled" | "disabled")) =
+ anthropic_thinking_type(request.provider_options.as_ref())
+ {
+ return Err(Error::Configuration {
+ message: format!(
+ "{} uses always-on adaptive thinking; provider_options.anthropic.thinking.type = \"{kind}\" is not supported. Omit thinking or set only display options.",
+ model.display_name()
+ ),
+ source: None,
+ });
+ }
+ }
+
+ Ok(())
+ }
+}
+```
+
+`refusal_error` — take the model ID and use it in the message (the response model in `complete`, the accumulator model in streaming):
+
+```rust
+fn refusal_error(
+ provider_name: &str,
+ model: &str,
+ raw: serde_json::Value,
+ stop_details: Option<&serde_json::Value>,
+) -> Error {
+ let model_label = if model.is_empty() { "model" } else { model };
+ let message = stop_details
+ .and_then(|details| details.get("explanation"))
+ .and_then(serde_json::Value::as_str)
+ .map_or_else(
+ || format!("{model_label} refused the request"),
+ |explanation| format!("{model_label} refused the request: {explanation}"),
+ );
+ // ... Error::Provider construction unchanged ...
+}
+```
+
+Call sites: in `complete()` pass `&api_resp.model`; in `process_sse_event_for_provider` pass `&acc.model` (populated by `message_start`, empty-string fallback handled above). `process_sse_event_for_provider` already receives `acc`.
+
+- [ ] **Step 5: Update the workflow test fixture message**
+
+`lib/crates/fabro-workflow/src/handler/llm/api.rs` (test module): in `refusal_llm_error()` change `message: "Claude Fable 5 refused the request".into()` to `message: "claude-fable-5 refused the request".into()`. In `classify_refusal_llm_returns_terminal_when_not_allowed` change `assert!(llm_err.to_string().contains("Claude Fable 5 refused"))` to `assert!(llm_err.to_string().contains("refused the request"))`.
+
+Also update the wire test `decode_refusal_returns_failover_eligible_content_filter_error` in `lib/crates/fabro-llm/tests/it/wire/anthropic.rs` if needed — its `detail.message.contains("declined")` assertion still passes (the explanation is embedded); optionally strengthen with `detail.message.contains("claude-fable-5")` since the canned body's `model` is `claude-fable-5`.
+
+- [ ] **Step 6: Run the full affected test suites**
+
+Run: `cargo nextest run -p fabro-llm -p fabro-model -p fabro-workflow`
+Expected: PASS, including all pre-existing Fable wire tests unchanged (`encode_fable_uses_api_id_effort_and_omits_1m_beta`, `encode_fable_without_effort_omits_default_thinking`, `fable_rejects_manual_enabled_or_disabled_thinking` — its `contains("Claude Fable 5")` assertion still passes because the message now uses the catalog `display_name`, which is "Claude Fable 5" — and the refusal decode/stream tests) plus the new opus test.
+
+- [ ] **Step 7: Verify no snapshot drift and no remaining leaks**
+
+Run: `cargo insta pending-snapshots`
+Expected: empty output (zero snapshot changes).
+
+Run: `rg -n "is_fable|CLAUDE_FABLE|claude-fable|Claude Fable|context-1m" lib/crates/fabro-llm/src lib/crates/fabro-workflow/src --type rust` — confirm every remaining hit is inside a `#[cfg(test)]` module or gone. Expected: no production-code hits. (`"Claude Fable 5"` remains only as catalog TOML data.)
+
+- [ ] **Step 8: Commit**
+
+```bash
+git add lib/crates/fabro-model/src/catalog/providers/anthropic.toml lib/crates/fabro-llm lib/crates/fabro-workflow
+git commit -m "refactor(llm): drive Fable request behaviors from catalog data"
+```
+
+---
+
+### Task 4: Regenerate TypeScript client; full verification
+
+**Files:**
+- Regenerate: `lib/packages/fabro-api-client` (generated)
+- Possibly modify: `apps/fabro-web` (only if typecheck flags exhaustive switches on `ReasoningEffortFeature`)
+
+- [ ] **Step 1: Regenerate the TS client**
+
+Run: `cd lib/packages/fabro-api-client && bun run generate`
+Expected: regenerated types include `"adaptive"` in the reasoning-effort-feature union and `sampling_params` on model features.
+
+- [ ] **Step 2: Web typecheck and tests**
+
+Run: `cd apps/fabro-web && bun run typecheck && bun test`
+Expected: PASS. If an exhaustive switch over the feature enum breaks, handle `"adaptive"` the same way as `"levels"`.
+
+- [ ] **Step 3: Full workspace verification**
+
+Run:
+```bash
+cargo build --workspace
+cargo nextest run --workspace
+cargo +nightly-2026-04-14 fmt --all
+cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings
+```
+Expected: all green. If fmt changes files, include them in the commit.
+
+- [ ] **Step 4: Commit**
+
+```bash
+git add lib/packages/fabro-api-client apps/fabro-web
+git commit -m "chore(api): regenerate TS client for adaptive effort and sampling_params"
+```
+
+---
+
+## Self-review notes
+
+- Tasks 1–2 are pure mechanism: no TOML uses the new fields until Task 3, so behavior and snapshots are provably unchanged at each commit.
+- Task 3's oracle: the existing Fable wire tests pass unchanged, the new opus test pins the deliberate header removal, and `cargo insta pending-snapshots` is empty.
+- The original plan had a per-model `provider_options.beta_headers` catalog mechanism; it was cut after verifying no catalog model needs the 1M beta header (GA per Anthropic docs). If a future model needs a per-model wire opt-in, resurrect that design from git history of this plan.
+- The `1c8fe39ab` changelog/docs files (`docs/public/changelog/2026-06-10.mdx`, `models.mdx`) describe user-visible behavior only and need no edits for this refactor.
diff --git a/lib/crates/fabro-agent/src/cli.rs b/lib/crates/fabro-agent/src/cli.rs
index fcbc6016d..ee1f05dc0 100644
--- a/lib/crates/fabro-agent/src/cli.rs
+++ b/lib/crates/fabro-agent/src/cli.rs
@@ -1073,6 +1073,7 @@ mod tests {
reasoning: Some(false),
reasoning_effort: None,
prompt_cache: None,
+ sampling_params: None,
}),
..ModelCatalogSettings::default()
});
@@ -1157,6 +1158,7 @@ mod tests {
reasoning: Some(false),
reasoning_effort: None,
prompt_cache: None,
+ sampling_params: None,
}),
..ModelCatalogSettings::default()
});
diff --git a/lib/crates/fabro-api/tests/model_features_round_trip.rs b/lib/crates/fabro-api/tests/model_features_round_trip.rs
index 033da8269..8e6f60aef 100644
--- a/lib/crates/fabro-api/tests/model_features_round_trip.rs
+++ b/lib/crates/fabro-api/tests/model_features_round_trip.rs
@@ -16,6 +16,7 @@ fn model_features_json_matches_openapi_shape() {
reasoning: true,
reasoning_effort: ReasoningEffortFeature::Levels,
prompt_cache: false,
+ sampling_params: true,
};
let json = serde_json::to_value(&features).unwrap();
@@ -24,6 +25,7 @@ fn model_features_json_matches_openapi_shape() {
assert_eq!(json["reasoning"], true);
assert_eq!(json["reasoning_effort"], "levels");
assert_eq!(json["prompt_cache"], false);
+ assert_eq!(json["sampling_params"], true);
let round_trip: ApiModelFeatures = serde_json::from_value(json).unwrap();
assert_eq!(round_trip, features);
diff --git a/lib/crates/fabro-api/tests/model_round_trip.rs b/lib/crates/fabro-api/tests/model_round_trip.rs
index ad896d4c4..727cd41da 100644
--- a/lib/crates/fabro-api/tests/model_round_trip.rs
+++ b/lib/crates/fabro-api/tests/model_round_trip.rs
@@ -29,6 +29,7 @@ fn model_json_matches_openapi_shape() {
reasoning: true,
reasoning_effort: ReasoningEffortFeature::Levels,
prompt_cache: true,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(5.0),
diff --git a/lib/crates/fabro-api/tests/provider_id_round_trip.rs b/lib/crates/fabro-api/tests/provider_id_round_trip.rs
index 21c5227c6..47a1a2ce3 100644
--- a/lib/crates/fabro-api/tests/provider_id_round_trip.rs
+++ b/lib/crates/fabro-api/tests/provider_id_round_trip.rs
@@ -39,6 +39,7 @@ fn provider_id_json_matches_openapi_shape_through_model() {
reasoning: false,
reasoning_effort: ReasoningEffortFeature::None,
prompt_cache: false,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: None,
diff --git a/lib/crates/fabro-cli/src/commands/model.rs b/lib/crates/fabro-cli/src/commands/model.rs
index 706cbbf84..e63b926f4 100644
--- a/lib/crates/fabro-cli/src/commands/model.rs
+++ b/lib/crates/fabro-cli/src/commands/model.rs
@@ -494,6 +494,7 @@ mod tests {
reasoning: false,
reasoning_effort: ReasoningEffortFeature::None,
prompt_cache: false,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(1.0),
@@ -527,6 +528,7 @@ mod tests {
reasoning: false,
reasoning_effort: ReasoningEffortFeature::None,
prompt_cache: false,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(1.0),
diff --git a/lib/crates/fabro-config/src/builders.rs b/lib/crates/fabro-config/src/builders.rs
index 38c022725..63672fc6f 100644
--- a/lib/crates/fabro-config/src/builders.rs
+++ b/lib/crates/fabro-config/src/builders.rs
@@ -397,6 +397,7 @@ fn model_features_to_catalog(features: &LlmModelFeatures) -> model_catalog::Sett
reasoning: features.reasoning,
reasoning_effort: features.reasoning_effort,
prompt_cache: features.prompt_cache,
+ sampling_params: features.sampling_params,
}
}
diff --git a/lib/crates/fabro-config/src/layers/llm.rs b/lib/crates/fabro-config/src/layers/llm.rs
index 9afb41d9e..d8037105f 100644
--- a/lib/crates/fabro-config/src/layers/llm.rs
+++ b/lib/crates/fabro-config/src/layers/llm.rs
@@ -161,6 +161,8 @@ pub struct ModelFeatures {
pub reasoning_effort: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub prompt_cache: Option,
+ #[serde(default, skip_serializing_if = "Option::is_none")]
+ pub sampling_params: Option,
}
/// User-facing allow-list for native control values Fabro accepts on this
diff --git a/lib/crates/fabro-dev/src/commands/docs_options_reference.rs b/lib/crates/fabro-dev/src/commands/docs_options_reference.rs
index 7507e431f..17a608288 100644
--- a/lib/crates/fabro-dev/src/commands/docs_options_reference.rs
+++ b/lib/crates/fabro-dev/src/commands/docs_options_reference.rs
@@ -325,14 +325,15 @@ cache_input_cost_per_mtok = 0.60
| `tools` | boolean | `false` | Whether the model supports tool calls. |
| `vision` | boolean | `false` | Whether the model accepts image inputs. |
| `reasoning` | boolean | `false` | Whether the model has reasoning behavior. |
-| `reasoning_effort` | `"levels"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. |
+| `reasoning_effort` | `"levels"` \| `"always_adaptive"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter. |
| `prompt_cache` | boolean | `false` | Whether prompt cache pricing/usage applies. |
+| `sampling_params` | boolean | `true` | Whether the model accepts classic sampling parameters (`temperature`, `top_p`). |
## `[llm.models..controls]`
| Key | Type / values | Default | Description |
|---|---|---|---|
-| `reasoning_effort` | array | all standard levels when feature is `"levels"` | User-facing reasoning effort values Fabro may send for this model. Can be set explicitly for reasoning models whose provider adapter maps effort to a non-native API shape. |
+| `reasoning_effort` | array | all standard levels when feature is `"levels"` or `"always_adaptive"` | User-facing reasoning effort values Fabro may send for this model. Can be set explicitly for reasoning models whose provider adapter maps effort to a non-native API shape. |
| `speed` | array | `[]` | Additional speeds beyond implicit `standard`; do not list `standard`. |
## `[llm.models..costs]`
diff --git a/lib/crates/fabro-llm/src/error.rs b/lib/crates/fabro-llm/src/error.rs
index fd21dd49d..fd2987047 100644
--- a/lib/crates/fabro-llm/src/error.rs
+++ b/lib/crates/fabro-llm/src/error.rs
@@ -209,7 +209,7 @@ impl Error {
/// same-provider retry) plus `QuotaExceeded` — a different provider won't
/// share the same quota.
#[must_use]
- pub const fn failover_eligible(&self) -> bool {
+ pub fn failover_eligible(&self) -> bool {
if self.retryable() {
return true;
}
@@ -219,6 +219,16 @@ impl Error {
kind: ProviderErrorKind::QuotaExceeded,
..
} | Self::RequestTimeout { .. }
+ ) || self.refusal_content_filter()
+ }
+
+ fn refusal_content_filter(&self) -> bool {
+ matches!(
+ self,
+ Self::Provider {
+ kind: ProviderErrorKind::ContentFilter,
+ detail,
+ } if detail.error_code.as_deref() == Some("refusal")
)
}
@@ -926,6 +936,35 @@ mod tests {
);
}
+ #[test]
+ fn failover_eligible_for_refusal_content_filter_only() {
+ assert!(
+ Error::Provider {
+ kind: ProviderErrorKind::ContentFilter,
+ detail: Box::new(ProviderErrorDetail {
+ error_code: Some("refusal".to_string()),
+ raw: Some(serde_json::json!({
+ "stop_reason": "refusal",
+ "stop_details": {"type": "refusal", "category": "cyber"}
+ })),
+ ..ProviderErrorDetail::new("declined", "anthropic")
+ }),
+ }
+ .failover_eligible()
+ );
+
+ assert!(
+ !Error::Provider {
+ kind: ProviderErrorKind::ContentFilter,
+ detail: Box::new(ProviderErrorDetail {
+ error_code: Some("safety".to_string()),
+ ..ProviderErrorDetail::new("blocked", "anthropic")
+ }),
+ }
+ .failover_eligible()
+ );
+ }
+
#[test]
fn failover_not_eligible_non_provider_errors() {
assert!(
diff --git a/lib/crates/fabro-llm/src/model_test.rs b/lib/crates/fabro-llm/src/model_test.rs
index e177901a2..14098f173 100644
--- a/lib/crates/fabro-llm/src/model_test.rs
+++ b/lib/crates/fabro-llm/src/model_test.rs
@@ -226,6 +226,7 @@ mod tests {
reasoning: true,
reasoning_effort: ReasoningEffortFeature::Levels,
prompt_cache: false,
+ sampling_params: true,
});
let outcome = run_model_test(&info, ModelTestMode::Deep, empty_test_client()).await;
diff --git a/lib/crates/fabro-llm/src/providers/anthropic.rs b/lib/crates/fabro-llm/src/providers/anthropic.rs
index 3674eef99..9191fb0ec 100644
--- a/lib/crates/fabro-llm/src/providers/anthropic.rs
+++ b/lib/crates/fabro-llm/src/providers/anthropic.rs
@@ -2,11 +2,11 @@ use std::sync::Arc;
use base64::Engine;
use base64::engine::general_purpose::STANDARD as BASE64_STANDARD;
-use fabro_model::{Catalog, ReasoningEffortFeature};
+use fabro_model::{Catalog, Model, ReasoningEffortFeature};
use futures::stream;
use crate::error::{Error, ProviderErrorDetail, ProviderErrorKind, error_from_status_code};
-use crate::provider::{ProviderAdapter, StreamEventStream};
+use crate::provider::{ProviderAdapter, StreamEventStream, validate_tool_choice};
use crate::providers::common::{
self as common, extract_system_prompt, parse_error_body, parse_rate_limit_headers,
parse_retry_after, send_and_read_response,
@@ -205,11 +205,13 @@ impl CacheControl {
#[derive(serde::Deserialize)]
struct ApiResponse {
- id: String,
- model: String,
- content: Vec,
- stop_reason: Option,
- usage: ApiUsage,
+ id: String,
+ model: String,
+ content: Vec,
+ stop_reason: Option,
+ #[serde(default)]
+ stop_details: Option,
+ usage: ApiUsage,
}
#[derive(serde::Deserialize)]
@@ -625,6 +627,14 @@ fn is_auto_cache_enabled(provider_options: Option<&serde_json::Value>) -> bool {
.unwrap_or(true)
}
+fn anthropic_thinking_type(provider_options: Option<&serde_json::Value>) -> Option<&str> {
+ provider_options
+ .and_then(|opts| opts.get("anthropic"))
+ .and_then(|anthropic| anthropic.get("thinking"))
+ .and_then(|thinking| thinking.get("type"))
+ .and_then(serde_json::Value::as_str)
+}
+
/// Wrap a system prompt string as an array of content blocks with
/// `cache_control` on the last block.
fn system_with_cache_control(system: &str) -> serde_json::Value {
@@ -674,13 +684,10 @@ fn apply_cache_control_to_conversation_prefix(messages: &mut [ApiMessage]) {
/// Collect beta headers from `provider_options` and merge with the caching
/// header when auto-caching is active.
-const CONTEXT_1M_BETA_HEADER: &str = "context-1m-2025-08-07";
-
fn build_beta_header(
provider_options: Option<&serde_json::Value>,
include_cache_header: bool,
include_fast_mode_header: bool,
- include_1m_context: bool,
) -> Option {
let mut headers: Vec = Vec::new();
@@ -708,11 +715,6 @@ fn build_beta_header(
headers.push(FAST_MODE_BETA_HEADER.to_string());
}
- // Add 1M context header for models with >= 1M context window
- if include_1m_context && !headers.iter().any(|h| h == CONTEXT_1M_BETA_HEADER) {
- headers.push(CONTEXT_1M_BETA_HEADER.to_string());
- }
-
if headers.is_empty() {
None
} else {
@@ -1043,11 +1045,67 @@ fn process_sse_event_for_provider(
) -> Result, Error> {
if event_type == "error" {
Err(stream_error_event_to_provider_error(data, provider_name))
+ } else if event_type == "message_delta"
+ && data
+ .get("delta")
+ .and_then(|delta| delta.get("stop_reason"))
+ .and_then(serde_json::Value::as_str)
+ == Some("refusal")
+ {
+ let stop_details = data
+ .get("delta")
+ .and_then(|delta| delta.get("stop_details"));
+ Err(refusal_error(
+ provider_name,
+ &acc.model,
+ refusal_stream_raw(data),
+ stop_details,
+ ))
} else {
Ok(process_sse_event(event_type, data, acc))
}
}
+fn refusal_stream_raw(data: &serde_json::Value) -> serde_json::Value {
+ serde_json::json!({
+ "stop_reason": "refusal",
+ "stop_details": data
+ .get("delta")
+ .and_then(|delta| delta.get("stop_details"))
+ .cloned()
+ .unwrap_or(serde_json::Value::Null),
+ "stream_event": data,
+ })
+}
+
+fn refusal_error(
+ provider_name: &str,
+ model: &str,
+ raw: serde_json::Value,
+ stop_details: Option<&serde_json::Value>,
+) -> Error {
+ let model_label = if model.is_empty() { "The model" } else { model };
+ let message = stop_details
+ .and_then(|details| details.get("explanation"))
+ .and_then(serde_json::Value::as_str)
+ .map_or_else(
+ || format!("{model_label} refused the request"),
+ |explanation| format!("{model_label} refused the request: {explanation}"),
+ );
+
+ Error::Provider {
+ kind: ProviderErrorKind::ContentFilter,
+ detail: Box::new(ProviderErrorDetail {
+ message,
+ provider: provider_name.to_string(),
+ status_code: None,
+ error_code: Some("refusal".to_string()),
+ retry_after: None,
+ raw: Some(raw),
+ }),
+ }
+}
+
fn stream_error_event_to_provider_error(data: &serde_json::Value, provider_name: &str) -> Error {
let error = data.get("error").unwrap_or(data);
let message = error
@@ -1223,6 +1281,7 @@ async fn build_api_request(
};
let model_info = common::catalog_model(adapter.catalog.as_deref(), &request.model);
+ let api_model = common::api_model_id(adapter.catalog.as_deref(), &request.model);
let supports_prompt_cache = model_info.is_some_and(|m| m.features.prompt_cache);
let auto_cache =
supports_prompt_cache && is_auto_cache_enabled(request.provider_options.as_ref());
@@ -1258,18 +1317,30 @@ async fn build_api_request(
// Check whether this model supports the `output_config.effort` parameter.
// Older reasoning models (e.g. claude-sonnet-4-5) need `thinking` with
// `budget_tokens` instead.
- let supports_effort =
- model_info.is_none_or(|m| m.features.reasoning_effort == ReasoningEffortFeature::Levels);
+ let supports_effort = model_info.is_none_or(Model::supports_reasoning_effort);
let mut resolved_max_tokens = request
.max_tokens
.or_else(|| model_info.and_then(|m| m.limits.max_output))
.unwrap_or(65536);
+ // Default thinking when none is configured explicitly: adaptive for
+ // `levels` models, with or without an effort level — effort is guidance
+ // for thinking allocation, not a replacement for it. Natively adaptive
+ // models don't need one injected (and reject a manual on/off toggle).
+ let default_thinking = || {
+ if model_info.is_some_and(|m| m.features.reasoning_effort == ReasoningEffortFeature::Levels)
+ {
+ Some(serde_json::json!({"type": "adaptive"}))
+ } else {
+ None
+ }
+ };
+
let (mut thinking, mut output_config) = if let Some(effort) = &request.reasoning_effort {
if supports_effort {
(
- explicit_thinking,
+ explicit_thinking.or_else(default_thinking),
Some(serde_json::json!({"effort": <&'static str>::from(*effort)})),
)
} else if explicit_thinking.is_none() {
@@ -1288,18 +1359,7 @@ async fn build_api_request(
(explicit_thinking, None)
}
} else {
- // Auto-set adaptive thinking for known effort-capable models when no
- // explicit thinking config or reasoning_effort is provided.
- let thinking = explicit_thinking.or_else(|| {
- if model_info
- .is_some_and(|m| m.features.reasoning_effort == ReasoningEffortFeature::Levels)
- {
- Some(serde_json::json!({"type": "adaptive"}))
- } else {
- None
- }
- });
- (thinking, None)
+ (explicit_thinking.or_else(default_thinking), None)
};
if tool_choice_forces_tool_use(tool_choice_json.as_ref()) {
@@ -1308,14 +1368,23 @@ async fn build_api_request(
}
let is_fast = request.speed == Some(Speed::Fast);
+ // Models with `sampling_params = false` reject classic sampling knobs.
+ // This gate covers only the typed request fields; values injected through
+ // `provider_options.anthropic` (e.g. `top_k`) are a raw escape hatch and
+ // pass through unfiltered.
+ let (temperature, top_p) = if model_info.is_none_or(Model::supports_sampling_params) {
+ (request.temperature, request.top_p)
+ } else {
+ (None, None)
+ };
let api_request = ApiRequest {
- model: common::api_model_id(adapter.catalog.as_deref(), &request.model),
+ model: api_model,
messages: api_messages,
max_tokens: resolved_max_tokens,
system: system_value,
- temperature: request.temperature,
- top_p: request.top_p,
+ temperature,
+ top_p,
stop_sequences: Some(request.stop_sequences.clone().unwrap_or_default()),
tools: api_tools,
tool_choice: tool_choice_json,
@@ -1343,13 +1412,9 @@ async fn build_api_request(
}
req_builder = req_builder.header("anthropic-version", "2023-06-01");
- let include_1m_context = model_info.is_some_and(|m| m.context_window() >= 1_000_000);
- if let Some(beta_str) = build_beta_header(
- request.provider_options.as_ref(),
- auto_cache,
- is_fast,
- include_1m_context,
- ) {
+ if let Some(beta_str) =
+ build_beta_header(request.provider_options.as_ref(), auto_cache, is_fast)
+ {
req_builder = req_builder.header("anthropic-beta", beta_str);
}
} else if let Some(api_key) = &adapter.http.api_key {
@@ -1386,7 +1451,6 @@ impl ProviderAdapter for Adapter {
let auto_cache =
supports_prompt_cache && is_auto_cache_enabled(request.provider_options.as_ref());
let is_fast = request.speed == Some(Speed::Fast);
- let include_1m_context = model_info.is_some_and(|m| m.context_window() >= 1_000_000);
let url = self.count_tokens_url();
let mut req = self.http.client.post(&url);
@@ -1397,12 +1461,9 @@ impl ProviderAdapter for Adapter {
req = req.header("x-api-key", api_key);
}
req = req.header("anthropic-version", "2023-06-01");
- if let Some(beta_str) = build_beta_header(
- request.provider_options.as_ref(),
- auto_cache,
- is_fast,
- include_1m_context,
- ) {
+ if let Some(beta_str) =
+ build_beta_header(request.provider_options.as_ref(), auto_cache, is_fast)
+ {
req = req.header("anthropic-beta", beta_str);
}
@@ -1447,12 +1508,27 @@ impl ProviderAdapter for Adapter {
}
let (body, headers) = send_and_read_response(req, &self.provider_name, "type").await?;
- let api_resp: ApiResponse = serde_json::from_str(&body).map_err(|e| {
+ let raw: serde_json::Value = serde_json::from_str(&body).map_err(|e| {
Error::network(
format!("failed to parse {} response: {e}", self.provider_name),
e,
)
})?;
+ let api_resp: ApiResponse = serde_json::from_value(raw.clone()).map_err(|e| {
+ Error::network(
+ format!("failed to parse {} response: {e}", self.provider_name),
+ e,
+ )
+ })?;
+
+ if api_resp.stop_reason.as_deref() == Some("refusal") {
+ return Err(refusal_error(
+ &self.provider_name,
+ &api_resp.model,
+ raw,
+ api_resp.stop_details.as_ref(),
+ ));
+ }
let content_parts: Vec = api_resp
.content
@@ -1486,7 +1562,7 @@ impl ProviderAdapter for Adapter {
},
finish_reason,
usage: token_counts_from_api_usage(&api_resp.usage),
- raw: serde_json::from_str(&body).ok(),
+ raw: Some(raw),
warnings: vec![],
rate_limit: parse_rate_limit_headers(&headers),
})
@@ -1582,6 +1658,31 @@ impl ProviderAdapter for Adapter {
fn supports_tool_choice(&self, mode: &str) -> bool {
matches!(mode, "auto" | "none" | "required" | "named")
}
+
+ fn validate_request(&self, request: &Request) -> Result<(), Error> {
+ if let Some(tool_choice) = &request.tool_choice {
+ validate_tool_choice(self, tool_choice)?;
+ }
+
+ let model_info = common::catalog_model(self.catalog.as_deref(), &request.model);
+ if let Some(model) = model_info
+ .filter(|m| m.features.reasoning_effort == ReasoningEffortFeature::AlwaysAdaptive)
+ {
+ if let Some(kind @ ("enabled" | "disabled")) =
+ anthropic_thinking_type(request.provider_options.as_ref())
+ {
+ return Err(Error::Configuration {
+ message: format!(
+ "{} uses always-on adaptive thinking; provider_options.anthropic.thinking.type = \"{kind}\" is not supported. Omit thinking or set only display options.",
+ model.display_name()
+ ),
+ source: None,
+ });
+ }
+ }
+
+ Ok(())
+ }
}
#[cfg(test)]
@@ -1924,13 +2025,13 @@ mod tests {
#[test]
fn beta_header_includes_cache_header() {
- let result = build_beta_header(None, true, false, false);
+ let result = build_beta_header(None, true, false);
assert_eq!(result, Some(CACHE_BETA_HEADER.to_string()));
}
#[test]
fn beta_header_no_cache_no_user_headers() {
- let result = build_beta_header(None, false, false, false);
+ let result = build_beta_header(None, false, false);
assert_eq!(result, None);
}
@@ -1941,7 +2042,7 @@ mod tests {
"beta_headers": ["interleaved-thinking-2025-05-14"]
}
});
- let result = build_beta_header(Some(&opts), true, false, false);
+ let result = build_beta_header(Some(&opts), true, false);
assert_eq!(
result,
Some(format!(
@@ -1957,7 +2058,7 @@ mod tests {
"beta_headers": [CACHE_BETA_HEADER]
}
});
- let result = build_beta_header(Some(&opts), true, false, false);
+ let result = build_beta_header(Some(&opts), true, false);
// Should not duplicate the header
assert_eq!(result, Some(CACHE_BETA_HEADER.to_string()));
}
@@ -1969,10 +2070,25 @@ mod tests {
"beta_headers": ["interleaved-thinking-2025-05-14"]
}
});
- let result = build_beta_header(Some(&opts), false, false, false);
+ let result = build_beta_header(Some(&opts), false, false);
assert_eq!(result, Some("interleaved-thinking-2025-05-14".to_string()));
}
+ #[test]
+ fn refusal_error_falls_back_to_generic_label_when_model_is_empty() {
+ let err = refusal_error("anthropic", "", serde_json::json!({}), None);
+ match err {
+ Error::Provider { detail, .. } => {
+ assert!(
+ detail.message.starts_with("The model refused"),
+ "unexpected message: {}",
+ detail.message
+ );
+ }
+ other => panic!("expected provider error, got {other:?}"),
+ }
+ }
+
#[test]
fn tool_serialization_includes_cache_control() {
let tool = ApiToolDef {
@@ -2515,7 +2631,7 @@ reasoning = true
];
// No user headers — only cache header should appear
- let header = build_beta_header(None, true, false, false).unwrap_or_default();
+ let header = build_beta_header(None, true, false).unwrap_or_default();
for dep in &deprecated {
assert!(
!header.contains(dep),
@@ -2529,7 +2645,7 @@ reasoning = true
"beta_headers": ["interleaved-thinking-2025-05-14"]
}
});
- let header = build_beta_header(Some(&opts), true, false, false).unwrap_or_default();
+ let header = build_beta_header(Some(&opts), true, false).unwrap_or_default();
for dep in &deprecated {
assert!(
!header.contains(dep),
@@ -2909,7 +3025,7 @@ reasoning_effort = "levels"
}
#[test]
fn beta_header_includes_both_cache_and_fast_mode() {
- let result = build_beta_header(None, true, true, false);
+ let result = build_beta_header(None, true, true);
let header = result.expect("should produce a header");
assert!(
header.contains(CACHE_BETA_HEADER),
diff --git a/lib/crates/fabro-llm/tests/it/wire/anthropic.rs b/lib/crates/fabro-llm/tests/it/wire/anthropic.rs
index 5f4822ed4..9e790f2d4 100644
--- a/lib/crates/fabro-llm/tests/it/wire/anthropic.rs
+++ b/lib/crates/fabro-llm/tests/it/wire/anthropic.rs
@@ -1,10 +1,16 @@
//! Wire snapshots for the Anthropic Messages dialect.
+use std::sync::Arc;
+
use fabro_llm::provider::ProviderAdapter;
use fabro_llm::providers::AnthropicAdapter;
use fabro_llm::types::{
- Message, Request, ResponseFormat, ResponseFormatType, ToolChoice, ToolDefinition,
+ Message, ReasoningEffort, Request, ResponseFormat, ResponseFormatType, StreamEvent, ToolChoice,
+ ToolDefinition,
};
+use fabro_llm::{Error, ProviderErrorKind};
+use fabro_model::Catalog;
+use futures::StreamExt;
use httpmock::prelude::*;
use crate::support::{
@@ -64,6 +70,18 @@ fn adapter() -> AnthropicAdapter {
AnthropicAdapter::new("test-key")
}
+fn builtin_catalog() -> Arc {
+ Arc::new(Catalog::from_builtin().expect("built-in catalog should build"))
+}
+
+fn header_value<'a>(capture: &'a WireCapture, name: &str) -> Option<&'a str> {
+ capture
+ .headers
+ .iter()
+ .find(|(header, _)| header == name)
+ .map(|(_, value)| value.as_str())
+}
+
// ---------------------------------------------------------------------------
// Round trip (encode + decode)
// ---------------------------------------------------------------------------
@@ -271,6 +289,131 @@ prompt_cache = false
fabro_test::fabro_json_snapshot!(capture.body);
}
+#[tokio::test]
+async fn encode_fable_uses_api_id_effort_and_omits_1m_beta() {
+ let request = Request {
+ reasoning_effort: Some(ReasoningEffort::XHigh),
+ temperature: Some(0.0),
+ top_p: Some(0.5),
+ ..base_request("fable")
+ };
+
+ let capture = encode_capture(adapter().with_catalog(builtin_catalog()), &request).await;
+
+ assert_eq!(capture.body["model"], "claude-fable-5");
+ assert_eq!(capture.body["output_config"]["effort"], "xhigh");
+ assert!(capture.body.get("thinking").is_none());
+ assert!(capture.body.get("temperature").is_none());
+ assert!(capture.body.get("top_p").is_none());
+ assert!(
+ !header_value(&capture, "anthropic-beta")
+ .unwrap_or("")
+ .contains("context-1m-2025-08-07"),
+ "Fable has 1M context by default and must not receive the legacy beta header"
+ );
+}
+
+#[tokio::test]
+async fn encode_opus_omits_1m_beta_header() {
+ let capture = encode_capture(
+ adapter().with_catalog(builtin_catalog()),
+ &base_request("claude-opus-4-8"),
+ )
+ .await;
+
+ assert!(
+ !header_value(&capture, "anthropic-beta")
+ .unwrap_or("")
+ .contains("context-1m-2025-08-07"),
+ "1M context is GA on opus; the legacy beta opt-in must not be sent"
+ );
+}
+
+#[tokio::test]
+async fn encode_opus_drops_sampling_params() {
+ let request = Request {
+ temperature: Some(0.0),
+ top_p: Some(0.5),
+ ..base_request("claude-opus-4-8")
+ };
+
+ let capture = encode_capture(adapter().with_catalog(builtin_catalog()), &request).await;
+
+ assert!(
+ capture.body.get("temperature").is_none(),
+ "Opus 4.7/4.8 reject temperature; it must not be sent"
+ );
+ assert!(
+ capture.body.get("top_p").is_none(),
+ "Opus 4.7/4.8 reject top_p; it must not be sent"
+ );
+}
+
+#[tokio::test]
+async fn encode_opus_effort_keeps_adaptive_thinking() {
+ let request = Request {
+ reasoning_effort: Some(ReasoningEffort::High),
+ ..base_request("claude-opus-4-8")
+ };
+
+ let capture = encode_capture(adapter().with_catalog(builtin_catalog()), &request).await;
+
+ assert_eq!(capture.body["output_config"]["effort"], "high");
+ assert_eq!(
+ capture.body["thinking"]["type"], "adaptive",
+ "asking for effort must not turn thinking off; Opus 4.7/4.8 run without thinking unless adaptive is sent"
+ );
+}
+
+#[tokio::test]
+async fn encode_opus_without_effort_injects_adaptive_thinking() {
+ let capture = encode_capture(
+ adapter().with_catalog(builtin_catalog()),
+ &base_request("claude-opus-4-8"),
+ )
+ .await;
+
+ assert_eq!(capture.body["thinking"]["type"], "adaptive");
+}
+
+#[tokio::test]
+async fn encode_fable_without_effort_omits_default_thinking() {
+ let capture = encode_capture(
+ adapter().with_catalog(builtin_catalog()),
+ &base_request("claude-fable-5"),
+ )
+ .await;
+
+ assert_eq!(capture.body["model"], "claude-fable-5");
+ assert!(capture.body.get("thinking").is_none());
+}
+
+#[test]
+fn fable_rejects_manual_enabled_or_disabled_thinking() {
+ let adapter = adapter().with_catalog(builtin_catalog());
+
+ for kind in ["enabled", "disabled"] {
+ let request = Request {
+ provider_options: Some(serde_json::json!({
+ "anthropic": {
+ "thinking": {"type": kind, "budget_tokens": 1024}
+ }
+ })),
+ ..base_request("claude-fable-5")
+ };
+
+ let err = adapter
+ .validate_request(&request)
+ .expect_err("manual Fable thinking mode should be rejected locally");
+ assert!(
+ err.to_string().contains("Claude Fable 5")
+ && err.to_string().contains("thinking")
+ && err.to_string().contains(kind),
+ "unexpected error: {err}"
+ );
+ }
+}
+
#[tokio::test]
async fn encode_prompt_cache_with_catalog() {
let catalog = support::catalog_from_toml(
@@ -351,6 +494,16 @@ async fn decode_response(body: serde_json::Value) -> fabro_llm::types::Response
response
}
+/// Runs `complete()` against a canned body and returns the adapter result.
+async fn complete_result(body: serde_json::Value) -> Result {
+ let server = MockServer::start();
+ let (mock, _slot) = mount_capture(&server, "/messages", body);
+ let adapter = adapter().with_base_url(server.base_url());
+ let result = adapter.complete(&base_request(MODEL)).await;
+ mock.assert();
+ result
+}
+
#[tokio::test]
async fn decode_tool_use_stop_reason() {
let response = decode_response(serde_json::json!({
@@ -411,6 +564,42 @@ async fn decode_max_tokens_stop_reason() {
fabro_test::fabro_json_snapshot!(response);
}
+#[tokio::test]
+async fn decode_refusal_returns_failover_eligible_content_filter_error() {
+ let err = complete_result(serde_json::json!({
+ "id": "msg_refusal",
+ "type": "message",
+ "role": "assistant",
+ "model": "claude-fable-5",
+ "content": [],
+ "stop_reason": "refusal",
+ "stop_details": {
+ "type": "refusal",
+ "category": "cyber",
+ "explanation": "This request was declined because it could enable cyber harm."
+ },
+ "usage": {"input_tokens": 412, "output_tokens": 0}
+ }))
+ .await
+ .expect_err("refusal should be returned as an LLM error");
+
+ assert!(err.failover_eligible());
+ match &err {
+ Error::Provider { kind, detail } => {
+ assert_eq!(*kind, ProviderErrorKind::ContentFilter);
+ assert_eq!(detail.provider, "anthropic");
+ assert_eq!(detail.error_code.as_deref(), Some("refusal"));
+ assert!(detail.message.contains("claude-fable-5"));
+ assert!(detail.message.contains("declined"));
+ assert_eq!(
+ detail.raw.as_ref().unwrap()["stop_details"]["category"],
+ "cyber"
+ );
+ }
+ other => panic!("expected provider content-filter error, got {other:?}"),
+ }
+}
+
// ---------------------------------------------------------------------------
// Stream
// ---------------------------------------------------------------------------
@@ -561,6 +750,58 @@ async fn stream_error_event_mid_stream() {
fabro_test::fabro_json_snapshot!(events);
}
+#[tokio::test]
+async fn stream_refusal_returns_error_without_final_response() {
+ let sse = support::sse_transcript(&[
+ (
+ "message_start",
+ r#"{"type":"message_start","message":{"id":"msg_stream_refusal","type":"message","role":"assistant","model":"claude-fable-5","content":[],"usage":{"input_tokens":412,"output_tokens":0}}}"#,
+ ),
+ (
+ "message_delta",
+ r#"{"type":"message_delta","delta":{"stop_reason":"refusal","stop_sequence":null,"stop_details":{"type":"refusal","category":"cyber","explanation":"This request was declined."}},"usage":{"output_tokens":0}}"#,
+ ),
+ ("message_stop", r#"{"type":"message_stop"}"#),
+ ]);
+ let server = MockServer::start();
+ let (mock, _slot) = mount_capture_sse(&server, "/messages", &sse);
+ let adapter = adapter().with_base_url(server.base_url());
+ let mut stream = adapter
+ .stream(&base_request("claude-fable-5"))
+ .await
+ .expect("stream should start");
+
+ let mut saw_finish = false;
+ let mut refusal = None;
+ while let Some(item) = stream.next().await {
+ match item {
+ Ok(StreamEvent::Finish { .. }) => saw_finish = true,
+ Ok(_) => {}
+ Err(err) => {
+ refusal = Some(err);
+ break;
+ }
+ }
+ }
+ mock.assert();
+
+ assert!(!saw_finish, "refusal stream must not emit a final response");
+ let err = refusal.expect("stream should yield a refusal error");
+ assert!(err.failover_eligible());
+ match &err {
+ Error::Provider { kind, detail } => {
+ assert_eq!(*kind, ProviderErrorKind::ContentFilter);
+ assert_eq!(detail.error_code.as_deref(), Some("refusal"));
+ assert!(detail.message.contains("claude-fable-5"));
+ assert_eq!(
+ detail.raw.as_ref().unwrap()["stop_details"]["category"],
+ "cyber"
+ );
+ }
+ other => panic!("expected provider content-filter error, got {other:?}"),
+ }
+}
+
/// The Anthropic decoder never synthesizes a `Finish` on byte-stream end:
/// `message_stop` is the only finisher. A transcript that ends without it
/// must produce no `Finish` event.
diff --git a/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__anthropic__encode_reasoning_effort_with_levels_catalog.snap b/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__anthropic__encode_reasoning_effort_with_levels_catalog.snap
index 92b9f5904..59e4b54ef 100644
--- a/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__anthropic__encode_reasoning_effort_with_levels_catalog.snap
+++ b/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__anthropic__encode_reasoning_effort_with_levels_catalog.snap
@@ -17,6 +17,9 @@ expression: rendered
],
"max_tokens": 128,
"stop_sequences": [],
+ "thinking": {
+ "type": "adaptive"
+ },
"output_config": {
"effort": "high"
}
diff --git a/lib/crates/fabro-model/src/catalog.rs b/lib/crates/fabro-model/src/catalog.rs
index 35f7a3907..1ed2d7ff6 100644
--- a/lib/crates/fabro-model/src/catalog.rs
+++ b/lib/crates/fabro-model/src/catalog.rs
@@ -123,6 +123,8 @@ pub struct SettingsModelFeatures {
pub reasoning_effort: Option,
#[serde(default)]
pub prompt_cache: Option,
+ #[serde(default)]
+ pub sampling_params: Option,
}
#[derive(Debug, Clone, Default, PartialEq, Deserialize)]
@@ -599,7 +601,7 @@ pub enum CatalogBuildError {
#[error("model '{model}' declares reasoning_effort feature but features.reasoning is false")]
ReasoningEffortWithoutReasoning { model: String },
#[error(
- "model '{model}' must declare at least one reasoning_effort when features.reasoning_effort is levels"
+ "model '{model}' must declare at least one reasoning_effort when features.reasoning_effort is levels or always_adaptive"
)]
EmptyReasoningEffortControls { model: String },
#[error("model '{model}' has invalid speed '{value}'")]
@@ -1189,6 +1191,7 @@ fn merge_model_features_settings(
reasoning: higher.reasoning.or(fallback.reasoning),
reasoning_effort: higher.reasoning_effort.or(fallback.reasoning_effort),
prompt_cache: higher.prompt_cache.or(fallback.prompt_cache),
+ sampling_params: higher.sampling_params.or(fallback.sampling_params),
}
}
@@ -1427,7 +1430,7 @@ fn build_model_features(
field: "features.reasoning",
})?;
let reasoning_effort = features.reasoning_effort.unwrap_or_default();
- if !reasoning && reasoning_effort == ReasoningEffortFeature::Levels {
+ if !reasoning && reasoning_effort != ReasoningEffortFeature::None {
return Err(CatalogBuildError::ReasoningEffortWithoutReasoning {
model: model_id.to_string(),
});
@@ -1449,6 +1452,7 @@ fn build_model_features(
reasoning,
reasoning_effort,
prompt_cache: features.prompt_cache.unwrap_or_default(),
+ sampling_params: features.sampling_params.unwrap_or(true),
})
}
@@ -1496,8 +1500,7 @@ fn build_model_controls(
features: &ModelFeatures,
settings: &ModelCatalogSettings,
) -> Result {
- let supports_native_reasoning_effort =
- features.reasoning_effort == ReasoningEffortFeature::Levels;
+ let supports_native_reasoning_effort = features.supports_reasoning_effort();
let reasoning_effort = match settings
.controls
.as_ref()
@@ -3448,6 +3451,52 @@ reasoning_effort = ["low", "medium"]
);
}
+ #[test]
+ fn catalog_from_settings_accepts_reasoning_effort_feature_always_adaptive() {
+ let settings = minimal_settings(
+ r#"
+[providers.test]
+display_name = "Test"
+adapter = "openai"
+agent_profile = "openai"
+
+[models.model]
+provider = "test"
+display_name = "Model"
+family = "test"
+default = true
+
+[models.model.limits]
+context_window = 1000
+
+[models.model.features]
+tools = true
+vision = false
+reasoning = true
+reasoning_effort = "always_adaptive"
+prompt_cache = true
+"#,
+ );
+
+ let catalog = Catalog::from_settings(&settings).unwrap();
+ let model = catalog.get("model").unwrap();
+ assert_eq!(
+ model.features.reasoning_effort,
+ crate::ReasoningEffortFeature::AlwaysAdaptive
+ );
+ assert!(model.supports_reasoning_effort());
+ // Always-adaptive models get the full default effort controls, same as
+ // Levels.
+ assert_eq!(
+ catalog
+ .model_settings("model")
+ .unwrap()
+ .controls
+ .reasoning_effort,
+ ReasoningEffort::VARIANTS.to_vec()
+ );
+ }
+
#[test]
fn catalog_from_settings_accepts_reasoning_effort_controls_without_native_effort_feature() {
let settings = minimal_settings(
@@ -3560,6 +3609,89 @@ reasoning_effort = "levels"
));
}
+ #[test]
+ fn catalog_from_settings_rejects_always_adaptive_effort_without_reasoning() {
+ let settings = minimal_settings(
+ r#"
+[providers.test]
+display_name = "Test"
+adapter = "openai"
+agent_profile = "openai"
+
+[models.model]
+provider = "test"
+display_name = "Model"
+family = "test"
+default = true
+
+[models.model.limits]
+context_window = 1000
+
+[models.model.features]
+tools = true
+vision = false
+reasoning = false
+reasoning_effort = "always_adaptive"
+"#,
+ );
+
+ assert!(matches!(
+ Catalog::from_settings(&settings).unwrap_err(),
+ CatalogBuildError::ReasoningEffortWithoutReasoning { model }
+ if model == "model"
+ ));
+ }
+
+ #[test]
+ fn catalog_from_settings_sampling_params_defaults_true_and_accepts_false() {
+ let settings = minimal_settings(
+ r#"
+[providers.test]
+display_name = "Test"
+adapter = "openai"
+agent_profile = "openai"
+
+[models.with-sampling]
+provider = "test"
+display_name = "With"
+family = "test"
+default = true
+
+[models.with-sampling.limits]
+context_window = 1000
+
+[models.with-sampling.features]
+tools = true
+vision = false
+reasoning = false
+
+[models.no-sampling]
+provider = "test"
+display_name = "Without"
+family = "test"
+
+[models.no-sampling.limits]
+context_window = 1000
+
+[models.no-sampling.features]
+tools = true
+vision = false
+reasoning = false
+sampling_params = false
+"#,
+ );
+
+ let catalog = Catalog::from_settings(&settings).unwrap();
+ assert!(
+ catalog
+ .get("with-sampling")
+ .unwrap()
+ .features
+ .sampling_params
+ );
+ assert!(!catalog.get("no-sampling").unwrap().features.sampling_params);
+ }
+
// ---- Provider / catalog data integrity tests ----
#[test]
@@ -3637,6 +3769,7 @@ reasoning_effort = "levels"
reasoning: true,
reasoning_effort: Levels,
prompt_cache: true,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(
@@ -3692,6 +3825,7 @@ reasoning_effort = "levels"
reasoning: false,
reasoning_effort: None,
prompt_cache: false,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(
@@ -3755,6 +3889,7 @@ reasoning_effort = "levels"
reasoning: true,
reasoning_effort: Levels,
prompt_cache: false,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(
@@ -3810,6 +3945,7 @@ reasoning_effort = "levels"
reasoning: true,
reasoning_effort: Levels,
prompt_cache: false,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(
diff --git a/lib/crates/fabro-model/src/catalog/providers/anthropic.toml b/lib/crates/fabro-model/src/catalog/providers/anthropic.toml
index eb924e85c..0a1587197 100644
--- a/lib/crates/fabro-model/src/catalog/providers/anthropic.toml
+++ b/lib/crates/fabro-model/src/catalog/providers/anthropic.toml
@@ -9,6 +9,30 @@ priority = 100
credentials = ["env:ANTHROPIC_API_KEY", "vault:ANTHROPIC_API_KEY"]
header = { custom = "x-api-key" }
+[models."claude-fable-5"]
+provider = "anthropic"
+api_id = "claude-fable-5"
+display_name = "Claude Fable 5"
+family = "claude-5"
+aliases = ["fable", "claude-fable"]
+
+[models."claude-fable-5".limits]
+context_window = 1000000
+max_output = 128000
+
+[models."claude-fable-5".features]
+tools = true
+vision = true
+reasoning = true
+reasoning_effort = "always_adaptive"
+prompt_cache = true
+sampling_params = false
+
+[models."claude-fable-5".costs]
+input_cost_per_mtok = 10.0
+output_cost_per_mtok = 50.0
+cache_input_cost_per_mtok = 1.0
+
[models."claude-opus-4-8"]
provider = "anthropic"
api_id = "claude-opus-4-8"
@@ -29,6 +53,7 @@ vision = true
reasoning = true
reasoning_effort = "levels"
prompt_cache = true
+sampling_params = false
[models."claude-opus-4-8".controls]
speed = ["fast"]
@@ -62,6 +87,7 @@ vision = true
reasoning = true
reasoning_effort = "levels"
prompt_cache = true
+sampling_params = false
[models."claude-opus-4-7".controls]
speed = ["fast"]
diff --git a/lib/crates/fabro-model/src/types.rs b/lib/crates/fabro-model/src/types.rs
index 130c0a702..8dcbd3270 100644
--- a/lib/crates/fabro-model/src/types.rs
+++ b/lib/crates/fabro-model/src/types.rs
@@ -17,10 +17,14 @@ use crate::ids::ProviderId;
strum::EnumString,
strum::IntoStaticStr,
)]
-#[serde(rename_all = "lowercase")]
-#[strum(serialize_all = "lowercase")]
+#[serde(rename_all = "snake_case")]
+#[strum(serialize_all = "snake_case")]
pub enum ReasoningEffortFeature {
Levels,
+ /// Effort levels are supported, and thinking is natively always-on
+ /// adaptive at the endpoint; a manual thinking on/off toggle is not
+ /// accepted.
+ AlwaysAdaptive,
#[default]
None,
}
@@ -31,6 +35,10 @@ pub struct ModelLimits {
pub max_output: Option,
}
+fn default_true() -> bool {
+ true
+}
+
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
pub struct ModelFeatures {
pub tools: bool,
@@ -43,6 +51,22 @@ pub struct ModelFeatures {
/// Whether this model endpoint supports prompt caching annotations.
#[serde(default)]
pub prompt_cache: bool,
+ /// Whether the model endpoint accepts classic sampling parameters
+ /// (`temperature`, `top_p`). Models with always-on adaptive behavior
+ /// reject them.
+ #[serde(default = "default_true")]
+ pub sampling_params: bool,
+}
+
+impl ModelFeatures {
+ /// Whether the model endpoint accepts a native reasoning-effort level.
+ #[must_use]
+ pub fn supports_reasoning_effort(&self) -> bool {
+ matches!(
+ self.reasoning_effort,
+ ReasoningEffortFeature::Levels | ReasoningEffortFeature::AlwaysAdaptive
+ )
+ }
}
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
@@ -114,13 +138,17 @@ impl Model {
}
pub fn supports_reasoning_effort(&self) -> bool {
- self.features.reasoning_effort == ReasoningEffortFeature::Levels
+ self.features.supports_reasoning_effort()
}
pub fn supports_prompt_cache(&self) -> bool {
self.features.prompt_cache
}
+ pub fn supports_sampling_params(&self) -> bool {
+ self.features.sampling_params
+ }
+
pub fn training(&self) -> Option<&str> {
self.training.as_deref()
}
@@ -163,6 +191,22 @@ mod tests {
use super::*;
use crate::ids::ProviderId;
+ #[test]
+ fn reasoning_effort_feature_always_adaptive_round_trips() {
+ let parsed: ReasoningEffortFeature =
+ serde_json::from_value(serde_json::json!("always_adaptive")).unwrap();
+ assert_eq!(parsed, ReasoningEffortFeature::AlwaysAdaptive);
+ assert_eq!(
+ serde_json::to_value(parsed).unwrap(),
+ serde_json::json!("always_adaptive")
+ );
+ assert_eq!(parsed.to_string(), "always_adaptive");
+ assert_eq!(
+ "always_adaptive".parse::().unwrap(),
+ parsed
+ );
+ }
+
#[test]
fn inherent_methods_return_correct_values() {
let info = Model {
@@ -182,6 +226,7 @@ mod tests {
reasoning: true,
reasoning_effort: ReasoningEffortFeature::Levels,
prompt_cache: true,
+ sampling_params: true,
},
costs: ModelCosts {
input_cost_per_mtok: Some(1.0),
@@ -206,6 +251,7 @@ mod tests {
assert!(info.supports_reasoning());
assert!(info.supports_reasoning_effort());
assert!(info.supports_prompt_cache());
+ assert!(info.supports_sampling_params());
assert_eq!(info.training(), Some("training"));
assert_eq!(info.knowledge_cutoff(), Some("knowledge-cutoff"));
assert_eq!(info.input_cost_per_mtok(), Some(1.0));
diff --git a/lib/crates/fabro-workflow/src/handler/llm/api.rs b/lib/crates/fabro-workflow/src/handler/llm/api.rs
index 91753f4c9..61649380c 100644
--- a/lib/crates/fabro-workflow/src/handler/llm/api.rs
+++ b/lib/crates/fabro-workflow/src/handler/llm/api.rs
@@ -1677,6 +1677,63 @@ mod tests {
}
}
+ struct RefusalTestProvider;
+
+ #[async_trait]
+ impl ProviderAdapter for RefusalTestProvider {
+ fn name(&self) -> &str {
+ "anthropic"
+ }
+
+ async fn complete(
+ &self,
+ _request: &Request,
+ ) -> Result {
+ Err(refusal_llm_error())
+ }
+
+ async fn stream(&self, _request: &Request) -> Result {
+ Ok(Box::pin(stream::empty()))
+ }
+ }
+
+ struct TextTestProvider {
+ provider: &'static str,
+ text: &'static str,
+ }
+
+ #[async_trait]
+ impl ProviderAdapter for TextTestProvider {
+ fn name(&self) -> &str {
+ self.provider
+ }
+
+ async fn complete(
+ &self,
+ request: &Request,
+ ) -> Result {
+ Ok(fabro_llm::types::Response {
+ id: "msg_fallback".to_string(),
+ model: request.model.clone(),
+ provider: self.provider.to_string(),
+ message: Message::assistant(self.text),
+ finish_reason: fabro_llm::types::FinishReason::Stop,
+ usage: TokenCounts {
+ input_tokens: 3,
+ output_tokens: 2,
+ ..TokenCounts::default()
+ },
+ raw: None,
+ warnings: vec![],
+ rate_limit: None,
+ })
+ }
+
+ async fn stream(&self, _request: &Request) -> Result {
+ Ok(Box::pin(stream::empty()))
+ }
+ }
+
fn mock_llm_catalog(server: &MockServer) -> Arc {
let settings: LlmCatalogSettings = toml::from_str(&format!(
r#"
@@ -2644,6 +2701,70 @@ reasoning = false
assert_eq!(client.provider_names(), vec!["anthropic"]);
}
+ #[tokio::test]
+ async fn one_shot_falls_back_after_refusal_error() {
+ let fallback_chain = vec![FallbackTarget {
+ provider: "openai".to_string(),
+ model: "gpt-5.5".to_string(),
+ }];
+ let backend = AgentApiBackend::new_from_env(
+ "claude-fable-5".to_string(),
+ ProviderId::anthropic(),
+ fallback_chain.clone(),
+ SteeringHub::for_tests(),
+ );
+ let mut providers = HashMap::new();
+ providers.insert(
+ "anthropic".to_string(),
+ Arc::new(RefusalTestProvider) as Arc,
+ );
+ providers.insert(
+ "openai".to_string(),
+ Arc::new(TextTestProvider {
+ provider: "openai",
+ text: "fallback ok",
+ }) as Arc,
+ );
+ let client = Client::new(providers, Some("anthropic".to_string()), Vec::new());
+ let node = Node::new("ask");
+ let context = Context::new();
+ let stage_scope = StageScope::for_handler(&context, &node.id);
+ let emitter = Arc::new(Emitter::new(fabro_types::RunId::new()));
+ let request = Request {
+ model: "claude-fable-5".to_string(),
+ messages: vec![Message::user("Hello")],
+ provider: Some("anthropic".to_string()),
+ tools: None,
+ tool_choice: None,
+ response_format: None,
+ temperature: None,
+ top_p: None,
+ max_tokens: Some(128),
+ stop_sequences: None,
+ reasoning_effort: None,
+ speed: None,
+ metadata: None,
+ provider_options: None,
+ };
+
+ let completion = backend
+ .complete_one_shot_request(
+ &client,
+ &node,
+ &emitter,
+ &stage_scope,
+ &request,
+ EffectiveRequestControls::default(),
+ &fallback_chain,
+ )
+ .await
+ .unwrap();
+
+ assert_eq!(completion.response.text(), "fallback ok");
+ assert_eq!(completion.model.provider, ProviderId::openai());
+ assert_eq!(completion.model.model_id, "gpt-5.5");
+ }
+
#[tokio::test]
async fn one_shot_repairs_custom_output_schema_with_previous_assistant_message() {
let server = MockServer::start();
@@ -2838,6 +2959,27 @@ reasoning = false
}
}
+ fn refusal_llm_error() -> LlmError {
+ LlmError::Provider {
+ kind: ProviderErrorKind::ContentFilter,
+ detail: Box::new(ProviderErrorDetail {
+ message: "claude-fable-5 refused the request".into(),
+ provider: "anthropic".into(),
+ status_code: None,
+ error_code: Some("refusal".into()),
+ retry_after: None,
+ raw: Some(serde_json::json!({
+ "stop_reason": "refusal",
+ "stop_details": {
+ "type": "refusal",
+ "category": "cyber",
+ "explanation": "This request was declined."
+ }
+ })),
+ }),
+ }
+ }
+
#[tokio::test]
async fn spawn_bridge_task_sets_cancelled_and_cancels_session_token() {
let run_token = CancellationToken::new();
@@ -3016,6 +3158,26 @@ reasoning = false
}
}
+ #[test]
+ fn classify_refusal_llm_returns_failover_when_allowed() {
+ let err = fabro_agent::Error::Llm(refusal_llm_error());
+ assert!(matches!(
+ classify_agent_error(err, true),
+ AgentApiErrorDisposition::FailoverEligible(_)
+ ));
+ }
+
+ #[test]
+ fn classify_refusal_llm_returns_terminal_when_not_allowed() {
+ let err = fabro_agent::Error::Llm(refusal_llm_error());
+ match classify_agent_error(err, false) {
+ AgentApiErrorDisposition::Terminal(Error::Llm(llm_err)) => {
+ assert!(llm_err.to_string().contains("claude-fable-5 refused"));
+ }
+ _ => panic!("expected Terminal(Error::Llm) when refusal failover is disallowed"),
+ }
+ }
+
#[test]
fn classify_session_closed_is_terminal_precondition() {
let err = fabro_agent::Error::SessionClosed;
diff --git a/lib/packages/fabro-api-client/src/models/model-features.ts b/lib/packages/fabro-api-client/src/models/model-features.ts
index 213cad50f..7eb314d8c 100644
--- a/lib/packages/fabro-api-client/src/models/model-features.ts
+++ b/lib/packages/fabro-api-client/src/models/model-features.ts
@@ -38,4 +38,8 @@ export interface ModelFeatures {
* Whether the model endpoint supports prompt caching.
*/
'prompt_cache': boolean;
+ /**
+ * Whether the model accepts classic sampling parameters (temperature, top_p).
+ */
+ 'sampling_params': boolean;
}
diff --git a/lib/packages/fabro-api-client/src/models/reasoning-effort-feature.ts b/lib/packages/fabro-api-client/src/models/reasoning-effort-feature.ts
index 5cd70ad45..1aa9d6734 100644
--- a/lib/packages/fabro-api-client/src/models/reasoning-effort-feature.ts
+++ b/lib/packages/fabro-api-client/src/models/reasoning-effort-feature.ts
@@ -15,11 +15,12 @@
/**
- * Whether the model endpoint supports a native reasoning-effort parameter.
+ * Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter.
*/
export const ReasoningEffortFeature = {
LEVELS: 'levels',
+ ALWAYS_ADAPTIVE: 'always_adaptive',
NONE: 'none'
} as const;