diff --git a/docs/public/core-concepts/models.mdx b/docs/public/core-concepts/models.mdx index 722cf62ca..a8c3df629 100644 --- a/docs/public/core-concepts/models.mdx +++ b/docs/public/core-concepts/models.mdx @@ -126,12 +126,21 @@ Provider auth is declared in `[llm.providers..auth]` with ordered `env: Provider fields in configuration, APIs, and model routing are provider ID strings. Built-in names like `anthropic`, `openai`, and `gemini` still work, but custom IDs like `proxy` work anywhere a provider ID is accepted. +### OpenRouter + +Fabro ships an [OpenRouter](/integrations/openrouter) provider definition with a curated model catalog, disabled by default. Enable it in settings and store an API key with `fabro provider login openrouter`: + +```toml title="settings.toml" +[llm.providers.openrouter] +enabled = true +``` + ### Ollama Fabro ships an Ollama provider definition that is disabled by default. Enable it in settings when you want Fabro to route through a local Ollama server: diff --git a/docs/public/docs.json b/docs/public/docs.json index 92545efe5..b2f737ba1 100644 --- a/docs/public/docs.json +++ b/docs/public/docs.json @@ -95,6 +95,7 @@ "integrations/github", "integrations/daytona", "integrations/litellm", + "integrations/openrouter", "integrations/slack", "integrations/brave-search" ] diff --git a/docs/public/integrations/openrouter.mdx b/docs/public/integrations/openrouter.mdx new file mode 100644 index 000000000..66b2eab05 --- /dev/null +++ b/docs/public/integrations/openrouter.mdx @@ -0,0 +1,148 @@ +--- +title: "OpenRouter" +description: "Route Fabro models through OpenRouter's multi-provider gateway" +--- + +[OpenRouter](https://openrouter.ai/) is an aggregator that fronts hundreds of models behind one OpenAI-compatible API. Fabro ships a disabled `openrouter` provider entry with a curated model catalog, so you can opt in from `settings.toml` without changing Fabro code. + +## Prerequisites + +- An [OpenRouter account](https://openrouter.ai/) with credit +- An API key from [openrouter.ai/keys](https://openrouter.ai/keys) + +## Enable the provider + +Add the provider override to `~/.fabro/settings.toml`: + +```toml title="settings.toml" +_version = 1 + +[llm.providers.openrouter] +enabled = true +``` + +## Configure credentials + +For server-backed runs, store the key in the Fabro server vault: + +```bash +fabro provider login openrouter +# or +fabro secret set OPENROUTER_API_KEY sk-or-v1-... +``` + +Standalone local SDK/CLI runs can use an env-backed credential source explicitly: + +```bash +export OPENROUTER_API_KEY=sk-or-v1-... +``` + +## Included models + +The built-in catalog curates frontier and open-weights models under vendor-namespaced IDs: + +| Fabro model ID | Notes | +| --- | --- | +| `anthropic/claude-opus-4-7` | Claude via OpenRouter, Anthropic-style cache billing | +| `anthropic/claude-sonnet-4-6` | Provider default | +| `anthropic/claude-haiku-4-5` | Provider small default | +| `openai/gpt-5.4`, `openai/gpt-5.5` | | +| `google/gemini-3.1-pro-preview`, `google/gemini-3.5-flash` | | +| `deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash` | | +| `moonshotai/kimi-k2.6`, `qwen/qwen3-coder`, `qwen/qwen3.6-flash` | | +| `z-ai/glm-4.6`, `minimax/minimax-m2.7`, `xiaomi/mimo-v2.5-pro` | | +| `nvidia/nemotron-3-super-120b-a12b`, `mistralai/devstral-2512` | | + +Any other OpenRouter model can be added as a settings model entry with `provider = "openrouter"` and the OpenRouter slug as `api_id`: + +```toml title="settings.toml" +[llm.models."meta-llama/llama-4-maverick"] +provider = "openrouter" +display_name = "Llama 4 Maverick" +family = "llama-4" + +[llm.models."meta-llama/llama-4-maverick".limits] +context_window = 1000000 + +[llm.models."meta-llama/llama-4-maverick".features] +tools = true +vision = false +reasoning = false +``` + +## Use OpenRouter models + +```bash +fabro model list --provider openrouter +fabro model test --model anthropic/claude-sonnet-4-6 +fabro run workflow.fabro --model deepseek/deepseek-v4-flash +``` + +In workflow stylesheets: + +```dot title="workflow.fabro" +digraph Example { + graph [ + model_stylesheet=" + * { model: anthropic/claude-sonnet-4-6; } + " + ] + + start [shape=Mdiamond, label="Start"] + work [label="Work", prompt="Use the configured OpenRouter model."] + exit [shape=Msquare, label="Exit"] + + start -> work -> exit +} +``` + +## Cost telemetry + +Every OpenRouter response includes an inline `usage.cost` with authoritative USD billing. Fabro surfaces it as `cost_usd` with `cost_source = "authoritative"` on completion responses. Other providers populate the same fields from catalog price estimates with `cost_source = "estimated"`. + +The catalog prices on OpenRouter model rows are best-effort estimates used only before the authoritative figure arrives (for example, mid-stream rollups). + +## Provider routing + +OpenRouter's [provider routing preferences](https://openrouter.ai/docs/guides/routing/provider-selection) pass through verbatim via `provider_options.openrouter` on API/SDK requests — the keys merge into the top level of the request body: + +```json +{ + "model": "anthropic/claude-sonnet-4-6", + "provider_options": { + "openrouter": { + "provider": { "sort": "throughput", "data_collection": "deny" }, + "models": ["anthropic/claude-sonnet-4.6", "deepseek/deepseek-v4-pro"] + } + } +} +``` + +## Attribution headers + +Fabro does not send OpenRouter's optional attribution headers (`HTTP-Referer`, `X-Title`) by default, so self-hosted installations stay anonymous on OpenRouter's public app leaderboard. To opt in: + +```toml title="settings.toml" +[llm.providers.openrouter.extra_headers] +"HTTP-Referer" = { literal = "https://your-site.example" } +"X-Title" = { literal = "Your App" } +``` + +## Troubleshooting + +**"No API key configured"** — For server-backed runs, set the key with `fabro provider login openrouter` or `fabro secret set OPENROUTER_API_KEY ...`. For standalone local usage, export `OPENROUTER_API_KEY` in the invoking shell. + +**402 / insufficient credits** — OpenRouter requires prepaid credit; check your balance at [openrouter.ai/credits](https://openrouter.ai/credits). + +**Unknown model** — Confirm the model's `api_id` matches an OpenRouter slug exactly (including the vendor prefix), then run `fabro model test --model `. + +## Further reading + + + + How Fabro routes model IDs, providers, and fallbacks. + + + Full reference for `[llm.providers.]` and `[llm.models.]`. + + diff --git a/docs/public/reference/user-configuration.mdx b/docs/public/reference/user-configuration.mdx index d14c91631..8f4e0c38e 100644 --- a/docs/public/reference/user-configuration.mdx +++ b/docs/public/reference/user-configuration.mdx @@ -255,6 +255,7 @@ cache_input_cost_per_mtok = 0.60 | `provider` | string | None | Provider ID this model belongs to. | | `api_id` | string | model ID | Identifier sent to the provider API. | | `agent_profile` | `"anthropic"` \| `"openai"` \| `"gemini"` | provider profile | Agent profile override for this model. Model overrides take precedence over provider overrides. | +| `billing_policy` | `"openai"` \| `"anthropic"` \| `"gemini"` \| `"none"` | provider policy | Billing algorithm override for this model — for models whose billing family differs from their provider's (e.g. Claude served through OpenRouter bills Anthropic-style cache reads/writes). | | `display_name` | string | model ID | Human-readable model name. | | `family` | string | model ID | Family label used for catalog display and matching. | | `training` | string | None | Training data cutoff label. | diff --git a/lib/crates/fabro-config/src/builders.rs b/lib/crates/fabro-config/src/builders.rs index 27c9725cb..3ce36d54b 100644 --- a/lib/crates/fabro-config/src/builders.rs +++ b/lib/crates/fabro-config/src/builders.rs @@ -348,6 +348,7 @@ fn model_settings_to_catalog(settings: ModelSettings) -> model_catalog::ModelCat provider, api_id, codec, + billing_policy, agent_profile, display_name, family, @@ -368,6 +369,7 @@ fn model_settings_to_catalog(settings: ModelSettings) -> model_catalog::ModelCat provider, api_id, codec, + billing_policy, agent_profile, display_name, family, diff --git a/lib/crates/fabro-config/src/layers/llm.rs b/lib/crates/fabro-config/src/layers/llm.rs index 56f83d071..7e531686d 100644 --- a/lib/crates/fabro-config/src/layers/llm.rs +++ b/lib/crates/fabro-config/src/layers/llm.rs @@ -104,6 +104,11 @@ pub struct ModelSettings { /// catalog build. #[serde(default, skip_serializing_if = "Option::is_none")] pub codec: Option, + /// Billing family for this model, overriding the provider's policy + /// (e.g. Anthropic cache billing for a Claude model served through an + /// aggregator). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub billing_policy: Option, /// Agent profile used for routing/profile-specific behavior. #[serde(default, skip_serializing_if = "Option::is_none")] pub agent_profile: Option, @@ -361,6 +366,23 @@ codec = "anthropic_messages" ); } + #[test] + fn model_billing_policy_parses_from_toml() { + let parsed: LlmLayer = toml::from_str( + r#" +[models.acme_claude] +provider = "acme" +billing_policy = "anthropic" +"#, + ) + .unwrap(); + + assert_eq!( + parsed.models.get("acme_claude").unwrap().billing_policy, + Some(fabro_model::BillingPolicy::Anthropic) + ); + } + #[test] fn model_agent_profile_parses_from_toml() { let parsed: LlmLayer = toml::from_str( diff --git a/lib/crates/fabro-dev/src/commands/docs_options_reference.rs b/lib/crates/fabro-dev/src/commands/docs_options_reference.rs index 17a608288..d5a52370f 100644 --- a/lib/crates/fabro-dev/src/commands/docs_options_reference.rs +++ b/lib/crates/fabro-dev/src/commands/docs_options_reference.rs @@ -301,6 +301,7 @@ cache_input_cost_per_mtok = 0.60 | `provider` | string | None | Provider ID this model belongs to. | | `api_id` | string | model ID | Identifier sent to the provider API. | | `agent_profile` | `"anthropic"` \| `"openai"` \| `"gemini"` | provider profile | Agent profile override for this model. Model overrides take precedence over provider overrides. | +| `billing_policy` | `"openai"` \| `"anthropic"` \| `"gemini"` \| `"none"` | provider policy | Billing algorithm override for this model — for models whose billing family differs from their provider's (e.g. Claude served through OpenRouter bills Anthropic-style cache reads/writes). | | `display_name` | string | model ID | Human-readable model name. | | `family` | string | model ID | Family label used for catalog display and matching. | | `training` | string | None | Training data cutoff label. | diff --git a/lib/crates/fabro-llm/src/adapter_registry.rs b/lib/crates/fabro-llm/src/adapter_registry.rs index a5cc6f535..bf88770d3 100644 --- a/lib/crates/fabro-llm/src/adapter_registry.rs +++ b/lib/crates/fabro-llm/src/adapter_registry.rs @@ -251,7 +251,7 @@ pub fn resolve_route(catalog: &Catalog, model_id_or_alias: &str) -> Option, + /// In-band USD cost from the usage chunk (OpenRouter), surfaced as + /// authoritative on the final response. + cost_usd: Option, } impl StreamState { @@ -46,6 +49,7 @@ impl StreamState { custom_tool_names: super::translate::custom_tool_names(ctx.request), finished: false, rate_limit, + cost_usd: None, } } @@ -65,11 +69,9 @@ impl StreamState { // Capture usage if present (often in a dedicated chunk). if let Some(usage) = &chunk.usage { - self.usage = TokenCounts { - input_tokens: usage.prompt_tokens, - output_tokens: usage.completion_tokens, - ..TokenCounts::default() - }; + self.usage = usage.token_counts(); + // Keep a previously seen cost when a later usage chunk omits it. + self.cost_usd = usage.cost.or(self.cost_usd); } let choices = chunk.choices.as_ref()?; @@ -222,8 +224,8 @@ impl StreamState { raw: None, warnings: vec![], rate_limit: self.rate_limit.clone(), - cost_usd: None, - cost_source: None, + cost_usd: self.cost_usd, + cost_source: super::translate::authoritative_cost_source(self.cost_usd), }; events.push(StreamEvent::finish( diff --git a/lib/crates/fabro-llm/src/codec/openai_compatible/translate.rs b/lib/crates/fabro-llm/src/codec/openai_compatible/translate.rs index d2e5e3422..99c751a22 100644 --- a/lib/crates/fabro-llm/src/codec/openai_compatible/translate.rs +++ b/lib/crates/fabro-llm/src/codec/openai_compatible/translate.rs @@ -2,10 +2,16 @@ use super::wire::{ChatFunction, ChatMessage, ChatToolCall}; use crate::types::{ - ContentPart, FinishReason, Message, Request, ResponseFormat, ResponseFormatType, Role, - ToolChoice, ToolDefinition, + ContentPart, CostSource, FinishReason, Message, Request, ResponseFormat, ResponseFormatType, + Role, ToolChoice, ToolDefinition, }; +/// In-band cost (OpenRouter) is authoritative billing data; the client's +/// catalog estimate never overwrites it. +pub(super) fn authoritative_cost_source(cost_usd: Option) -> Option { + cost_usd.is_some().then_some(CostSource::Authoritative) +} + pub(super) fn map_finish_reason(reason: Option<&str>) -> FinishReason { match reason { Some("stop") | None => FinishReason::Stop, diff --git a/lib/crates/fabro-llm/src/codec/openai_compatible/wire.rs b/lib/crates/fabro-llm/src/codec/openai_compatible/wire.rs index fd4750f0a..f2e859b13 100644 --- a/lib/crates/fabro-llm/src/codec/openai_compatible/wire.rs +++ b/lib/crates/fabro-llm/src/codec/openai_compatible/wire.rs @@ -1,5 +1,7 @@ //! Serde types mirroring the OpenAI Chat Completions wire shapes. +use crate::types::TokenCounts; + #[derive(serde::Serialize)] pub(super) struct ApiRequest { pub model: String, @@ -92,8 +94,65 @@ pub(super) struct ApiFunction { reason = "Field names mirror the provider API payload." )] pub(super) struct ApiUsage { - pub prompt_tokens: i64, + pub prompt_tokens: i64, pub completion_tokens: i64, + /// Tolerant superset: aggregator dialects (OpenRouter) report in-band + /// USD cost and cache/reasoning token detail. Absent on plain providers. + #[serde(default)] + pub cost: Option, + #[serde(default)] + pub prompt_tokens_details: Option, + #[serde(default)] + pub completion_tokens_details: Option, +} + +#[derive(serde::Deserialize)] +pub(super) struct PromptTokensDetails { + #[serde(default)] + pub cached_tokens: Option, + /// OpenRouter-specific: explicit-cache write tokens. + #[serde(default)] + pub cache_write_tokens: Option, +} + +#[derive(serde::Deserialize)] +pub(super) struct CompletionTokensDetails { + #[serde(default)] + pub reasoning_tokens: Option, +} + +impl ApiUsage { + /// Normalize into disjoint [`TokenCounts`] buckets: cached and + /// cache-write detail tokens are subtracted out of `input_tokens`, and + /// reasoning tokens out of `output_tokens`, mirroring the + /// `openai_responses` convention. + pub(super) fn token_counts(&self) -> TokenCounts { + let cached = self + .prompt_tokens_details + .as_ref() + .and_then(|d| d.cached_tokens) + .unwrap_or(0); + let cache_write = self + .prompt_tokens_details + .as_ref() + .and_then(|d| d.cache_write_tokens) + .unwrap_or(0); + let reasoning = self + .completion_tokens_details + .as_ref() + .and_then(|d| d.reasoning_tokens) + .unwrap_or(0); + TokenCounts { + input_tokens: self + .prompt_tokens + .saturating_sub(cached) + .saturating_sub(cache_write), + output_tokens: self.completion_tokens.saturating_sub(reasoning), + reasoning_tokens: reasoning, + cache_read_tokens: cached, + cache_write_tokens: cache_write, + } + } } // --- Streaming response types --- diff --git a/lib/crates/fabro-llm/tests/integration.rs b/lib/crates/fabro-llm/tests/integration.rs index fff3d4050..1407f0f8c 100644 --- a/lib/crates/fabro-llm/tests/integration.rs +++ b/lib/crates/fabro-llm/tests/integration.rs @@ -7,8 +7,10 @@ use std::sync::Arc; use fabro_llm::error::ProviderErrorKind; use fabro_llm::provider::ProviderAdapter; -use fabro_llm::providers::{AnthropicAdapter, GeminiAdapter, OpenAiAdapter}; -use fabro_llm::types::{FinishReason, Message, Request}; +use fabro_llm::providers::{ + AnthropicAdapter, GeminiAdapter, OpenAiAdapter, OpenAiCompatibleAdapter, +}; +use fabro_llm::types::{CostSource, FinishReason, Message, Request}; use fabro_model::Catalog; use fabro_static::EnvVars; @@ -174,6 +176,29 @@ async fn gemini_complete() { assert_eq!(response.provider, "gemini"); } +#[fabro_macros::e2e_test(live("OPENROUTER_API_KEY"))] +async fn openrouter_complete() { + let api_key = + std::env::var(EnvVars::OPENROUTER_API_KEY).expect("OPENROUTER_API_KEY must be set"); + let adapter = OpenAiCompatibleAdapter::new(api_key, "https://openrouter.ai/api/v1") + .with_name("openrouter"); + let request = make_request("deepseek/deepseek-v4-flash"); + let response = adapter.complete(&request).await.unwrap(); + + assert!( + !response.text().is_empty(), + "response text should not be empty" + ); + assert!(response.usage.input_tokens > 0); + assert!(response.usage.output_tokens > 0); + assert_eq!(response.provider, "openrouter"); + assert!( + response.cost_usd.is_some(), + "OpenRouter responses should carry an authoritative usage.cost", + ); + assert_eq!(response.cost_source, Some(CostSource::Authoritative)); +} + async fn run_multi_turn_cache_test( adapter: &dyn ProviderAdapter, model: &str, diff --git a/lib/crates/fabro-llm/tests/it/wire/openai_compatible.rs b/lib/crates/fabro-llm/tests/it/wire/openai_compatible.rs index 23af82687..86d1c51c6 100644 --- a/lib/crates/fabro-llm/tests/it/wire/openai_compatible.rs +++ b/lib/crates/fabro-llm/tests/it/wire/openai_compatible.rs @@ -331,10 +331,10 @@ async fn decode_reasoning_content_as_thinking() { fabro_test::fabro_json_snapshot!(response); } -/// Compat usage reads only prompt/completion tokens; cached-token details are -/// ignored today (parsing them is a 438-redo behavior change). +/// Cached and reasoning detail tokens are split into their own disjoint +/// buckets and subtracted out of input/output. #[tokio::test] -async fn decode_usage_ignores_token_details() { +async fn decode_usage_parses_token_details() { let response = decode_response(serde_json::json!({ "id": "chatcmpl_test", "object": "chat.completion", @@ -357,6 +357,38 @@ async fn decode_usage_ignores_token_details() { fabro_test::fabro_json_snapshot!(response); } +/// OpenRouter usage superset: in-band `cost` becomes an authoritative +/// `cost_usd`, and `cache_write_tokens` lands in its own disjoint bucket. +/// Unmodeled fields (`cost_details`, `audio_tokens`, top-level `provider`, +/// `native_finish_reason`) are tolerated and ignored. +#[tokio::test] +async fn decode_usage_openrouter_cost_and_cache_write() { + let response = decode_response(serde_json::json!({ + "id": "gen_or_test", + "object": "chat.completion", + "created": CREATED_TS, + "model": MODEL, + "provider": "Anthropic", + "choices": [{ + "index": 0, + "message": {"role": "assistant", "content": "ok"}, + "finish_reason": "stop", + "native_finish_reason": "end_turn" + }], + "usage": { + "prompt_tokens": 200, + "completion_tokens": 10, + "total_tokens": 210, + "cost": 0.0042, + "cost_details": {"upstream_inference_cost": null}, + "prompt_tokens_details": {"cached_tokens": 50, "cache_write_tokens": 100, "audio_tokens": 0}, + "completion_tokens_details": {"reasoning_tokens": 0} + } + })) + .await; + fabro_test::fabro_json_snapshot!(response); +} + // --------------------------------------------------------------------------- // Stream // --------------------------------------------------------------------------- @@ -387,6 +419,20 @@ async fn stream_text_happy_path_events() { fabro_test::fabro_json_snapshot!(events); } +/// OpenRouter streams report `cost` in the usage chunk; the Finish response +/// carries it as authoritative, with cached tokens in their own bucket. +#[tokio::test] +async fn stream_usage_openrouter_cost() { + let sse = support::sse_data_transcript(&[ + r#"{"id":"gen_or_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{"role":"assistant","content":"Hi"},"finish_reason":null}]}"#, + r#"{"id":"gen_or_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}"#, + r#"{"id":"gen_or_stream","object":"chat.completion.chunk","created":1700000000,"model":"test-model","choices":[],"usage":{"prompt_tokens":12,"completion_tokens":2,"total_tokens":14,"cost":0.00031,"prompt_tokens_details":{"cached_tokens":4,"cache_write_tokens":0}}}"#, + "[DONE]", + ]); + let (_capture, events) = stream_capture(&base_request(MODEL), &sse).await; + fabro_test::fabro_json_snapshot!(events); +} + #[tokio::test] async fn stream_tool_call_deltas() { let sse = support::sse_data_transcript(&[ diff --git a/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_openrouter_cost_and_cache_write.snap b/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_openrouter_cost_and_cache_write.snap new file mode 100644 index 000000000..f473db75a --- /dev/null +++ b/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_openrouter_cost_and_cache_write.snap @@ -0,0 +1,65 @@ +--- +source: lib/crates/fabro-llm/tests/it/wire/openai_compatible.rs +expression: rendered +--- +{ + "id": "gen_or_test", + "model": "test-model", + "provider": "openai-compatible", + "message": { + "role": "assistant", + "content": [ + { + "kind": "text", + "data": "ok" + } + ] + }, + "finish_reason": "stop", + "usage": { + "input_tokens": 50, + "output_tokens": 10, + "reasoning_tokens": 0, + "cache_read_tokens": 50, + "cache_write_tokens": 100 + }, + "raw": { + "id": "gen_or_test", + "object": "chat.completion", + "created": 1700000000, + "model": "test-model", + "provider": "Anthropic", + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": "ok" + }, + "finish_reason": "stop", + "native_finish_reason": "end_turn" + } + ], + "usage": { + "prompt_tokens": 200, + "completion_tokens": 10, + "total_tokens": 210, + "cost": 0.0042, + "cost_details": { + "upstream_inference_cost": null + }, + "prompt_tokens_details": { + "cached_tokens": 50, + "cache_write_tokens": 100, + "audio_tokens": 0 + }, + "completion_tokens_details": { + "reasoning_tokens": 0 + } + } + }, + "warnings": [], + "rate_limit": null, + "cost_usd": 0.0042, + "cost_source": "authoritative" +} diff --git a/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_ignores_token_details.snap b/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_parses_token_details.snap similarity index 90% rename from lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_ignores_token_details.snap rename to lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_parses_token_details.snap index fecf6e4be..c212ac7c2 100644 --- a/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_ignores_token_details.snap +++ b/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__decode_usage_parses_token_details.snap @@ -17,10 +17,10 @@ expression: rendered }, "finish_reason": "length", "usage": { - "input_tokens": 100, - "output_tokens": 50, - "reasoning_tokens": 0, - "cache_read_tokens": 0, + "input_tokens": 20, + "output_tokens": 30, + "reasoning_tokens": 20, + "cache_read_tokens": 80, "cache_write_tokens": 0 }, "raw": { diff --git a/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__stream_usage_openrouter_cost.snap b/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__stream_usage_openrouter_cost.snap new file mode 100644 index 000000000..da24b9b97 --- /dev/null +++ b/lib/crates/fabro-llm/tests/it/wire/snapshots/it__wire__openai_compatible__stream_usage_openrouter_cost.snap @@ -0,0 +1,57 @@ +--- +source: lib/crates/fabro-llm/tests/it/wire/openai_compatible.rs +expression: rendered +--- +[ + { + "type": "text_start", + "text_id": null + }, + { + "type": "text_delta", + "delta": "Hi", + "text_id": null + }, + { + "type": "text_end", + "text_id": null + }, + { + "type": "finish", + "finish_reason": "stop", + "usage": { + "input_tokens": 8, + "output_tokens": 2, + "reasoning_tokens": 0, + "cache_read_tokens": 4, + "cache_write_tokens": 0 + }, + "response": { + "id": "gen_or_stream", + "model": "test-model", + "provider": "openai-compatible", + "message": { + "role": "assistant", + "content": [ + { + "kind": "text", + "data": "Hi" + } + ] + }, + "finish_reason": "stop", + "usage": { + "input_tokens": 8, + "output_tokens": 2, + "reasoning_tokens": 0, + "cache_read_tokens": 4, + "cache_write_tokens": 0 + }, + "raw": null, + "warnings": [], + "rate_limit": null, + "cost_usd": 0.00031, + "cost_source": "authoritative" + } + } +] diff --git a/lib/crates/fabro-model/src/billing.rs b/lib/crates/fabro-model/src/billing.rs index 022ba313d..d75992d7a 100644 --- a/lib/crates/fabro-model/src/billing.rs +++ b/lib/crates/fabro-model/src/billing.rs @@ -446,7 +446,7 @@ impl Catalog { pricing_for_model_costs( model, provider.id.clone(), - provider.billing_policy, + settings.billing_policy, model_ref.speed, &costs, ) @@ -458,8 +458,9 @@ impl Catalog { model_ref: &ModelRef, tokens: &TokenCounts, ) -> Option { - self.provider(&model_ref.provider) - .and_then(|provider| ModelBillingFacts::for_policy(provider.billing_policy, tokens)) + let policy = + self.effective_billing_policy(&model_ref.provider, Some(&model_ref.model_id))?; + ModelBillingFacts::for_policy(policy, tokens) } /// Price a partial token sample for `model` using catalog pricing. @@ -729,6 +730,77 @@ mod tests { } } + #[test] + fn model_billing_policy_override_changes_the_billing_algorithm() { + let catalog = catalog_from_toml( + r#" +[providers.aggregator] +display_name = "Aggregator" +adapter = "openai_compatible" +base_url = "https://aggregator.test/v1" + +[models."claude-via-aggregator"] +provider = "aggregator" +billing_policy = "anthropic" +display_name = "Claude (via Aggregator)" +family = "claude" +default = true + +[models."claude-via-aggregator".limits] +context_window = 200000 + +[models."claude-via-aggregator".features] +tools = true +vision = false +reasoning = false + +[models."claude-via-aggregator".costs] +input_cost_per_mtok = 3.0 +output_cost_per_mtok = 15.0 +cache_input_cost_per_mtok = 0.3 + +[models."plain-model"] +provider = "aggregator" +display_name = "Plain" +family = "plain" + +[models."plain-model".limits] +context_window = 100000 + +[models."plain-model".features] +tools = false +vision = false +reasoning = false + +[models."plain-model".costs] +input_cost_per_mtok = 3.0 +output_cost_per_mtok = 15.0 +cache_input_cost_per_mtok = 0.3 +"#, + ); + + let tokens = TokenCounts { + cache_write_tokens: 1_000_000, + ..TokenCounts::default() + }; + let claude = ModelRef { + provider: ProviderId::new("aggregator"), + model_id: "claude-via-aggregator".to_string(), + speed: None, + }; + let plain = ModelRef { + provider: ProviderId::new("aggregator"), + model_id: "plain-model".to_string(), + speed: None, + }; + + // The override bills Anthropic-style: cache writes at 1.25x input + // ($3/MTok -> $3.75/MTok -> $3.75 for 1M write tokens). + assert_eq!(catalog.price_tokens(&claude, &tokens), Some(3_750_000)); + // The provider's default OpenAI policy has no cache-write charge. + assert_eq!(catalog.price_tokens(&plain, &tokens), Some(0)); + } + #[test] fn billed_token_counts_add_counts_accumulates_cost_when_known() { let mut counts = BilledTokenCounts { diff --git a/lib/crates/fabro-model/src/catalog.rs b/lib/crates/fabro-model/src/catalog.rs index 636ef1ea1..480759d58 100644 --- a/lib/crates/fabro-model/src/catalog.rs +++ b/lib/crates/fabro-model/src/catalog.rs @@ -79,6 +79,11 @@ pub struct ModelCatalogSettings { /// accepted today. #[serde(default)] pub codec: Option, + /// Billing family for this model, overriding the provider's policy + /// (e.g. Anthropic cache billing for a Claude model served through an + /// aggregator whose other models bill OpenAI-style). + #[serde(default)] + pub billing_policy: Option, #[serde(default)] pub agent_profile: Option, #[serde(default)] @@ -519,14 +524,17 @@ pub struct CatalogModelControls { #[derive(Debug, Clone, PartialEq)] pub struct CatalogModelSettings { - pub api_id: String, + pub api_id: String, /// Wire dialect for this model's route (the provider codec unless the /// model row overrides it). - pub codec: CodecKind, - pub agent_profile: AgentProfileKind, - pub controls: CatalogModelControls, - pub speed_costs: HashMap, - probe: bool, + pub codec: CodecKind, + /// Billing family for this model (the provider policy unless the model + /// row overrides it). + pub billing_policy: BillingPolicy, + pub agent_profile: AgentProfileKind, + pub controls: CatalogModelControls, + pub speed_costs: HashMap, + probe: bool, } #[derive(Debug, thiserror::Error)] @@ -933,6 +941,24 @@ impl Catalog { Some(model_codec.unwrap_or(provider.codec)) } + /// The billing family for `model_id_or_alias` on `provider_id`: the model + /// row's policy when one is configured, otherwise the provider's (unknown + /// passthrough model ids keep the provider policy). + #[must_use] + pub fn effective_billing_policy( + &self, + provider_id: &ProviderId, + model_id_or_alias: Option<&str>, + ) -> Option { + let provider = self.provider(provider_id)?; + let model_policy = model_id_or_alias + .and_then(|model_id| self.get(model_id)) + .filter(|model| model.provider == provider.id) + .and_then(|model| self.model_settings.get(&model.id)) + .map(|settings| settings.billing_policy); + Some(model_policy.unwrap_or(provider.billing_policy)) + } + /// List all models, optionally filtered by provider. #[must_use] pub fn list(&self, provider: Option<&ProviderId>) -> Vec<&Model> { @@ -1185,6 +1211,7 @@ fn merge_model_settings( provider: higher.provider.or(fallback.provider), api_id: higher.api_id.or(fallback.api_id), codec: higher.codec.or(fallback.codec), + billing_policy: higher.billing_policy.or(fallback.billing_policy), agent_profile: higher.agent_profile.or(fallback.agent_profile), display_name: higher.display_name.or(fallback.display_name), family: higher.family.or(fallback.family), @@ -1485,6 +1512,7 @@ fn build_model( .clone() .unwrap_or_else(|| model_id.to_string()), codec: resolve_model_codec(model_id, provider, settings.codec)?, + billing_policy: settings.billing_policy.unwrap_or(provider.billing_policy), agent_profile: settings.agent_profile.unwrap_or(provider.agent_profile), controls, speed_costs, @@ -1886,6 +1914,57 @@ reasoning = false assert_eq!(model.provider, ProviderId::new("acme")); } + #[test] + fn builtin_openrouter_provider_is_opt_in() { + let openrouter = ProviderId::new("openrouter"); + let builtin = Catalog::builtin(); + + assert!(builtin.provider(&openrouter).is_none()); + assert!(builtin.list(Some(&openrouter)).is_empty()); + + let catalog = Catalog::from_builtin_with_overrides(&minimal_settings( + r" +[providers.openrouter] +enabled = true +", + )) + .expect("enabled OpenRouter override should build from the built-in provider settings"); + + let provider = catalog + .provider(&openrouter) + .expect("enabled OpenRouter provider should be present"); + assert_eq!(provider.adapter, AdapterKind::OpenAiCompatible); + assert_eq!(provider.codec, CodecKind::OpenAiCompatible); + assert_eq!( + provider.base_url.as_deref(), + Some("https://openrouter.ai/api/v1") + ); + assert_eq!(provider.billing_policy, BillingPolicy::OpenAi); + + // Claude rows override the provider's OpenAI billing default; + // open-weights rows inherit it. + assert_eq!( + catalog + .model_settings("anthropic/claude-sonnet-4-6") + .unwrap() + .billing_policy, + BillingPolicy::Anthropic + ); + assert_eq!( + catalog + .model_settings("deepseek/deepseek-v4-flash") + .unwrap() + .billing_policy, + BillingPolicy::OpenAi + ); + assert_eq!( + catalog + .default_for_provider(&openrouter) + .map(|model| model.id.as_str()), + Some("anthropic/claude-sonnet-4-6") + ); + } + #[test] fn builtin_ollama_provider_is_opt_in() { let ollama = ProviderId::new("ollama"); diff --git a/lib/crates/fabro-model/src/catalog/providers/openrouter.toml b/lib/crates/fabro-model/src/catalog/providers/openrouter.toml new file mode 100644 index 000000000..957f6aac2 --- /dev/null +++ b/lib/crates/fabro-model/src/catalog/providers/openrouter.toml @@ -0,0 +1,374 @@ +[providers.openrouter] +display_name = "OpenRouter" +adapter = "openai_compatible" +api_key_url = "https://openrouter.ai/keys" +base_url = "https://openrouter.ai/api/v1" +priority = 25 +enabled = false + +[providers.openrouter.auth] +credentials = ["env:OPENROUTER_API_KEY", "vault:OPENROUTER_API_KEY"] + +# Attribution headers (HTTP-Referer, X-Title) are NOT sent by default. +# Self-hosted Fabro installations stay anonymous on OpenRouter's public +# leaderboard unless the operator opts in. To advertise, add to +# settings.toml: +# +# [llm.providers.openrouter.extra_headers] +# "HTTP-Referer" = { literal = "https://your-site.example" } +# "X-Title" = { literal = "Your App" } +# +# To enable OpenRouter, add the following to ~/.fabro/settings.toml: +# +# [llm.providers.openrouter] +# enabled = true +# +# Then run `fabro provider login openrouter` to store the API key, +# or set the OPENROUTER_API_KEY environment variable. + +# ---------- Anthropic via OpenRouter ---------- +# +# Claude models bill Anthropic-style (cache read/write pricing), so these +# rows override the provider's OpenAI-default billing_policy. Costs are +# best-effort estimates; OpenRouter returns the authoritative usage.cost +# in-band on every response. + +[models."anthropic/claude-opus-4-7"] +provider = "openrouter" +api_id = "anthropic/claude-opus-4.7" +display_name = "Claude Opus 4.7 (via OpenRouter)" +family = "claude-4" +billing_policy = "anthropic" + +[models."anthropic/claude-opus-4-7".limits] +context_window = 1000000 +max_output = 128000 + +[models."anthropic/claude-opus-4-7".features] +tools = true +vision = true +reasoning = true +prompt_cache = true + +[models."anthropic/claude-opus-4-7".costs] +input_cost_per_mtok = 5.0 +output_cost_per_mtok = 25.0 +cache_input_cost_per_mtok = 0.5 + +[models."anthropic/claude-sonnet-4-6"] +provider = "openrouter" +api_id = "anthropic/claude-sonnet-4.6" +display_name = "Claude Sonnet 4.6 (via OpenRouter)" +family = "claude-4" +billing_policy = "anthropic" +default = true + +[models."anthropic/claude-sonnet-4-6".limits] +context_window = 1000000 +max_output = 64000 + +[models."anthropic/claude-sonnet-4-6".features] +tools = true +vision = true +reasoning = true +prompt_cache = true + +[models."anthropic/claude-sonnet-4-6".costs] +input_cost_per_mtok = 3.0 +output_cost_per_mtok = 15.0 +cache_input_cost_per_mtok = 0.3 + +[models."anthropic/claude-haiku-4-5"] +provider = "openrouter" +api_id = "anthropic/claude-haiku-4.5" +display_name = "Claude Haiku 4.5 (via OpenRouter)" +family = "claude-4" +billing_policy = "anthropic" +small_default = true + +[models."anthropic/claude-haiku-4-5".limits] +context_window = 200000 +max_output = 8192 + +[models."anthropic/claude-haiku-4-5".features] +tools = true +vision = true +reasoning = false +prompt_cache = true + +[models."anthropic/claude-haiku-4-5".costs] +input_cost_per_mtok = 1.0 +output_cost_per_mtok = 5.0 +cache_input_cost_per_mtok = 0.1 + +# ---------- OpenAI via OpenRouter ---------- + +[models."openai/gpt-5.4"] +provider = "openrouter" +api_id = "openai/gpt-5.4" +display_name = "GPT-5.4 (via OpenRouter)" +family = "gpt-5" + +[models."openai/gpt-5.4".limits] +context_window = 1050000 +max_output = 32768 + +[models."openai/gpt-5.4".features] +tools = true +vision = true +reasoning = true + +[models."openai/gpt-5.4".costs] +input_cost_per_mtok = 2.5 +output_cost_per_mtok = 15.0 + +[models."openai/gpt-5.5"] +provider = "openrouter" +api_id = "openai/gpt-5.5" +display_name = "GPT-5.5 (via OpenRouter)" +family = "gpt-5" + +[models."openai/gpt-5.5".limits] +context_window = 1050000 +max_output = 32768 + +[models."openai/gpt-5.5".features] +tools = true +vision = true +reasoning = true + +[models."openai/gpt-5.5".costs] +input_cost_per_mtok = 5.0 +output_cost_per_mtok = 30.0 + +# ---------- Google Gemini via OpenRouter ---------- + +[models."google/gemini-3.1-pro-preview"] +provider = "openrouter" +api_id = "google/gemini-3.1-pro-preview" +display_name = "Gemini 3.1 Pro Preview (via OpenRouter)" +family = "gemini-3" + +[models."google/gemini-3.1-pro-preview".limits] +context_window = 1048576 +max_output = 65536 + +[models."google/gemini-3.1-pro-preview".features] +tools = true +vision = true +reasoning = true + +[models."google/gemini-3.1-pro-preview".costs] +input_cost_per_mtok = 2.0 +output_cost_per_mtok = 12.0 + +[models."google/gemini-3.5-flash"] +provider = "openrouter" +api_id = "google/gemini-3.5-flash" +display_name = "Gemini 3.5 Flash (via OpenRouter)" +family = "gemini-3" + +[models."google/gemini-3.5-flash".limits] +context_window = 1048576 +max_output = 65536 + +[models."google/gemini-3.5-flash".features] +tools = true +vision = true +reasoning = false + +[models."google/gemini-3.5-flash".costs] +input_cost_per_mtok = 1.5 +output_cost_per_mtok = 9.0 + +# ---------- Open-weights models ---------- + +[models."xiaomi/mimo-v2.5-pro"] +provider = "openrouter" +api_id = "xiaomi/mimo-v2.5-pro" +display_name = "Xiaomi MiMo v2.5 Pro" +family = "mimo-v2" + +[models."xiaomi/mimo-v2.5-pro".limits] +context_window = 1050000 +max_output = 16384 + +[models."xiaomi/mimo-v2.5-pro".features] +tools = true +vision = false +reasoning = false + +[models."xiaomi/mimo-v2.5-pro".costs] +input_cost_per_mtok = 0.435 +output_cost_per_mtok = 0.87 + +[models."minimax/minimax-m2.7"] +provider = "openrouter" +api_id = "minimax/minimax-m2.7" +display_name = "MiniMax M2.7" +family = "minimax-m2" + +[models."minimax/minimax-m2.7".limits] +context_window = 200000 +max_output = 16384 + +[models."minimax/minimax-m2.7".features] +tools = true +vision = false +reasoning = false + +[models."minimax/minimax-m2.7".costs] +input_cost_per_mtok = 0.28 +output_cost_per_mtok = 1.20 + +[models."deepseek/deepseek-v4-pro"] +provider = "openrouter" +api_id = "deepseek/deepseek-v4-pro" +display_name = "DeepSeek V4 Pro" +family = "deepseek-v4" + +[models."deepseek/deepseek-v4-pro".limits] +context_window = 1050000 +max_output = 16384 + +[models."deepseek/deepseek-v4-pro".features] +tools = true +vision = false +reasoning = true + +[models."deepseek/deepseek-v4-pro".costs] +input_cost_per_mtok = 0.435 +output_cost_per_mtok = 0.87 + +[models."deepseek/deepseek-v4-flash"] +provider = "openrouter" +api_id = "deepseek/deepseek-v4-flash" +display_name = "DeepSeek V4 Flash" +family = "deepseek-v4" + +[models."deepseek/deepseek-v4-flash".limits] +context_window = 1050000 +max_output = 16384 + +[models."deepseek/deepseek-v4-flash".features] +tools = true +vision = false +reasoning = false + +[models."deepseek/deepseek-v4-flash".costs] +input_cost_per_mtok = 0.10 +output_cost_per_mtok = 0.20 + +[models."moonshotai/kimi-k2.6"] +provider = "openrouter" +api_id = "moonshotai/kimi-k2.6" +display_name = "Kimi K2.6" +family = "kimi-k2" + +[models."moonshotai/kimi-k2.6".limits] +context_window = 262144 +max_output = 16384 + +[models."moonshotai/kimi-k2.6".features] +tools = true +vision = false +reasoning = false + +[models."moonshotai/kimi-k2.6".costs] +input_cost_per_mtok = 0.73 +output_cost_per_mtok = 3.49 + +[models."qwen/qwen3-coder"] +provider = "openrouter" +api_id = "qwen/qwen3-coder" +display_name = "Qwen3 Coder" +family = "qwen3" + +[models."qwen/qwen3-coder".limits] +context_window = 1050000 +max_output = 16384 + +[models."qwen/qwen3-coder".features] +tools = true +vision = false +reasoning = false + +[models."qwen/qwen3-coder".costs] +input_cost_per_mtok = 0.22 +output_cost_per_mtok = 1.80 + +[models."qwen/qwen3.6-flash"] +provider = "openrouter" +api_id = "qwen/qwen3.6-flash" +display_name = "Qwen3.6 Flash" +family = "qwen3" + +[models."qwen/qwen3.6-flash".limits] +context_window = 1000000 +max_output = 16384 + +[models."qwen/qwen3.6-flash".features] +tools = true +vision = false +reasoning = false + +[models."qwen/qwen3.6-flash".costs] +input_cost_per_mtok = 0.1875 +output_cost_per_mtok = 1.125 + +[models."z-ai/glm-4.6"] +provider = "openrouter" +api_id = "z-ai/glm-4.6" +display_name = "GLM 4.6" +family = "glm-4" + +[models."z-ai/glm-4.6".limits] +context_window = 203000 +max_output = 16384 + +[models."z-ai/glm-4.6".features] +tools = true +vision = false +reasoning = false + +[models."z-ai/glm-4.6".costs] +input_cost_per_mtok = 0.43 +output_cost_per_mtok = 1.74 + +[models."nvidia/nemotron-3-super-120b-a12b"] +provider = "openrouter" +api_id = "nvidia/nemotron-3-super-120b-a12b" +display_name = "NVIDIA Nemotron 3 Super 120B" +family = "nemotron-3" + +[models."nvidia/nemotron-3-super-120b-a12b".limits] +context_window = 1000000 +max_output = 16384 + +[models."nvidia/nemotron-3-super-120b-a12b".features] +tools = true +vision = false +reasoning = false + +[models."nvidia/nemotron-3-super-120b-a12b".costs] +input_cost_per_mtok = 0.09 +output_cost_per_mtok = 0.45 + +[models."mistralai/devstral-2512"] +provider = "openrouter" +api_id = "mistralai/devstral-2512" +display_name = "Devstral 2512" +family = "devstral" + +[models."mistralai/devstral-2512".limits] +context_window = 262144 +max_output = 16384 + +[models."mistralai/devstral-2512".features] +tools = true +vision = false +reasoning = false + +[models."mistralai/devstral-2512".costs] +input_cost_per_mtok = 0.40 +output_cost_per_mtok = 2.00 diff --git a/lib/crates/fabro-redact/data/gitleaks.toml b/lib/crates/fabro-redact/data/gitleaks.toml index 055e09305..00b7cb9ca 100644 --- a/lib/crates/fabro-redact/data/gitleaks.toml +++ b/lib/crates/fabro-redact/data/gitleaks.toml @@ -2546,6 +2546,13 @@ regex = '''\b(sk-[a-zA-Z0-9]{20}T3BlbkFJ[a-zA-Z0-9]{20})(?:['|\"|\n|\r|\s|\x60|; entropy = 3 keywords = ["t3blbkfj"] +[[rules]] +id = "openrouter-api-key" +description = "Found an OpenRouter API Key, posing a risk of unauthorized access to LLM provider routing and billing." +regex = '''\b(sk-or-v1-[0-9a-f]{64})(?:['|\"|\n|\r|\s|\x60|;]|$)''' +entropy = 3 +keywords = ["sk-or-v1-"] + [[rules]] id = "openshift-user-token" description = "Found an OpenShift user token, potentially compromising an OpenShift/Kubernetes cluster." diff --git a/lib/crates/fabro-static/src/env_vars.rs b/lib/crates/fabro-static/src/env_vars.rs index eec275f72..315de6477 100644 --- a/lib/crates/fabro-static/src/env_vars.rs +++ b/lib/crates/fabro-static/src/env_vars.rs @@ -56,6 +56,7 @@ impl EnvVars { pub const OPENAI_PROJECT: &'static str = "OPENAI_PROJECT"; pub const OPENAI_ORG_ID: &'static str = "OPENAI_ORG_ID"; pub const OPENAI_PROJECT_ID: &'static str = "OPENAI_PROJECT_ID"; + pub const OPENROUTER_API_KEY: &'static str = "OPENROUTER_API_KEY"; pub const ZAI_API_KEY: &'static str = "ZAI_API_KEY"; // GitHub, OAuth, and Slack @@ -193,6 +194,7 @@ mod tests { EnvVars::OPENAI_PROJECT, EnvVars::OPENAI_ORG_ID, EnvVars::OPENAI_PROJECT_ID, + EnvVars::OPENROUTER_API_KEY, EnvVars::ZAI_API_KEY, EnvVars::GH_TOKEN, EnvVars::GITHUB_APP_CLIENT_SECRET, diff --git a/lib/crates/fabro-static/src/secret_registry.rs b/lib/crates/fabro-static/src/secret_registry.rs index 8534636bd..b6770aeab 100644 --- a/lib/crates/fabro-static/src/secret_registry.rs +++ b/lib/crates/fabro-static/src/secret_registry.rs @@ -28,6 +28,7 @@ const OPTIONAL_VAULT_SECRETS: &[&str] = &[ EnvVars::KIMI_API_KEY, EnvVars::MINIMAX_API_KEY, EnvVars::OPENAI_API_KEY, + EnvVars::OPENROUTER_API_KEY, EnvVars::ZAI_API_KEY, EnvVars::DAYTONA_API_KEY, ]; @@ -90,6 +91,7 @@ mod tests { EnvVars::KIMI_API_KEY, EnvVars::MINIMAX_API_KEY, EnvVars::OPENAI_API_KEY, + EnvVars::OPENROUTER_API_KEY, EnvVars::ZAI_API_KEY, ] { assert_eq!(