Expand OpenAI twin coverage across integration tests

Add shared twin scenario helpers and use them to cover OpenAI-backed
CLI, agent parity, workflow, and exec integration paths. This brings the
worktree implementation back into the main checkout as a single commit.
This commit is contained in:
Bryan Helmkamp 2026-04-01 10:13:24 -04:00
parent 74344f8e9f
commit a6e83f551e
No known key found for this signature in database
8 changed files with 1037 additions and 43 deletions

1
Cargo.lock generated
View file

@ -1944,6 +1944,7 @@ dependencies = [
"insta",
"regex",
"reqwest",
"serde_json",
"tempfile",
"tokio",
"twin-github",

View file

@ -1,3 +1,4 @@
use std::collections::HashMap;
use std::fmt::Write as _;
use std::path::Path;
use std::sync::Arc;
@ -8,10 +9,18 @@ use fabro_agent::{
SessionConfig, SubAgentManager, WebFetchSummarizer,
};
use fabro_llm::client::Client;
use fabro_llm::provider::Provider;
use fabro_llm::provider::{Provider, ProviderAdapter};
use fabro_llm::providers::OpenAiAdapter;
use fabro_model::ModelRef;
use fabro_test::{TwinScenario, TwinScenarios, TwinToolCall, twin_openai};
use tokio::sync::Mutex as AsyncMutex;
#[derive(Clone)]
struct OpenAiTwinConfig {
base_url: String,
api_key: String,
}
fn summarizer_model_id(provider: Provider) -> ModelRef {
match provider {
Provider::OpenAi
@ -57,8 +66,13 @@ fn build_profile(provider: Provider, model: &str, client: &Client) -> Box<dyn Ag
}
}
async fn make_session(provider: Provider, model: &str, cwd: &Path) -> Session {
let client = Client::from_env().await.expect("Client::from_env failed");
async fn make_session(
provider: Provider,
model: &str,
cwd: &Path,
twin: Option<OpenAiTwinConfig>,
) -> Session {
let client = make_client(provider, twin.as_ref()).await;
let mut profile = build_profile(provider, model, &client);
let env = Arc::new(LocalSandbox::new(cwd.to_path_buf()));
@ -115,20 +129,62 @@ async fn make_session_with_config(
model: &str,
cwd: &Path,
config: SessionConfig,
twin: Option<OpenAiTwinConfig>,
) -> Session {
let client = Client::from_env().await.expect("Client::from_env failed");
let client = make_client(provider, twin.as_ref()).await;
let profile: Arc<dyn AgentProfile> = Arc::from(build_profile(provider, model, &client));
let env = Arc::new(LocalSandbox::new(cwd.to_path_buf()));
Session::new(client, profile, env, config, None)
}
async fn make_client(provider: Provider, twin: Option<&OpenAiTwinConfig>) -> Client {
if provider == Provider::OpenAi && fabro_test::TestMode::from_env().is_twin() {
return make_twin_client(twin.expect("openai twin config should be provided")).await;
}
Client::from_env().await.expect("Client::from_env failed")
}
async fn make_twin_client(twin: &OpenAiTwinConfig) -> Client {
let adapter: Arc<dyn ProviderAdapter> =
Arc::new(OpenAiAdapter::new(twin.api_key.clone()).with_base_url(twin.base_url.clone()));
let mut providers: HashMap<String, Arc<dyn ProviderAdapter>> = HashMap::new();
providers.insert("openai".to_string(), adapter);
Client::new(providers, Some("openai".to_string()), Vec::new())
}
macro_rules! provider_test {
($scenario:ident, $provider:expr, $model:expr, $prefix:ident, keys = [$($key:expr),+ $(,)?]) => {
paste::paste! {
#[fabro_macros::e2e_test($(live($key)),+)]
async fn [<$prefix _ $scenario>]() {
let tmp = tempfile::tempdir().expect("failed to create tempdir");
let mut session = make_session($provider, $model, tmp.path()).await;
let mut session = make_session($provider, $model, tmp.path(), None).await;
session.initialize().await;
[<scenario_ $scenario>](&mut session, tmp.path()).await;
}
}
};
}
macro_rules! openai_twin_provider_test {
($scenario:ident) => {
paste::paste! {
#[fabro_macros::e2e_test(twin, live("OPENAI_API_KEY"))]
async fn [<openai_twin_ $scenario>]() {
let tmp = tempfile::tempdir().expect("failed to create tempdir");
let (base_url, api_key) = fabro_test::e2e_openai!();
let twin = OpenAiTwinConfig { base_url, api_key };
if fabro_test::TestMode::from_env().is_twin() {
load_openai_twin_scenario(stringify!($scenario), &twin.api_key, tmp.path())
.await;
}
let mut session = make_session(
Provider::OpenAi,
"gpt-5.4-mini",
tmp.path(),
Some(twin),
).await;
session.initialize().await;
[<scenario_ $scenario>](&mut session, tmp.path()).await;
}
@ -145,13 +201,6 @@ macro_rules! provider_tests {
anthropic,
keys = ["ANTHROPIC_API_KEY"]
);
provider_test!(
$scenario,
Provider::OpenAi,
"gpt-5.4-mini",
openai,
keys = ["OPENAI_API_KEY"]
);
provider_test!(
$scenario,
Provider::Gemini,
@ -193,16 +242,26 @@ macro_rules! provider_tests {
}
provider_tests!(simple_file_creation);
openai_twin_provider_test!(simple_file_creation);
provider_tests!(read_and_edit_file);
openai_twin_provider_test!(read_and_edit_file);
provider_tests!(multi_file_edit);
openai_twin_provider_test!(multi_file_edit);
provider_tests!(shell_execution);
openai_twin_provider_test!(shell_execution);
provider_tests!(shell_timeout);
openai_twin_provider_test!(shell_timeout);
provider_tests!(grep_and_glob);
openai_twin_provider_test!(grep_and_glob);
provider_tests!(tool_output_truncation);
openai_twin_provider_test!(tool_output_truncation);
provider_tests!(parallel_tool_calls);
openai_twin_provider_test!(parallel_tool_calls);
provider_tests!(steering_before_input);
openai_twin_provider_test!(steering_before_input);
provider_tests!(steering_mid_task);
provider_tests!(follow_up);
openai_twin_provider_test!(follow_up);
provider_tests!(subagent_spawn);
provider_test!(
@ -316,6 +375,7 @@ provider_test!(
// - loop_detection: needs custom config, tested separately below.
provider_tests!(error_recovery);
openai_twin_provider_test!(error_recovery);
// gpt-5-mini is too weak to reliably apply precise file edits (uses apply_patch, not edit_file).
macro_rules! non_openai_provider_tests {
@ -595,7 +655,8 @@ macro_rules! reasoning_effort_tests {
reasoning_effort: Some(fabro_llm::types::ReasoningEffort::Low),
..SessionConfig::default()
};
let mut session = make_session_with_config($provider, $model, tmp.path(), config).await;
let mut session =
make_session_with_config($provider, $model, tmp.path(), config, None).await;
session.initialize().await;
session
.process_input("Say hello")
@ -672,7 +733,8 @@ macro_rules! loop_detection_tests {
loop_detection_window: 3,
..SessionConfig::default()
};
let mut session = make_session_with_config($provider, $model, tmp.path(), config).await;
let mut session =
make_session_with_config($provider, $model, tmp.path(), config, None).await;
session.initialize().await;
session
.process_input("Repeatedly read the file /dev/null")
@ -727,6 +789,126 @@ loop_detection_tests!(
keys = ["INCEPTION_API_KEY"]
);
async fn load_openai_twin_scenario(name: &str, namespace: &str, cwd: &Path) {
let scenarios = match name {
"simple_file_creation" => TwinScenarios::new(namespace.to_string()).scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains("Create a file called hello.txt containing 'Hello'")
.tool_call(TwinToolCall::write_file("hello.txt", "Hello"))
.text("Done."),
),
"read_and_edit_file" => TwinScenarios::new(namespace.to_string())
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains("Read data.txt and replace its content with 'new content'")
.tool_call(TwinToolCall::read_file("data.txt")),
)
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.tool_call(TwinToolCall::write_file("data.txt", "new content"))
.text("Done."),
),
"multi_file_edit" => TwinScenarios::new(namespace.to_string())
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains(
"Read a.txt and b.txt, then replace the content of a.txt with 'AAA' and b.txt with 'BBB'",
)
.tool_calls(vec![
TwinToolCall::read_file("a.txt"),
TwinToolCall::read_file("b.txt"),
]),
)
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.tool_calls(vec![
TwinToolCall::write_file("a.txt", "AAA"),
TwinToolCall::write_file("b.txt", "BBB"),
])
.text("Done."),
),
"shell_execution" => TwinScenarios::new(namespace.to_string()).scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains(
"Run the command `echo hello_from_shell` in the shell and tell me what it printed",
)
.tool_call(TwinToolCall::shell("echo hello_from_shell"))
.text("It printed hello_from_shell."),
),
"shell_timeout" => TwinScenarios::new(namespace.to_string()).scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains("Run the command `sleep 999` with a 1-second timeout")
.tool_call(TwinToolCall::shell_with_timeout("sleep 999", 1000))
.text("The command timed out."),
),
"grep_and_glob" => TwinScenarios::new(namespace.to_string())
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains(
"Search for files containing 'needle_pattern_xyz' and tell me which file has it",
)
.tool_calls(vec![
TwinToolCall::glob_pattern("*.txt", cwd.display().to_string()),
TwinToolCall::grep_pattern("needle_pattern_xyz", "."),
]),
)
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.text("target.txt contains needle_pattern_xyz."),
),
"tool_output_truncation" => TwinScenarios::new(namespace.to_string()).scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains("Read the file big.txt and tell me how many lines it has")
.tool_call(TwinToolCall::read_file("big.txt"))
.text("The file has 10000 lines."),
),
"parallel_tool_calls" => TwinScenarios::new(namespace.to_string()).scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains("Read one.txt, two.txt, and three.txt and tell me what each contains")
.tool_calls(vec![
TwinToolCall::read_file("one.txt"),
TwinToolCall::read_file("two.txt"),
TwinToolCall::read_file("three.txt"),
])
.text("one: content_one, two: content_two, three: content_three"),
),
"steering_before_input" => TwinScenarios::new(namespace.to_string()).scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains("Count from 1 to 100, one number per line")
.text("DONE"),
),
"follow_up" => TwinScenarios::new(namespace.to_string())
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains("Create a file called first.txt containing 'first'")
.tool_call(TwinToolCall::write_file("first.txt", "first"))
.text("Created first.txt."),
)
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains("Create a file called second.txt containing 'second'")
.tool_call(TwinToolCall::write_file("second.txt", "second"))
.text("Created second.txt."),
),
"error_recovery" => TwinScenarios::new(namespace.to_string())
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.input_contains(
"Try to read a file called nonexistent_file.txt. If it doesn't exist, create it with the content 'recovered'",
)
.tool_call(TwinToolCall::read_file("nonexistent_file.txt")),
)
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.tool_call(TwinToolCall::write_file("nonexistent_file.txt", "recovered"))
.text("Created the file."),
),
other => panic!("missing openai twin scenario for {other}"),
};
scenarios.load(twin_openai().await).await;
}
// ---------------------------------------------------------------------------
// Scenario 14: error_recovery
// ---------------------------------------------------------------------------

View file

@ -1,6 +1,14 @@
use fabro_test::{fabro_snapshot, test_context};
use std::process::Output;
use fabro_test::{fabro_snapshot, test_context, twin_openai};
use predicates::prelude::*;
async fn run_success_output(mut cmd: assert_cmd::Command) -> Output {
tokio::task::spawn_blocking(move || cmd.assert().success().get_output().clone())
.await
.expect("blocking command task should complete")
}
#[test]
fn help() {
let context = test_context!();
@ -69,6 +77,32 @@ fn live_doctor() {
context.doctor().assert().success();
}
#[fabro_macros::e2e_test(twin)]
async fn twin_doctor() {
let context = test_context!();
let twin = twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
let mut cmd = context.doctor();
cmd.arg("--verbose");
cmd.env_clear();
cmd.env("NO_COLOR", "1");
cmd.env("HOME", &context.home_dir);
cmd.env("FABRO_NO_UPGRADE_CHECK", "true");
cmd.env("FABRO_STORAGE_DIR", &context.storage_dir);
cmd.env(
"PATH",
"/usr/local/bin:/opt/homebrew/bin:/usr/bin:/bin:/usr/sbin:/sbin",
);
twin.configure_command(&mut cmd, &namespace);
let output = run_success_output(cmd).await;
let stdout = String::from_utf8(output.stdout).unwrap();
assert!(
stdout.to_lowercase().contains("openai connectivity: ok"),
"expected verbose doctor output to include openai probe success, got: {stdout}"
);
}
#[test]
fn doctor_no_color_when_no_color_set() {
let context = test_context!();

View file

@ -1,5 +1,13 @@
use std::process::Output;
use fabro_test::{fabro_snapshot, test_context};
async fn run_success_output(mut cmd: assert_cmd::Command) -> Output {
tokio::task::spawn_blocking(move || cmd.assert().success().get_output().clone())
.await
.expect("blocking command task should complete")
}
#[test]
fn help() {
let context = test_context!();
@ -241,3 +249,163 @@ fn exec_read_and_edit() {
"Expected 'new content' in data.txt, got: {content}"
);
}
#[fabro_macros::e2e_test(twin)]
async fn twin_exec_creates_file() {
let context = test_context!();
let twin = fabro_test::twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
fabro_test::TwinScenarios::new(namespace.clone())
.scenario(
fabro_test::TwinScenario::responses("gpt-5.4-mini")
.input_contains("Create a file called hello.txt containing exactly 'Hello'")
.tool_call(fabro_test::TwinToolCall::write_file("hello.txt", "Hello"))
.text("Done."),
)
.load(twin)
.await;
let mut cmd = context.exec_cmd();
twin.configure_command(&mut cmd, &namespace);
cmd.args([
"--auto-approve",
"--permissions",
"full",
"--provider",
"openai",
"--model",
"gpt-5.4-mini",
"Create a file called hello.txt containing exactly 'Hello'",
]);
let _output = run_success_output(cmd).await;
let content =
std::fs::read_to_string(context.temp_dir.join("hello.txt")).expect("read hello.txt");
assert_eq!(content, "Hello");
}
#[fabro_macros::e2e_test(twin)]
async fn twin_exec_shell_command() {
let context = test_context!();
let twin = fabro_test::twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
fabro_test::TwinScenarios::new(namespace.clone())
.scenario(
fabro_test::TwinScenario::responses("gpt-5.4-mini")
.input_contains(
"Run the shell command `echo hello_from_shell` and tell me what it printed",
)
.tool_call(fabro_test::TwinToolCall::shell("echo hello_from_shell"))
.text("It printed hello_from_shell."),
)
.load(twin)
.await;
let mut cmd = context.exec_cmd();
twin.configure_command(&mut cmd, &namespace);
cmd.args([
"--auto-approve",
"--permissions",
"full",
"--provider",
"openai",
"--model",
"gpt-5.4-mini",
"Run the shell command `echo hello_from_shell` and tell me what it printed",
]);
let output = run_success_output(cmd).await;
let stdout = String::from_utf8(output.stdout).expect("valid utf8");
assert!(
stdout.contains("hello_from_shell"),
"expected shell marker in output, got: {stdout}"
);
}
#[fabro_macros::e2e_test(twin)]
async fn twin_exec_json_output() {
let context = test_context!();
let twin = fabro_test::twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
fabro_test::TwinScenarios::new(namespace.clone())
.scenario(
fabro_test::TwinScenario::responses("gpt-5.4-mini")
.input_contains("Create a file called test.txt containing 'test'")
.tool_call(fabro_test::TwinToolCall::write_file("test.txt", "test"))
.text("Done."),
)
.load(twin)
.await;
let mut cmd = context.exec_cmd();
twin.configure_command(&mut cmd, &namespace);
cmd.args([
"--auto-approve",
"--permissions",
"full",
"--output-format",
"json",
"--provider",
"openai",
"--model",
"gpt-5.4-mini",
"Create a file called test.txt containing 'test'",
]);
let output = run_success_output(cmd).await;
let stdout = String::from_utf8(output.stdout).expect("valid utf8");
let lines: Vec<&str> = stdout
.lines()
.filter(|line| !line.trim().is_empty())
.collect();
assert!(!lines.is_empty(), "json output should not be empty");
let parsed: Vec<serde_json::Value> = lines
.iter()
.map(|line| serde_json::from_str(line).expect("each line should be valid JSON"))
.collect();
let first = &parsed[0];
assert!(
first.get("event").is_some() || first.get("type").is_some(),
"NDJSON line should have an event or type field, got: {first}"
);
}
#[fabro_macros::e2e_test(twin)]
async fn twin_exec_read_and_edit() {
let context = test_context!();
context.write_temp("data.txt", "old content");
let twin = fabro_test::twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
fabro_test::TwinScenarios::new(namespace.clone())
.scenario(
fabro_test::TwinScenario::responses("gpt-5.4-mini")
.input_contains("Read data.txt then replace its entire content with 'new content'")
.tool_call(fabro_test::TwinToolCall::read_file("data.txt")),
)
.scenario(
fabro_test::TwinScenario::responses("gpt-5.4-mini")
.tool_call(fabro_test::TwinToolCall::write_file(
"data.txt",
"new content",
))
.text("Done."),
)
.load(twin)
.await;
let mut cmd = context.exec_cmd();
twin.configure_command(&mut cmd, &namespace);
cmd.args([
"--auto-approve",
"--permissions",
"full",
"--provider",
"openai",
"--model",
"gpt-5.4-mini",
"Read data.txt then replace its entire content with 'new content'",
]);
let _output = run_success_output(cmd).await;
let content =
std::fs::read_to_string(context.temp_dir.join("data.txt")).expect("read data.txt");
assert_eq!(content, "new content");
}

View file

@ -1,6 +1,14 @@
use fabro_test::{fabro_snapshot, test_context};
use std::process::Output;
use fabro_test::{TwinScenario, TwinScenarios, fabro_snapshot, test_context, twin_openai};
use predicates::prelude::*;
async fn run_success_output(mut cmd: assert_cmd::Command) -> Output {
tokio::task::spawn_blocking(move || cmd.assert().success().get_output().clone())
.await
.expect("blocking command task should complete")
}
#[test]
fn prompt_bad_option() {
let context = test_context!();
@ -105,6 +113,22 @@ fn prompt_no_stream_generates_response() {
.stdout(predicate::str::is_empty().not());
}
#[fabro_macros::e2e_test(twin)]
async fn twin_prompt_no_stream() {
let context = test_context!();
let (base_url, api_key) = fabro_test::e2e_openai!();
let mut cmd = context.llm();
cmd.env("OPENAI_BASE_URL", base_url);
cmd.env("OPENAI_API_KEY", api_key);
cmd.args(["prompt", "--no-stream", "-m", "gpt-5.4-mini", "Say hello"]);
cmd.write_stdin("");
let output = run_success_output(cmd).await;
assert_eq!(
String::from_utf8(output.stdout).unwrap().trim(),
"deterministic: Say hello"
);
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
fn prompt_stream_generates_response() {
let context = test_context!();
@ -121,6 +145,22 @@ fn prompt_stream_generates_response() {
.stdout(predicate::str::is_empty().not());
}
#[fabro_macros::e2e_test(twin)]
async fn twin_prompt_stream() {
let context = test_context!();
let (base_url, api_key) = fabro_test::e2e_openai!();
let mut cmd = context.llm();
cmd.env("OPENAI_BASE_URL", base_url);
cmd.env("OPENAI_API_KEY", api_key);
cmd.args(["prompt", "-m", "gpt-5.4-mini", "Say hello"]);
cmd.write_stdin("");
let output = run_success_output(cmd).await;
assert_eq!(
String::from_utf8(output.stdout).unwrap().trim(),
"deterministic: Say hello"
);
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
fn prompt_usage_shows_tokens() {
let context = test_context!();
@ -139,6 +179,35 @@ fn prompt_usage_shows_tokens() {
.stderr(predicate::str::contains("Tokens:"));
}
#[fabro_macros::e2e_test(twin)]
async fn twin_prompt_usage() {
let context = test_context!();
let (base_url, api_key) = fabro_test::e2e_openai!();
let mut cmd = context.llm();
cmd.env("OPENAI_BASE_URL", base_url);
cmd.env("OPENAI_API_KEY", api_key);
cmd.args([
"prompt",
"--no-stream",
"-u",
"-m",
"gpt-5.4-mini",
"Say hello",
]);
cmd.write_stdin("");
let output = run_success_output(cmd).await;
assert_eq!(
String::from_utf8(output.stdout).unwrap().trim(),
"deterministic: Say hello"
);
assert!(
String::from_utf8(output.stderr)
.unwrap()
.contains("Tokens:"),
"stderr should include token usage"
);
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
fn prompt_schema_no_stream_generates_json() {
let context = test_context!();
@ -161,6 +230,41 @@ fn prompt_schema_no_stream_generates_json() {
);
}
#[fabro_macros::e2e_test(twin)]
async fn twin_prompt_schema_no_stream() {
let context = test_context!();
let twin = twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
TwinScenarios::new(namespace.clone())
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.stream(false)
.input_contains("Return JSON")
.text(r#"{"greeting":"hello"}"#),
)
.load(twin)
.await;
let mut cmd = context.llm();
twin.configure_command(&mut cmd, &namespace);
cmd.args([
"prompt",
"--no-stream",
"-m",
"gpt-5.4-mini",
"--schema",
r#"{"type":"object","properties":{"greeting":{"type":"string"}},"required":["greeting"]}"#,
"Return JSON",
]);
cmd.write_stdin("");
let output = run_success_output(cmd).await;
let stdout = String::from_utf8(output.stdout).unwrap();
let parsed: serde_json::Value =
serde_json::from_str(stdout.trim()).expect("stdout should be valid JSON");
assert_eq!(parsed["greeting"], "hello");
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
fn prompt_schema_stream_generates_json() {
let context = test_context!();
@ -183,6 +287,40 @@ fn prompt_schema_stream_generates_json() {
);
}
#[fabro_macros::e2e_test(twin)]
async fn twin_prompt_schema_stream() {
let context = test_context!();
let twin = twin_openai().await;
let namespace = format!("{}::{}", module_path!(), line!());
TwinScenarios::new(namespace.clone())
.scenario(
TwinScenario::responses("gpt-5.4-mini")
.stream(true)
.input_contains("Return JSON")
.text(r#"{"greeting":"hello"}"#),
)
.load(twin)
.await;
let mut cmd = context.llm();
twin.configure_command(&mut cmd, &namespace);
cmd.args([
"prompt",
"-m",
"gpt-5.4-mini",
"--schema",
r#"{"type":"object","properties":{"greeting":{"type":"string"}},"required":["greeting"]}"#,
"Return JSON",
]);
cmd.write_stdin("");
let output = run_success_output(cmd).await;
let stdout = String::from_utf8(output.stdout).unwrap();
let parsed: serde_json::Value =
serde_json::from_str(stdout.trim()).expect("stdout should be valid JSON");
assert_eq!(parsed["greeting"], "hello");
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
fn chat_multi_turn_with_system_prompt() {
let context = test_context!();
@ -223,3 +361,30 @@ fn chat_multi_turn_with_system_prompt() {
"second response should show multi-turn context, got: {stdout}"
);
}
#[fabro_macros::e2e_test(twin)]
async fn twin_chat_multi_turn() {
let context = test_context!();
let (base_url, api_key) = fabro_test::e2e_openai!();
let mut cmd = context.command();
cmd.env("OPENAI_BASE_URL", base_url);
cmd.env("OPENAI_API_KEY", api_key);
cmd.args([
"llm",
"chat",
"-m",
"gpt-5.4-mini",
"-s",
"You are a pilot. End every response with 'Roger that.'",
]);
cmd.write_stdin("What is your profession?\nWhat did I just ask you?\n");
let output = run_success_output(cmd).await;
let stdout = String::from_utf8(output.stdout).unwrap();
let stderr = String::from_utf8(output.stderr).unwrap();
assert!(!stdout.trim().is_empty(), "stdout should not be empty");
assert!(
stderr.contains("Using model:"),
"stderr should show model info"
);
}

View file

@ -18,6 +18,7 @@ axum = { workspace = true }
insta = { workspace = true, features = ["filters"] }
regex = { workspace = true }
reqwest = { workspace = true }
serde_json = { workspace = true }
tempfile = "3"
tokio = { workspace = true }
twin-openai = { workspace = true }

View file

@ -3,6 +3,7 @@ use std::process::Output;
use assert_cmd::Command;
use regex::Regex;
use serde_json::{Map, Value, json};
/// Walk up from `start` to find the repo-level `test/` fixtures directory.
pub fn find_test_fixtures_dir(start: &Path) -> Option<PathBuf> {
@ -516,6 +517,252 @@ impl TwinGitHub {
}
}
impl TwinOpenAi {
pub fn configure_command(&self, cmd: &mut Command, namespace: &str) {
cmd.env("OPENAI_BASE_URL", &self.base_url);
cmd.env("OPENAI_API_KEY", namespace);
}
#[must_use]
pub fn admin_url(&self) -> String {
self.base_url.trim_end_matches("/v1").to_string()
}
pub async fn reset_namespace(&self, namespace: &str) {
let response = reqwest::Client::new()
.post(format!("{}/__admin/reset", self.admin_url()))
.bearer_auth(namespace)
.send()
.await
.expect("reset twin-openai namespace");
assert!(
response.status().is_success(),
"reset twin-openai namespace failed: {}",
response.status()
);
}
}
#[derive(Debug, Default, Clone)]
pub struct TwinScenarios {
namespace: String,
scenarios: Vec<TwinScenario>,
}
impl TwinScenarios {
#[must_use]
pub fn new(namespace: impl Into<String>) -> Self {
Self {
namespace: namespace.into(),
scenarios: Vec::new(),
}
}
#[must_use]
pub fn scenario(mut self, scenario: TwinScenario) -> Self {
self.scenarios.push(scenario);
self
}
pub async fn load(self, twin: &TwinOpenAi) {
twin.reset_namespace(&self.namespace).await;
let response = reqwest::Client::new()
.post(format!("{}/__admin/scenarios", twin.admin_url()))
.bearer_auth(&self.namespace)
.json(&json!({
"scenarios": self.scenarios.into_iter().map(TwinScenario::into_json).collect::<Vec<_>>(),
}))
.send()
.await
.expect("load twin-openai scenarios");
assert!(
response.status().is_success(),
"load twin-openai scenarios failed: {}",
response.status()
);
}
}
#[derive(Debug, Clone)]
pub struct TwinScenario {
matcher: Map<String, Value>,
script: Value,
}
impl TwinScenario {
#[must_use]
pub fn responses(model: impl Into<String>) -> Self {
Self {
matcher: Map::from_iter([
(
"endpoint".to_string(),
Value::String("responses".to_string()),
),
("model".to_string(), Value::String(model.into())),
]),
script: json!({ "kind": "success" }),
}
}
#[must_use]
pub fn text(mut self, text: impl Into<String>) -> Self {
self.assert_script_kind("success", "text");
self.script["response_text"] = Value::String(text.into());
self
}
#[must_use]
pub fn tool_call(self, tool_call: TwinToolCall) -> Self {
self.tool_calls(vec![tool_call])
}
#[must_use]
pub fn tool_calls(mut self, tool_calls: Vec<TwinToolCall>) -> Self {
self.assert_script_kind("success", "tool_calls");
self.script["tool_calls"] = Value::Array(
tool_calls
.into_iter()
.map(TwinToolCall::into_json)
.collect::<Vec<_>>(),
);
self
}
#[must_use]
pub fn error(mut self, status: u16, message: impl Into<String>) -> Self {
self.script = json!({
"kind": "error",
"status": status,
"message": message.into(),
"error_type": "invalid_request_error",
"code": "twin_error",
});
self
}
#[must_use]
pub fn retry_after(mut self, retry_after: impl Into<String>) -> Self {
self.assert_script_kind("error", "retry_after");
self.script["retry_after"] = Value::String(retry_after.into());
self
}
#[must_use]
pub fn stream(mut self, stream: bool) -> Self {
self.matcher
.insert("stream".to_string(), Value::Bool(stream));
self
}
#[must_use]
pub fn input_contains(mut self, needle: impl Into<String>) -> Self {
self.matcher
.insert("input_contains".to_string(), Value::String(needle.into()));
self
}
#[must_use]
pub fn metadata(mut self, key: impl Into<String>, value: Value) -> Self {
let metadata = self
.matcher
.entry("metadata".to_string())
.or_insert_with(|| Value::Object(Map::new()));
metadata
.as_object_mut()
.expect("metadata should be an object")
.insert(key.into(), value);
self
}
fn into_json(self) -> Value {
json!({
"matcher": self.matcher,
"script": self.script,
})
}
fn assert_script_kind(&self, expected: &str, method: &str) {
let actual = self.script["kind"]
.as_str()
.expect("twin scenario script must have a kind");
assert_eq!(
actual, expected,
"TwinScenario::{method} requires a {expected} script, got {actual}"
);
}
}
#[derive(Debug, Clone)]
pub struct TwinToolCall {
name: String,
arguments: Value,
}
impl TwinToolCall {
#[must_use]
pub fn new(name: impl Into<String>, arguments: Value) -> Self {
Self {
name: name.into(),
arguments,
}
}
#[must_use]
pub fn write_file(path: impl Into<String>, content: impl Into<String>) -> Self {
Self::new(
"write_file",
json!({ "file_path": path.into(), "content": content.into() }),
)
}
#[must_use]
pub fn read_file(path: impl Into<String>) -> Self {
Self::new("read_file", json!({ "file_path": path.into() }))
}
#[must_use]
pub fn shell(command: impl Into<String>) -> Self {
Self::new("shell", json!({ "command": command.into() }))
}
#[must_use]
pub fn shell_with_timeout(command: impl Into<String>, timeout_ms: u64) -> Self {
Self::new(
"shell",
json!({ "command": command.into(), "timeout_ms": timeout_ms }),
)
}
#[must_use]
pub fn grep_pattern(pattern: impl Into<String>, path: impl Into<String>) -> Self {
Self::new(
"grep",
json!({ "pattern": pattern.into(), "path": path.into() }),
)
}
#[must_use]
pub fn glob_pattern(pattern: impl Into<String>, path: impl Into<String>) -> Self {
Self::new(
"glob",
json!({ "pattern": pattern.into(), "path": path.into() }),
)
}
#[must_use]
pub fn apply_patch(patch: impl Into<String>) -> Self {
Self::new("apply_patch", json!({ "patch": patch.into() }))
}
fn into_json(self) -> Value {
json!({
"name": self.name,
"arguments": self.arguments,
})
}
}
static TWIN_OPENAI: OnceCell<TwinOpenAi> = OnceCell::const_new();
/// Returns a shared twin-openai server, starting it on first call.
@ -578,3 +825,66 @@ macro_rules! e2e_openai {
}
}};
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn twin_admin_url_removes_v1_suffix() {
let twin = TwinOpenAi {
base_url: "http://127.0.0.1:3000/v1".to_string(),
};
assert_eq!(twin.admin_url(), "http://127.0.0.1:3000");
}
#[test]
fn twin_configure_command_sets_openai_env() {
let twin = TwinOpenAi {
base_url: "http://127.0.0.1:3000/v1".to_string(),
};
let mut cmd = Command::new("env");
twin.configure_command(&mut cmd, "test-namespace");
let envs = cmd.get_envs().collect::<Vec<_>>();
assert!(envs.iter().any(|(key, value)| {
*key == std::ffi::OsStr::new("OPENAI_BASE_URL")
&& *value == Some(std::ffi::OsStr::new("http://127.0.0.1:3000/v1"))
}),);
assert!(envs.iter().any(|(key, value)| {
*key == std::ffi::OsStr::new("OPENAI_API_KEY")
&& *value == Some(std::ffi::OsStr::new("test-namespace"))
}),);
}
#[test]
fn twin_scenario_builder_matches_admin_contract() {
let scenario = TwinScenario::responses("gpt-5.4-mini")
.stream(false)
.input_contains("Return JSON")
.tool_call(TwinToolCall::write_file("hello.txt", "Hello"))
.text(r#"{"greeting":"hello"}"#)
.into_json();
assert_eq!(scenario["matcher"]["endpoint"], "responses");
assert_eq!(scenario["matcher"]["model"], "gpt-5.4-mini");
assert_eq!(scenario["matcher"]["stream"], false);
assert_eq!(scenario["matcher"]["input_contains"], "Return JSON");
assert_eq!(scenario["script"]["kind"], "success");
assert_eq!(
scenario["script"]["response_text"],
r#"{"greeting":"hello"}"#
);
assert_eq!(scenario["script"]["tool_calls"][0]["name"], "write_file");
assert_eq!(
scenario["script"]["tool_calls"][0]["arguments"]["file_path"],
"hello.txt"
);
}
#[test]
#[should_panic(expected = "TwinScenario::retry_after requires a error script")]
fn twin_scenario_rejects_retry_after_on_success() {
let _ = TwinScenario::responses("gpt-5.4-mini").retry_after("30");
}
}

View file

@ -5817,6 +5817,7 @@ async fn fidelity_resume_preserves_context_values_across_checkpoint() {
// ===========================================================================
mod real_llm {
use std::collections::HashMap;
use std::sync::Arc;
use async_trait::async_trait;
@ -5828,11 +5829,13 @@ mod real_llm {
use fabro_workflow::handler::agent::{AgentHandler, CodergenBackend, CodergenResult};
use fabro_llm::client::Client;
use fabro_llm::providers::OpenAiAdapter;
use fabro_llm::types::{Message, Request};
struct LlmCodergenBackend {
client: Arc<Client>,
model: String,
provider: String,
}
#[async_trait]
@ -5867,7 +5870,7 @@ mod real_llm {
let request = Request {
model: self.model.clone(),
messages: vec![Message::user(prompt)],
provider: Some("anthropic".to_string()),
provider: Some(self.provider.clone()),
tools: None,
tool_choice: None,
response_format: None,
@ -5894,18 +5897,50 @@ mod real_llm {
}
}
fn test_llm_model() -> &'static str {
if fabro_test::TestMode::from_env().is_twin() {
"gpt-5.4-mini"
} else {
"claude-haiku-4-5"
}
}
fn test_llm_provider() -> &'static str {
if fabro_test::TestMode::from_env().is_twin() {
"openai"
} else {
"anthropic"
}
}
async fn make_llm_client() -> Option<Arc<Client>> {
if fabro_test::TestMode::from_env().is_twin() {
let (base_url, api_key) = fabro_test::e2e_openai!();
let adapter: Arc<dyn fabro_llm::provider::ProviderAdapter> =
Arc::new(OpenAiAdapter::new(api_key).with_base_url(base_url));
let mut providers: HashMap<String, Arc<dyn fabro_llm::provider::ProviderAdapter>> =
HashMap::new();
providers.insert("openai".to_string(), adapter);
return Some(Arc::new(Client::new(
providers,
Some("openai".to_string()),
Vec::new(),
)));
}
fabro_test::require_env("ANTHROPIC_API_KEY")?;
let client = Client::from_env()
.await
.expect("unified-llm client should initialize from env");
Some(Arc::new(client))
Some(Arc::new(
Client::from_env()
.await
.expect("unified-llm client should initialize from env"),
))
}
fn make_llm_backend(client: Arc<Client>) -> Box<LlmCodergenBackend> {
Box::new(LlmCodergenBackend {
client,
model: "claude-haiku-4-5".to_string(),
model: test_llm_model().to_string(),
provider: test_llm_provider().to_string(),
})
}
@ -5922,7 +5957,7 @@ mod real_llm {
use fabro_workflow::run_options::RunOptions;
use fabro_workflow::test_support::WorkflowRunner;
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn real_llm_linear_pipeline() {
let client = make_llm_client().await.unwrap();
@ -6031,7 +6066,7 @@ mod real_llm {
);
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn real_llm_two_stage_pipeline() {
let client = make_llm_client().await.unwrap();
@ -6122,7 +6157,7 @@ mod real_llm {
assert_eq!(last_stage, Some("review"));
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn real_llm_human_gate_auto_approve() {
let client = make_llm_client().await.unwrap();
@ -6265,7 +6300,7 @@ mod real_llm {
);
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn real_llm_one_shot_pipeline() {
let client = make_llm_client().await.unwrap();
@ -6301,7 +6336,7 @@ mod real_llm {
);
classify.attrs.insert(
"model".to_string(),
AttrValue::String("claude-haiku-4-5".to_string()),
AttrValue::String(test_llm_model().to_string()),
);
graph.nodes.insert("classify".to_string(), classify);
@ -7217,6 +7252,62 @@ fn simple_linear_dot() -> &'static str {
}"#
}
struct EnvVarGuard {
saved: Vec<(&'static str, Option<String>)>,
}
impl EnvVarGuard {
fn new() -> Self {
Self { saved: Vec::new() }
}
fn set(&mut self, name: &'static str, value: Option<String>) {
self.saved.push((name, std::env::var(name).ok()));
match value {
Some(value) => unsafe { std::env::set_var(name, value) },
None => unsafe { std::env::remove_var(name) },
}
}
}
impl Drop for EnvVarGuard {
fn drop(&mut self) {
while let Some((name, value)) = self.saved.pop() {
match value {
Some(value) => unsafe { std::env::set_var(name, value) },
None => unsafe { std::env::remove_var(name) },
}
}
}
}
fn test_hook_model() -> &'static str {
if fabro_test::TestMode::from_env().is_twin() {
"gpt-5.4-mini"
} else {
"haiku"
}
}
async fn configure_twin_hook_env(scenarios: Vec<fabro_test::TwinScenario>) -> Option<EnvVarGuard> {
if !fabro_test::TestMode::from_env().is_twin() {
return None;
}
let (base_url, api_key) = fabro_test::e2e_openai!();
let mut batch = fabro_test::TwinScenarios::new(api_key.clone());
for scenario in scenarios {
batch = batch.scenario(scenario);
}
batch.load(fabro_test::twin_openai().await).await;
let mut guard = EnvVarGuard::new();
guard.set("ANTHROPIC_API_KEY", None);
guard.set("OPENAI_API_KEY", Some(api_key));
guard.set("OPENAI_BASE_URL", Some(base_url));
Some(guard)
}
fn two_step_dot() -> &'static str {
r#"digraph HookTest {
graph [goal="Test hooks"]
@ -8078,15 +8169,19 @@ timeout_ms = 120000
// --- Prompt/Agent hook E2E with real LLM ---
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn hook_prompt_proceed_allows_run() {
let _guard = configure_twin_hook_env(vec![
fabro_test::TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#),
])
.await;
let hooks = vec![fabro_hooks::HookDefinition {
name: Some("prompt-proceed".into()),
event: fabro_hooks::HookEvent::RunStart,
command: None,
hook_type: Some(fabro_hooks::HookType::Prompt {
prompt: "A workflow is starting. Always approve. Respond with {\"ok\": true}.".into(),
model: Some("haiku".into()),
model: Some(test_hook_model().into()),
}),
matcher: None,
blocking: None,
@ -8102,8 +8197,13 @@ async fn hook_prompt_proceed_allows_run() {
assert_eq!(outcome.status, StageStatus::Success);
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn hook_prompt_block_prevents_run() {
let _guard = configure_twin_hook_env(vec![
fabro_test::TwinScenario::responses("gpt-5.4-mini")
.text(r#"{"ok":false,"reason":"math check failed"}"#),
])
.await;
// Use a factual question that evaluates to false: "Is 2+2=5?"
let hooks = vec![fabro_hooks::HookDefinition {
name: Some("prompt-block".into()),
@ -8111,7 +8211,7 @@ async fn hook_prompt_block_prevents_run() {
command: None,
hook_type: Some(fabro_hooks::HookType::Prompt {
prompt: "Check: is 2+2 equal to 5? If the statement is true, respond {\"ok\": true}. If false, respond {\"ok\": false, \"reason\": \"math check failed\"}.".into(),
model: Some("haiku".into()),
model: Some(test_hook_model().into()),
}),
matcher: None,
blocking: None,
@ -8130,15 +8230,19 @@ async fn hook_prompt_block_prevents_run() {
);
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn hook_agent_proceed_allows_run() {
let _guard = configure_twin_hook_env(vec![
fabro_test::TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#),
])
.await;
let hooks = vec![fabro_hooks::HookDefinition {
name: Some("agent-proceed".into()),
event: fabro_hooks::HookEvent::RunStart,
command: None,
hook_type: Some(fabro_hooks::HookType::Agent {
prompt: "A workflow is starting. Always approve. Respond with {\"ok\": true}. Do not use any tools.".into(),
model: Some("haiku".into()),
model: Some(test_hook_model().into()),
max_tool_rounds: Some(1),
}),
matcher: None,
@ -8155,11 +8259,18 @@ async fn hook_agent_proceed_allows_run() {
assert_eq!(outcome.status, StageStatus::Success);
}
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn hook_agent_with_tool_use() {
let dir = tempfile::tempdir().unwrap();
let marker = dir.path().join("hook_check.txt");
std::fs::write(&marker, "READY").unwrap();
let _guard = configure_twin_hook_env(vec![
fabro_test::TwinScenario::responses("gpt-5.4-mini").tool_call(
fabro_test::TwinToolCall::read_file(marker.display().to_string()),
),
fabro_test::TwinScenario::responses("gpt-5.4-mini").text(r#"{"ok":true}"#),
])
.await;
let hooks = vec![fabro_hooks::HookDefinition {
name: Some("agent-tools".into()),
@ -8170,7 +8281,7 @@ async fn hook_agent_with_tool_use() {
"Read the file at {} using the read_file tool. If it contains 'READY', respond with {{\"ok\": true}}. Otherwise respond with {{\"ok\": false, \"reason\": \"not ready\"}}.",
marker.display()
),
model: Some("haiku".into()),
model: Some(test_hook_model().into()),
max_tool_rounds: Some(5),
}),
matcher: None,
@ -8228,7 +8339,7 @@ async fn hooks_do_not_duplicate_workflow_events() {
// E2E test with real LLM
// ---------------------------------------------------------------------------
#[fabro_macros::e2e_test(live("ANTHROPIC_API_KEY"))]
#[fabro_macros::e2e_test(twin, live("ANTHROPIC_API_KEY"))]
async fn arc_e2e_with_real_llm() {
let dir = tempfile::tempdir().unwrap();
let dir_path = dir.path().to_str().unwrap().to_string();
@ -8252,15 +8363,37 @@ async fn arc_e2e_with_real_llm() {
validate_or_raise(&graph, &[]).expect("validation should pass");
let interviewer: Arc<dyn Interviewer> = Arc::new(AutoApproveInterviewer);
let model = "claude-haiku-4-5".to_string();
let (_guard, model, provider) = if fabro_test::TestMode::from_env().is_twin() {
let (base_url, api_key) = fabro_test::e2e_openai!();
fabro_test::TwinScenarios::new(api_key.clone())
.scenario(
fabro_test::TwinScenario::responses("gpt-5.4-mini")
.input_contains(&format!(
"Create a file called hello.txt in {dir_path} containing exactly 'Hello from LLM'. Do not output anything else."
))
.tool_call(fabro_test::TwinToolCall::write_file(
format!("{dir_path}/hello.txt"),
"Hello from LLM",
))
.text("Done."),
)
.load(fabro_test::twin_openai().await)
.await;
let mut guard = EnvVarGuard::new();
guard.set("ANTHROPIC_API_KEY", None);
guard.set("OPENAI_API_KEY", Some(api_key));
guard.set("OPENAI_BASE_URL", Some(base_url));
(Some(guard), "gpt-5.4-mini".to_string(), Provider::OpenAi)
} else {
(None, "claude-haiku-4-5".to_string(), Provider::Anthropic)
};
let registry = default_registry(interviewer, move || {
Some(Box::new(AgentApiBackend::new(
model.clone(),
Provider::Anthropic,
Vec::new(),
))
as Box<dyn fabro_workflow::handler::agent::CodergenBackend>)
Some(
Box::new(AgentApiBackend::new(model.clone(), provider, Vec::new()))
as Box<dyn fabro_workflow::handler::agent::CodergenBackend>,
)
});
let run_dir = tempfile::tempdir().unwrap();