mirror of
https://github.com/fabro-sh/fabro.git
synced 2026-09-15 23:32:46 +00:00
Cover the pebble agent loop with workflow-level tests
Each behaviour the pebble backend owes a run now has one test that drives it through the workflow engine against a scripted OpenAI-compatible model: - an agent stage under every harness profile (openai, anthropic, claude-5, gemini, kimi) writing a file with that profile's own tool spelling, with the event sequence, files touched, response, usage, and cost checked; the codex vocabulary applies a patch through the OpenAI twin's custom tool call, which `TwinToolCall::custom` now scripts - steering delivered mid-stage, an interrupt with a steer, run cancellation, and the executor-enforced stage timeout - a question answered through the interviewer, a subagent whose events carry its parent's session id, and an MCP tool served by a stdio server - failover to a second provider after a tool ran, continuing the recorded conversation without running the tool again - a failing event sink ending the stage with the sink's error - Ask Fabro resuming a stored record across turns, with the cursor moved past the run's event log when the record's own cursor fell behind - Docker and Daytona smokes running an agent stage through the provider sandboxes Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
18a3c4741e
commit
6599f7cdc0
4 changed files with 1642 additions and 0 deletions
|
|
@ -1960,3 +1960,288 @@ enabled = true
|
|||
assert!(input.ends_with("User question:\nWhy did it fail?"));
|
||||
}
|
||||
}
|
||||
|
||||
/// Ask Fabro across turns and processes: a second turn resumes the stored
|
||||
/// pebble record, and a record whose cursor fell behind the run's event log
|
||||
/// (a crash between the two writes) is moved past the log before it answers.
|
||||
#[cfg(test)]
|
||||
mod resume_tests {
|
||||
use std::sync::Arc;
|
||||
|
||||
use axum::body::{Body, to_bytes};
|
||||
use axum::http::{Request, StatusCode};
|
||||
use fabro_config::daemon::ServerDaemon;
|
||||
use fabro_config::{RunEnvironmentLayer, RunLayer, Storage};
|
||||
use fabro_static::EnvVars;
|
||||
use fabro_test::{TwinScenario, TwinScenarios, twin_openai};
|
||||
use fabro_types::{RunId, SessionId};
|
||||
use tower::ServiceExt;
|
||||
|
||||
use crate::server::{AppState, spawn_scheduler};
|
||||
use crate::test_support::{
|
||||
TestAppStateBuilder, build_test_router, default_test_server_settings,
|
||||
llm_overlay_with_provider_base_url,
|
||||
};
|
||||
|
||||
const MODEL: &str = "gpt-5.4-mini";
|
||||
const DOT: &str = r#"digraph Test {
|
||||
graph [goal="Test"]
|
||||
start [shape=Mdiamond]
|
||||
exit [shape=Msquare]
|
||||
start -> exit
|
||||
}"#;
|
||||
|
||||
fn api(path: &str) -> String {
|
||||
format!("/api/v1{path}")
|
||||
}
|
||||
|
||||
async fn json_response(
|
||||
app: &axum::Router,
|
||||
request: Request<Body>,
|
||||
expected: StatusCode,
|
||||
) -> serde_json::Value {
|
||||
let response = app.clone().oneshot(request).await.unwrap();
|
||||
let status = response.status();
|
||||
let bytes = to_bytes(response.into_body(), usize::MAX).await.unwrap();
|
||||
assert_eq!(
|
||||
status,
|
||||
expected,
|
||||
"unexpected status, body {}",
|
||||
String::from_utf8_lossy(&bytes)
|
||||
);
|
||||
if bytes.is_empty() {
|
||||
serde_json::Value::Null
|
||||
} else {
|
||||
serde_json::from_slice(&bytes).expect("response should be JSON")
|
||||
}
|
||||
}
|
||||
|
||||
fn post_json(path: &str, body: &serde_json::Value) -> Request<Body> {
|
||||
Request::builder()
|
||||
.method("POST")
|
||||
.uri(api(path))
|
||||
.header("content-type", "application/json")
|
||||
.body(Body::from(body.to_string()))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// A server whose `openai` provider is the twin under `namespace`, whose
|
||||
/// runs execute in place, and whose own address Ask Fabro can resolve.
|
||||
fn twin_backed_state(base_url: String, namespace: &str) -> Arc<AppState> {
|
||||
let api_key = namespace.to_string();
|
||||
let state = TestAppStateBuilder::new()
|
||||
.runtime_settings(default_test_server_settings(), RunLayer {
|
||||
environment: Some(RunEnvironmentLayer {
|
||||
id: Some("local".to_string()),
|
||||
..RunEnvironmentLayer::default()
|
||||
}),
|
||||
..RunLayer::default()
|
||||
})
|
||||
.max_concurrent_runs(2)
|
||||
// A registry factory runs the dry run in this process, so no
|
||||
// worker executable is needed.
|
||||
.registry_factory(|interviewer| {
|
||||
fabro_workflow::handler::default_registry(interviewer, || None)
|
||||
})
|
||||
.llm_overlay(llm_overlay_with_provider_base_url("openai", base_url))
|
||||
.vault_entries([(EnvVars::OPENAI_API_KEY, namespace.to_string())])
|
||||
.env_lookup(move |name| (name == EnvVars::OPENAI_API_KEY).then(|| api_key.clone()))
|
||||
.build();
|
||||
let runtime_directory = Storage::new(state.server_storage_dir()).runtime_directory();
|
||||
ServerDaemon::new(
|
||||
std::process::id(),
|
||||
fabro_config::bind::Bind::Tcp("127.0.0.1:32277".parse().unwrap()),
|
||||
runtime_directory.log_path(),
|
||||
)
|
||||
.write(&runtime_directory)
|
||||
.expect("test server record should be written");
|
||||
state
|
||||
}
|
||||
|
||||
/// A completed local dry run, so the session has a sandbox to reconnect.
|
||||
async fn completed_run(app: &axum::Router) -> RunId {
|
||||
let manifest = serde_json::json!({
|
||||
"version": 1,
|
||||
"cwd": std::env::temp_dir().display().to_string(),
|
||||
"args": { "dry_run": true },
|
||||
"target": { "path": "workflow.fabro" },
|
||||
"workflows": { "workflow.fabro": { "source": DOT, "files": {} } },
|
||||
});
|
||||
let created = json_response(app, post_json("/runs", &manifest), StatusCode::CREATED).await;
|
||||
let run_id = created["id"].as_str().unwrap().to_string();
|
||||
let start = Request::builder()
|
||||
.method("POST")
|
||||
.uri(api(&format!("/runs/{run_id}/start")))
|
||||
.body(Body::empty())
|
||||
.unwrap();
|
||||
json_response(app, start, StatusCode::OK).await;
|
||||
for _ in 0..500 {
|
||||
let get = Request::builder()
|
||||
.method("GET")
|
||||
.uri(api(&format!("/runs/{run_id}")))
|
||||
.body(Body::empty())
|
||||
.unwrap();
|
||||
let run = json_response(app, get, StatusCode::OK).await;
|
||||
match run["lifecycle"]["status"]["kind"].as_str() {
|
||||
Some("succeeded") => return run_id.parse().unwrap(),
|
||||
Some("failed") => panic!("the dry run failed: {run}"),
|
||||
_ => tokio::time::sleep(std::time::Duration::from_millis(10)).await,
|
||||
}
|
||||
}
|
||||
panic!("run {run_id} did not complete");
|
||||
}
|
||||
|
||||
/// Submits one turn and returns the streamed session events.
|
||||
async fn turn(
|
||||
app: &axum::Router,
|
||||
session_id: SessionId,
|
||||
input: &str,
|
||||
) -> Vec<serde_json::Value> {
|
||||
let response = app
|
||||
.clone()
|
||||
.oneshot(post_json(
|
||||
&format!("/sessions/{session_id}/turns"),
|
||||
&serde_json::json!({ "input": input }),
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(response.status(), StatusCode::OK);
|
||||
let bytes = to_bytes(response.into_body(), usize::MAX).await.unwrap();
|
||||
let body = String::from_utf8(bytes.to_vec()).unwrap();
|
||||
let events: Vec<serde_json::Value> = body
|
||||
.lines()
|
||||
.filter_map(|line| line.strip_prefix("data: "))
|
||||
.map(|data| serde_json::from_str(data).unwrap())
|
||||
.collect();
|
||||
assert!(
|
||||
events
|
||||
.iter()
|
||||
.any(|event| event["event"] == "run.session.turn.succeeded"),
|
||||
"the turn should succeed: {events:#?}"
|
||||
);
|
||||
events
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn a_resumed_session_continues_its_conversation_past_the_event_log() {
|
||||
let twin = twin_openai().await;
|
||||
let namespace = format!("{}::{}", module_path!(), line!());
|
||||
TwinScenarios::new(namespace.clone())
|
||||
.scenario(
|
||||
TwinScenario::responses(MODEL)
|
||||
.input_contains("First question")
|
||||
.text("First answer"),
|
||||
)
|
||||
.scenario(
|
||||
TwinScenario::responses(MODEL)
|
||||
.input_contains("Second question")
|
||||
.text("Second answer"),
|
||||
)
|
||||
.load(twin)
|
||||
.await;
|
||||
let state = twin_backed_state(twin.base_url.clone(), &namespace);
|
||||
spawn_scheduler(Arc::clone(&state));
|
||||
let app = build_test_router(Arc::clone(&state));
|
||||
let run_id = completed_run(&app).await;
|
||||
|
||||
let created = json_response(
|
||||
&app,
|
||||
post_json(
|
||||
&format!("/runs/{run_id}/sessions"),
|
||||
&serde_json::json!({ "title": "Ask Fabro", "model": MODEL }),
|
||||
),
|
||||
StatusCode::CREATED,
|
||||
)
|
||||
.await;
|
||||
let session_id: SessionId = created["id"].as_str().unwrap().parse().unwrap();
|
||||
|
||||
turn(&app, session_id, "First question").await;
|
||||
let after_first = state
|
||||
.stores
|
||||
.session_records
|
||||
.get(session_id)
|
||||
.await
|
||||
.unwrap()
|
||||
.expect("the first turn persists the record");
|
||||
assert!(
|
||||
after_first.record.last_event_seq > 0,
|
||||
"the record carries the committed event cursor"
|
||||
);
|
||||
|
||||
// The crash: the run's events were written, the record's cursor was
|
||||
// not. Drop the agent so the next turn resumes from the stale record
|
||||
// the way a new process would.
|
||||
let mut stale = after_first.record.clone();
|
||||
stale.last_event_seq = 0;
|
||||
state
|
||||
.stores
|
||||
.session_records
|
||||
.put(session_id, run_id, &stale, chrono::Utc::now())
|
||||
.await
|
||||
.unwrap();
|
||||
state
|
||||
.session_runtimes()
|
||||
.load_or_create_runtime(session_id)
|
||||
.clear_agent()
|
||||
.await;
|
||||
let log_head_before_resume = state
|
||||
.store_ref()
|
||||
.open_run_reader(&run_id)
|
||||
.await
|
||||
.unwrap()
|
||||
.last_event_seq()
|
||||
.await
|
||||
.unwrap()
|
||||
.expect("the run has events");
|
||||
|
||||
turn(&app, session_id, "Second question").await;
|
||||
|
||||
let after_second = state
|
||||
.stores
|
||||
.session_records
|
||||
.get(session_id)
|
||||
.await
|
||||
.unwrap()
|
||||
.expect("the second turn persists the record");
|
||||
assert!(
|
||||
after_second.record.last_event_seq > u64::from(log_head_before_resume),
|
||||
"the resumed session numbers past the log head {log_head_before_resume}, got {}",
|
||||
after_second.record.last_event_seq
|
||||
);
|
||||
assert!(
|
||||
after_second.record.last_event_seq > after_first.record.last_event_seq,
|
||||
"the cursor only moves forward"
|
||||
);
|
||||
assert_eq!(
|
||||
after_second.record.messages.len(),
|
||||
2 * after_first.record.messages.len(),
|
||||
"the record holds both turns"
|
||||
);
|
||||
|
||||
// The run's title generator also calls the model; the turns are the
|
||||
// streamed requests. The twin logs the user side of the input, so the
|
||||
// resumed turn shows as carrying the first question ahead of the
|
||||
// second.
|
||||
let logs = twin.request_logs(&namespace).await;
|
||||
let turns: Vec<&str> = logs["requests"]
|
||||
.as_array()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.filter(|request| request["stream"] == true)
|
||||
.map(|request| request["input_text"].as_str().unwrap_or_default())
|
||||
.collect();
|
||||
assert_eq!(turns.len(), 2, "one model call per turn: {logs}");
|
||||
assert!(
|
||||
!turns[0].contains("Second question"),
|
||||
"the first turn knows nothing of the second, got {}",
|
||||
turns[0]
|
||||
);
|
||||
let first_at = turns[1]
|
||||
.find("User question: First question")
|
||||
.expect("the resumed turn replays the first question");
|
||||
let second_at = turns[1]
|
||||
.find("User question: Second question")
|
||||
.expect("the resumed turn ends with the second question");
|
||||
assert!(first_at < second_at, "got {}", turns[1]);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,3 +3,4 @@ mod cp_integration;
|
|||
mod daytona_integration;
|
||||
mod git_integration;
|
||||
mod integration;
|
||||
mod pebble_agent;
|
||||
|
|
|
|||
1337
lib/components/fabro-workflow/tests/it/pebble_agent.rs
Normal file
1337
lib/components/fabro-workflow/tests/it/pebble_agent.rs
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -2340,6 +2340,7 @@ pub struct TwinToolCall {
|
|||
name: String,
|
||||
arguments: Value,
|
||||
raw_arguments: Option<String>,
|
||||
custom: bool,
|
||||
}
|
||||
|
||||
impl TwinToolCall {
|
||||
|
|
@ -2349,6 +2350,7 @@ impl TwinToolCall {
|
|||
name: name.into(),
|
||||
arguments,
|
||||
raw_arguments: None,
|
||||
custom: false,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -2362,6 +2364,7 @@ impl TwinToolCall {
|
|||
name: name.into(),
|
||||
arguments,
|
||||
raw_arguments: Some(raw_arguments.into()),
|
||||
custom: false,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -2417,6 +2420,19 @@ impl TwinToolCall {
|
|||
Self::new_raw_arguments("apply_patch", Value::Null, patch.into())
|
||||
}
|
||||
|
||||
/// A free-form `custom_tool_call` on the Responses API, carrying `input`
|
||||
/// as text rather than JSON arguments. This is how the codex harness's
|
||||
/// `apply_patch` reaches the model.
|
||||
#[must_use]
|
||||
pub fn custom(name: impl Into<String>, input: impl Into<String>) -> Self {
|
||||
Self {
|
||||
name: name.into(),
|
||||
arguments: Value::String(input.into()),
|
||||
raw_arguments: None,
|
||||
custom: true,
|
||||
}
|
||||
}
|
||||
|
||||
fn into_json(self) -> Value {
|
||||
let mut value = json!({
|
||||
"name": self.name,
|
||||
|
|
@ -2425,6 +2441,9 @@ impl TwinToolCall {
|
|||
if let Some(raw_arguments) = self.raw_arguments {
|
||||
value["raw_arguments"] = Value::String(raw_arguments);
|
||||
}
|
||||
if self.custom {
|
||||
value["kind"] = Value::String("custom".to_string());
|
||||
}
|
||||
value
|
||||
}
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue