diff --git a/Cargo.lock b/Cargo.lock index 12036503a..7e297f549 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -5869,7 +5869,7 @@ dependencies = [ [[package]] name = "pebble-agent" version = "0.1.0" -source = "git+https://github.com/lithoscomputer/pebble?rev=69969420c9017ca15ac6c175c820a0cb8090866a#69969420c9017ca15ac6c175c820a0cb8090866a" +source = "git+https://github.com/lithoscomputer/pebble?rev=6d802a9d2e9c356e76371089e94d4d80c0ba16c5#6d802a9d2e9c356e76371089e94d4d80c0ba16c5" dependencies = [ "async-trait", "futures-util", @@ -5886,7 +5886,7 @@ dependencies = [ [[package]] name = "pebble-cli-core" version = "0.1.0" -source = "git+https://github.com/lithoscomputer/pebble?rev=69969420c9017ca15ac6c175c820a0cb8090866a#69969420c9017ca15ac6c175c820a0cb8090866a" +source = "git+https://github.com/lithoscomputer/pebble?rev=6d802a9d2e9c356e76371089e94d4d80c0ba16c5#6d802a9d2e9c356e76371089e94d4d80c0ba16c5" dependencies = [ "anyhow", "async-trait", @@ -5915,7 +5915,7 @@ dependencies = [ [[package]] name = "pebble-coding-agent" version = "0.1.0" -source = "git+https://github.com/lithoscomputer/pebble?rev=69969420c9017ca15ac6c175c820a0cb8090866a#69969420c9017ca15ac6c175c820a0cb8090866a" +source = "git+https://github.com/lithoscomputer/pebble?rev=6d802a9d2e9c356e76371089e94d4d80c0ba16c5#6d802a9d2e9c356e76371089e94d4d80c0ba16c5" dependencies = [ "async-trait", "futures-util", diff --git a/Cargo.toml b/Cargo.toml index c932182a1..b539ed615 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -122,9 +122,9 @@ sandbox-driver-testing = { git = "https://github.com/lithoscomputer/sandbox-driv # sandbox, so the pebble and sandbox-driver pins move independently. Pebble # pins the same lithos-llm rev as fabro, and its lockfile policy is that # every shared crate resolves to the version lithos-llm locks. -pebble-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "69969420c9017ca15ac6c175c820a0cb8090866a" } -pebble-coding-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "69969420c9017ca15ac6c175c820a0cb8090866a", features = ["mcp", "search-providers"] } -pebble-cli-core = { git = "https://github.com/lithoscomputer/pebble", rev = "69969420c9017ca15ac6c175c820a0cb8090866a" } +pebble-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "6d802a9d2e9c356e76371089e94d4d80c0ba16c5" } +pebble-coding-agent = { git = "https://github.com/lithoscomputer/pebble", rev = "6d802a9d2e9c356e76371089e94d4d80c0ba16c5", features = ["mcp", "search-providers"] } +pebble-cli-core = { git = "https://github.com/lithoscomputer/pebble", rev = "6d802a9d2e9c356e76371089e94d4d80c0ba16c5" } sentry = { version = "0.35", default-features = false, features = ["backtrace", "contexts", "ureq", "rustls"] } fork = "0.2" exec = "0.3" diff --git a/lib/apps/fabro-cli/src/args.rs b/lib/apps/fabro-cli/src/args.rs index ac90101d9..09d7a01dc 100644 --- a/lib/apps/fabro-cli/src/args.rs +++ b/lib/apps/fabro-cli/src/args.rs @@ -1198,7 +1198,8 @@ pub(crate) struct AgentArgs { #[arg(long)] pub(crate) debug: bool, - /// Print full LLM request/response JSON to stderr + /// Print tool results, the transcript, and full LLM request/response JSON + /// to stderr #[arg(long)] pub(crate) verbose: bool, diff --git a/lib/apps/fabro-cli/src/commands/exec.rs b/lib/apps/fabro-cli/src/commands/exec.rs index 2f3a4d5a2..8e8101319 100644 --- a/lib/apps/fabro-cli/src/commands/exec.rs +++ b/lib/apps/fabro-cli/src/commands/exec.rs @@ -2,9 +2,12 @@ //! //! The session is pebble's coding agent over a local sandbox, run through //! pebble's own command-line session: its event renderer, closing summary, -//! and terminal approval prompt. What is fabro's here is the client (model -//! calls go either straight to the provider with the CLI's credentials or -//! through a Fabro server's completions endpoint when a server target is +//! and terminal approval prompt. `--verbose` asks that renderer for each +//! tool call's arguments and result in full and for the transcript, and adds +//! fabro's own dump of every model request and response; without it the +//! renderer prints what it always has. What is fabro's here is the client +//! (model calls go either straight to the provider with the CLI's credentials +//! or through a Fabro server's completions endpoint when a server target is //! set), the sandbox, the MCP servers, skills, search, and redaction. use std::collections::HashMap; @@ -31,8 +34,8 @@ use fabro_util::terminal::Styles; use fabro_workflow::web_search::{self, SearchSecrets}; use lithos_llm::catalog::ProviderId; use pebble_cli_core::approval::TerminalApproval; -use pebble_cli_core::render::{self, JsonStream, Style}; -use pebble_cli_core::session::{SessionOptions, run_prompt}; +use pebble_cli_core::render::{self, JsonStream, RenderOptions, Style}; +use pebble_cli_core::session::{SessionOptions, run_prompt_with}; use pebble_coding_agent::environment::Environment; use pebble_coding_agent::subagents::SubagentOptions; use pebble_coding_agent::tools::{PermissionLevelPolicy, PermissionMiddleware}; @@ -497,7 +500,16 @@ async fn run_session( sigint_token.cancel(); }); - let report = run_prompt(agent, args.prompt.as_str(), &cancel_token, session).await?; + // `--verbose` asks the renderer for everything it can say: each tool + // call's arguments and result in full, and the transcript. The answer on + // stdout is the session's and is not changed by it. + let render = if args.verbose { + RenderOptions::verbose() + } else { + RenderOptions::default() + }; + let report = + run_prompt_with(agent, args.prompt.as_str(), &cancel_token, session, render).await?; report .result .map(|_| ()) diff --git a/lib/apps/fabro-cli/tests/it/cmd/exec.rs b/lib/apps/fabro-cli/tests/it/cmd/exec.rs index 7f2d74720..d1acba902 100644 --- a/lib/apps/fabro-cli/tests/it/cmd/exec.rs +++ b/lib/apps/fabro-cli/tests/it/cmd/exec.rs @@ -260,7 +260,7 @@ fn help() { --quiet Suppress non-essential output [env: FABRO_QUIET=] --auto-approve Skip interactive prompts; deny tools outside permission level --debug Print LLM request/response debug info to stderr - --verbose Print full LLM request/response JSON to stderr + --verbose Print tool results, the transcript, and full LLM request/response JSON to stderr --skills-dir Directory containing skill files (overrides default discovery) --output-format Output format (text for human-readable, json for NDJSON event stream) [possible values: text, json] -h, --help Print help @@ -914,7 +914,10 @@ async fn twin_exec_shell_command() { .input_contains( "Run the shell command `echo hello_from_shell` and tell me what it printed", ) - .tool_call(fabro_test::TwinToolCall::shell("echo hello_from_shell")) + .tool_call(fabro_test::TwinToolCall::shell("echo hello_from_shell")), + ) + .scenario( + fabro_test::TwinScenario::responses("gpt-5.4-mini") .text("It printed hello_from_shell."), ) .load(twin) @@ -934,9 +937,80 @@ async fn twin_exec_shell_command() { ]); let output = run_success_output(cmd).await; let stdout = String::from_utf8(output.stdout).expect("valid utf8"); + let stderr = String::from_utf8(output.stderr).expect("valid utf8"); + let stderr = console::strip_ansi_codes(&stderr); + assert_eq!( + stdout, "It printed hello_from_shell.\n", + "without --verbose, stdout carries the final answer only; stderr:\n{stderr}" + ); + for absent in ["[result]", "[reasoning]", "[verbose]"] { + assert!( + !stderr.contains(absent), + "without --verbose, stderr should not carry {absent}:\n{stderr}" + ); + } +} + +#[fabro_macros::e2e_test(twin)] +async fn twin_exec_verbose_prints_tool_calls_and_results() { + let context = test_context!(); + let twin = fabro_test::twin_openai().await; + let namespace = format!("{}::{}", module_path!(), line!()); + fabro_test::TwinScenarios::new(namespace.clone()) + .scenario( + fabro_test::TwinScenario::responses("gpt-5.4-mini") + .input_contains( + "Run the shell command `echo verbose_marker` and tell me what it printed", + ) + .tool_call(fabro_test::TwinToolCall::shell("echo verbose_marker")), + ) + .scenario( + fabro_test::TwinScenario::responses("gpt-5.4-mini").text("It printed verbose_marker."), + ) + .load(twin) + .await; + + let mut cmd = context.exec_cmd(); + twin.configure_command(&mut cmd, &namespace); + cmd.args([ + "--auto-approve", + "--permissions", + "full", + "--verbose", + "--provider", + "openai", + "--model", + "gpt-5.4-mini", + "Run the shell command `echo verbose_marker` and tell me what it printed", + ]); + let output = run_success_output(cmd).await; + let stdout = String::from_utf8(output.stdout).expect("valid utf8"); + let stderr = String::from_utf8(output.stderr).expect("valid utf8"); + let stderr = console::strip_ansi_codes(&stderr); + assert_eq!( + stdout, "It printed verbose_marker.\n", + "--verbose should leave the answer on stdout unchanged; stderr:\n{stderr}" + ); + // Pebble prints the call's arguments in full under its `[tool]` line and + // what the call answered under a `[result]` line; fabro's middleware + // still dumps each model request. (A streamed response is not dumped.) + for expected in [ + "[tool] shell\n", + "\"command\": \"echo verbose_marker\"", + "[result] shell\n", + "[verbose] request:", + ] { + assert!( + stderr.contains(expected), + "--verbose should print {expected:?} on stderr, got:\n{stderr}" + ); + } + let (_, result) = stderr + .split_once("[result] shell\n") + .expect("the result block should follow the tool line"); assert!( - stdout.contains("hello_from_shell"), - "expected shell marker in output, got: {stdout}" + result.contains("verbose_marker"), + "the [result] block should carry what the shell printed, got:\n{result}" ); }