mirror of
https://github.com/fabro-sh/fabro.git
synced 2026-10-11 03:40:05 +00:00
parent
ebf67381b0
commit
86ef1dd6e2
6 changed files with 1685 additions and 40 deletions
398
run.json
398
run.json
File diff suppressed because one or more lines are too long
614
stages/006-simplify_opus@1/diff.patch
Normal file
614
stages/006-simplify_opus@1/diff.patch
Normal file
|
|
@ -0,0 +1,614 @@
|
|||
diff --git a/apps/fabro-web/app/lib/query-keys.test.ts b/apps/fabro-web/app/lib/query-keys.test.ts
|
||||
index c5ac22a03..4ac4f438e 100644
|
||||
--- a/apps/fabro-web/app/lib/query-keys.test.ts
|
||||
+++ b/apps/fabro-web/app/lib/query-keys.test.ts
|
||||
@@ -69,7 +69,6 @@ describe("queryKeys", () => {
|
||||
queryKeys.runs.graph("run-1", "TB"),
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "stage-1"),
|
||||
- queryKeys.runs.stageContextWindow("run-1", "stage-1"),
|
||||
]);
|
||||
expect(queryKeysForRunEvent("run-1", "run.title.updated")).toEqual([
|
||||
queryKeys.runs.detail("run-1"),
|
||||
@@ -87,7 +86,6 @@ describe("queryKeys", () => {
|
||||
]) {
|
||||
expect(queryKeysForRunEvent("run-1", event, "stage-1")).toEqual([
|
||||
queryKeys.runs.stageEvents("run-1", "stage-1"),
|
||||
- queryKeys.runs.stageContextWindow("run-1", "stage-1"),
|
||||
]);
|
||||
}
|
||||
});
|
||||
diff --git a/apps/fabro-web/app/lib/run-events.test.tsx b/apps/fabro-web/app/lib/run-events.test.tsx
|
||||
index 306ddf6b1..141826e41 100644
|
||||
--- a/apps/fabro-web/app/lib/run-events.test.tsx
|
||||
+++ b/apps/fabro-web/app/lib/run-events.test.tsx
|
||||
@@ -61,7 +61,6 @@ describe("queryKeysForRunEvent", () => {
|
||||
queryKeys.runs.graph("run-1", "TB"),
|
||||
queryKeys.runs.detail("run-1"),
|
||||
queryKeys.runs.stageEvents("run-1", "verify@2"),
|
||||
- queryKeys.runs.stageContextWindow("run-1", "verify@2"),
|
||||
]);
|
||||
});
|
||||
|
||||
@@ -69,7 +68,6 @@ describe("queryKeysForRunEvent", () => {
|
||||
expect(queryKeysForRunEvent("run-1", "agent.session.activated", "agent@1")).toEqual([
|
||||
queryKeys.runs.events("run-1", 1000),
|
||||
queryKeys.runs.stageEvents("run-1", "agent@1"),
|
||||
- queryKeys.runs.stageContextWindow("run-1", "agent@1"),
|
||||
]);
|
||||
});
|
||||
|
||||
@@ -77,18 +75,15 @@ describe("queryKeysForRunEvent", () => {
|
||||
expect(queryKeysForRunEvent("run-1", "agent.interrupt.injected", "nap@1")).toEqual([
|
||||
queryKeys.runs.events("run-1", 1000),
|
||||
queryKeys.runs.stageEvents("run-1", "nap@1"),
|
||||
- queryKeys.runs.stageContextWindow("run-1", "nap@1"),
|
||||
]);
|
||||
});
|
||||
|
||||
test("pair messages invalidate the stage events query", () => {
|
||||
expect(queryKeysForRunEvent("run-1", "agent.pair.user_message", "nap@1")).toEqual([
|
||||
queryKeys.runs.stageEvents("run-1", "nap@1"),
|
||||
- queryKeys.runs.stageContextWindow("run-1", "nap@1"),
|
||||
]);
|
||||
expect(queryKeysForRunEvent("run-1", "agent.pair.system_message", "nap@1")).toEqual([
|
||||
queryKeys.runs.stageEvents("run-1", "nap@1"),
|
||||
- queryKeys.runs.stageContextWindow("run-1", "nap@1"),
|
||||
]);
|
||||
});
|
||||
|
||||
@@ -116,7 +111,6 @@ describe("queryKeysForRunEvent", () => {
|
||||
queryKeys.runs.state("run-1"),
|
||||
queryKeys.runs.events("run-1", 1000),
|
||||
queryKeys.runs.stageEvents("run-1", "code@1"),
|
||||
- queryKeys.runs.stageContextWindow("run-1", "code@1"),
|
||||
]);
|
||||
});
|
||||
});
|
||||
diff --git a/apps/fabro-web/app/lib/run-events.ts b/apps/fabro-web/app/lib/run-events.ts
|
||||
index 7c8898257..70bacc7d7 100644
|
||||
--- a/apps/fabro-web/app/lib/run-events.ts
|
||||
+++ b/apps/fabro-web/app/lib/run-events.ts
|
||||
@@ -155,7 +155,6 @@ export function queryKeysForRunEvent(
|
||||
];
|
||||
if (stageId) {
|
||||
keys.push(queryKeys.runs.stageEvents(runId, stageId));
|
||||
- keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
@@ -164,7 +163,6 @@ export function queryKeysForRunEvent(
|
||||
const keys: Key[] = [queryKeys.runs.events(runId, 1000)];
|
||||
if (stageId) {
|
||||
keys.push(queryKeys.runs.stageEvents(runId, stageId));
|
||||
- keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
@@ -179,12 +177,7 @@ export function queryKeysForRunEvent(
|
||||
}
|
||||
|
||||
if (STAGE_ACTIVITY_EVENTS.has(event)) {
|
||||
- return stageId
|
||||
- ? [
|
||||
- queryKeys.runs.stageEvents(runId, stageId),
|
||||
- queryKeys.runs.stageContextWindow(runId, stageId),
|
||||
- ]
|
||||
- : [];
|
||||
+ return stageId ? [queryKeys.runs.stageEvents(runId, stageId)] : [];
|
||||
}
|
||||
|
||||
if (TODO_EVENTS.has(event)) {
|
||||
@@ -194,7 +187,6 @@ export function queryKeysForRunEvent(
|
||||
];
|
||||
if (stageId) {
|
||||
keys.push(queryKeys.runs.stageEvents(runId, stageId));
|
||||
- keys.push(queryKeys.runs.stageContextWindow(runId, stageId));
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
diff --git a/docs/public/api-reference/fabro-api.yaml b/docs/public/api-reference/fabro-api.yaml
|
||||
index 7094c1ca5..92fe2dd46 100644
|
||||
--- a/docs/public/api-reference/fabro-api.yaml
|
||||
+++ b/docs/public/api-reference/fabro-api.yaml
|
||||
@@ -7935,16 +7935,11 @@ components:
|
||||
type: object
|
||||
required:
|
||||
- category
|
||||
- - label
|
||||
- tokens
|
||||
- usage_percent
|
||||
- - source
|
||||
properties:
|
||||
category:
|
||||
$ref: "#/components/schemas/StageContextWindowCategory"
|
||||
- label:
|
||||
- type: string
|
||||
- example: System prompt
|
||||
tokens:
|
||||
type: integer
|
||||
format: uint64
|
||||
@@ -7955,10 +7950,6 @@ components:
|
||||
format: double
|
||||
minimum: 0
|
||||
example: 7.5
|
||||
- source:
|
||||
- type: string
|
||||
- description: Content-free source label for the category count.
|
||||
- example: scaled_local_estimate
|
||||
|
||||
StageContextWindowProjection:
|
||||
description: Durable content-free context-window snapshot projected onto an agent stage.
|
||||
diff --git a/lib/crates/fabro-agent/src/context_window.rs b/lib/crates/fabro-agent/src/context_window.rs
|
||||
index cb9e1dd51..b441c4367 100644
|
||||
--- a/lib/crates/fabro-agent/src/context_window.rs
|
||||
+++ b/lib/crates/fabro-agent/src/context_window.rs
|
||||
@@ -7,18 +7,14 @@ use fabro_llm::token_count::{
|
||||
};
|
||||
use fabro_llm::types::{Request, Role, Warning as LlmWarning};
|
||||
use fabro_types::{
|
||||
- StageContextWindowBreakdownProjection, StageContextWindowCategory,
|
||||
- StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness,
|
||||
- StageContextWindowWarning,
|
||||
+ StageContextWindowBreakdownItem, StageContextWindowCategory, StageContextWindowCountMethod,
|
||||
+ StageContextWindowProjection, StageContextWindowStaleness, StageContextWindowWarning,
|
||||
};
|
||||
|
||||
use crate::memory::MemoryDocument;
|
||||
use crate::skills::{Skill, format_skills_prompt_section};
|
||||
use crate::tool_registry::{ToolDefinitionWithSource, ToolSource};
|
||||
|
||||
-const LOCAL_BREAKDOWN_SOURCE: &str = "local_estimate";
|
||||
-const SCALED_BREAKDOWN_SOURCE: &str = "scaled_local_estimate";
|
||||
-
|
||||
#[derive(Clone, Copy)]
|
||||
pub(crate) struct ContextWindowSnapshotInput<'a> {
|
||||
pub request: &'a Request,
|
||||
@@ -57,7 +53,6 @@ pub(crate) fn build_local_snapshot(
|
||||
context_window_tokens: u64::try_from(input.context_window_tokens).unwrap_or(u64::MAX),
|
||||
count_method: StageContextWindowCountMethod::LocalEstimate,
|
||||
staleness: StageContextWindowStaleness::Live,
|
||||
- breakdown_source: LOCAL_BREAKDOWN_SOURCE,
|
||||
warnings,
|
||||
})
|
||||
}
|
||||
@@ -206,12 +201,10 @@ impl BreakdownBuilder {
|
||||
let breakdown = self
|
||||
.tokens
|
||||
.into_iter()
|
||||
- .map(|(category, tokens)| StageContextWindowBreakdownProjection {
|
||||
+ .map(|(category, tokens)| StageContextWindowBreakdownItem {
|
||||
category,
|
||||
- label: category_label(category).to_string(),
|
||||
tokens,
|
||||
usage_percent: usage_percent(tokens, meta.context_window_tokens),
|
||||
- source: meta.breakdown_source.to_string(),
|
||||
})
|
||||
.collect();
|
||||
StageContextWindowProjection {
|
||||
@@ -236,84 +229,55 @@ struct SnapshotMeta {
|
||||
context_window_tokens: u64,
|
||||
count_method: StageContextWindowCountMethod,
|
||||
staleness: StageContextWindowStaleness,
|
||||
- breakdown_source: &'static str,
|
||||
warnings: Vec<StageContextWindowWarning>,
|
||||
}
|
||||
|
||||
+/// Proportionally scale a local breakdown so it sums to `target_total`. Any
|
||||
+/// rounding leftover is absorbed by the last bucket; this is a best-effort
|
||||
+/// estimate, not exact apportionment.
|
||||
fn scale_breakdown(
|
||||
- breakdown: &[StageContextWindowBreakdownProjection],
|
||||
+ breakdown: &[StageContextWindowBreakdownItem],
|
||||
target_total: u64,
|
||||
context_window_tokens: u64,
|
||||
-) -> Vec<StageContextWindowBreakdownProjection> {
|
||||
+) -> Vec<StageContextWindowBreakdownItem> {
|
||||
let local_total = breakdown.iter().map(|item| item.tokens).sum::<u64>();
|
||||
- if target_total == local_total {
|
||||
- return breakdown
|
||||
- .iter()
|
||||
- .cloned()
|
||||
- .map(|mut item| {
|
||||
- item.source = SCALED_BREAKDOWN_SOURCE.to_string();
|
||||
- item
|
||||
- })
|
||||
- .collect();
|
||||
- }
|
||||
if breakdown.is_empty() || local_total == 0 {
|
||||
return (target_total > 0)
|
||||
- .then(|| StageContextWindowBreakdownProjection {
|
||||
+ .then(|| StageContextWindowBreakdownItem {
|
||||
category: StageContextWindowCategory::Other,
|
||||
- label: category_label(StageContextWindowCategory::Other).to_string(),
|
||||
tokens: target_total,
|
||||
usage_percent: usage_percent(target_total, context_window_tokens),
|
||||
- source: SCALED_BREAKDOWN_SOURCE.to_string(),
|
||||
})
|
||||
.into_iter()
|
||||
.collect();
|
||||
}
|
||||
|
||||
- let mut scaled = Vec::with_capacity(breakdown.len());
|
||||
- let mut allocated = 0u64;
|
||||
- let mut remainders = Vec::with_capacity(breakdown.len());
|
||||
- let local_total_u128 = u128::from(local_total);
|
||||
- for (index, item) in breakdown.iter().enumerate() {
|
||||
- let numerator = u128::from(item.tokens).saturating_mul(u128::from(target_total));
|
||||
- let tokens = u64::try_from(numerator / local_total_u128).unwrap_or(u64::MAX);
|
||||
- allocated = allocated.saturating_add(tokens);
|
||||
- remainders.push((index, numerator % local_total_u128));
|
||||
- scaled.push(StageContextWindowBreakdownProjection {
|
||||
- category: item.category,
|
||||
- label: item.label.clone(),
|
||||
- tokens,
|
||||
- usage_percent: 0.0,
|
||||
- source: SCALED_BREAKDOWN_SOURCE.to_string(),
|
||||
- });
|
||||
- }
|
||||
-
|
||||
- let mut remaining = target_total.saturating_sub(allocated);
|
||||
- remainders.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(&right.0)));
|
||||
- for (index, _) in remainders {
|
||||
- if remaining == 0 {
|
||||
- break;
|
||||
+ let mut scaled: Vec<_> = breakdown
|
||||
+ .iter()
|
||||
+ .map(|item| {
|
||||
+ let scaled = u128::from(item.tokens).saturating_mul(u128::from(target_total))
|
||||
+ / u128::from(local_total);
|
||||
+ let tokens = u64::try_from(scaled).unwrap_or(u64::MAX);
|
||||
+ StageContextWindowBreakdownItem {
|
||||
+ category: item.category,
|
||||
+ tokens,
|
||||
+ usage_percent: usage_percent(tokens, context_window_tokens),
|
||||
+ }
|
||||
+ })
|
||||
+ .collect();
|
||||
+
|
||||
+ // Push any rounding leftover into the last bucket so totals match exactly.
|
||||
+ let allocated: u64 = scaled.iter().map(|item| item.tokens).sum();
|
||||
+ if let Some(last) = scaled.last_mut() {
|
||||
+ let leftover = target_total.saturating_sub(allocated);
|
||||
+ if leftover > 0 {
|
||||
+ last.tokens = last.tokens.saturating_add(leftover);
|
||||
+ last.usage_percent = usage_percent(last.tokens, context_window_tokens);
|
||||
}
|
||||
- scaled[index].tokens = scaled[index].tokens.saturating_add(1);
|
||||
- remaining -= 1;
|
||||
- }
|
||||
- for item in &mut scaled {
|
||||
- item.usage_percent = usage_percent(item.tokens, context_window_tokens);
|
||||
}
|
||||
scaled
|
||||
}
|
||||
|
||||
-fn category_label(category: StageContextWindowCategory) -> &'static str {
|
||||
- match category {
|
||||
- StageContextWindowCategory::SystemPrompt => "System prompt",
|
||||
- StageContextWindowCategory::Tools => "Tools",
|
||||
- StageContextWindowCategory::McpTools => "MCP tools",
|
||||
- StageContextWindowCategory::Skills => "Skills",
|
||||
- StageContextWindowCategory::Memory => "Memory",
|
||||
- StageContextWindowCategory::Conversation => "Conversation",
|
||||
- StageContextWindowCategory::Other => "Other",
|
||||
- }
|
||||
-}
|
||||
-
|
||||
fn usage_percent(tokens: u64, denominator: u64) -> f64 {
|
||||
if denominator == 0 {
|
||||
0.0
|
||||
@@ -444,19 +408,15 @@ mod tests {
|
||||
generated_at: Utc::now(),
|
||||
event_seq: None,
|
||||
breakdown: vec![
|
||||
- StageContextWindowBreakdownProjection {
|
||||
+ StageContextWindowBreakdownItem {
|
||||
category: StageContextWindowCategory::SystemPrompt,
|
||||
- label: "System prompt".to_string(),
|
||||
tokens: 10,
|
||||
usage_percent: 0.0,
|
||||
- source: LOCAL_BREAKDOWN_SOURCE.to_string(),
|
||||
},
|
||||
- StageContextWindowBreakdownProjection {
|
||||
+ StageContextWindowBreakdownItem {
|
||||
category: StageContextWindowCategory::Conversation,
|
||||
- label: "Conversation".to_string(),
|
||||
tokens: 20,
|
||||
usage_percent: 0.0,
|
||||
- source: LOCAL_BREAKDOWN_SOURCE.to_string(),
|
||||
},
|
||||
],
|
||||
warnings: Vec::new(),
|
||||
@@ -474,11 +434,5 @@ mod tests {
|
||||
scaled.breakdown.iter().map(|item| item.tokens).sum::<u64>(),
|
||||
101
|
||||
);
|
||||
- assert!(
|
||||
- scaled
|
||||
- .breakdown
|
||||
- .iter()
|
||||
- .all(|item| item.source == SCALED_BREAKDOWN_SOURCE)
|
||||
- );
|
||||
}
|
||||
}
|
||||
diff --git a/lib/crates/fabro-agent/src/tool_registry.rs b/lib/crates/fabro-agent/src/tool_registry.rs
|
||||
index df2be273e..3e43d7e34 100644
|
||||
--- a/lib/crates/fabro-agent/src/tool_registry.rs
|
||||
+++ b/lib/crates/fabro-agent/src/tool_registry.rs
|
||||
@@ -131,18 +131,9 @@ impl ToolRegistry {
|
||||
policy: Option<&dyn ToolAccessPolicy>,
|
||||
exposure_mode: ToolExposureMode,
|
||||
) -> Vec<ToolDefinition> {
|
||||
- let Some(policy) = policy else {
|
||||
- return self.definitions();
|
||||
- };
|
||||
-
|
||||
- self.tools
|
||||
- .values()
|
||||
- .filter(|tool| {
|
||||
- policy
|
||||
- .access_for_tool(&tool.definition.name)
|
||||
- .is_exposed(exposure_mode)
|
||||
- })
|
||||
- .map(|tool| tool.definition.clone())
|
||||
+ self.definitions_with_source_for_policy(policy, exposure_mode)
|
||||
+ .into_iter()
|
||||
+ .map(|tool| tool.definition)
|
||||
.collect()
|
||||
}
|
||||
|
||||
@@ -152,16 +143,14 @@ impl ToolRegistry {
|
||||
policy: Option<&dyn ToolAccessPolicy>,
|
||||
exposure_mode: ToolExposureMode,
|
||||
) -> Vec<ToolDefinitionWithSource> {
|
||||
- let Some(policy) = policy else {
|
||||
- return self.definitions_with_source();
|
||||
- };
|
||||
-
|
||||
self.tools
|
||||
.values()
|
||||
.filter(|tool| {
|
||||
- policy
|
||||
- .access_for_tool(&tool.definition.name)
|
||||
- .is_exposed(exposure_mode)
|
||||
+ policy.is_none_or(|policy| {
|
||||
+ policy
|
||||
+ .access_for_tool(&tool.definition.name)
|
||||
+ .is_exposed(exposure_mode)
|
||||
+ })
|
||||
})
|
||||
.map(|tool| ToolDefinitionWithSource {
|
||||
definition: tool.definition.clone(),
|
||||
diff --git a/lib/crates/fabro-api/src/lib.rs b/lib/crates/fabro-api/src/lib.rs
|
||||
index 346d0f036..0f9d0e6a1 100644
|
||||
--- a/lib/crates/fabro-api/src/lib.rs
|
||||
+++ b/lib/crates/fabro-api/src/lib.rs
|
||||
@@ -50,12 +50,11 @@ pub mod types {
|
||||
SecretMetadata, SecretType, ServerSettings, SessionDetail, SessionId, SessionMessage,
|
||||
SessionRecord, SessionStatus, SessionSummary, SessionTurn, SkillsProjection,
|
||||
StageCompletion, StageContextWindow, StageContextWindowBreakdownItem,
|
||||
- StageContextWindowBreakdownProjection, StageContextWindowCategory,
|
||||
- StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness,
|
||||
- StageContextWindowUnavailableReason, StageContextWindowWarning, StageHandler,
|
||||
- StageModelUsage, StageOutcome, StageProjection, StageState, SubAgentProjection,
|
||||
- SubAgentStatus, SystemActorKind, TodoListProjection, TurnId, UserPrincipal,
|
||||
- WorkflowSettings,
|
||||
+ StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection,
|
||||
+ StageContextWindowStaleness, StageContextWindowUnavailableReason,
|
||||
+ StageContextWindowWarning, StageHandler, StageModelUsage, StageOutcome, StageProjection,
|
||||
+ StageState, SubAgentProjection, SubAgentStatus, SystemActorKind, TodoListProjection,
|
||||
+ TurnId, UserPrincipal, WorkflowSettings,
|
||||
};
|
||||
|
||||
pub use crate::generated::types::*;
|
||||
diff --git a/lib/crates/fabro-api/tests/stage_projection_round_trip.rs b/lib/crates/fabro-api/tests/stage_projection_round_trip.rs
|
||||
index ff27814d8..d6ca201f8 100644
|
||||
--- a/lib/crates/fabro-api/tests/stage_projection_round_trip.rs
|
||||
+++ b/lib/crates/fabro-api/tests/stage_projection_round_trip.rs
|
||||
@@ -161,10 +161,8 @@ fn stage_projection_round_trips_representative_json() {
|
||||
"breakdown": [
|
||||
{
|
||||
"category": "system_prompt",
|
||||
- "label": "System prompt",
|
||||
"tokens": 30000,
|
||||
- "usage_percent": 7.5,
|
||||
- "source": "scaled_local_estimate"
|
||||
+ "usage_percent": 7.5
|
||||
}
|
||||
],
|
||||
"warnings": []
|
||||
@@ -194,10 +192,8 @@ fn stage_context_window_response_round_trips_representative_json() {
|
||||
"breakdown": [
|
||||
{
|
||||
"category": "system_prompt",
|
||||
- "label": "System prompt",
|
||||
"tokens": 30000,
|
||||
- "usage_percent": 7.5,
|
||||
- "source": "scaled_local_estimate"
|
||||
+ "usage_percent": 7.5
|
||||
}
|
||||
],
|
||||
"warnings": [
|
||||
diff --git a/lib/crates/fabro-server/src/server/handler/runs.rs b/lib/crates/fabro-server/src/server/handler/runs.rs
|
||||
index a8aff0c48..1b34ca888 100644
|
||||
--- a/lib/crates/fabro-server/src/server/handler/runs.rs
|
||||
+++ b/lib/crates/fabro-server/src/server/handler/runs.rs
|
||||
@@ -22,7 +22,7 @@ use fabro_llm::client::Client as LlmClient;
|
||||
use fabro_types::{
|
||||
Principal, RunClientProvenance, RunId, RunProvenance, RunServerProvenance, StageContextWindow,
|
||||
StageContextWindowStaleness, StageContextWindowUnavailableReason, StageHandler,
|
||||
- StageModelUsage, StageProjection, StageState, SystemActorKind, parse_blob_ref,
|
||||
+ StageModelUsage, StageProjection, SystemActorKind, parse_blob_ref,
|
||||
};
|
||||
use fabro_util::version::FABRO_VERSION;
|
||||
use fabro_workflow::command_log::{command_log_path, read_json_string_blob, read_log_slice};
|
||||
@@ -1059,7 +1059,7 @@ async fn get_run_stage_context_window(
|
||||
};
|
||||
|
||||
let mut response = StageContextWindow::available(stage_id, snapshot);
|
||||
- if !is_live_stage(stage.state) {
|
||||
+ if stage.state.is_terminal() {
|
||||
response.staleness = StageContextWindowStaleness::Stored;
|
||||
}
|
||||
Json(response).into_response()
|
||||
@@ -1077,13 +1077,6 @@ fn is_agent_context_window_stage(stage: &StageProjection) -> bool {
|
||||
})
|
||||
}
|
||||
|
||||
-fn is_live_stage(state: StageState) -> bool {
|
||||
- matches!(
|
||||
- state,
|
||||
- StageState::Pending | StageState::Running | StageState::Retrying
|
||||
- )
|
||||
-}
|
||||
-
|
||||
async fn get_run_stage_command_log(
|
||||
RequireCommandLog(id, stage_id): RequireCommandLog,
|
||||
State(state): State<Arc<AppState>>,
|
||||
diff --git a/lib/crates/fabro-server/src/server/tests.rs b/lib/crates/fabro-server/src/server/tests.rs
|
||||
index 19bbfca35..833e0c1cf 100644
|
||||
--- a/lib/crates/fabro-server/src/server/tests.rs
|
||||
+++ b/lib/crates/fabro-server/src/server/tests.rs
|
||||
@@ -21,7 +21,7 @@ use fabro_types::settings::ServerAuthMethod;
|
||||
use fabro_types::{
|
||||
AgentBackend, AttrValue, AuthMethod, CommandTermination, FailureCategory, FailureDetail, Graph,
|
||||
InterviewQuestionRecord, Node, Outcome, QuestionType, RunBlobId, RunId, RunSpec,
|
||||
- SandboxProvider, StageContextWindowBreakdownProjection, StageContextWindowCategory,
|
||||
+ SandboxProvider, StageContextWindowBreakdownItem, StageContextWindowCategory,
|
||||
StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness,
|
||||
StageContextWindowWarning, StageModelUsage, StageTiming, SuccessReason, SystemActorKind,
|
||||
WorkflowSettings, fixtures,
|
||||
@@ -3039,12 +3039,10 @@ fn context_window_snapshot(
|
||||
staleness: StageContextWindowStaleness::Live,
|
||||
generated_at: Utc::now(),
|
||||
event_seq: None,
|
||||
- breakdown: vec![StageContextWindowBreakdownProjection {
|
||||
+ breakdown: vec![StageContextWindowBreakdownItem {
|
||||
category: StageContextWindowCategory::Conversation,
|
||||
- label: "Conversation".to_string(),
|
||||
tokens: input_tokens,
|
||||
usage_percent: input_tokens as f64 * 100.0 / 400_000.0,
|
||||
- source: "scaled_local_estimate".to_string(),
|
||||
}],
|
||||
warnings,
|
||||
}
|
||||
diff --git a/lib/crates/fabro-store/src/run_state.rs b/lib/crates/fabro-store/src/run_state.rs
|
||||
index 9cd2a3716..0fb20e238 100644
|
||||
--- a/lib/crates/fabro-store/src/run_state.rs
|
||||
+++ b/lib/crates/fabro-store/src/run_state.rs
|
||||
@@ -1116,7 +1116,7 @@ mod tests {
|
||||
CheckpointRecord, CommandTermination, EventBody, FailureCategory, FailureDetail,
|
||||
FailureReason, Graph, McpServerStatus, Outcome, PendingReason, PullRequestLink,
|
||||
QuestionType, ReasoningEffort, RunApprovalState, RunBlobId, RunControlAction, RunDiff,
|
||||
- RunEvent, RunSize, RunSpec, RunStatus, Speed, StageContextWindowBreakdownProjection,
|
||||
+ RunEvent, RunSize, RunSpec, RunStatus, Speed, StageContextWindowBreakdownItem,
|
||||
StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection,
|
||||
StageContextWindowStaleness, StageContextWindowWarning, StageModelUsage, StageOutcome,
|
||||
StageState, SubAgentStatus, SuccessReason, WorkflowSettings, first_event_seq, fixtures,
|
||||
@@ -4189,12 +4189,10 @@ mod tests {
|
||||
staleness: StageContextWindowStaleness::Live,
|
||||
generated_at: Utc::now(),
|
||||
event_seq: None,
|
||||
- breakdown: vec![StageContextWindowBreakdownProjection {
|
||||
+ breakdown: vec![StageContextWindowBreakdownItem {
|
||||
category: StageContextWindowCategory::Conversation,
|
||||
- label: "Conversation".to_string(),
|
||||
tokens: input_tokens,
|
||||
usage_percent: input_tokens as f64 * 100.0 / 400_000.0,
|
||||
- source: "local_estimate".to_string(),
|
||||
}],
|
||||
warnings: vec![StageContextWindowWarning {
|
||||
code: "local_token_estimate".to_string(),
|
||||
diff --git a/lib/crates/fabro-types/src/lib.rs b/lib/crates/fabro-types/src/lib.rs
|
||||
index 9f3104e6d..622e8729f 100644
|
||||
--- a/lib/crates/fabro-types/src/lib.rs
|
||||
+++ b/lib/crates/fabro-types/src/lib.rs
|
||||
@@ -105,10 +105,9 @@ pub use run_id::{RunId, fixtures};
|
||||
pub use run_projection::{
|
||||
ActivatedSkill, CheckpointRecord, McpServerProjection, McpServerStatus, PendingInterviewRecord,
|
||||
RunProjection, SkillsProjection, StageContextWindow, StageContextWindowBreakdownItem,
|
||||
- StageContextWindowBreakdownProjection, StageContextWindowCategory,
|
||||
- StageContextWindowCountMethod, StageContextWindowProjection, StageContextWindowStaleness,
|
||||
- StageContextWindowUnavailableReason, StageContextWindowWarning, StageModelUsage,
|
||||
- StageProjection, SubAgentProjection, SubAgentStatus, first_event_seq,
|
||||
+ StageContextWindowCategory, StageContextWindowCountMethod, StageContextWindowProjection,
|
||||
+ StageContextWindowStaleness, StageContextWindowUnavailableReason, StageContextWindowWarning,
|
||||
+ StageModelUsage, StageProjection, SubAgentProjection, SubAgentStatus, first_event_seq,
|
||||
};
|
||||
pub use run_sandbox::{RunSandbox, RunSandboxRuntime};
|
||||
pub use run_summary::{
|
||||
diff --git a/lib/crates/fabro-types/src/run_event/mod.rs b/lib/crates/fabro-types/src/run_event/mod.rs
|
||||
index 35e8e72b7..84a58ef16 100644
|
||||
--- a/lib/crates/fabro-types/src/run_event/mod.rs
|
||||
+++ b/lib/crates/fabro-types/src/run_event/mod.rs
|
||||
@@ -2173,12 +2173,10 @@ mod tests {
|
||||
.unwrap()
|
||||
.with_timezone(&Utc),
|
||||
event_seq: None,
|
||||
- breakdown: vec![crate::StageContextWindowBreakdownProjection {
|
||||
+ breakdown: vec![crate::StageContextWindowBreakdownItem {
|
||||
category: crate::StageContextWindowCategory::SystemPrompt,
|
||||
- label: "System prompt".to_string(),
|
||||
tokens: 30_000,
|
||||
usage_percent: 7.5,
|
||||
- source: "scaled_local_estimate".to_string(),
|
||||
}],
|
||||
warnings: vec![crate::StageContextWindowWarning {
|
||||
code: "local_token_estimate".to_string(),
|
||||
diff --git a/lib/crates/fabro-types/src/run_projection.rs b/lib/crates/fabro-types/src/run_projection.rs
|
||||
index 46c13a4b1..b07146cfc 100644
|
||||
--- a/lib/crates/fabro-types/src/run_projection.rs
|
||||
+++ b/lib/crates/fabro-types/src/run_projection.rs
|
||||
@@ -215,16 +215,12 @@ pub struct StageContextWindowWarning {
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)]
|
||||
-pub struct StageContextWindowBreakdownProjection {
|
||||
+pub struct StageContextWindowBreakdownItem {
|
||||
pub category: StageContextWindowCategory,
|
||||
- pub label: String,
|
||||
pub tokens: u64,
|
||||
pub usage_percent: f64,
|
||||
- pub source: String,
|
||||
}
|
||||
|
||||
-pub type StageContextWindowBreakdownItem = StageContextWindowBreakdownProjection;
|
||||
-
|
||||
#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)]
|
||||
pub struct StageContextWindowProjection {
|
||||
pub provider: String,
|
||||
@@ -238,7 +234,7 @@ pub struct StageContextWindowProjection {
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub event_seq: Option<u32>,
|
||||
#[serde(default)]
|
||||
- pub breakdown: Vec<StageContextWindowBreakdownProjection>,
|
||||
+ pub breakdown: Vec<StageContextWindowBreakdownItem>,
|
||||
#[serde(default)]
|
||||
pub warnings: Vec<StageContextWindowWarning>,
|
||||
}
|
||||
@@ -267,7 +263,7 @@ pub struct StageContextWindow {
|
||||
#[serde(default)]
|
||||
pub event_seq: Option<u32>,
|
||||
#[serde(default)]
|
||||
- pub breakdown: Vec<StageContextWindowBreakdownProjection>,
|
||||
+ pub breakdown: Vec<StageContextWindowBreakdownItem>,
|
||||
#[serde(default)]
|
||||
pub warnings: Vec<StageContextWindowWarning>,
|
||||
}
|
||||
diff --git a/lib/packages/fabro-api-client/src/models/stage-context-window-breakdown-item.ts b/lib/packages/fabro-api-client/src/models/stage-context-window-breakdown-item.ts
|
||||
index 0798a16d5..52c59712e 100644
|
||||
--- a/lib/packages/fabro-api-client/src/models/stage-context-window-breakdown-item.ts
|
||||
+++ b/lib/packages/fabro-api-client/src/models/stage-context-window-breakdown-item.ts
|
||||
@@ -21,11 +21,6 @@ import type { StageContextWindowCategory } from './stage-context-window-category
|
||||
*/
|
||||
export interface StageContextWindowBreakdownItem {
|
||||
'category': StageContextWindowCategory;
|
||||
- 'label': string;
|
||||
'tokens': number;
|
||||
'usage_percent': number;
|
||||
- /**
|
||||
- * Content-free source label for the category count.
|
||||
- */
|
||||
- 'source': string;
|
||||
}
|
||||
6
stages/006-simplify_opus@1/status.json
Normal file
6
stages/006-simplify_opus@1/status.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"outcome": "succeeded",
|
||||
"notes": "Stage completed: simplify_opus",
|
||||
"failure_reason": null,
|
||||
"timestamp": "2026-05-23T21:28:54.802143Z"
|
||||
}
|
||||
679
stages/007-simplify_gpt@1/prompt.md
Normal file
679
stages/007-simplify_gpt@1/prompt.md
Normal file
|
|
@ -0,0 +1,679 @@
|
|||
Goal: # Context Window Breakdown Endpoint Plan
|
||||
|
||||
Date: 2026-05-23
|
||||
|
||||
## Context
|
||||
|
||||
Agent stage pages already have enough projected data for todos, subagents,
|
||||
skills, and MCP servers through `StageProjection` in
|
||||
`lib/crates/fabro-types/src/run_projection.rs` and the reducer in
|
||||
`lib/crates/fabro-store/src/run_state.rs`. Context-window usage is different:
|
||||
Fabro emits context-window warnings and compaction events today, but it does not
|
||||
store a category breakdown of the model-visible request context.
|
||||
|
||||
`fabro-llm` already exposes the right counting primitive:
|
||||
`Client::count_input_tokens(request, InputTokenCountPreference::PreferProvider)`
|
||||
in `lib/crates/fabro-llm/src/client.rs`. Provider adapters can call native count
|
||||
endpoints for OpenAI, Anthropic, and Gemini, and the client already falls back
|
||||
to local estimates for fallback-eligible failures.
|
||||
|
||||
## Goal
|
||||
|
||||
Add a best-effort context-window API for agent stages:
|
||||
|
||||
```text
|
||||
GET /api/v1/runs/{id}/stages/{stageId}/context-window
|
||||
```
|
||||
|
||||
The endpoint should return the best context-window usage Fabro can produce with
|
||||
no caller-controlled count or accuracy parameters. It may call the configured
|
||||
LLM provider by default. If provider counting is not possible, it should degrade
|
||||
to a local estimate or the latest stored snapshot instead of making the sidebar
|
||||
treat ordinary count gaps as hard errors.
|
||||
|
||||
## Scope
|
||||
|
||||
In scope:
|
||||
|
||||
- OpenAPI contract and generated Rust/TypeScript clients.
|
||||
- Content-free context-window DTOs in the run projection.
|
||||
- A typed agent event for latest context-window snapshots.
|
||||
- Server endpoint that combines live provider counting, local estimates, and
|
||||
stored-snapshot fallback.
|
||||
- Web query key, hook, and SSE invalidation support so the future sidebar can
|
||||
consume the endpoint.
|
||||
|
||||
Out of scope:
|
||||
|
||||
- Building the full new left sidebar UI.
|
||||
- Persisting raw prompt, memory, tool arguments, or message contents for later
|
||||
token counting.
|
||||
- Adding user-visible count-mode or accuracy knobs.
|
||||
- Retrofitting exact historical context-window counts for older completed runs.
|
||||
|
||||
Before implementing, read:
|
||||
|
||||
- `docs/internal/events-strategy.md`
|
||||
- `docs/internal/error-handling-strategy.md`
|
||||
- `docs/internal/testing-strategy.md`
|
||||
|
||||
## API Contract
|
||||
|
||||
Add the route under the existing Run Internals tag in
|
||||
`docs/public/api-reference/fabro-api.yaml`:
|
||||
|
||||
```text
|
||||
GET /runs/{id}/stages/{stageId}/context-window
|
||||
```
|
||||
|
||||
Proposed response shape:
|
||||
|
||||
```json
|
||||
{
|
||||
"stage_id": "implement@1",
|
||||
"available": true,
|
||||
"unavailable_reason": null,
|
||||
"provider": "openai",
|
||||
"model": "gpt-5.4",
|
||||
"context_window_tokens": 400000,
|
||||
"input_tokens": 123456,
|
||||
"usage_percent": 30.86,
|
||||
"count_method": "provider_api_scaled_breakdown",
|
||||
"staleness": "live",
|
||||
"generated_at": "2026-05-23T12:34:56Z",
|
||||
"event_seq": 42,
|
||||
"breakdown": [
|
||||
{
|
||||
"category": "system_prompt",
|
||||
"label": "System prompt",
|
||||
"tokens": 30000,
|
||||
"usage_percent": 7.5,
|
||||
"source": "scaled_local_estimate"
|
||||
}
|
||||
],
|
||||
"warnings": []
|
||||
}
|
||||
```
|
||||
|
||||
Required schemas:
|
||||
|
||||
- `StageContextWindow`
|
||||
- `StageContextWindowBreakdownItem`
|
||||
- `StageContextWindowCategory`
|
||||
- `StageContextWindowCountMethod`
|
||||
- `StageContextWindowStaleness`
|
||||
- `StageContextWindowUnavailableReason`
|
||||
- `StageContextWindowWarning`
|
||||
|
||||
Enums:
|
||||
|
||||
```text
|
||||
StageContextWindowCategory:
|
||||
system_prompt
|
||||
tools
|
||||
mcp_tools
|
||||
skills
|
||||
memory
|
||||
conversation
|
||||
other
|
||||
|
||||
StageContextWindowCountMethod:
|
||||
provider_api_scaled_breakdown
|
||||
response_usage_scaled_breakdown
|
||||
local_estimate
|
||||
|
||||
StageContextWindowStaleness:
|
||||
live
|
||||
stored
|
||||
unavailable
|
||||
|
||||
StageContextWindowUnavailableReason:
|
||||
not_agent_stage
|
||||
not_observed
|
||||
provider_unconfigured
|
||||
```
|
||||
|
||||
Use `available: false` for a real run/stage where Fabro has no context-window
|
||||
data yet. Missing runs and missing stages should still return 404. For
|
||||
`available: false`, return `breakdown: []`, `warnings` explaining the gap, and
|
||||
nullable token fields.
|
||||
|
||||
Define context-window usage as model-visible input/context tokens only:
|
||||
|
||||
- Include prompt input, system/developer content, tool definitions, MCP tool
|
||||
definitions, skills, memory files, conversation history, tool results, and
|
||||
cache input tokens when response usage is the source.
|
||||
- Exclude output tokens and reasoning tokens.
|
||||
- Do not report cost/billing totals here. Existing billing APIs own billing.
|
||||
|
||||
## Data Model
|
||||
|
||||
Add projection-only, content-free context-window types to
|
||||
`lib/crates/fabro-types/src/run_projection.rs`:
|
||||
|
||||
- `StageContextWindowProjection`
|
||||
- `StageContextWindowBreakdownProjection`
|
||||
- category/count/staleness/warning enums, shared with the API through
|
||||
`fabro-api` replacements if the serde shape matches.
|
||||
|
||||
Extend `StageProjection` with:
|
||||
|
||||
```rust
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub context_window: Option<StageContextWindowProjection>,
|
||||
```
|
||||
|
||||
Do not store raw content. The stored snapshot may contain only:
|
||||
|
||||
- provider and model
|
||||
- context window size
|
||||
- input token total
|
||||
- category token counts
|
||||
- count method
|
||||
- generated timestamp
|
||||
- source event sequence
|
||||
- warning codes and messages
|
||||
|
||||
## Events
|
||||
|
||||
Add a typed event in `lib/crates/fabro-types/src/run_event/agent.rs` and
|
||||
`lib/crates/fabro-types/src/run_event/mod.rs`:
|
||||
|
||||
```text
|
||||
agent.context_window.snapshot
|
||||
```
|
||||
|
||||
Event payload should carry the same content-free counts as the projection plus
|
||||
the stage id. The reducer in `lib/crates/fabro-store/src/run_state.rs` should
|
||||
replace the selected stage's `context_window` with the latest snapshot.
|
||||
|
||||
This event is the durable fallback for inactive stages. It should be emitted
|
||||
when the agent assembles or refreshes the LLM request context, before the
|
||||
provider request is sent. If a provider response later supplies better input
|
||||
usage for the same request, emit another snapshot using
|
||||
`response_usage_scaled_breakdown`.
|
||||
|
||||
## Live Counting Design
|
||||
|
||||
The agent should produce snapshots using this decision order whenever it builds
|
||||
an LLM request:
|
||||
|
||||
1. Build a content-free local category breakdown from the same inputs used to
|
||||
assemble the `fabro_llm::Request`.
|
||||
2. Emit an immediate `agent.context_window.snapshot` with `local_estimate` so a
|
||||
sidebar has data even if provider counting is slow or unavailable.
|
||||
3. Attempt provider counting with
|
||||
`Client::count_input_tokens(..., InputTokenCountPreference::PreferProvider)`
|
||||
using the exact in-memory request. This work must not persist or log the raw
|
||||
request.
|
||||
4. If provider count succeeds, scale the local category estimates to the
|
||||
provider total and emit a replacement snapshot with
|
||||
`provider_api_scaled_breakdown`.
|
||||
5. If provider count falls back or fails, keep the local snapshot and include a
|
||||
warning code on the next emitted snapshot. Count failures must not block the
|
||||
agent's normal LLM request.
|
||||
|
||||
The API endpoint should follow this decision order:
|
||||
|
||||
1. Validate the run and stage exist.
|
||||
2. If the stage is not an agent stage, return `available: false` with
|
||||
`unavailable_reason: not_agent_stage`.
|
||||
3. Return the latest projected `StageContextWindowProjection`.
|
||||
4. If no snapshot has ever been observed, return `available: false` with
|
||||
`unavailable_reason: not_observed`.
|
||||
|
||||
Do not make the HTTP server own raw LLM requests. The existing server state
|
||||
tracks live run control and durable projections, while the exact request exists
|
||||
inside the active agent session. Provider counting should therefore happen in
|
||||
the agent/worker process at request-assembly time, and the server endpoint
|
||||
should expose the latest durable snapshot.
|
||||
|
||||
If a future implementation needs user-triggered refreshes, add a separate
|
||||
worker request/response control path. Do not tunnel raw request content through
|
||||
run events or store it in `ManagedRun`.
|
||||
|
||||
The live request snapshot needs to be short-lived and content-safe:
|
||||
|
||||
- Hold raw `Request` content only inside active agent sessions.
|
||||
- Never write that raw request to run events, projection state, logs, or API
|
||||
responses.
|
||||
- Cache provider-count results by run id, stage id, provider, model, and request
|
||||
fingerprint or source event seq so each LLM request is counted at most once.
|
||||
- Clear the live request handle when the session/stage deactivates.
|
||||
|
||||
## Resolved Handoff Decisions
|
||||
|
||||
### Decision 1: Category-Aware Estimation Ownership
|
||||
|
||||
Options:
|
||||
|
||||
- Agent-only estimator: build all category counts in `fabro-agent`.
|
||||
- LLM-only estimator: move the full breakdown model into `fabro-llm`.
|
||||
- Hybrid estimator: keep category ownership in `fabro-agent`, but expose small
|
||||
reusable token-estimation helpers from `fabro-llm`.
|
||||
|
||||
Recommendation: use the hybrid estimator.
|
||||
|
||||
Justification:
|
||||
|
||||
- `fabro-agent` has the category knowledge. It sees memory documents, skills,
|
||||
MCP registration, tool registry policy, and the final session history before
|
||||
`Session::build_request` flattens everything into a generic LLM request.
|
||||
- `fabro-llm` has the token math and provider-neutral request structures. It
|
||||
already owns local count behavior in `token_count.rs`, so duplicating that
|
||||
estimator in `fabro-agent` would drift.
|
||||
- `fabro-llm` should not learn Fabro-specific categories like `skills` or
|
||||
`memory`; that would couple a provider abstraction crate to agent UI
|
||||
semantics.
|
||||
|
||||
Implementation guidance:
|
||||
|
||||
- Add a `fabro-agent/src/context_window.rs` builder that owns the category
|
||||
taxonomy and content-free snapshot assembly.
|
||||
- Expose narrow helpers from `fabro-llm::token_count`, such as tool-definition
|
||||
and message/content-part estimators, instead of making the private estimator
|
||||
logic public wholesale.
|
||||
- Keep the existing `Client::count_input_tokens` provider call as the
|
||||
authoritative total when available.
|
||||
|
||||
### Decision 2: Provider Count Location
|
||||
|
||||
Options:
|
||||
|
||||
- Server-side count: store or reconstruct the exact `fabro_llm::Request` in the
|
||||
HTTP server and call the provider from the endpoint.
|
||||
- Synchronous worker query: add a request/response control channel so the HTTP
|
||||
server can ask the active worker for a fresh count on demand.
|
||||
- Agent-side count on request assembly: the active session counts the exact
|
||||
request it already has and emits content-free snapshots; the endpoint returns
|
||||
the latest projection.
|
||||
|
||||
Recommendation: use agent-side count on request assembly for the first
|
||||
implementation.
|
||||
|
||||
Justification:
|
||||
|
||||
- It satisfies the requirement that provider counting is attempted by default,
|
||||
because every active agent request can be counted once as it is assembled.
|
||||
- It avoids moving raw prompts, memory, tool results, or message history into
|
||||
server-managed state.
|
||||
- The existing subprocess control path is one-way JSONL for actions like steer,
|
||||
interrupt, and pair. Adding a synchronous query path just for this read model
|
||||
would be more complex than emitting the durable projection the UI already
|
||||
needs.
|
||||
- It works for both subprocess and in-process runs because the agent session is
|
||||
the common place where the exact request exists.
|
||||
|
||||
Implementation guidance:
|
||||
|
||||
- Emit a local snapshot immediately, then emit a provider-scaled replacement if
|
||||
provider counting succeeds.
|
||||
- Do not delay the LLM stream on provider counting unless implementation finds
|
||||
that provider count latency is consistently negligible. A spawned count task
|
||||
with the cloned request is acceptable as long as it is cancelled/ignored when
|
||||
the session closes.
|
||||
- Treat provider count errors as snapshot warnings, not stage failures.
|
||||
|
||||
### Decision 3: Memory and Skills Attribution
|
||||
|
||||
Options:
|
||||
|
||||
- Keep the current flattened system prompt and count all prompt additions as
|
||||
`system_prompt`.
|
||||
- Refactor prompt assembly to retain component boundaries, then estimate
|
||||
memory and skills before the final prompt string is concatenated.
|
||||
- Add origin metadata to every history message and tool result so activated
|
||||
skill instructions can be attributed even after entering the conversation.
|
||||
|
||||
Recommendation: refactor prompt assembly for this first slice; defer full
|
||||
history-origin metadata.
|
||||
|
||||
Justification:
|
||||
|
||||
- `assemble_system_prompt` already receives `memory` and `skills` separately,
|
||||
then concatenates them with the core prompt. Returning component metadata from
|
||||
that boundary is a small, local change.
|
||||
- This gives useful and accurate first-slice attribution for loaded memory
|
||||
files and the available-skills prompt without changing the persisted
|
||||
conversation format.
|
||||
- Activated skill instructions are harder: slash expansion becomes a user turn,
|
||||
and `use_skill` returns a tool result. The current `Message` enum does not
|
||||
preserve source metadata. Adding it is possible, but it is a broader history
|
||||
serialization migration and should not block the endpoint.
|
||||
|
||||
Implementation guidance:
|
||||
|
||||
- Count the base/core prompt as `system_prompt`.
|
||||
- Count memory document text appended by prompt assembly as `memory`.
|
||||
- Count the available-skills section and the `use_skill` tool definition as
|
||||
`skills`.
|
||||
- Count slash-expanded skill templates and `use_skill` tool results as
|
||||
`conversation` in this first implementation, with a warning such as
|
||||
`activated_skill_context_counted_as_conversation` when such activations are
|
||||
present.
|
||||
|
||||
### Decision 4: Tool vs MCP Tool Attribution
|
||||
|
||||
Options:
|
||||
|
||||
- Split MCP tools by name prefix, such as `mcp__`.
|
||||
- Add source metadata to `RegisteredTool` / `ToolRegistry`.
|
||||
- Recompute MCP membership from `McpConnectionManager` at request time.
|
||||
|
||||
Recommendation: add source metadata to `RegisteredTool` / `ToolRegistry`.
|
||||
|
||||
Justification:
|
||||
|
||||
- The current registry stores only `ToolDefinition` plus executor, so origin is
|
||||
lost after registration.
|
||||
- Prefix-based classification matches today's naming convention but is brittle
|
||||
and will misclassify any future native tool that shares the prefix or any MCP
|
||||
naming change.
|
||||
- `McpConnectionManager` has source knowledge during registration, but the
|
||||
request builder only sees the final registry and policy-filtered tool
|
||||
definitions.
|
||||
|
||||
Implementation guidance:
|
||||
|
||||
- Add a small `ToolSource` enum, for example `Native`, `Mcp { server_name }`,
|
||||
and `Skill`.
|
||||
- Set `ToolSource::Mcp` in `mcp_integration::make_mcp_tools`.
|
||||
- Set `ToolSource::Skill` for `make_use_skill_tool`.
|
||||
- Keep existing public `definitions()` behavior unchanged; add a parallel
|
||||
method that returns definitions with source metadata for context-window
|
||||
accounting.
|
||||
|
||||
### Decision 5: Unavailable and Error Semantics
|
||||
|
||||
Options:
|
||||
|
||||
- Return 404/409 for non-agent stages or stages with no snapshot.
|
||||
- Return 200 with `available: false` for known stages where context-window data
|
||||
is not applicable or not observed.
|
||||
- Return partial data with warnings for provider-count failures.
|
||||
|
||||
Recommendation: return 200 with `available: false` for known-but-unavailable
|
||||
data, and reserve HTTP errors for missing run/stage or malformed requests.
|
||||
|
||||
Justification:
|
||||
|
||||
- The sidebar needs to render stable empty states without treating normal
|
||||
projection gaps as transport errors.
|
||||
- Provider-count support varies by provider and credentials; those are data
|
||||
quality issues, not endpoint availability issues.
|
||||
- This matches the broader best-effort contract and avoids UI retry loops when
|
||||
a completed older run simply has no context-window snapshot.
|
||||
|
||||
Implementation guidance:
|
||||
|
||||
- Use 404 only for missing run or missing stage.
|
||||
- Use `available: false` + `unavailable_reason` for `not_agent_stage`,
|
||||
`not_observed`, or `provider_unconfigured`.
|
||||
- Use `warnings` for local estimate, provider fallback, ambiguous categories,
|
||||
and activated skill content counted as conversation.
|
||||
|
||||
The previous open questions are resolved by these decisions.
|
||||
|
||||
## Implementation Units
|
||||
|
||||
### Unit 1: OpenAPI and Generated Clients
|
||||
|
||||
Files:
|
||||
|
||||
- `docs/public/api-reference/fabro-api.yaml`
|
||||
- `lib/crates/fabro-api/build.rs`
|
||||
- `lib/crates/fabro-api/tests/`
|
||||
- `lib/packages/fabro-api-client/`
|
||||
|
||||
Tasks:
|
||||
|
||||
- Add the path and schemas under Run Internals.
|
||||
- Prefer reusing hand-written Rust projection types through
|
||||
`with_replacement(...)` when serde shape and semantics are identical.
|
||||
- Add a `fabro-api` test proving type identity and JSON parity for any new
|
||||
replacement.
|
||||
- Regenerate Rust and TypeScript clients.
|
||||
|
||||
Tests:
|
||||
|
||||
- `cargo build -p fabro-api`
|
||||
- `cargo nextest run -p fabro-api`
|
||||
- `cd lib/packages/fabro-api-client && bun run generate`
|
||||
|
||||
### Unit 2: Content-Free Snapshot Builder
|
||||
|
||||
Files:
|
||||
|
||||
- `lib/crates/fabro-llm/src/token_count.rs`
|
||||
- `lib/crates/fabro-agent/src/context_window.rs` (new)
|
||||
- `lib/crates/fabro-agent/src/session.rs`
|
||||
- `lib/crates/fabro-agent/src/profiles/mod.rs`
|
||||
- `lib/crates/fabro-agent/src/tool_registry.rs`
|
||||
- `lib/crates/fabro-agent/src/mcp_integration.rs`
|
||||
- `lib/crates/fabro-agent/src/skills.rs`
|
||||
- `lib/crates/fabro-agent/src/compaction.rs`
|
||||
- `lib/crates/fabro-agent/src/lib.rs`
|
||||
|
||||
Tasks:
|
||||
|
||||
- Build category estimates at the same boundary where `Session::build_request`
|
||||
assembles the `fabro_llm::Request`.
|
||||
- Expose narrow reusable token-estimation helpers from `fabro-llm` rather than
|
||||
duplicating the estimator in `fabro-agent`.
|
||||
- Refactor prompt assembly enough to retain component boundaries for core
|
||||
system prompt, memory, available skills, and user instructions before the
|
||||
final prompt string is concatenated.
|
||||
- Add source metadata to registered tools so the builder can split native tools,
|
||||
MCP tools, and skill-related tools after policy filtering.
|
||||
- Classify the system prompt separately from conversation history.
|
||||
- Count slash-expanded skill templates and `use_skill` tool results as
|
||||
`conversation` for the first implementation, with a warning when activated
|
||||
skill context is present.
|
||||
- Emit a local snapshot immediately and a provider-scaled replacement snapshot
|
||||
when provider counting succeeds.
|
||||
- Keep local estimates deterministic and content-free.
|
||||
|
||||
Tests:
|
||||
|
||||
- system prompt, tools, MCP tools, skills, memory, conversation, and other
|
||||
categories are counted into the expected buckets.
|
||||
- category totals equal the local total before provider scaling.
|
||||
- provider-scaled totals add up to the provider total.
|
||||
- snapshots do not serialize prompt text, memory contents, tool arguments, or
|
||||
message text.
|
||||
- opaque/ambiguous inputs produce warnings rather than silent misclassification.
|
||||
- provider count failure does not fail the agent turn.
|
||||
- each request fingerprint is counted by the provider at most once.
|
||||
|
||||
### Unit 3: Event and Projection
|
||||
|
||||
Files:
|
||||
|
||||
- `lib/crates/fabro-types/src/run_event/agent.rs`
|
||||
- `lib/crates/fabro-types/src/run_event/mod.rs`
|
||||
- `lib/crates/fabro-types/src/run_projection.rs`
|
||||
- `lib/crates/fabro-store/src/run_state.rs`
|
||||
|
||||
Tasks:
|
||||
|
||||
- Add `agent.context_window.snapshot`.
|
||||
- Emit snapshots from the agent session path when requests are assembled and
|
||||
when later response usage improves the count.
|
||||
- Project the latest snapshot onto `StageProjection.context_window`.
|
||||
- Preserve backwards-compatible deserialization for run projections that do not
|
||||
have the new field.
|
||||
|
||||
Tests:
|
||||
|
||||
- event `type_name()` returns `agent.context_window.snapshot`.
|
||||
- reducer updates only the matching stage.
|
||||
- later snapshots replace earlier snapshots for the same stage.
|
||||
- old projection JSON without `context_window` still deserializes.
|
||||
|
||||
### Unit 4: Server Endpoint
|
||||
|
||||
Files:
|
||||
|
||||
- `lib/crates/fabro-server/src/server/handler/mod.rs`
|
||||
- `lib/crates/fabro-server/src/server/handler/runs.rs` or a new
|
||||
`context_window.rs` handler module
|
||||
- `lib/crates/fabro-server/src/server/tests.rs`
|
||||
|
||||
Tasks:
|
||||
|
||||
- Add `GET /runs/{id}/stages/{stageId}/context-window`.
|
||||
- Use the same run-scoped authorization pattern as adjacent run internals.
|
||||
- Resolve run/stage from the cached projection first for fast 404s and stored
|
||||
fallback.
|
||||
- Return the latest projected context-window snapshot for the stage.
|
||||
- Return `available: false` for known stages where context-window data is not
|
||||
applicable or has not been observed.
|
||||
- Return 404 only for missing run/stage. Provider count failures are represented
|
||||
as snapshot warnings because provider counting happens in the agent.
|
||||
|
||||
Tests:
|
||||
|
||||
- missing run returns 404.
|
||||
- missing stage returns 404.
|
||||
- non-agent stage returns 200 with `available: false`.
|
||||
- projected provider-count success returns `staleness: live` and
|
||||
`count_method: provider_api_scaled_breakdown`.
|
||||
- inactive stage returns the latest stored projection snapshot.
|
||||
- no observed snapshot returns `available: false` with `not_observed`.
|
||||
- projected warning payloads are returned without changing HTTP status.
|
||||
|
||||
### Unit 5: Web Query Support
|
||||
|
||||
Files:
|
||||
|
||||
- `apps/fabro-web/app/lib/query-keys.ts`
|
||||
- `apps/fabro-web/app/lib/queries.ts`
|
||||
- `apps/fabro-web/app/lib/run-events.ts`
|
||||
- `apps/fabro-web/app/lib/query-keys.test.ts`
|
||||
- `apps/fabro-web/app/lib/run-events.test.tsx`
|
||||
|
||||
Tasks:
|
||||
|
||||
- Add `queryKeys.runs.stageContextWindow(id, stageId)`.
|
||||
- Add `useRunStageContextWindow(runId, stageId)` using the generated
|
||||
TypeScript client.
|
||||
- Invalidate the context-window key for:
|
||||
- `agent.context_window.snapshot`
|
||||
- stage lifecycle events for the same stage
|
||||
- agent activity events that can change the request context
|
||||
- Keep the full sidebar UI as follow-up work, but make the hook ready for the
|
||||
agent-node page.
|
||||
|
||||
Tests:
|
||||
|
||||
- query key encodes run id and stage id stably.
|
||||
- snapshot event invalidates the context-window key, run events, and stage
|
||||
events.
|
||||
- stage lifecycle events invalidate the context-window key for the selected
|
||||
stage.
|
||||
- activity events without a stage id do not invalidate unrelated stage context
|
||||
windows.
|
||||
|
||||
## Security and Privacy
|
||||
|
||||
- Do not persist raw request content to make inactive-stage provider counting
|
||||
possible.
|
||||
- Do not log prompt, memory, tool args, or message contents while computing
|
||||
counts.
|
||||
- Warning messages should identify count quality, not repeat provider error
|
||||
bodies if those bodies may contain request excerpts.
|
||||
- The endpoint should expose counts and category labels only.
|
||||
|
||||
## Validation
|
||||
|
||||
Expected validation after implementation:
|
||||
|
||||
```bash
|
||||
cargo build -p fabro-api
|
||||
cargo nextest run -p fabro-api -p fabro-agent -p fabro-store -p fabro-server
|
||||
cd apps/fabro-web && bun test
|
||||
cd apps/fabro-web && bun run typecheck
|
||||
cargo +nightly-2026-04-14 fmt --check --all
|
||||
git diff --check
|
||||
```
|
||||
|
||||
## Remaining Follow-Ups
|
||||
|
||||
- Full sidebar visualization.
|
||||
- Optional history-origin metadata if we later want activated skill templates
|
||||
to move from `conversation` into `skills`.
|
||||
- Optional user-triggered live refresh path if future product needs require
|
||||
provider counting on demand rather than at request assembly time.
|
||||
|
||||
|
||||
## Completed stages
|
||||
- **toolchain**: succeeded
|
||||
- Script: `command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1`
|
||||
- Output:
|
||||
```
|
||||
cargo 1.95.0 (f2d3ce0bd 2026-03-21)
|
||||
```
|
||||
- **preflight_compile**: succeeded
|
||||
- Script: `cargo check -q --workspace 2>&1`
|
||||
- Output: (empty)
|
||||
- **preflight_lint**: succeeded
|
||||
- Script: `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1`
|
||||
- Output: (empty)
|
||||
- **implement**: succeeded
|
||||
- Model: gpt-5.5, 919.5k tokens in / 65.6k out
|
||||
- Files: /home/daytona/workspace/fabro/lib/crates/fabro-agent/src/context_window.rs, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window-breakdown-item.ts, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window-category.ts, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window-count-method.ts, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window-projection.ts, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window-staleness.ts, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window-unavailable-reason.ts, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window-warning.ts, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window.ts
|
||||
- **simplify_opus**: succeeded
|
||||
- Model: claude-opus-4-7, 158.5k tokens in / 47.7k out
|
||||
- Files: /home/daytona/workspace/fabro/apps/fabro-web/app/lib/query-keys.test.ts, /home/daytona/workspace/fabro/apps/fabro-web/app/lib/run-events.test.tsx, /home/daytona/workspace/fabro/apps/fabro-web/app/lib/run-events.ts, /home/daytona/workspace/fabro/docs/public/api-reference/fabro-api.yaml, /home/daytona/workspace/fabro/lib/crates/fabro-agent/src/context_window.rs, /home/daytona/workspace/fabro/lib/crates/fabro-agent/src/tool_registry.rs, /home/daytona/workspace/fabro/lib/crates/fabro-api/src/lib.rs, /home/daytona/workspace/fabro/lib/crates/fabro-api/tests/stage_projection_round_trip.rs, /home/daytona/workspace/fabro/lib/crates/fabro-server/src/server/handler/runs.rs, /home/daytona/workspace/fabro/lib/crates/fabro-server/src/server/tests.rs, /home/daytona/workspace/fabro/lib/crates/fabro-store/src/run_state.rs, /home/daytona/workspace/fabro/lib/crates/fabro-types/src/lib.rs, /home/daytona/workspace/fabro/lib/crates/fabro-types/src/run_event/mod.rs, /home/daytona/workspace/fabro/lib/crates/fabro-types/src/run_projection.rs, /home/daytona/workspace/fabro/lib/packages/fabro-api-client/src/models/stage-context-window-breakdown-item.ts
|
||||
|
||||
|
||||
# Simplify: Code Review and Cleanup
|
||||
|
||||
Review changes vs. origin for reuse, quality, and efficiency. Fix any issues found.
|
||||
|
||||
## Phase 1: Identify Changes
|
||||
|
||||
Run git diff (or git diff HEAD if there are staged changes) to see what changed. If there are no git changes, review the most recently modified files that the user mentioned or that you edited earlier in this conversation.
|
||||
|
||||
## Phase 2: Launch Three Review Agents in Parallel
|
||||
|
||||
Use the Agent tool to launch all three agents concurrently in a single message. Pass each agent the full diff so it has the complete context.
|
||||
|
||||
### Agent 1: Code Reuse Review
|
||||
|
||||
For each change:
|
||||
|
||||
1. Search for existing utilities and helpers that could replace newly written code. Use Grep to find similar patterns elsewhere in the codebase — common locations are utility directories, shared modules, and files adjacent to the changed ones.
|
||||
2. Flag any new function that duplicates existing functionality. Suggest the existing function to use instead.
|
||||
3. Flag any inline logic that could use an existing utility — hand-rolled string manipulation, manual path handling, custom environment checks, ad-hoc type guards, and similar patterns are common candidates.
|
||||
|
||||
Note: This is a greenfield app, so focus on maximizing simplicity and don't worry about changing things to achieve it.
|
||||
|
||||
### Agent 2: Code Quality Review
|
||||
|
||||
Review the same changes for hacky patterns:
|
||||
|
||||
1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls
|
||||
2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones
|
||||
3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction
|
||||
4. Leaky abstractions: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries
|
||||
5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase
|
||||
|
||||
Note: This is a greenfield app, so be aggressive in optimizing quality.
|
||||
|
||||
### Agent 3: Efficiency Review
|
||||
|
||||
Review the same changes for efficiency:
|
||||
|
||||
1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns
|
||||
2. Missed concurrency: independent operations run sequentially when they could run in parallel
|
||||
3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths
|
||||
4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error
|
||||
5. Memory: unbounded data structures, missing cleanup, event listener leaks
|
||||
6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one
|
||||
|
||||
## Phase 3: Fix Issues
|
||||
|
||||
Wait for all three agents to complete. Aggregate their findings and fix each issue directly. If a finding is a false positive or not worth addressing, note it and move on — do not argue with the finding, just skip it.
|
||||
|
||||
When done, briefly summarize what was fixed (or confirm the code was already clean).
|
||||
5
stages/007-simplify_gpt@1/provider_used.json
Normal file
5
stages/007-simplify_gpt@1/provider_used.json
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
{
|
||||
"mode": "agent",
|
||||
"provider": "openai",
|
||||
"model": "gpt-5.5"
|
||||
}
|
||||
23
stages/007-simplify_gpt@1/response.md
Normal file
23
stages/007-simplify_gpt@1/response.md
Normal file
|
|
@ -0,0 +1,23 @@
|
|||
Reviewed the branch diff and made two cleanup fixes:
|
||||
|
||||
- Prevented late async provider token-count results from overwriting a better response-usage context-window snapshot for the same request.
|
||||
- Added per-request fingerprint tracking for response-usage snapshots in `lib/crates/fabro-agent/src/session.rs`.
|
||||
- Provider count tasks now skip emitting if response usage has already produced the newer/better snapshot.
|
||||
|
||||
- Completed web SSE invalidation coverage for the new context-window query.
|
||||
- Stage lifecycle, steering/session, and stage activity events now invalidate `queryKeys.runs.stageContextWindow(runId, stageId)` in addition to stage events.
|
||||
- Updated tests in:
|
||||
- `apps/fabro-web/app/lib/query-keys.test.ts`
|
||||
- `apps/fabro-web/app/lib/run-events.test.tsx`
|
||||
|
||||
Validation run:
|
||||
|
||||
- `cargo check -q -p fabro-agent` ✅
|
||||
- `cargo nextest run -p fabro-agent context_window` ✅
|
||||
- `cargo check -q -p fabro-server` ✅
|
||||
- `cd apps/fabro-web && bun test app/lib/run-events.test.tsx app/lib/query-keys.test.ts` ✅
|
||||
- `cd apps/fabro-web && bun run typecheck` ✅
|
||||
- `cargo +nightly-2026-04-14 fmt --check --all` ✅
|
||||
- `git diff --check` ✅
|
||||
|
||||
One attempted focused clippy command failed on an existing unrelated `fabro-sandbox` `large_enum_variant` warning, not caused by these cleanup changes.
|
||||
Loading…
Add table
Reference in a new issue