diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml index 9e22303a35..34ac70a6cf 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.yml +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -76,7 +76,7 @@ body: label: API Provider (optional) options: - Anthropic - - AWS Bedrock + - Amazon Bedrock - Chutes AI - DeepSeek - Featherless AI diff --git a/.husky/pre-push b/.husky/pre-push index 3c206835b7..4cf91d9580 100644 --- a/.husky/pre-push +++ b/.husky/pre-push @@ -18,6 +18,19 @@ fi $pnpm_cmd run check-types +# Use dotenvx to securely load .env.local and run commands that depend on it +if [ -f ".env.local" ]; then + # Check if RUN_TESTS_ON_PUSH is set to true and run tests with dotenvx + if npx dotenvx get RUN_TESTS_ON_PUSH -f .env.local 2>/dev/null | grep -q "^true$"; then + npx dotenvx run -f .env.local -- $pnpm_cmd run test + fi +else + # Fallback: run tests if RUN_TESTS_ON_PUSH is set in regular environment + if [ "$RUN_TESTS_ON_PUSH" = "true" ]; then + $pnpm_cmd run test + fi +fi + # Check for new changesets. NEW_CHANGESETS=$(find .changeset -name "*.md" ! -name "README.md" | wc -l | tr -d ' ') echo "Changeset files: $NEW_CHANGESETS" diff --git a/.roo/commands/release.md b/.roo/commands/release.md index 8adf57e6f0..707844cdc4 100644 --- a/.roo/commands/release.md +++ b/.roo/commands/release.md @@ -16,14 +16,14 @@ argument-hint: patch | minor | major [list of changes] ``` -- Always include contributor attribution using format: (thanks @username!) -- For PRs that close issues, also include the issue number and reporter: "- Fix: Description (#123 by @reporter, PR by @contributor)" -- For PRs without linked issues, use the standard format: "- Add support for feature (thanks @contributor!)" +- Always include contributor attribution and the PR number: use "(PR # by @username)". +- For PRs that close issues, include both the issue number and the PR number and authors: "- Fix: Description (#123 by @reporter, PR #456 by @contributor)" +- For PRs without linked issues, include the PR number and author: "- Add support for feature (PR #456 by @contributor)" - Provide brief descriptions of each item to explain the change - Order the list from most important to least important - Example formats: - - With issue: "- Fix: Resolve memory leak in extension (#456 by @issueReporter, PR by @prAuthor)" - - Without issue: "- Add support for Gemini 2.5 Pro caching (thanks @contributor!)" + - With issue: "- Fix: Resolve memory leak in extension (#456 by @issueReporter, PR #789 by @prAuthor)" + - Without issue: "- Add support for Gemini 2.5 Pro caching (PR #789 by @contributor)" - CRITICAL: Include EVERY SINGLE PR in the changeset - don't assume you know which ones are important. Count the total PRs to verify completeness and cross-reference the list to ensure nothing is missed. 6. If the generate_image tool is available, create a release image at `releases/[version]-release.png` diff --git a/.roo/rules-issue-fixer/1_Workflow.xml b/.roo/rules-issue-fixer/1_Workflow.xml index 62a3fd7c26..c6f2b570a0 100644 --- a/.roo/rules-issue-fixer/1_Workflow.xml +++ b/.roo/rules-issue-fixer/1_Workflow.xml @@ -17,7 +17,7 @@ Then retrieve the issue: - gh issue view [issue-number] --repo [owner]/[repo] --json number,title,body,state,labels,assignees,milestone,createdAt,updatedAt,closedAt,author + gh api repos/[owner]/[repo]/issues/[issue-number] --jq '{number,title,body,state,labels,assignees,milestone,createdAt:.created_at,updatedAt:.updated_at,closedAt:.closed_at,author:.user.login}' If the command fails with an authentication error (e.g., "gh: Not authenticated" or "HTTP 401"), ask the user to authenticate: @@ -49,7 +49,7 @@ - Any decisions or changes to requirements - gh issue view [issue number] --repo [owner]/[repo] --comments + gh api repos/[owner]/[repo]/issues/[issue-number]/comments --paginate --jq '.[].body' Also check for: diff --git a/.roo/rules-issue-fixer/4_github_cli_usage.xml b/.roo/rules-issue-fixer/4_github_cli_usage.xml index e12fb06a5b..b4fa19b3cc 100644 --- a/.roo/rules-issue-fixer/4_github_cli_usage.xml +++ b/.roo/rules-issue-fixer/4_github_cli_usage.xml @@ -29,23 +29,23 @@ - Retrieve the issue details at the start + Retrieve the issue details at the start using the REST Issues API. Always use first to get the full issue content - gh issue view [issue-number] --repo [owner]/[repo] --json number,title,body,state,labels,assignees,milestone,createdAt,updatedAt,closedAt,author + gh api repos/[owner]/[repo]/issues/[issue-number] --jq '{number,title,body,state,labels,assignees,milestone,createdAt:.created_at,updatedAt:.updated_at,closedAt:.closed_at,author:.user.login}' - gh issue view 123 --repo octocat/hello-world --json number,title,body,state,labels,assignees,milestone,createdAt,updatedAt,closedAt,author + gh api repos/octocat/hello-world/issues/123 --jq '{number,title,body,state,labels,assignees,milestone,createdAt:.created_at,updatedAt:.updated_at,closedAt:.closed_at,author:.user.login}' - Get additional context and requirements from issue comments + Get additional context and requirements from issue comments. Always use after viewing issue to see full discussion - gh issue view [issue-number] --repo [owner]/[repo] --comments + gh api repos/[owner]/[repo]/issues/[issue-number]/comments --paginate --jq '.[].body' - gh issue view 123 --repo octocat/hello-world --comments + gh api repos/octocat/hello-world/issues/123/comments --paginate --jq '.[].body' @@ -109,6 +109,30 @@ + + + Inspect associations with GitHub Projects (new Projects experience) for a given issue + Use when project context is relevant to understanding priority, ownership, or workflow + gh api graphql -f query=' +query($owner:String!, $repo:String!, $number:Int!) { + repository(owner:$owner, name:$repo) { + issue(number:$number) { + projectsV2(first:20) { + nodes { + title + url + } + } + } + } +} +' -F owner=[owner] -F repo=[repo] -F number=[issue-number] + + This uses the projectsV2 field from the new GitHub Projects experience for issue-level project context. + + + + Create a pull request diff --git a/.roo/rules-translate/instructions-zh-cn.md b/.roo/rules-translate/instructions-zh-cn.md index 241ae338dc..6141038728 100644 --- a/.roo/rules-translate/instructions-zh-cn.md +++ b/.roo/rules-translate/instructions-zh-cn.md @@ -115,7 +115,7 @@ - 保留英文品牌名 - 技术术语保持一致性 - - 保留英文专有名词:如"AWS Bedrock ARN" + - 保留英文专有名词:如"Amazon Bedrock ARN" 4. **用户操作** - 操作动词统一: diff --git a/CHANGELOG.md b/CHANGELOG.md index b7d4e6c8e0..effbcbdc35 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,336 @@ # Roo Code Changelog +## [3.36.0] - 2025-12-04 + +![3.36.0 Release - Rewind Kangaroo](/releases/3.36.0-release.png) + +- Fix: Restore context when rewinding after condense (#8295 by @hannesrudolph, PR #9665 by @hannesrudolph) +- Add reasoning_details support to Roo provider for enhanced model reasoning visibility (PR #9796 by @app/roomote) +- Default to native tools for all models in the Roo provider for improved performance (PR #9811 by @mrubens) +- Enable search_and_replace for Minimax models (PR #9780 by @mrubens) +- Fix: Resolve Vercel AI Gateway model fetching issues (PR #9791 by @cte) +- Fix: Apply conservative max tokens for Cerebras provider (PR #9804 by @sebastiand-cerebras) +- Fix: Remove omission detection logic to eliminate false positives (#9785 by @Michaelzag, PR #9787 by @app/roomote) +- Refactor: Remove deprecated insert_content tool (PR #9751 by @daniel-lxs) +- Chore: Hide parallel tool calls experiment and disable feature (PR #9798 by @hannesrudolph) +- Update next.js documentation site dependencies (PR #9799 by @jr) +- Fix: Correct download count display on homepage (PR #9807 by @mrubens) + +## [3.35.5] - 2025-12-03 + +- Feat: Add provider routing selection for OpenRouter embeddings (#9144 by @SannidhyaSah, PR #9693 by @SannidhyaSah) +- Default Minimax M2 to native tool calling (PR #9778 by @mrubens) +- Sanitize the native tool calls to fix a bug with Gemini (PR #9769 by @mrubens) +- UX: Updates to CloudView (PR #9776 by @roomote) + +## [3.35.4] - 2025-12-02 + +- Fix: Handle malformed native tool calls to prevent hanging (PR #9758 by @daniel-lxs) +- Fix: Remove reasoning toggles for GLM-4.5 and GLM-4.6 on z.ai provider (PR #9752 by @roomote) +- Refactor: Remove line_count parameter from write_to_file tool (PR #9667 by @hannesrudolph) + +## [3.35.3] - 2025-12-02 + +- Switch to new welcome view for improved onboarding experience (PR #9741 by @mrubens) +- Update homepage with latest changes (PR #9675 by @brunobergher) +- Improve privacy for stealth models by adding vendor confidentiality section to system prompt (PR #9742 by @mrubens) + +## [3.35.2] - 2025-12-01 + +![3.35.2 Release - Model Default Temperatures](/releases/3.35.2-release.png) + +- Allow models to contain default temperature settings for provider-specific optimal defaults (PR #9734 by @mrubens) +- Add tag-based native tool calling detection for Roo provider models (PR #9735 by @mrubens) +- Enable native tool support for all LiteLLM models by default (PR #9736 by @mrubens) +- Pass app version to provider for improved request tracking (PR #9730 by @cte) + +## [3.35.1] - 2025-12-01 + +- Fix: Flush pending tool results before task delegation (PR #9726 by @daniel-lxs) +- Improve: Better IPC error logging for easier debugging (PR #9727 by @cte) + +## [3.35.0] - 2025-12-01 + +![3.35.0 Release - Subtasks & Native Tools](/releases/3.35.0-release.png) + +- Metadata-driven subtasks with automatic parent resume and single-open safety for improved task orchestration (#8081 by @hannesrudolph, PR #9090 by @hannesrudolph) +- Native tool calling support expanded across many providers: Bedrock (PR #9698 by @mrubens), Cerebras (PR #9692 by @mrubens), Chutes with auto-detection from API (PR #9715 by @daniel-lxs), DeepInfra (PR #9691 by @mrubens), DeepSeek and Doubao (PR #9671 by @daniel-lxs), Groq (PR #9673 by @daniel-lxs), LiteLLM (PR #9719 by @daniel-lxs), Ollama (PR #9696 by @mrubens), OpenAI-compatible providers (PR #9676 by @daniel-lxs), Requesty (PR #9672 by @daniel-lxs), Unbound (PR #9699 by @mrubens), Vercel AI Gateway (PR #9697 by @mrubens), Vertex Gemini (PR #9678 by @daniel-lxs), and xAI with new Grok 4 Fast and Grok 4.1 Fast models (PR #9690 by @mrubens) +- Fix: Preserve tool_use blocks in summary for parallel tool calls (#9700 by @SilentFlower, PR #9714 by @SilentFlower) +- Default Grok Code Fast to native tools for better performance (PR #9717 by @mrubens) +- UX improvements to the Roo Code Cloud provider-centric onboarding flow (PR #9709 by @brunobergher) +- UX toolbar cleanup and settings consolidation for a cleaner interface (PR #9710 by @brunobergher) +- Add model-specific tool customization via `excludedTools` and `includedTools` configuration (PR #9641 by @daniel-lxs) +- Add new `apply_patch` native tool for more efficient file editing operations (PR #9663 by @hannesrudolph) +- Add new `search_and_replace` tool for batch text replacements across files (PR #9549 by @hannesrudolph) +- Add debug buttons to view API and UI history for troubleshooting (PR #9684 by @hannesrudolph) +- Include tool format in environment details for better context awareness (PR #9661 by @mrubens) +- Fix: Display install count in millions instead of thousands (PR #9677 by @app/roomote) +- Web-evals improvements: add task log viewing, export failed logs, and new run options (PR #9637 by @hannesrudolph) +- Web-evals updates: add kill run functionality (PR #9681 by @hannesrudolph) +- Fix: Prevent navigation buttons from wrapping on smaller screens (PR #9721 by @app/roomote) + +## [3.34.8] - 2025-11-27 + +![3.34.8 Release - Race Condition Fix](/releases/3.34.8-release.png) + +- Fix: Race condition in new_task tool for native protocol (PR #9655 by @daniel-lxs) + +## [3.34.7] - 2025-11-27 + +![3.34.7 Release - More Native Tool Integrations](/releases/3.34.7-release.png) + +- Support native tools in the Anthropic provider for improved tool calling (PR #9644 by @mrubens) +- Enable native tool calling for z.ai models (PR #9645 by @mrubens) +- Enable native tool calling for Moonshot models (PR #9646 by @mrubens) +- Fix: OpenRouter tool calls handling improvements (PR #9642 by @mrubens) +- Fix: OpenRouter GPT-5 strict schema validation for read_file tool (PR #9633 by @daniel-lxs) +- Fix: Create parent directories early in write_to_file to prevent ENOENT errors (#9634 by @ivanenev, PR #9640 by @daniel-lxs) +- Fix: Disable native tools and temperature support for claude-code provider (PR #9643 by @hannesrudolph) +- Add 'taking you to cloud' screen after provider welcome for improved onboarding (PR #9652 by @mrubens) + +## [3.34.6] - 2025-11-26 + +![3.34.6 Release - Bedrock Embeddings](/releases/3.34.6-release.png) + +- Add support for AWS Bedrock embeddings in code indexing (#8658 by @kyle-hobbs, PR #9475 by @ggoranov-smar) +- Add native tool calling support for Mistral provider (PR #9625 by @hannesrudolph) +- Wire MULTIPLE_NATIVE_TOOL_CALLS experiment to OpenAI parallel_tool_calls for parallel tool execution (PR #9621 by @hannesrudolph) +- Add fine grained tool streaming for OpenRouter Anthropic (PR #9629 by @mrubens) +- Allow global inference selection for Bedrock when cross-region is enabled (PR #9616 by @roomote) +- Fix: Filter non-Anthropic content blocks before sending to Vertex API (#9583 by @cardil, PR #9618 by @hannesrudolph) +- Fix: Restore content undefined check in WriteToFileTool.handlePartial() (#9611 by @Lissanro, PR #9614 by @daniel-lxs) +- Fix: Prevent model cache from persisting empty API responses (#9597 by @zx2021210538, PR #9623 by @daniel-lxs) +- Fix: Exclude access_mcp_resource tool when MCP has no resources (PR #9615 by @daniel-lxs) +- Fix: Update default settings for inline terminal and codebase indexing (PR #9622 by @roomote) +- Fix: Convert line_ranges strings to lineRanges objects in native tool calls (PR #9627 by @daniel-lxs) +- Fix: Defer new_task tool_result until subtask completes for native protocol (PR #9628 by @daniel-lxs) + +## [3.34.5] - 2025-11-25 + +![3.34.5 Release - Experimental Parallel Tool Calling](/releases/3.34.5-release.png) + +- Experimental feature to enable multiple native tool calls per turn (PR #9273 by @daniel-lxs) +- Add Bedrock Opus 4.5 to global inference model list (PR #9595 by @roomote) +- Fix: Update API handler when toolProtocol changes (PR #9599 by @mrubens) +- Set native tools as default for minimax-m2 and claude-haiku-4.5 (PR #9586 by @daniel-lxs) +- Make single file read only apply to XML tools (PR #9600 by @mrubens) +- Enhance web-evals dashboard with dynamic tool columns and UX improvements (PR #9592 by @hannesrudolph) +- Revert "Add support for Roo Code Cloud as an embeddings provider" while we fix some issues (PR #9602 by @mrubens) + +## [3.34.4] - 2025-11-25 + +![3.34.4 Release - BFL Image Generation](/releases/3.34.4-release.png) + +- Add new Black Forest Labs image generation models, free on Roo Code Cloud and also available on OpenRouter (PR #9587 and #9589 by @mrubens) +- Fix: Preserve dynamic MCP tool names in native mode API history to prevent tool name mismatches (PR #9559 by @daniel-lxs) +- Fix: Preserve tool_use blocks in summary message during condensing with native tools to maintain conversation context (PR #9582 by @daniel-lxs) + +## [3.34.3] - 2025-11-25 + +![3.34.3 Release - Streaming and Opus 4.5](/releases/3.34.3-release.png) + +- Implement streaming for native tool calls, providing real-time feedback during tool execution (PR #9542 by @daniel-lxs) +- Add Claude Opus 4.5 model to Claude Code provider (PR #9560 by @mrubens) +- Add Claude Opus 4.5 model to Bedrock provider (#9571 by @pisicode, PR #9572 by @roomote) +- Enable caching for Opus 4.5 model to improve performance (#9567 by @iainRedro, PR #9568 by @roomote) +- Add support for Roo Code Cloud as an embeddings provider (PR #9543 by @mrubens) +- Fix ask_followup_question streaming issue and add missing tool cases (PR #9561 by @daniel-lxs) +- Add contact links to About Roo Code settings page (PR #9570 by @roomote) +- Switch from asdf to mise-en-place in bare-metal evals setup script (PR #9548 by @cte) + +## [3.34.2] - 2025-11-24 + +![3.34.2 Release - Opus Conductor](/releases/3.34.2-release.png) + +- Add support for Claude Opus 4.5 in Anthropic and Vertex providers (PR #9541 by @daniel-lxs) +- Add support for Claude Opus 4.5 in OpenRouter with prompt caching and reasoning budget (PR #9540 by @daniel-lxs) +- Add Roo Code Cloud as an image generation provider (PR #9528 by @mrubens) +- Fix: Gracefully skip unsupported content blocks in Gemini transformer (PR #9537 by @daniel-lxs) +- Fix: Flush LiteLLM cache when credentials change on refresh (PR #9536 by @daniel-lxs) +- Fix: Ensure XML parser state matches tool protocol on config update (PR #9535 by @daniel-lxs) +- Update Cerebras models (PR #9527 by @sebastiand-cerebras) +- Fix: Support reasoning_details format for Gemini 3 models (PR #9506 by @daniel-lxs) + +## [3.34.1] - 2025-11-23 + +- Show the prompt for image generation in the UI (PR #9505 by @mrubens) +- Fix double todo list display issue (PR #9517 by @mrubens) +- Add tracking for cloud synced messages (PR #9518 by @mrubens) +- Enable the Roo Code Cloud provider in evals (PR #9492 by @cte) + +## [3.34.0] - 2025-11-21 + +![3.34.0 Release - Browser Use 2.0](/releases/3.34.0-release.png) + +- Add Browser Use 2.0 with enhanced browser interaction capabilities (PR #8941 by @hannesrudolph) +- Add support for Baseten as a new AI provider (PR #9461 by @AlexKer) +- Improve base OpenAI compatible provider with better error handling and configuration (PR #9462 by @mrubens) +- Add provider-oriented welcome screen to improve onboarding experience (PR #9484 by @mrubens) +- Pin Roo provider to the top of the provider list for better discoverability (PR #9485 by @mrubens) +- Enhance native tool descriptions with examples and clarifications for better AI understanding (PR #9486 by @daniel-lxs) +- Fix: Make cancel button immediately responsive during streaming (#9435 by @jwadow, PR #9448 by @daniel-lxs) +- Fix: Resolve apply_diff performance regression from earlier changes (PR #9474 by @daniel-lxs) +- Fix: Implement model cache refresh to prevent stale disk cache issues (PR #9478 by @daniel-lxs) +- Fix: Copy model-level capabilities to OpenRouter endpoint models correctly (PR #9483 by @daniel-lxs) +- Fix: Add fallback to yield tool calls regardless of finish_reason (PR #9476 by @daniel-lxs) + +## [3.33.3] - 2025-11-20 + +![3.33.3 Release - Gemini 3 Pro Image Preview](/releases/3.33.3-release.png) + +- Add Google Gemini 3 Pro Image Preview to image generation models (PR #9440 by @app/roomote) +- Add support for Minimax as Anthropic-compatible provider (PR #9455 by @daniel-lxs) +- Store reasoning in conversation history for all providers (PR #9451 by @daniel-lxs) +- Fix: Improve preserveReasoning flag to control API reasoning inclusion (PR #9453 by @daniel-lxs) +- Fix: Prevent OpenAI Native parallel tool calls for native tool calling (PR #9433 by @hannesrudolph) +- Fix: Improve search and replace symbol parsing (PR #9456 by @daniel-lxs) +- Fix: Send tool_result blocks for skipped tools in native protocol (PR #9457 by @daniel-lxs) +- Fix: Improve markdown formatting and add reasoning support (PR #9458 by @daniel-lxs) +- Fix: Prevent duplicate environment_details when resuming cancelled tasks (PR #9442 by @daniel-lxs) +- Improve read_file tool description with examples (PR #9422 by @daniel-lxs) +- Update glob dependency to ^11.1.0 (PR #9449 by @jr) +- Update tar-fs to 3.1.1 via pnpm override (PR #9450 by @app/roomote) + +## [3.33.2] - 2025-11-19 + +- Enable native tool calling for Gemini provider (PR #9343 by @hannesrudolph) +- Add RCC credit balance display (PR #9386 by @jr) +- Fix: Preserve user images in native tool call results (PR #9401 by @daniel-lxs) +- Perf: Reduce excessive getModel() calls and implement disk cache fallback (PR #9410 by @daniel-lxs) +- Show zero price for free models (PR #9419 by @mrubens) + +## [3.33.1] - 2025-11-18 + +![3.33.1 Release - Native Tool Protocol Fixes](/releases/3.33.1-release.png) + +- Add native tool calling support to OpenAI-compatible (PR #9369 by @mrubens) +- Fix: Resolve native tool protocol race condition causing 400 errors (PR #9363 by @daniel-lxs) +- Fix: Update tools to return structured JSON for native protocol (PR #9373 by @daniel-lxs) +- Fix: Include nativeArgs in tool repetition detection (PR #9377 by @daniel-lxs) +- Fix: Ensure no XML parsing when protocol is native (PR #9371 by @daniel-lxs) +- Fix: Gemini maxOutputTokens and reasoning config (PR #9375 by @hannesrudolph) +- Fix: Gemini thought signature validation and token counting errors (PR #9380 by @hannesrudolph) +- Fix: Exclude XML tool examples from MODES section when native protocol enabled (PR #9367 by @daniel-lxs) +- Retry eval tasks if API instability detected (PR #9365 by @cte) +- Add toolProtocol property to PostHog tool usage telemetry (PR #9374 by @app/roomote) + +## [3.33.0] - 2025-11-18 + +![v3.33.0 Release - Twin Kangaroos and the Gemini Constellation](/releases/v3.33.0-release.png) + +- Add Gemini 3 Pro Preview model (PR #9357 by @hannesrudolph) +- Improve Google Gemini defaults with better temperature and cost reporting (PR #9327 by @hannesrudolph) +- Enable native tool calling for openai-native provider (PR #9348 by @hannesrudolph) +- Add git status information to environment details (PR #9310 by @daniel-lxs) +- Add tool protocol selector to advanced settings (PR #9324 by @daniel-lxs) +- Implement dynamic tool protocol resolution with proper precedence hierarchy (PR #9286 by @daniel-lxs) +- Move Import/Export functionality to Modes view toolbar and cleanup Mode Edit view (PR #9077 by @hannesrudolph) +- Update cloud agent CTA to point to setup page (PR #9338 by @app/roomote) +- Fix: Prevent duplicate tool_result blocks in native tool protocol (PR #9248 by @daniel-lxs) +- Fix: Format tool responses properly for native protocol (PR #9270 by @daniel-lxs) +- Fix: Centralize toolProtocol configuration checks (PR #9279 by @daniel-lxs) +- Fix: Preserve tool blocks for native protocol in conversation history (PR #9319 by @daniel-lxs) +- Fix: Prevent infinite loop when task_done succeeds (PR #9325 by @daniel-lxs) +- Fix: Sync parser state with profile/model changes (PR #9355 by @daniel-lxs) +- Fix: Pass tool protocol parameter to lineCountTruncationError (PR #9358 by @daniel-lxs) +- Use VSCode theme color for outline button borders (PR #9336 by @app/roomote) +- Replace broken badgen.net badges with shields.io (PR #9318 by @app/roomote) +- Add max git status files setting to evals (PR #9322 by @mrubens) +- Roo Code Cloud Provider pricing page and changes elsewhere (PR #9195 by @brunobergher) + +## [3.32.1] - 2025-11-14 + +![3.32.1 Release - Bug Fixes](/releases/3.32.1-release.png) + +- Fix: Add abort controller for request cancellation in OpenAI native protocol (PR #9276 by @daniel-lxs) +- Fix: Resolve duplicate tool blocks causing 'tool has already been used' error in native protocol mode (PR #9275 by @daniel-lxs) +- Fix: Prevent duplicate tool_result blocks in native protocol mode for read_file (PR #9272 by @daniel-lxs) +- Fix: Correct OpenAI Native handling of encrypted reasoning blocks to prevent errors during condensing (PR #9263 by @hannesrudolph) +- Fix: Disable XML parser for native tool protocol to prevent parsing conflicts (PR #9277 by @daniel-lxs) + +## [3.32.0] - 2025-11-14 + +![3.32.0 Release - GPT-5.1 models and OpenAI prompt caching](/releases/3.32.0-release.png) + +- Feature: Add GPT-5.1 models to OpenAI provider (PR #9252 by @hannesrudolph) +- Feature: Support for OpenAI Responses 24 hour prompt caching (PR #9259 by @hannesrudolph) +- Fix: Repair the share button in the UI (PR #9253 by @hannesrudolph) +- Docs: Include PR numbers in the release guide to improve traceability (PR #9236 by @hannesrudolph) + +## [3.31.3] - 2025-11-13 + +![3.31.3 Release - Kangaroo Decrypting a Message](/releases/3.31.3-release.png) + +- Fix: OpenAI Native encrypted_content handling and remove gpt-5-chat-latest verbosity flag (#9225 by @politsin, PR by @hannesrudolph) +- Fix: Roo Code Cloud provider Anthropic input token normalization to avoid double-counting (thanks @hannesrudolph!) +- Refactor: Rename sliding-window to context-management and truncateConversationIfNeeded to manageContext (thanks @hannesrudolph!) + +## [3.31.2] - 2025-11-12 + +- Fix: Apply updated API profile settings when provider/model unchanged (#9208 by @hannesrudolph, PR by @hannesrudolph) +- Migrate conversation continuity to plugin-side encrypted reasoning items using Responses API for improved reliability (thanks @hannesrudolph!) +- Fix: Include mcpServers in getState() for auto-approval (#9190 by @bozoweed, PR by @daniel-lxs) +- Batch settings updates from the webview to the extension host for improved performance (thanks @cte!) +- Fix: Replace rate-limited badges with badgen.net to improve README reliability (thanks @daniel-lxs!) + +## [3.31.1] - 2025-11-11 + +![3.31.1 Release - Kangaroo Stuck in the Clouds](/releases/3.31.1-release.png) + +- Fix: Prevent command_output ask from blocking in cloud/headless environments (thanks @daniel-lxs!) +- Add IPC command for sending messages to the current task (thanks @mrubens!) +- Fix: Model switch re-applies selected profile, ensuring task configuration stays in sync (#9179 by @hannesrudolph, PR by @hannesrudolph) +- Move auto-approval logic from `ChatView` to `Task` for better architecture (thanks @cte!) +- Add custom Button component with variant system (thanks @brunobergher!) + +## [3.31.0] - 2025-11-07 + +![3.31.0 Release - Todo List and Task Header Improvements](/releases/3.31.0-release.png) + +- Improvements to to-do lists and task headers (thanks @brunobergher!) +- Fix: Prevent crash when streaming chunks have null choices array (thanks @daniel-lxs!) +- Fix: Prevent context condensing on settings save when provider/model unchanged (#4430 by @hannesrudolph, PR by @daniel-lxs) +- Fix: Respect custom OpenRouter URL for all API operations (#8947 by @sstraus, PR by @roomote) +- Add comprehensive error logging to Roo Cloud provider (thanks @daniel-lxs!) +- UX: Less caffeinated kangaroo (thanks @brunobergher!) + +## [3.30.3] - 2025-11-06 + +![3.30.3 Release - Moonshot Brain](/releases/3.30.3-release.png) + +- Feat: Add kimi-k2-thinking model to Moonshot provider (thanks @daniel-lxs!) +- Fix: Auto-retry on empty assistant response to prevent task failures (#9076 by @Akillatech, PR by @daniel-lxs) +- Fix: Use system role for OpenAI Compatible provider when streaming is disabled (#8215 by @whitfin, PR by @roomote) +- Fix: Prevent notification sound on attempt_completion with queued messages (#8537 by @hannesrudolph, PR by @roomote) +- Feat: Auto-switch to imported mode with architect fallback for better mode detection (#8239 by @hannesrudolph, PR by @daniel-lxs) +- Feat: Add MiniMax-M2-Stable model and enable prompt caching (#9070 by @nokaka, PR by @roomote) +- Feat: Improve diff appearance in main chat view (thanks @hannesrudolph!) +- UX: Home screen visuals (thanks @brunobergher!) +- Docs: Clarify that setting 0 disables Error & Repetition Limit (thanks @roomote!) +- Chore: Update dependency @changesets/cli to v2.29.7 (thanks @renovate!) + +## [3.30.2] - 2025-11-05 + +![3.30.2 Release - Eliminating UI Flicker](/releases/3.30.2-release.png) + +- Fix: eliminate UI flicker during task cancellation (thanks @daniel-lxs!) +- Add Global Inference support for Bedrock models (#8750 by @ronyblum, PR by @hannesrudolph) +- Add Qwen3 embedding models (0.6B and 4B) to OpenRouter support (#9058 by @dmarkey, PR by @app/roomote) +- Fix: resolve incorrect commit location when GIT_DIR set in Dev Containers (#4567 by @nonsleepr, PR by @heyseth) +- Fix: keep pinned models fixed at top of scrollable list (#8812 by @XiaoYingYo, PR by @app/roomote) +- Fix: update Opus 4.1 max tokens from 8K to 32K (#9045 by @kaveh-deriv, PR by @app/roomote) +- Set Claude Sonnet 4.5 as default for key providers (thanks @hannesrudolph!) +- Fix: dynamic provider model validation to prevent cross-contamination (#9047 by @NotADev137, PR by @daniel-lxs) +- Fix: Bedrock user agent to report full SDK details (#9031 by @ajjuaire, PR by @ajjuaire) +- Add file path tooltips with centralized PathTooltip component (#8278 by @da2ce7, PR by @daniel-lxs) +- Add conditional test running to pre-push hook (thanks @daniel-lxs!) +- Update Cerebras integration (thanks @sebastiand-cerebras!) + +## [3.30.1] - 2025-11-04 + +- Fix: Correct OpenRouter Mistral model embedding dimension from 3072 to 1536 (thanks @daniel-lxs!) +- Revert: Previous UI flicker fix that caused issues with task resumption (thanks @mrubens!) + ## [3.30.0] - 2025-11-03 ![3.30.0 Release - PR Fixer](/releases/3.30.0-release.png) @@ -148,7 +479,7 @@ ## [3.28.11] - 2025-09-29 -- Fix: Correct AWS Bedrock Claude Sonnet 4.5 model identifier (#8371 by @sunhyung, PR by @app/roomote) +- Fix: Correct Amazon Bedrock Claude Sonnet 4.5 model identifier (#8371 by @sunhyung, PR by @app/roomote) - Fix: Correct Claude Sonnet 4.5 model ID format (thanks @daniel-lxs!) ## [3.28.10] - 2025-09-29 @@ -480,7 +811,7 @@ ## [3.25.14] - 2025-08-13 - Fix: Only include verbosity parameter for models that support it (#7054 by @eastonmeth, PR by @app/roomote) -- Fix: AWS Bedrock 1M context - Move anthropic_beta to additionalModelRequestFields (thanks @daniel-lxs!) +- Fix: Amazon Bedrock 1M context - Move anthropic_beta to additionalModelRequestFields (thanks @daniel-lxs!) - Fix: Make cancelling requests more responsive by reverting recent changes ## [3.25.13] - 2025-08-12 @@ -845,7 +1176,7 @@ - Add user-configurable search score threshold slider for semantic search (thanks @hannesrudolph!) - Add default headers and testing for litellm fetcher (thanks @andrewshu2000!) - Fix consistent cancellation error messages for thinking vs streaming phases -- Fix AWS Bedrock cross-region inference profile mapping (thanks @KevinZhao!) +- Fix Amazon Bedrock cross-region inference profile mapping (thanks @KevinZhao!) - Fix URL loading timeout issues in @ mentions (thanks @MuriloFP!) - Fix API retry exponential backoff capped at 10 minutes (thanks @MuriloFP!) - Fix Qdrant URL field auto-filling with default value (thanks @SannidhyaSah!) @@ -859,7 +1190,7 @@ - Suppress Mermaid error rendering - Improve Mermaid buttons with light background in light mode (thanks @chrarnoldus!) - Add .vscode/ to write-protected files/directories -- Update AWS Bedrock cross-region inference profile mapping (thanks @KevinZhao!) +- Update Amazon Bedrock cross-region inference profile mapping (thanks @KevinZhao!) ## [3.22.5] - 2025-06-28 @@ -1483,7 +1814,7 @@ - Improved display of diff errors + easy copying for investigation - Fixes to .vscodeignore (thanks @franekp!) - Fix a zh-CN translation for model capabilities (thanks @zhangtony239!) -- Rename AWS Bedrock to Amazon Bedrock (thanks @ronyblum!) +- Rename Amazon Bedrock to Amazon Bedrock (thanks @ronyblum!) - Update extension title and description (thanks @StevenTCramer!) ## [3.11.12] - 2025-04-09 @@ -1732,12 +2063,12 @@ - PowerShell-specific command handling (thanks @KJ7LNW!) - OpenAI-compatible DeepSeek/QwQ reasoning support (thanks @lightrabbit!) - Anthropic-style prompt caching in the OpenAI-compatible provider (thanks @dleen!) -- Add Deepseek R1 for AWS Bedrock (thanks @ATempsch!) +- Add Deepseek R1 for Amazon Bedrock (thanks @ATempsch!) - Fix MarkdownBlock text color for Dark High Contrast theme (thanks @cannuri!) - Add gemini-2.0-pro-exp-02-05 model to vertex (thanks @shohei-ihaya!) - Bring back progress status for multi-diff edits (thanks @qdaxb!) - Refactor alert dialog styles to use the correct vscode theme (thanks @cannuri!) -- Custom ARNs in AWS Bedrock (thanks @Smartsheet-JB-Brown!) +- Custom ARNs in Amazon Bedrock (thanks @Smartsheet-JB-Brown!) - Update MCP servers directory path for platform compatibility (thanks @hannesrudolph!) - Fix browser system prompt inclusion rules (thanks @cannuri!) - Publish git tags to github from CI (thanks @pdecat!) @@ -1875,7 +2206,7 @@ ## [3.7.1] - 2025-02-24 -- Add AWS Bedrock support for Sonnet 3.7 and update some defaults to Sonnet 3.7 instead of 3.5 +- Add Amazon Bedrock support for Sonnet 3.7 and update some defaults to Sonnet 3.7 instead of 3.5 ## [3.7.0] - 2025-02-24 @@ -1892,7 +2223,7 @@ ## [3.3.24] - 2025-02-20 -- Fixed a bug with region selection preventing AWS Bedrock profiles from being saved (thanks @oprstchn!) +- Fixed a bug with region selection preventing Amazon Bedrock profiles from being saved (thanks @oprstchn!) - Updated the price of gpt-4o (thanks @marvijo-code!) ## [3.3.23] - 2025-02-20 @@ -2076,7 +2407,7 @@ - Reverts provider key entry back to checking onInput instead of onChange to hopefully address issues entering API keys (thanks @samhvw8!) - Added explicit checkbox to use Azure for OpenAI compatible providers (thanks @samhvw8!) - Fixed Glama usage reporting (thanks @punkpeye!) -- Added Llama 3.3 70B Instruct model to the AWS Bedrock provider options (thanks @Premshay!) +- Added Llama 3.3 70B Instruct model to the Amazon Bedrock provider options (thanks @Premshay!) ## [3.2.7] diff --git a/README.md b/README.md index 14de4ab109..1be0603a7c 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,5 @@

- VS Code + VS Code Marketplace X YouTube Join Discord @@ -35,7 +35,7 @@ - [简体中文](locales/zh-CN/README.md) - [繁體中文](locales/zh-TW/README.md) - ... - + --- @@ -167,16 +167,6 @@ We love community contributions! Get started by reading our [CONTRIBUTING.md](CO --- -## Contributors - -Thanks to all our contributors who have helped make Roo Code better! - - - -[![Contributors](https://contrib.rocks/image?repo=RooCodeInc/roo-code&max=120&columns=12&cacheBust=0000000000)](https://github.com/RooCodeInc/roo-code/graphs/contributors) - - - ## License [Apache 2.0 © 2025 Roo Code, Inc.](./LICENSE) diff --git a/apps/vscode-e2e/package.json b/apps/vscode-e2e/package.json index 1d19ffebf2..d366f72a2d 100644 --- a/apps/vscode-e2e/package.json +++ b/apps/vscode-e2e/package.json @@ -18,7 +18,7 @@ "@types/vscode": "^1.95.0", "@vscode/test-cli": "^0.0.11", "@vscode/test-electron": "^2.4.0", - "glob": "^11.0.1", + "glob": "^11.1.0", "mocha": "^11.1.0", "rimraf": "^6.0.1", "typescript": "5.8.3" diff --git a/apps/vscode-e2e/src/suite/extension.test.ts b/apps/vscode-e2e/src/suite/extension.test.ts index e7a92521cf..5d59e003ef 100644 --- a/apps/vscode-e2e/src/suite/extension.test.ts +++ b/apps/vscode-e2e/src/suite/extension.test.ts @@ -15,8 +15,6 @@ suite("Roo Code Extension", function () { "SidebarProvider.removeView", "activationCompleted", "plusButtonClicked", - "mcpButtonClicked", - "promptsButtonClicked", "popoutButtonClicked", "openInNewTab", "settingsButtonClicked", diff --git a/apps/vscode-e2e/src/suite/tools/insert-content.test.ts b/apps/vscode-e2e/src/suite/tools/insert-content.test.ts deleted file mode 100644 index a3a3abb186..0000000000 --- a/apps/vscode-e2e/src/suite/tools/insert-content.test.ts +++ /dev/null @@ -1,628 +0,0 @@ -import * as assert from "assert" -import * as fs from "fs/promises" -import * as path from "path" -import * as vscode from "vscode" - -import { RooCodeEventName, type ClineMessage } from "@roo-code/types" - -import { waitFor, sleep } from "../utils" -import { setDefaultSuiteTimeout } from "../test-utils" - -suite.skip("Roo Code insert_content Tool", function () { - setDefaultSuiteTimeout(this) - - let workspaceDir: string - - // Pre-created test files that will be used across tests - const testFiles = { - simpleText: { - name: `test-insert-simple-${Date.now()}.txt`, - content: "Line 1\nLine 2\nLine 3", - path: "", - }, - jsFile: { - name: `test-insert-js-${Date.now()}.js`, - content: `function hello() { - console.log("Hello World") -} - -function goodbye() { - console.log("Goodbye World") -}`, - path: "", - }, - emptyFile: { - name: `test-insert-empty-${Date.now()}.txt`, - content: "", - path: "", - }, - pythonFile: { - name: `test-insert-python-${Date.now()}.py`, - content: `def main(): - print("Start") - print("End")`, - path: "", - }, - } - - // Get the actual workspace directory that VSCode is using and create all test files - suiteSetup(async function () { - // Get the workspace folder from VSCode - const workspaceFolders = vscode.workspace.workspaceFolders - if (!workspaceFolders || workspaceFolders.length === 0) { - throw new Error("No workspace folder found") - } - workspaceDir = workspaceFolders[0]!.uri.fsPath - console.log("Using workspace directory:", workspaceDir) - - // Create all test files before any tests run - console.log("Creating test files in workspace...") - for (const [key, file] of Object.entries(testFiles)) { - file.path = path.join(workspaceDir, file.name) - await fs.writeFile(file.path, file.content) - console.log(`Created ${key} test file at:`, file.path) - } - - // Verify all files exist - for (const [key, file] of Object.entries(testFiles)) { - const exists = await fs - .access(file.path) - .then(() => true) - .catch(() => false) - if (!exists) { - throw new Error(`Failed to create ${key} test file at ${file.path}`) - } - } - }) - - // Clean up after all tests - suiteTeardown(async () => { - // Cancel any running tasks before cleanup - test("Should insert content at the beginning of a file (line 1)", async function () { - const api = globalThis.api - // Clean up before each test - setup(async () => { - // Cancel any previous task - try { - await globalThis.api.cancelCurrentTask() - } catch { - // Task might not be running - } - - // Small delay to ensure clean state - await sleep(100) - }) - - // Clean up after each test - teardown(async () => { - // Cancel the current task - try { - await globalThis.api.cancelCurrentTask() - } catch { - // Task might not be running - } - - // Small delay to ensure clean state - await sleep(100) - }) - const messages: ClineMessage[] = [] - const testFile = testFiles.simpleText - const insertContent = "New first line" - const expectedContent = `${insertContent} -${testFile.content}` - let taskStarted = false - let taskCompleted = false - let errorOccurred: string | null = null - let insertContentExecuted = false - - // Listen for messages - const messageHandler = ({ message }: { message: ClineMessage }) => { - messages.push(message) - - // Log important messages for debugging - if (message.type === "say" && message.say === "error") { - errorOccurred = message.text || "Unknown error" - console.error("Error:", message.text) - } - if (message.type === "ask" && message.ask === "tool") { - console.log("Tool request:", message.text?.substring(0, 200)) - } - if (message.type === "say" && (message.say === "completion_result" || message.say === "text")) { - console.log("AI response:", message.text?.substring(0, 200)) - } - - // Check for tool execution - if (message.type === "say" && message.say === "api_req_started" && message.text) { - console.log("API request started:", message.text.substring(0, 200)) - try { - const requestData = JSON.parse(message.text) - if (requestData.request && requestData.request.includes("insert_content")) { - insertContentExecuted = true - console.log("insert_content tool executed!") - } - } catch (e) { - console.log("Failed to parse api_req_started message:", e) - } - } - } - api.on(RooCodeEventName.Message, messageHandler) - - // Listen for task events - const taskStartedHandler = (id: string) => { - if (id === taskId) { - taskStarted = true - console.log("Task started:", id) - } - } - api.on(RooCodeEventName.TaskStarted, taskStartedHandler) - - const taskCompletedHandler = (id: string) => { - if (id === taskId) { - taskCompleted = true - console.log("Task completed:", id) - } - } - api.on(RooCodeEventName.TaskCompleted, taskCompletedHandler) - - let taskId: string - try { - // Start the task - taskId = await api.startNewTask({ - configuration: { - mode: "code", - autoApprovalEnabled: true, - alwaysAllowWrite: true, - alwaysAllowReadOnly: true, - alwaysAllowReadOnlyOutsideWorkspace: true, - }, - text: `Use insert_content to add "${insertContent}" at line 1 (beginning) of the file ${testFile.name}. The file already exists with this content: -${testFile.content} - -Assume the file exists and you can modify it directly.`, - }) - - console.log("Task ID:", taskId) - console.log("Test filename:", testFile.name) - - // Wait for task to start - await waitFor(() => taskStarted, { timeout: 45_000 }) - - // Check for early errors - if (errorOccurred) { - console.error("Early error detected:", errorOccurred) - } - - // Wait for task completion - await waitFor(() => taskCompleted, { timeout: 45_000 }) - - // Give extra time for file system operations - await sleep(2000) - - // Check if the file was modified correctly - const actualContent = await fs.readFile(testFile.path, "utf-8") - console.log("File content after insertion:", actualContent) - - // Verify tool was executed - assert.strictEqual(insertContentExecuted, true, "insert_content tool should have been executed") - - // Verify file content - assert.strictEqual( - actualContent.trim(), - expectedContent.trim(), - "Content should be inserted at the beginning of the file", - ) - - // Verify no errors occurred - assert.strictEqual( - errorOccurred, - null, - `Task should complete without errors, but got: ${errorOccurred}`, - ) - - console.log("Test passed! insert_content tool executed and content inserted at beginning successfully") - } finally { - api.off(RooCodeEventName.Message, messageHandler) - api.off(RooCodeEventName.TaskStarted, taskStartedHandler) - api.off(RooCodeEventName.TaskCompleted, taskCompletedHandler) - } - }) - try { - await globalThis.api.cancelCurrentTask() - } catch { - // Task might not be running - } - - // Clean up all test files - console.log("Cleaning up test files...") - for (const [key, file] of Object.entries(testFiles)) { - try { - await fs.unlink(file.path) - console.log(`Cleaned up ${key} test file`) - } catch (error) { - console.log(`Failed to clean up ${key} test file:`, error) - } - } - }) - - test("Should insert content at the end of a file (line 0)", async function () { - const api = globalThis.api - const messages: ClineMessage[] = [] - const testFile = testFiles.simpleText - const insertContent = "New last line" - const expectedContent = `${testFile.content} -${insertContent}` - let taskStarted = false - let taskCompleted = false - let errorOccurred: string | null = null - let insertContentExecuted = false - - // Listen for messages - const messageHandler = ({ message }: { message: ClineMessage }) => { - messages.push(message) - - // Log important messages for debugging - if (message.type === "say" && message.say === "error") { - errorOccurred = message.text || "Unknown error" - console.error("Error:", message.text) - } - if (message.type === "ask" && message.ask === "tool") { - console.log("Tool request:", message.text?.substring(0, 200)) - } - if (message.type === "say" && (message.say === "completion_result" || message.say === "text")) { - console.log("AI response:", message.text?.substring(0, 200)) - } - - // Check for tool execution - if (message.type === "say" && message.say === "api_req_started" && message.text) { - console.log("API request started:", message.text.substring(0, 200)) - try { - const requestData = JSON.parse(message.text) - if (requestData.request && requestData.request.includes("insert_content")) { - insertContentExecuted = true - console.log("insert_content tool executed!") - } - } catch (e) { - console.log("Failed to parse api_req_started message:", e) - } - } - } - api.on(RooCodeEventName.Message, messageHandler) - - // Listen for task events - const taskStartedHandler = (id: string) => { - if (id === taskId) { - taskStarted = true - console.log("Task started:", id) - } - } - api.on(RooCodeEventName.TaskStarted, taskStartedHandler) - - const taskCompletedHandler = (id: string) => { - if (id === taskId) { - taskCompleted = true - console.log("Task completed:", id) - } - } - api.on(RooCodeEventName.TaskCompleted, taskCompletedHandler) - - let taskId: string - try { - // Start the task - taskId = await api.startNewTask({ - configuration: { - mode: "code", - autoApprovalEnabled: true, - alwaysAllowWrite: true, - alwaysAllowReadOnly: true, - alwaysAllowReadOnlyOutsideWorkspace: true, - }, - text: `Use insert_content to add "${insertContent}" at line 0 (end of file) of the file ${testFile.name}. The file already exists with this content: -${testFile.content} - -Assume the file exists and you can modify it directly.`, - }) - - console.log("Task ID:", taskId) - console.log("Test filename:", testFile.name) - - // Wait for task to start - await waitFor(() => taskStarted, { timeout: 45_000 }) - - // Check for early errors - if (errorOccurred) { - console.error("Early error detected:", errorOccurred) - } - - // Wait for task completion - await waitFor(() => taskCompleted, { timeout: 45_000 }) - - // Give extra time for file system operations - await sleep(2000) - - // Check if the file was modified correctly - const actualContent = await fs.readFile(testFile.path, "utf-8") - console.log("File content after insertion:", actualContent) - - // Verify tool was executed - test("Should insert multiline content into a JavaScript file", async function () { - const api = globalThis.api - const messages: ClineMessage[] = [] - const testFile = testFiles.jsFile - const insertContent = `// New import statements -import { utils } from './utils' -import { helpers } from './helpers'` - const expectedContent = `${insertContent} -${testFile.content}` - let taskStarted = false - let taskCompleted = false - let errorOccurred: string | null = null - let insertContentExecuted = false - - // Listen for messages - const messageHandler = ({ message }: { message: ClineMessage }) => { - messages.push(message) - - // Log important messages for debugging - if (message.type === "say" && message.say === "error") { - errorOccurred = message.text || "Unknown error" - console.error("Error:", message.text) - } - if (message.type === "ask" && message.ask === "tool") { - console.log("Tool request:", message.text?.substring(0, 200)) - } - if (message.type === "say" && (message.say === "completion_result" || message.say === "text")) { - console.log("AI response:", message.text?.substring(0, 200)) - } - - // Check for tool execution - if (message.type === "say" && message.say === "api_req_started" && message.text) { - console.log("API request started:", message.text.substring(0, 200)) - try { - const requestData = JSON.parse(message.text) - if (requestData.request && requestData.request.includes("insert_content")) { - insertContentExecuted = true - console.log("insert_content tool executed!") - } - } catch (e) { - console.log("Failed to parse api_req_started message:", e) - } - } - } - api.on(RooCodeEventName.Message, messageHandler) - - // Listen for task events - const taskStartedHandler = (id: string) => { - if (id === taskId) { - taskStarted = true - console.log("Task started:", id) - } - } - api.on(RooCodeEventName.TaskStarted, taskStartedHandler) - - const taskCompletedHandler = (id: string) => { - if (id === taskId) { - taskCompleted = true - console.log("Task completed:", id) - } - } - api.on(RooCodeEventName.TaskCompleted, taskCompletedHandler) - - let taskId: string - try { - // Start the task - taskId = await api.startNewTask({ - configuration: { - mode: "code", - autoApprovalEnabled: true, - alwaysAllowWrite: true, - alwaysAllowReadOnly: true, - alwaysAllowReadOnlyOutsideWorkspace: true, - }, - text: `Use insert_content to add import statements at the beginning (line 1) of the JavaScript file ${testFile.name}. Add these lines: -${insertContent} - -The file already exists with this content: -${testFile.content} - -Assume the file exists and you can modify it directly.`, - }) - - console.log("Task ID:", taskId) - console.log("Test filename:", testFile.name) - - // Wait for task to start - await waitFor(() => taskStarted, { timeout: 45_000 }) - - // Check for early errors - if (errorOccurred) { - console.error("Early error detected:", errorOccurred) - } - - // Wait for task completion - await waitFor(() => taskCompleted, { timeout: 45_000 }) - - // Give extra time for file system operations - await sleep(2000) - - test("Should insert content into an empty file", async function () { - const api = globalThis.api - const messages: ClineMessage[] = [] - const testFile = testFiles.emptyFile - const insertContent = `# My New File -This is the first line of content -And this is the second line` - const expectedContent = insertContent - let taskStarted = false - let taskCompleted = false - let errorOccurred: string | null = null - let insertContentExecuted = false - - // Listen for messages - const messageHandler = ({ message }: { message: ClineMessage }) => { - messages.push(message) - - // Log important messages for debugging - if (message.type === "say" && message.say === "error") { - errorOccurred = message.text || "Unknown error" - console.error("Error:", message.text) - } - if (message.type === "ask" && message.ask === "tool") { - console.log("Tool request:", message.text?.substring(0, 200)) - } - if ( - message.type === "say" && - (message.say === "completion_result" || message.say === "text") - ) { - console.log("AI response:", message.text?.substring(0, 200)) - } - - // Check for tool execution - if (message.type === "say" && message.say === "api_req_started" && message.text) { - console.log("API request started:", message.text.substring(0, 200)) - try { - const requestData = JSON.parse(message.text) - if (requestData.request && requestData.request.includes("insert_content")) { - insertContentExecuted = true - console.log("insert_content tool executed!") - } - } catch (e) { - console.log("Failed to parse api_req_started message:", e) - } - } - } - api.on(RooCodeEventName.Message, messageHandler) - - // Listen for task events - const taskStartedHandler = (id: string) => { - if (id === taskId) { - taskStarted = true - console.log("Task started:", id) - } - } - api.on(RooCodeEventName.TaskStarted, taskStartedHandler) - - const taskCompletedHandler = (id: string) => { - if (id === taskId) { - taskCompleted = true - console.log("Task completed:", id) - } - } - api.on(RooCodeEventName.TaskCompleted, taskCompletedHandler) - - let taskId: string - try { - // Start the task - taskId = await api.startNewTask({ - configuration: { - mode: "code", - autoApprovalEnabled: true, - alwaysAllowWrite: true, - alwaysAllowReadOnly: true, - alwaysAllowReadOnlyOutsideWorkspace: true, - }, - text: `Use insert_content to add content to the empty file ${testFile.name}. Add this content at line 0 (end of file): -${insertContent} - -The file is currently empty. Assume the file exists and you can modify it directly.`, - }) - - console.log("Task ID:", taskId) - console.log("Test filename:", testFile.name) - - // Wait for task to start - await waitFor(() => taskStarted, { timeout: 45_000 }) - - // Check for early errors - if (errorOccurred) { - console.error("Early error detected:", errorOccurred) - } - - // Wait for task completion - await waitFor(() => taskCompleted, { timeout: 45_000 }) - - // Give extra time for file system operations - await sleep(2000) - - // Check if the file was modified correctly - const actualContent = await fs.readFile(testFile.path, "utf-8") - console.log("File content after insertion:", actualContent) - - // Verify tool was executed - assert.strictEqual( - insertContentExecuted, - true, - "insert_content tool should have been executed", - ) - - // Verify file content - assert.strictEqual( - actualContent.trim(), - expectedContent.trim(), - "Content should be inserted into the empty file", - ) - - // Verify no errors occurred - assert.strictEqual( - errorOccurred, - null, - `Task should complete without errors, but got: ${errorOccurred}`, - ) - - console.log( - "Test passed! insert_content tool executed and content inserted into empty file successfully", - ) - } finally { - api.off(RooCodeEventName.Message, messageHandler) - api.off(RooCodeEventName.TaskStarted, taskStartedHandler) - api.off(RooCodeEventName.TaskCompleted, taskCompletedHandler) - } - }) - // Check if the file was modified correctly - const actualContent = await fs.readFile(testFile.path, "utf-8") - console.log("File content after insertion:", actualContent) - - // Verify tool was executed - assert.strictEqual(insertContentExecuted, true, "insert_content tool should have been executed") - - // Verify file content - assert.strictEqual( - actualContent.trim(), - expectedContent.trim(), - "Multiline content should be inserted at the beginning of the JavaScript file", - ) - - // Verify no errors occurred - assert.strictEqual( - errorOccurred, - null, - `Task should complete without errors, but got: ${errorOccurred}`, - ) - - console.log("Test passed! insert_content tool executed and multiline content inserted successfully") - } finally { - api.off(RooCodeEventName.Message, messageHandler) - api.off(RooCodeEventName.TaskStarted, taskStartedHandler) - api.off(RooCodeEventName.TaskCompleted, taskCompletedHandler) - } - }) - assert.strictEqual(insertContentExecuted, true, "insert_content tool should have been executed") - - // Verify file content - assert.strictEqual( - actualContent.trim(), - expectedContent.trim(), - "Content should be inserted at the end of the file", - ) - - // Verify no errors occurred - assert.strictEqual(errorOccurred, null, `Task should complete without errors, but got: ${errorOccurred}`) - - console.log("Test passed! insert_content tool executed and content inserted at end successfully") - } finally { - api.off(RooCodeEventName.Message, messageHandler) - api.off(RooCodeEventName.TaskStarted, taskStartedHandler) - api.off(RooCodeEventName.TaskCompleted, taskCompletedHandler) - } - }) - // Tests will be added here one by one -}) diff --git a/apps/web-evals/package.json b/apps/web-evals/package.json index 3774016332..b2ac0d4346 100644 --- a/apps/web-evals/package.json +++ b/apps/web-evals/package.json @@ -14,6 +14,7 @@ "dependencies": { "@hookform/resolvers": "^5.1.1", "@radix-ui/react-alert-dialog": "^1.1.7", + "@radix-ui/react-checkbox": "^1.1.5", "@radix-ui/react-dialog": "^1.1.6", "@radix-ui/react-dropdown-menu": "^2.1.7", "@radix-ui/react-label": "^2.1.2", @@ -28,12 +29,13 @@ "@roo-code/evals": "workspace:^", "@roo-code/types": "workspace:^", "@tanstack/react-query": "^5.69.0", + "archiver": "^7.0.1", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", "cmdk": "^1.1.0", "fuzzysort": "^3.1.0", "lucide-react": "^0.518.0", - "next": "^15.2.5", + "next": "~15.2.6", "next-themes": "^0.4.6", "p-map": "^7.0.3", "react": "^18.3.1", @@ -51,6 +53,7 @@ "@roo-code/config-eslint": "workspace:^", "@roo-code/config-typescript": "workspace:^", "@tailwindcss/postcss": "^4", + "@types/archiver": "^7.0.0", "@types/ps-tree": "^1.1.6", "@types/react": "^18.3.23", "@types/react-dom": "^18.3.5", diff --git a/apps/web-evals/src/actions/__tests__/killRun.spec.ts b/apps/web-evals/src/actions/__tests__/killRun.spec.ts new file mode 100644 index 0000000000..814d70d9fc --- /dev/null +++ b/apps/web-evals/src/actions/__tests__/killRun.spec.ts @@ -0,0 +1,207 @@ +// npx vitest run src/actions/__tests__/killRun.spec.ts + +import { execFileSync } from "child_process" + +// Mock child_process +vi.mock("child_process", () => ({ + execFileSync: vi.fn(), + spawn: vi.fn(), +})) + +// Mock next/cache +vi.mock("next/cache", () => ({ + revalidatePath: vi.fn(), +})) + +// Mock redis client +vi.mock("@/lib/server/redis", () => ({ + redisClient: vi.fn().mockResolvedValue({ + del: vi.fn().mockResolvedValue(1), + }), +})) + +// Mock @roo-code/evals +vi.mock("@roo-code/evals", () => ({ + createRun: vi.fn(), + deleteRun: vi.fn(), + createTask: vi.fn(), + exerciseLanguages: [], + getExercisesForLanguage: vi.fn().mockResolvedValue([]), +})) + +// Mock timers to speed up tests +vi.useFakeTimers() + +// Import after mocks +import { killRun } from "../runs" + +const mockExecFileSync = execFileSync as ReturnType + +describe("killRun", () => { + beforeEach(() => { + vi.clearAllMocks() + }) + + afterEach(() => { + vi.clearAllTimers() + }) + + it("should kill controller first, wait, then kill task containers", async () => { + const runId = 123 + + // execFileSync is used for all docker commands + mockExecFileSync + .mockReturnValueOnce("") // docker kill controller + .mockReturnValueOnce("evals-task-123-456.0\nevals-task-123-789.1\n") // docker ps + .mockReturnValueOnce("") // docker kill evals-task-123-456.0 + .mockReturnValueOnce("") // docker kill evals-task-123-789.1 + + const resultPromise = killRun(runId) + + // Fast-forward past the 10 second sleep + await vi.advanceTimersByTimeAsync(10000) + + const result = await resultPromise + + expect(result.success).toBe(true) + expect(result.killedContainers).toContain("evals-controller-123") + expect(result.killedContainers).toContain("evals-task-123-456.0") + expect(result.killedContainers).toContain("evals-task-123-789.1") + expect(result.errors).toHaveLength(0) + + // Verify execFileSync was called for docker kill + expect(mockExecFileSync).toHaveBeenNthCalledWith( + 1, + "docker", + ["kill", "evals-controller-123"], + expect.any(Object), + ) + // Verify execFileSync was called for docker ps with run-specific filter + expect(mockExecFileSync).toHaveBeenNthCalledWith( + 2, + "docker", + ["ps", "--format", "{{.Names}}", "--filter", "name=evals-task-123-"], + expect.any(Object), + ) + }) + + it("should continue killing runners even if controller is not running", async () => { + const runId = 456 + + mockExecFileSync + .mockImplementationOnce(() => { + throw new Error("No such container") + }) // controller kill fails + .mockReturnValueOnce("evals-task-456-100.0\n") // docker ps + .mockReturnValueOnce("") // docker kill task + + const resultPromise = killRun(runId) + await vi.advanceTimersByTimeAsync(10000) + const result = await resultPromise + + expect(result.success).toBe(true) + expect(result.killedContainers).toContain("evals-task-456-100.0") + // Controller not in list since it failed + expect(result.killedContainers).not.toContain("evals-controller-456") + }) + + it("should clear Redis state after killing containers", async () => { + const runId = 789 + + const mockDel = vi.fn().mockResolvedValue(1) + const { redisClient } = await import("@/lib/server/redis") + vi.mocked(redisClient).mockResolvedValue({ del: mockDel } as never) + + mockExecFileSync + .mockReturnValueOnce("") // controller kill + .mockReturnValueOnce("") // docker ps (no tasks) + + const resultPromise = killRun(runId) + await vi.advanceTimersByTimeAsync(10000) + await resultPromise + + expect(mockDel).toHaveBeenCalledWith("heartbeat:789") + expect(mockDel).toHaveBeenCalledWith("runners:789") + }) + + it("should handle docker ps failure gracefully", async () => { + const runId = 111 + + mockExecFileSync + .mockReturnValueOnce("") // controller kill succeeds + .mockImplementationOnce(() => { + throw new Error("Docker error") + }) // docker ps fails + + const resultPromise = killRun(runId) + await vi.advanceTimersByTimeAsync(10000) + const result = await resultPromise + + // Should still be successful because controller was killed + expect(result.success).toBe(true) + expect(result.killedContainers).toContain("evals-controller-111") + expect(result.errors).toContain("Failed to list Docker task containers") + }) + + it("should handle individual task kill failures", async () => { + const runId = 222 + + mockExecFileSync + .mockReturnValueOnce("") // controller kill + .mockReturnValueOnce("evals-task-222-300.0\nevals-task-222-400.0\n") // docker ps + .mockImplementationOnce(() => { + throw new Error("Kill failed") + }) // first task kill fails + .mockReturnValueOnce("") // second task kill succeeds + + const resultPromise = killRun(runId) + await vi.advanceTimersByTimeAsync(10000) + const result = await resultPromise + + expect(result.success).toBe(true) + expect(result.killedContainers).toContain("evals-controller-222") + expect(result.killedContainers).toContain("evals-task-222-400.0") + expect(result.errors.length).toBe(1) + expect(result.errors[0]).toContain("evals-task-222-300.0") + }) + + it("should return success with no containers when nothing is running", async () => { + const runId = 333 + + mockExecFileSync + .mockImplementationOnce(() => { + throw new Error("No such container") + }) // controller not running + .mockReturnValueOnce("") // no task containers + + const resultPromise = killRun(runId) + await vi.advanceTimersByTimeAsync(10000) + const result = await resultPromise + + expect(result.success).toBe(true) + expect(result.killedContainers).toHaveLength(0) + expect(result.errors).toHaveLength(0) + }) + + it("should only kill containers belonging to the specific run", async () => { + const runId = 555 + + mockExecFileSync + .mockReturnValueOnce("") // controller kill + .mockReturnValueOnce("evals-task-555-100.0\n") // docker ps + .mockReturnValueOnce("") // docker kill task + + const resultPromise = killRun(runId) + await vi.advanceTimersByTimeAsync(10000) + const result = await resultPromise + + expect(result.success).toBe(true) + // Verify execFileSync was called for docker ps with run-specific filter + expect(mockExecFileSync).toHaveBeenNthCalledWith( + 2, + "docker", + ["ps", "--format", "{{.Names}}", "--filter", "name=evals-task-555-"], + expect.any(Object), + ) + }) +}) diff --git a/apps/web-evals/src/actions/runs.ts b/apps/web-evals/src/actions/runs.ts index 2eae1f6804..a3fb3feccc 100644 --- a/apps/web-evals/src/actions/runs.ts +++ b/apps/web-evals/src/actions/runs.ts @@ -3,7 +3,7 @@ import * as path from "path" import fs from "fs" import { fileURLToPath } from "url" -import { spawn } from "child_process" +import { spawn, execFileSync } from "child_process" import { revalidatePath } from "next/cache" import pMap from "p-map" @@ -18,11 +18,11 @@ import { } from "@roo-code/evals" import { CreateRun } from "@/lib/schemas" +import { redisClient } from "@/lib/server/redis" const EVALS_REPO_PATH = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../../../../../evals") -// eslint-disable-next-line @typescript-eslint/no-unused-vars -export async function createRun({ suite, exercises = [], systemPrompt, timeout, ...values }: CreateRun) { +export async function createRun({ suite, exercises = [], timeout, iterations = 1, ...values }: CreateRun) { const run = await _createRun({ ...values, timeout, @@ -37,15 +37,34 @@ export async function createRun({ suite, exercises = [], systemPrompt, timeout, throw new Error("Invalid exercise path: " + path) } - await createTask({ ...values, runId: run.id, language: language as ExerciseLanguage, exercise }) + // Create multiple tasks for each iteration + for (let iteration = 1; iteration <= iterations; iteration++) { + await createTask({ + ...values, + runId: run.id, + language: language as ExerciseLanguage, + exercise, + iteration, + }) + } } } else { for (const language of exerciseLanguages) { - const exercises = await getExercisesForLanguage(EVALS_REPO_PATH, language) + const languageExercises = await getExercisesForLanguage(EVALS_REPO_PATH, language) - await pMap(exercises, (exercise) => createTask({ runId: run.id, language, exercise }), { - concurrency: 10, - }) + // Create tasks for all iterations of each exercise + const tasksToCreate: Array<{ language: ExerciseLanguage; exercise: string; iteration: number }> = [] + for (const exercise of languageExercises) { + for (let iteration = 1; iteration <= iterations; iteration++) { + tasksToCreate.push({ language, exercise, iteration }) + } + } + + await pMap( + tasksToCreate, + ({ language, exercise, iteration }) => createTask({ runId: run.id, language, exercise, iteration }), + { concurrency: 10 }, + ) } } @@ -98,3 +117,100 @@ export async function deleteRun(runId: number) { await _deleteRun(runId) revalidatePath("/runs") } + +export type KillRunResult = { + success: boolean + killedContainers: string[] + errors: string[] +} + +const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms)) + +/** + * Kill all Docker containers associated with a run (controller and task runners). + * Kills the controller first, waits 10 seconds, then kills runners. + * Also clears Redis state for heartbeat and runners. + * + * Container naming conventions: + * - Controller: evals-controller-{runId} + * - Task runners: evals-task-{runId}-{taskId}.{attempt} + */ +export async function killRun(runId: number): Promise { + const killedContainers: string[] = [] + const errors: string[] = [] + const controllerPattern = `evals-controller-${runId}` + const taskPattern = `evals-task-${runId}-` + + try { + // Step 1: Kill the controller first + console.log(`Killing controller: ${controllerPattern}`) + try { + execFileSync("docker", ["kill", controllerPattern], { encoding: "utf-8", timeout: 10000 }) + killedContainers.push(controllerPattern) + console.log(`Killed controller container: ${controllerPattern}`) + } catch (_error) { + // Controller might not be running - that's ok, continue to kill runners + console.log(`Controller ${controllerPattern} not running or already stopped`) + } + + // Step 2: Wait 10 seconds before killing runners + console.log("Waiting 10 seconds before killing runners...") + await sleep(10000) + + // Step 3: Find and kill all task runner containers for THIS run only + let taskContainerNames: string[] = [] + + try { + const output = execFileSync("docker", ["ps", "--format", "{{.Names}}", "--filter", `name=${taskPattern}`], { + encoding: "utf-8", + timeout: 10000, + }) + taskContainerNames = output + .split("\n") + .map((name) => name.trim()) + .filter((name) => name.length > 0 && name.startsWith(taskPattern)) + } catch (error) { + console.error("Failed to list task containers:", error) + errors.push("Failed to list Docker task containers") + } + + // Kill each task runner container + for (const containerName of taskContainerNames) { + try { + execFileSync("docker", ["kill", containerName], { encoding: "utf-8", timeout: 10000 }) + killedContainers.push(containerName) + console.log(`Killed task container: ${containerName}`) + } catch (error) { + // Container might have already stopped + console.error(`Failed to kill container ${containerName}:`, error) + errors.push(`Failed to kill container: ${containerName}`) + } + } + + // Step 4: Clear Redis state + try { + const redis = await redisClient() + const heartbeatKey = `heartbeat:${runId}` + const runnersKey = `runners:${runId}` + + await redis.del(heartbeatKey) + await redis.del(runnersKey) + console.log(`Cleared Redis keys: ${heartbeatKey}, ${runnersKey}`) + } catch (error) { + console.error("Failed to clear Redis state:", error) + errors.push("Failed to clear Redis state") + } + } catch (error) { + console.error("Error in killRun:", error) + errors.push("Unexpected error while killing containers") + } + + revalidatePath(`/runs/${runId}`) + revalidatePath("/runs") + + return { + success: killedContainers.length > 0 || errors.length === 0, + killedContainers, + errors, + } +} diff --git a/apps/web-evals/src/app/api/health/route.ts b/apps/web-evals/src/app/api/health/route.ts deleted file mode 100644 index ca8a833942..0000000000 --- a/apps/web-evals/src/app/api/health/route.ts +++ /dev/null @@ -1,24 +0,0 @@ -import { NextResponse } from "next/server" - -export async function GET() { - try { - return NextResponse.json( - { - status: "healthy", - timestamp: new Date().toISOString(), - uptime: process.uptime(), - environment: process.env.NODE_ENV || "production", - }, - { status: 200 }, - ) - } catch (error) { - return NextResponse.json( - { - status: "unhealthy", - timestamp: new Date().toISOString(), - error: error instanceof Error ? error.message : "Unknown error", - }, - { status: 503 }, - ) - } -} diff --git a/apps/web-evals/src/app/api/runs/[id]/logs/[taskId]/route.ts b/apps/web-evals/src/app/api/runs/[id]/logs/[taskId]/route.ts new file mode 100644 index 0000000000..e5ec8751ab --- /dev/null +++ b/apps/web-evals/src/app/api/runs/[id]/logs/[taskId]/route.ts @@ -0,0 +1,74 @@ +import { NextResponse } from "next/server" +import type { NextRequest } from "next/server" +import * as fs from "node:fs/promises" +import * as path from "node:path" + +import { findTask, findRun } from "@roo-code/evals" + +export const dynamic = "force-dynamic" + +const LOG_BASE_PATH = "/tmp/evals/runs" + +// Sanitize path components to prevent path traversal attacks +function sanitizePathComponent(component: string): string { + // Remove any path separators, null bytes, and other dangerous characters + return component.replace(/[/\\:\0*?"<>|]/g, "_") +} + +export async function GET(request: NextRequest, { params }: { params: Promise<{ id: string; taskId: string }> }) { + const { id, taskId } = await params + + try { + const runId = Number(id) + const taskIdNum = Number(taskId) + + if (isNaN(runId) || isNaN(taskIdNum)) { + return NextResponse.json({ error: "Invalid run ID or task ID" }, { status: 400 }) + } + + // Verify the run exists + await findRun(runId) + + // Get the task to find its language and exercise + const task = await findTask(taskIdNum) + + // Verify the task belongs to this run + if (task.runId !== runId) { + return NextResponse.json({ error: "Task does not belong to this run" }, { status: 404 }) + } + + // Sanitize language and exercise to prevent path traversal + const safeLanguage = sanitizePathComponent(task.language) + const safeExercise = sanitizePathComponent(task.exercise) + + // Construct the log file path + const logFileName = `${safeLanguage}-${safeExercise}.log` + const logFilePath = path.join(LOG_BASE_PATH, String(runId), logFileName) + + // Verify the resolved path is within the expected directory (defense in depth) + const resolvedPath = path.resolve(logFilePath) + const expectedBase = path.resolve(LOG_BASE_PATH) + if (!resolvedPath.startsWith(expectedBase)) { + return NextResponse.json({ error: "Invalid log path" }, { status: 400 }) + } + + // Check if the log file exists and read it (async) + try { + const logContent = await fs.readFile(logFilePath, "utf-8") + return NextResponse.json({ logContent }) + } catch (err) { + if ((err as NodeJS.ErrnoException).code === "ENOENT") { + return NextResponse.json({ error: "Log file not found", logContent: null }, { status: 200 }) + } + throw err + } + } catch (error) { + console.error("Error reading task log:", error) + + if (error instanceof Error && error.name === "RecordNotFoundError") { + return NextResponse.json({ error: "Task or run not found" }, { status: 404 }) + } + + return NextResponse.json({ error: "Failed to read log file" }, { status: 500 }) + } +} diff --git a/apps/web-evals/src/app/api/runs/[id]/logs/failed/route.ts b/apps/web-evals/src/app/api/runs/[id]/logs/failed/route.ts new file mode 100644 index 0000000000..f8c6cec06b --- /dev/null +++ b/apps/web-evals/src/app/api/runs/[id]/logs/failed/route.ts @@ -0,0 +1,129 @@ +import { NextResponse } from "next/server" +import type { NextRequest } from "next/server" +import * as fs from "node:fs" +import * as path from "node:path" +import archiver from "archiver" + +import { findRun, getTasks } from "@roo-code/evals" + +export const dynamic = "force-dynamic" + +const LOG_BASE_PATH = "/tmp/evals/runs" + +// Sanitize path components to prevent path traversal attacks +function sanitizePathComponent(component: string): string { + // Remove any path separators, null bytes, and other dangerous characters + return component.replace(/[/\\:\0*?"<>|]/g, "_") +} + +export async function GET(request: NextRequest, { params }: { params: Promise<{ id: string }> }) { + const { id } = await params + + try { + const runId = Number(id) + + if (isNaN(runId)) { + return NextResponse.json({ error: "Invalid run ID" }, { status: 400 }) + } + + // Verify the run exists + await findRun(runId) + + // Get all tasks for this run + const tasks = await getTasks(runId) + + // Filter for failed tasks only + const failedTasks = tasks.filter((task) => task.passed === false) + + if (failedTasks.length === 0) { + return NextResponse.json({ error: "No failed tasks to export" }, { status: 400 }) + } + + // Create a zip archive + const archive = archiver("zip", { zlib: { level: 9 } }) + + // Collect chunks to build the response + const chunks: Buffer[] = [] + + archive.on("data", (chunk: Buffer) => { + chunks.push(chunk) + }) + + // Track archive errors + let archiveError: Error | null = null + archive.on("error", (err: Error) => { + archiveError = err + }) + + // Set up the end promise before finalizing (proper event listener ordering) + const archiveEndPromise = new Promise((resolve, reject) => { + archive.on("end", resolve) + archive.on("error", reject) + }) + + // Add each failed task's log file to the archive + const logDir = path.join(LOG_BASE_PATH, String(runId)) + let filesAdded = 0 + + for (const task of failedTasks) { + // Sanitize language and exercise to prevent path traversal + const safeLanguage = sanitizePathComponent(task.language) + const safeExercise = sanitizePathComponent(task.exercise) + const logFileName = `${safeLanguage}-${safeExercise}.log` + const logFilePath = path.join(logDir, logFileName) + + // Verify the resolved path is within the expected directory (defense in depth) + const resolvedPath = path.resolve(logFilePath) + const expectedBase = path.resolve(LOG_BASE_PATH) + if (!resolvedPath.startsWith(expectedBase)) { + continue // Skip files with suspicious paths + } + + if (fs.existsSync(logFilePath)) { + archive.file(logFilePath, { name: logFileName }) + filesAdded++ + } + } + + // Check if any files were actually added + if (filesAdded === 0) { + archive.abort() + return NextResponse.json( + { error: "No log files found - they may have been cleared from disk" }, + { status: 404 }, + ) + } + + // Finalize the archive + await archive.finalize() + + // Wait for all data to be collected + await archiveEndPromise + + // Check for archive errors + if (archiveError) { + throw archiveError + } + + // Combine all chunks into a single buffer + const zipBuffer = Buffer.concat(chunks) + + // Return the zip file + return new NextResponse(zipBuffer, { + status: 200, + headers: { + "Content-Type": "application/zip", + "Content-Disposition": `attachment; filename="run-${runId}-failed-logs.zip"`, + "Content-Length": String(zipBuffer.length), + }, + }) + } catch (error) { + console.error("Error exporting failed logs:", error) + + if (error instanceof Error && error.name === "RecordNotFoundError") { + return NextResponse.json({ error: "Run not found" }, { status: 404 }) + } + + return NextResponse.json({ error: "Failed to export logs" }, { status: 500 }) + } +} diff --git a/apps/web-evals/src/app/runs/[id]/page.tsx b/apps/web-evals/src/app/runs/[id]/page.tsx index aae3fc70f9..8b993eec8a 100644 --- a/apps/web-evals/src/app/runs/[id]/page.tsx +++ b/apps/web-evals/src/app/runs/[id]/page.tsx @@ -7,7 +7,7 @@ export default async function Page({ params }: { params: Promise<{ id: string }> const run = await findRun(Number(id)) return ( -

+
) diff --git a/apps/web-evals/src/app/runs/[id]/run-status.tsx b/apps/web-evals/src/app/runs/[id]/run-status.tsx index 4b94ef14fa..e05b1b51eb 100644 --- a/apps/web-evals/src/app/runs/[id]/run-status.tsx +++ b/apps/web-evals/src/app/runs/[id]/run-status.tsx @@ -1,55 +1,79 @@ "use client" +import { Link2, Link2Off, CheckCircle2 } from "lucide-react" import type { RunStatus as _RunStatus } from "@/hooks/use-run-status" import { cn } from "@/lib/utils" +import { Tooltip, TooltipContent, TooltipTrigger } from "@/components/ui" -export const RunStatus = ({ runStatus: { sseStatus, heartbeat, runners = [] } }: { runStatus: _RunStatus }) => ( -
-
-
-
Task Stream:
-
{sseStatus}
-
-
-
-
-
-
-
-
-
Task Controller:
-
{heartbeat ?? "dead"}
-
-
-
-
-
-
-
-
Task Runners:
- {runners.length > 0 &&
{runners?.join(", ")}
} -
-
-) +function StreamIcon({ status }: { status: "connected" | "waiting" | "error" }) { + if (status === "connected") { + return + } + return +} + +export const RunStatus = ({ + runStatus: { sseStatus, heartbeat, runners = [] }, + isComplete = false, +}: { + runStatus: _RunStatus + isComplete?: boolean +}) => { + // For completed runs, show a simple "Complete" badge + if (isComplete) { + return ( + + +
+ +
+
+ + Run complete + +
+ ) + } + + return ( + + +
+ {/* Task Stream status icon */} + + + {/* Task Controller ID */} + {heartbeat ?? "-"} + + {/* Task Runners count */} + 0 ? "text-green-500" : "text-rose-500"}> + {runners.length > 0 ? `${runners.length}r` : "0r"} + +
+
+ +
+
+ + Task Stream: {sseStatus} +
+
+ + Task Controller: {heartbeat ?? "dead"} +
+
+ 0 ? "text-green-500" : "text-rose-500"}>● + Task Runners: {runners.length > 0 ? runners.length : "none"} +
+ {runners.length > 0 && ( +
+ {runners.map((runner) => ( +
{runner}
+ ))} +
+ )} +
+
+
+ ) +} diff --git a/apps/web-evals/src/app/runs/[id]/run.tsx b/apps/web-evals/src/app/runs/[id]/run.tsx index b6c5290b13..a4b3910024 100644 --- a/apps/web-evals/src/app/runs/[id]/run.tsx +++ b/apps/web-evals/src/app/runs/[id]/run.tsx @@ -1,22 +1,339 @@ "use client" -import { useMemo } from "react" -import { LoaderCircle } from "lucide-react" +import { useMemo, useState, useCallback, useEffect } from "react" +import { toast } from "sonner" +import { LoaderCircle, FileText, Copy, Check, StopCircle } from "lucide-react" -import type { Run, TaskMetrics as _TaskMetrics } from "@roo-code/evals" +import type { Run, TaskMetrics as _TaskMetrics, Task } from "@roo-code/evals" +import type { ToolName } from "@roo-code/types" -import { formatCurrency, formatDuration, formatTokens } from "@/lib/formatters" +import { formatCurrency, formatDuration, formatTokens, formatToolUsageSuccessRate } from "@/lib/formatters" import { useRunStatus } from "@/hooks/use-run-status" -import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui" +import { killRun } from "@/actions/runs" +import { + Table, + TableBody, + TableCell, + TableHead, + TableHeader, + TableRow, + Tooltip, + TooltipContent, + TooltipTrigger, + Dialog, + DialogContent, + DialogHeader, + DialogTitle, + ScrollArea, + Button, + AlertDialog, + AlertDialogAction, + AlertDialogCancel, + AlertDialogContent, + AlertDialogDescription, + AlertDialogFooter, + AlertDialogHeader, + AlertDialogTitle, +} from "@/components/ui" import { TaskStatus } from "./task-status" import { RunStatus } from "./run-status" type TaskMetrics = Pick<_TaskMetrics, "tokensIn" | "tokensOut" | "tokensContext" | "duration" | "cost"> +type ToolUsageEntry = { attempts: number; failures: number } +type ToolUsage = Record + +// Generate abbreviation from tool name (e.g., "read_file" -> "RF", "list_code_definition_names" -> "LCDN") +function getToolAbbreviation(toolName: string): string { + return toolName + .split("_") + .map((word) => word[0]?.toUpperCase() ?? "") + .join("") +} + +// Pattern definitions for syntax highlighting +type HighlightPattern = { + pattern: RegExp + className: string + // If true, wraps the entire match; if a number, wraps that capture group + wrapGroup?: number +} + +const HIGHLIGHT_PATTERNS: HighlightPattern[] = [ + // Log levels - styled as badges + { pattern: /\|\s*(INFO)\s*\|/g, className: "text-green-400", wrapGroup: 1 }, + { pattern: /\|\s*(WARN|WARNING)\s*\|/g, className: "text-yellow-400", wrapGroup: 1 }, + { pattern: /\|\s*(ERROR)\s*\|/g, className: "text-red-400 font-semibold", wrapGroup: 1 }, + { pattern: /\|\s*(DEBUG)\s*\|/g, className: "text-gray-400", wrapGroup: 1 }, + // Task identifiers - important events + { + pattern: /(taskCreated|taskFocused|taskStarted|taskCompleted|taskAborted|taskResumable)/g, + className: "text-purple-400 font-medium", + }, + // Tool failures - highlight in red + { pattern: /(taskToolFailed)/g, className: "text-red-400 font-bold" }, + { pattern: /(Tool execution failed|tool.*failed|failed.*tool)/gi, className: "text-red-400" }, + { pattern: /(EvalPass)/g, className: "text-green-400 font-bold" }, + { pattern: /(EvalFail)/g, className: "text-red-400 font-bold" }, + // Message arrows + { pattern: /→/g, className: "text-cyan-400" }, + // Tool names in quotes + { pattern: /"(tool)":\s*"([^"]+)"/g, className: "text-orange-400" }, + // JSON keys + { pattern: /"([^"]+)":/g, className: "text-sky-300" }, + // Boolean values + { pattern: /:\s*(true|false)/g, className: "text-amber-400", wrapGroup: 1 }, + // Numbers + { pattern: /:\s*(-?\d+\.?\d*)/g, className: "text-emerald-400", wrapGroup: 1 }, +] + +// Extract timestamp from a log line and return elapsed time from baseline +function formatElapsedTime(timestamp: string, baselineMs: number): string { + const currentMs = new Date(timestamp).getTime() + const elapsedMs = currentMs - baselineMs + const totalSeconds = Math.floor(elapsedMs / 1000) + const minutes = Math.floor(totalSeconds / 60) + const seconds = totalSeconds % 60 + return `${minutes.toString().padStart(2, "0")}:${seconds.toString().padStart(2, "0")}` +} + +// Extract the first timestamp from the log to use as baseline +function extractFirstTimestamp(log: string): number | null { + // Match timestamp at start of line: [2025-11-28T09:35:23.187Z | ... or [2025-11-28T09:35:23.187Z] + const match = log.match(/\[(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z)[\s|\]]/) + const isoString = match?.[1] + if (!isoString) return null + return new Date(isoString).getTime() +} + +// Simplify log line by removing redundant metadata +function simplifyLogLine(line: string, baselineMs: number | null): { timestamp: string; simplified: string } { + // Extract timestamp - matches [2025-11-28T09:35:23.187Z | ... format + const timestampMatch = line.match(/\[(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z)[\s|\]]/) + const isoTimestamp = timestampMatch?.[1] + if (!isoTimestamp) { + return { timestamp: "", simplified: line } + } + + const timestamp = baselineMs !== null ? formatElapsedTime(isoTimestamp, baselineMs) : isoTimestamp.slice(11, 19) + + // Remove the timestamp from the line (handles both [timestamp] and [timestamp | formats) + let simplified = line.replace(/\[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z\s*\|?\s*/, "") + + // Remove redundant metadata: pid, run, task IDs (they're same for entire log) + simplified = simplified.replace(/\|\s*pid:\d+\s*/g, "") + simplified = simplified.replace(/\|\s*run:\d+\s*/g, "") + simplified = simplified.replace(/\|\s*task:\d+\s*/g, "") + simplified = simplified.replace(/runTask\s*\|\s*/g, "") + + // Clean up extra pipes, spaces, and trailing brackets + simplified = simplified.replace(/\|\s*\|/g, "|") + simplified = simplified.replace(/^\s*\|\s*/, "") + simplified = simplified.replace(/\]\s*$/, "") // Remove trailing bracket if present + + return { timestamp, simplified } +} + +// Format a single line with syntax highlighting using React elements (XSS-safe) +function formatLine(line: string): React.ReactNode[] { + // Find all matches with their positions + type Match = { start: number; end: number; text: string; className: string } + const matches: Match[] = [] + + for (const { pattern, className, wrapGroup } of HIGHLIGHT_PATTERNS) { + // Reset regex state + pattern.lastIndex = 0 + let regexMatch + while ((regexMatch = pattern.exec(line)) !== null) { + const capturedText = wrapGroup !== undefined ? regexMatch[wrapGroup] : regexMatch[0] + // Skip if capture group didn't match + if (!capturedText) continue + const start = + wrapGroup !== undefined ? regexMatch.index + regexMatch[0].indexOf(capturedText) : regexMatch.index + matches.push({ + start, + end: start + capturedText.length, + text: capturedText, + className, + }) + } + } + + // Sort matches by position and filter overlapping ones + matches.sort((a, b) => a.start - b.start) + const filteredMatches: Match[] = [] + for (const m of matches) { + const lastMatch = filteredMatches[filteredMatches.length - 1] + if (!lastMatch || m.start >= lastMatch.end) { + filteredMatches.push(m) + } + } + + // Build result with highlighted spans + const result: React.ReactNode[] = [] + let currentPos = 0 + + for (const [i, m] of filteredMatches.entries()) { + // Add text before this match + if (m.start > currentPos) { + result.push(line.slice(currentPos, m.start)) + } + // Add highlighted match + result.push( + + {m.text} + , + ) + currentPos = m.end + } + + // Add remaining text + if (currentPos < line.length) { + result.push(line.slice(currentPos)) + } + + return result.length > 0 ? result : [line] +} + +// Determine the visual style for a log line based on its content +function getLineStyle(line: string): string { + if (line.includes("ERROR")) return "bg-red-950/30 border-l-2 border-red-500" + if (line.includes("WARN") || line.includes("WARNING")) return "bg-yellow-950/20 border-l-2 border-yellow-500" + if (line.includes("taskToolFailed")) return "bg-red-950/30 border-l-2 border-red-500" + if (line.includes("taskStarted") || line.includes("taskCreated")) return "bg-purple-950/20" + if (line.includes("EvalPass")) return "bg-green-950/30 border-l-2 border-green-500" + if (line.includes("EvalFail")) return "bg-red-950/30 border-l-2 border-red-500" + if (line.includes("taskCompleted") || line.includes("taskAborted")) return "bg-blue-950/20" + return "" +} + +// Format log content with basic highlighting (XSS-safe - no dangerouslySetInnerHTML) +function formatLogContent(log: string): React.ReactNode[] { + const lines = log.split("\n") + const baselineMs = extractFirstTimestamp(log) + + return lines.map((line, index) => { + if (!line.trim()) { + return ( +
+ {" "} +
+ ) + } + + const parsed = simplifyLogLine(line, baselineMs) + const lineStyle = getLineStyle(line) + + return ( +
+ {/* Elapsed time */} + + {parsed.timestamp} + + {/* Log content - pl-12 ensures wrapped lines are indented under the timestamp */} + + {formatLine(parsed.simplified)} + +
+ ) + }) +} + export function Run({ run }: { run: Run }) { const runStatus = useRunStatus(run) - const { tasks, tokenUsage, usageUpdatedAt } = runStatus + const { tasks, tokenUsage, usageUpdatedAt, heartbeat, runners } = runStatus + + const [selectedTask, setSelectedTask] = useState(null) + const [taskLog, setTaskLog] = useState(null) + const [isLoadingLog, setIsLoadingLog] = useState(false) + const [copied, setCopied] = useState(false) + const [showKillDialog, setShowKillDialog] = useState(false) + const [isKilling, setIsKilling] = useState(false) + + // Determine if run is still active (has heartbeat or runners) + const isRunActive = !run.taskMetricsId && (!!heartbeat || (runners && runners.length > 0)) + + const onKillRun = useCallback(async () => { + setIsKilling(true) + try { + const result = await killRun(run.id) + if (result.killedContainers.length > 0) { + toast.success(`Killed ${result.killedContainers.length} container(s)`) + } else if (result.errors.length === 0) { + toast.info("No running containers found") + } else { + toast.error(result.errors.join(", ")) + } + } catch (error) { + console.error("Failed to kill run:", error) + toast.error("Failed to kill run") + } finally { + setIsKilling(false) + setShowKillDialog(false) + } + }, [run.id]) + + const onCopyLog = useCallback(async () => { + if (!taskLog) return + + try { + await navigator.clipboard.writeText(taskLog) + setCopied(true) + toast.success("Log copied to clipboard") + setTimeout(() => setCopied(false), 2000) + } catch (error) { + console.error("Failed to copy log:", error) + toast.error("Failed to copy log") + } + }, [taskLog]) + + // Handle ESC key to close the dialog + useEffect(() => { + const handleKeyDown = (e: KeyboardEvent) => { + if (e.key === "Escape" && selectedTask) { + setSelectedTask(null) + } + } + + document.addEventListener("keydown", handleKeyDown) + return () => document.removeEventListener("keydown", handleKeyDown) + }, [selectedTask]) + + const onViewTaskLog = useCallback( + async (task: Task) => { + // Only allow viewing logs for tasks that have started + if (!task.startedAt && !tokenUsage.get(task.id)) { + toast.error("Task has not started yet") + return + } + + setSelectedTask(task) + setIsLoadingLog(true) + setTaskLog(null) + + try { + const response = await fetch(`/api/runs/${run.id}/logs/${task.id}`) + + if (!response.ok) { + const error = await response.json() + toast.error(error.error || "Failed to load log") + setSelectedTask(null) + return + } + + const data = await response.json() + setTaskLog(data.logContent) + } catch (error) { + console.error("Error loading task log:", error) + toast.error("Failed to load log") + setSelectedTask(null) + } finally { + setIsLoadingLog(false) + } + }, + [run.id, tokenUsage], + ) const taskMetrics: Record = useMemo(() => { const metrics: Record = {} @@ -41,16 +358,239 @@ export function Run({ run }: { run: Run }) { // eslint-disable-next-line react-hooks/exhaustive-deps }, [tasks, tokenUsage, usageUpdatedAt]) + // Collect all unique tool names from all tasks and sort by total attempts + const toolColumns = useMemo(() => { + if (!tasks) return [] + + const toolTotals = new Map() + + for (const task of tasks) { + if (task.taskMetrics?.toolUsage) { + for (const [toolName, usage] of Object.entries(task.taskMetrics.toolUsage)) { + const tool = toolName as ToolName + const current = toolTotals.get(tool) ?? 0 + toolTotals.set(tool, current + usage.attempts) + } + } + } + + // Sort by total attempts descending + return Array.from(toolTotals.entries()) + .sort((a, b) => b[1] - a[1]) + .map(([name]): ToolName => name) + }, [tasks]) + + // Compute aggregate stats + const stats = useMemo(() => { + if (!tasks) return null + + const passed = tasks.filter((t) => t.passed === true).length + const failed = tasks.filter((t) => t.passed === false).length + const completed = passed + failed + + let totalTokensIn = 0 + let totalTokensOut = 0 + let totalCost = 0 + let totalDuration = 0 + + // Aggregate tool usage from completed tasks + const toolUsage: ToolUsage = {} + + for (const task of tasks) { + const metrics = taskMetrics[task.id] + if (metrics) { + totalTokensIn += metrics.tokensIn + totalTokensOut += metrics.tokensOut + totalCost += metrics.cost + totalDuration += metrics.duration + } + + // Aggregate tool usage from finished tasks with taskMetrics + if (task.finishedAt && task.taskMetrics?.toolUsage) { + for (const [key, usage] of Object.entries(task.taskMetrics.toolUsage)) { + const tool = key as keyof ToolUsage + if (!toolUsage[tool]) { + toolUsage[tool] = { attempts: 0, failures: 0 } + } + toolUsage[tool].attempts += usage.attempts + toolUsage[tool].failures += usage.failures + } + } + } + + return { + passed, + failed, + completed, + passRate: completed > 0 ? ((passed / completed) * 100).toFixed(1) : null, + totalTokensIn, + totalTokensOut, + totalCost, + totalDuration, + toolUsage, + } + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [tasks, taskMetrics, tokenUsage, usageUpdatedAt]) + + // Calculate elapsed time (wall-clock time from run creation to completion or now) + const elapsedTime = useMemo(() => { + if (!tasks || tasks.length === 0) return null + + const startTime = new Date(run.createdAt).getTime() + + // If run is complete, find the latest finishedAt from tasks + if (run.taskMetricsId) { + const latestFinish = tasks.reduce((latest, task) => { + if (task.finishedAt) { + const finishTime = new Date(task.finishedAt).getTime() + return finishTime > latest ? finishTime : latest + } + return latest + }, startTime) + return latestFinish - startTime + } + + // If still running, use current time + return Date.now() - startTime + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [tasks, run.createdAt, run.taskMetricsId, usageUpdatedAt]) + return ( <>
-
-
-
{run.model}
- {run.description &&
{run.description}
} + {stats && ( +
+ {/* Provider, Model title and status */} +
+ {run.settings?.apiProvider && ( + {run.settings.apiProvider} + )} +
{run.model}
+ + {run.description && ( + - {run.description} + )} + {isRunActive && ( + + + + + Stop all containers for this run + + )} +
+ {/* Main Stats Row */} +
+ {/* Passed/Failed */} +
+
+ {stats.passed} + / + {stats.failed} +
+
Passed / Failed
+
+ + {/* Pass Rate */} +
+
= 80 + ? "text-yellow-500" + : "text-red-500" + }`}> + {stats.passRate ? `${stats.passRate}%` : "-"} +
+
Pass Rate
+
+ + {/* Tokens */} +
+
+ {formatTokens(stats.totalTokensIn)} + / + {formatTokens(stats.totalTokensOut)} +
+
Tokens In / Out
+
+ + {/* Cost */} +
+
{formatCurrency(stats.totalCost)}
+
Cost
+
+ + {/* Duration */} +
+
+ {stats.totalDuration > 0 ? formatDuration(stats.totalDuration) : "-"} +
+
Duration
+
+ + {/* Elapsed Time */} +
+
+ {elapsedTime !== null ? formatDuration(elapsedTime) : "-"} +
+
Elapsed
+
+
+ + {/* Tool Usage Row */} + {Object.keys(stats.toolUsage).length > 0 && ( +
+ {Object.entries(stats.toolUsage) + .sort(([, a], [, b]) => b.attempts - a.attempts) + .map(([toolName, usage]) => { + const abbr = getToolAbbreviation(toolName) + const successRate = + usage.attempts > 0 + ? ((usage.attempts - usage.failures) / usage.attempts) * 100 + : 100 + const rateColor = + successRate === 100 + ? "text-green-500" + : successRate >= 80 + ? "text-yellow-500" + : "text-red-500" + return ( + + +
+ + {abbr} + + {usage.attempts} + + {formatToolUsageSuccessRate(usage)} + +
+
+ {toolName} +
+ ) + })} +
+ )}
- {!run.taskMetricsId && } -
+ )} {!tasks ? ( ) : ( @@ -60,53 +600,206 @@ export function Run({ run }: { run: Run }) { Exercise Tokens In / Out Context + {toolColumns.map((toolName) => ( + + + {getToolAbbreviation(toolName)} + {toolName} + + + ))} Duration Cost - {tasks.map((task) => ( - - -
- -
- {task.language}/{task.exercise} -
-
-
- {taskMetrics[task.id] ? ( - <> - -
-
{formatTokens(taskMetrics[task.id]!.tokensIn)}
/ -
{formatTokens(taskMetrics[task.id]!.tokensOut)}
+ {tasks.map((task) => { + const hasStarted = !!task.startedAt || !!tokenUsage.get(task.id) + return ( + hasStarted && onViewTaskLog(task)}> + +
+ +
+ + {task.language}/{task.exercise} + {task.iteration > 1 && ( + + (#{task.iteration}) + + )} + + {hasStarted && ( + + + + + Click to view log + + )}
- - - {formatTokens(taskMetrics[task.id]!.tokensContext)} - - - {taskMetrics[task.id]!.duration - ? formatDuration(taskMetrics[task.id]!.duration) - : "-"} - - - {formatCurrency(taskMetrics[task.id]!.cost)} - - - ) : ( - - )} - - ))} +
+
+ {taskMetrics[task.id] ? ( + <> + +
+
{formatTokens(taskMetrics[task.id]!.tokensIn)}
/ +
{formatTokens(taskMetrics[task.id]!.tokensOut)}
+
+
+ + {formatTokens(taskMetrics[task.id]!.tokensContext)} + + {toolColumns.map((toolName) => { + const usage = task.taskMetrics?.toolUsage?.[toolName] + const successRate = + usage && usage.attempts > 0 + ? ((usage.attempts - usage.failures) / usage.attempts) * 100 + : 100 + const rateColor = + successRate === 100 + ? "text-muted-foreground" + : successRate >= 80 + ? "text-yellow-500" + : "text-red-500" + return ( + + {usage ? ( +
+ + {usage.attempts} + + + {formatToolUsageSuccessRate(usage)} + +
+ ) : ( + - + )} +
+ ) + })} + + {taskMetrics[task.id]!.duration + ? formatDuration(taskMetrics[task.id]!.duration) + : "-"} + + + {formatCurrency(taskMetrics[task.id]!.cost)} + + + ) : ( + + )} +
+ ) + })} )}
+ + {/* Task Log Dialog - Full Screen */} + setSelectedTask(null)}> + + +
+ + + {selectedTask?.language}/{selectedTask?.exercise} + {selectedTask?.iteration && selectedTask.iteration > 1 && ( + (#{selectedTask.iteration}) + )} + + ( + {selectedTask?.passed === true + ? "Passed" + : selectedTask?.passed === false + ? "Failed" + : "Running"} + ) + + + {taskLog && ( + + )} +
+
+
+ {isLoadingLog ? ( +
+ +
+ ) : taskLog ? ( + +
+ {formatLogContent(taskLog)} +
+
+ ) : ( +
+ Log file not available (may have been cleared) +
+ )} +
+
+
+ + {/* Kill Run Confirmation Dialog */} + + + + Kill Run? + + This will stop the controller and all task runner containers for this run. Any running tasks + will be terminated immediately. This action cannot be undone. + + + + Cancel + + {isKilling ? ( + <> + + Killing... + + ) : ( + "Kill Run" + )} + + + + ) } diff --git a/apps/web-evals/src/app/runs/new/new-run.tsx b/apps/web-evals/src/app/runs/new/new-run.tsx index 41d35f3c4c..561c3ceb27 100644 --- a/apps/web-evals/src/app/runs/new/new-run.tsx +++ b/apps/web-evals/src/app/runs/new/new-run.tsx @@ -1,39 +1,55 @@ "use client" -import { useCallback, useRef, useState } from "react" +import { useCallback, useEffect, useMemo, useState } from "react" import { useRouter } from "next/navigation" import { z } from "zod" import { useQuery } from "@tanstack/react-query" import { useForm, FormProvider } from "react-hook-form" import { zodResolver } from "@hookform/resolvers/zod" -import fuzzysort from "fuzzysort" import { toast } from "sonner" -import { X, Rocket, Check, ChevronsUpDown, SlidersHorizontal, CircleCheck } from "lucide-react" +import { X, Rocket, Check, ChevronsUpDown, SlidersHorizontal, Info } from "lucide-react" -import { globalSettingsSchema, providerSettingsSchema, EVALS_SETTINGS, getModelId } from "@roo-code/types" +import { + globalSettingsSchema, + providerSettingsSchema, + EVALS_SETTINGS, + getModelId, + type ProviderSettings, + type GlobalSettings, + type ReasoningEffort, +} from "@roo-code/types" import { createRun } from "@/actions/runs" import { getExercises } from "@/actions/exercises" + import { - createRunSchema, type CreateRun, - MODEL_DEFAULT, + createRunSchema, CONCURRENCY_MIN, CONCURRENCY_MAX, CONCURRENCY_DEFAULT, TIMEOUT_MIN, TIMEOUT_MAX, TIMEOUT_DEFAULT, + ITERATIONS_MIN, + ITERATIONS_MAX, + ITERATIONS_DEFAULT, } from "@/lib/schemas" import { cn } from "@/lib/utils" + import { useOpenRouterModels } from "@/hooks/use-open-router-models" +import { useRooCodeCloudModels } from "@/hooks/use-roo-code-cloud-models" + import { Button, + Checkbox, FormControl, + FormDescription, FormField, FormItem, FormLabel, FormMessage, + Input, Textarea, Tabs, TabsList, @@ -48,36 +64,66 @@ import { Popover, PopoverContent, PopoverTrigger, - ScrollArea, - ScrollBar, Slider, + Label, + Select, + SelectContent, + SelectItem, + SelectTrigger, + SelectValue, + Tooltip, + TooltipContent, + TooltipTrigger, } from "@/components/ui" import { SettingsDiff } from "./settings-diff" +type ImportedSettings = { + apiConfigs: Record + globalSettings: GlobalSettings + currentApiConfigName: string +} + export function NewRun() { const router = useRouter() - const [mode, setMode] = useState<"openrouter" | "settings">("openrouter") - const [modelSearchValue, setModelSearchValue] = useState("") + const [provider, setModelSource] = useState<"roo" | "openrouter" | "other">("other") const [modelPopoverOpen, setModelPopoverOpen] = useState(false) + const [useNativeToolProtocol, setUseNativeToolProtocol] = useState(true) + const [useMultipleNativeToolCalls, setUseMultipleNativeToolCalls] = useState(false) + const [reasoningEffort, setReasoningEffort] = useState("") + const [commandExecutionTimeout, setCommandExecutionTimeout] = useState(20) + const [terminalShellIntegrationTimeout, setTerminalShellIntegrationTimeout] = useState(30) // seconds - const modelSearchResultsRef = useRef>(new Map()) - const modelSearchValueRef = useRef("") + // State for imported settings with config selection + const [importedSettings, setImportedSettings] = useState(null) + const [selectedConfigName, setSelectedConfigName] = useState("") + const [configPopoverOpen, setConfigPopoverOpen] = useState(false) + + const openRouter = useOpenRouterModels() + const rooCodeCloud = useRooCodeCloudModels() + const models = provider === "openrouter" ? openRouter.data : rooCodeCloud.data + const searchValue = provider === "openrouter" ? openRouter.searchValue : rooCodeCloud.searchValue + const setSearchValue = provider === "openrouter" ? openRouter.setSearchValue : rooCodeCloud.setSearchValue + const onFilter = provider === "openrouter" ? openRouter.onFilter : rooCodeCloud.onFilter - const models = useOpenRouterModels() const exercises = useQuery({ queryKey: ["getExercises"], queryFn: () => getExercises() }) + // State for selected exercises (needed for language toggle buttons) + const [selectedExercises, setSelectedExercises] = useState([]) + const form = useForm({ resolver: zodResolver(createRunSchema), defaultValues: { - model: MODEL_DEFAULT, + model: "", description: "", suite: "full", exercises: [], settings: undefined, concurrency: CONCURRENCY_DEFAULT, timeout: TIMEOUT_DEFAULT, + iterations: ITERATIONS_DEFAULT, + jobToken: "", }, }) @@ -90,11 +136,169 @@ export function NewRun() { const [model, suite, settings] = watch(["model", "suite", "settings", "concurrency"]) + // Load settings from localStorage on mount + useEffect(() => { + const savedConcurrency = localStorage.getItem("evals-concurrency") + if (savedConcurrency) { + const parsed = parseInt(savedConcurrency, 10) + if (!isNaN(parsed) && parsed >= CONCURRENCY_MIN && parsed <= CONCURRENCY_MAX) { + setValue("concurrency", parsed) + } + } + const savedTimeout = localStorage.getItem("evals-timeout") + if (savedTimeout) { + const parsed = parseInt(savedTimeout, 10) + if (!isNaN(parsed) && parsed >= TIMEOUT_MIN && parsed <= TIMEOUT_MAX) { + setValue("timeout", parsed) + } + } + const savedCommandTimeout = localStorage.getItem("evals-command-execution-timeout") + if (savedCommandTimeout) { + const parsed = parseInt(savedCommandTimeout, 10) + if (!isNaN(parsed) && parsed >= 20 && parsed <= 60) { + setCommandExecutionTimeout(parsed) + } + } + const savedShellTimeout = localStorage.getItem("evals-shell-integration-timeout") + if (savedShellTimeout) { + const parsed = parseInt(savedShellTimeout, 10) + if (!isNaN(parsed) && parsed >= 30 && parsed <= 60) { + setTerminalShellIntegrationTimeout(parsed) + } + } + // Load saved exercises selection + const savedSuite = localStorage.getItem("evals-suite") + if (savedSuite === "partial") { + setValue("suite", "partial") + const savedExercises = localStorage.getItem("evals-exercises") + if (savedExercises) { + try { + const parsed = JSON.parse(savedExercises) as string[] + if (Array.isArray(parsed)) { + setSelectedExercises(parsed) + setValue("exercises", parsed) + } + } catch { + // Invalid JSON, ignore + } + } + } + }, [setValue]) + + // Extract unique languages from exercises + const languages = useMemo(() => { + if (!exercises.data) return [] + const langs = new Set() + for (const path of exercises.data) { + const lang = path.split("/")[0] + if (lang) langs.add(lang) + } + return Array.from(langs).sort() + }, [exercises.data]) + + // Get exercises for a specific language + const getExercisesForLanguage = useCallback( + (lang: string) => { + if (!exercises.data) return [] + return exercises.data.filter((path) => path.startsWith(`${lang}/`)) + }, + [exercises.data], + ) + + // Toggle all exercises for a language + const toggleLanguage = useCallback( + (lang: string) => { + const langExercises = getExercisesForLanguage(lang) + const allSelected = langExercises.every((ex) => selectedExercises.includes(ex)) + + let newSelected: string[] + if (allSelected) { + // Remove all exercises for this language + newSelected = selectedExercises.filter((ex) => !ex.startsWith(`${lang}/`)) + } else { + // Add all exercises for this language (avoiding duplicates) + const existing = new Set(selectedExercises) + for (const ex of langExercises) { + existing.add(ex) + } + newSelected = Array.from(existing) + } + + setSelectedExercises(newSelected) + setValue("exercises", newSelected) + localStorage.setItem("evals-exercises", JSON.stringify(newSelected)) + }, + [getExercisesForLanguage, selectedExercises, setValue], + ) + + // Check if all exercises for a language are selected + const isLanguageSelected = useCallback( + (lang: string) => { + const langExercises = getExercisesForLanguage(lang) + return langExercises.length > 0 && langExercises.every((ex) => selectedExercises.includes(ex)) + }, + [getExercisesForLanguage, selectedExercises], + ) + + // Check if some (but not all) exercises for a language are selected + const isLanguagePartiallySelected = useCallback( + (lang: string) => { + const langExercises = getExercisesForLanguage(lang) + const selectedCount = langExercises.filter((ex) => selectedExercises.includes(ex)).length + return selectedCount > 0 && selectedCount < langExercises.length + }, + [getExercisesForLanguage, selectedExercises], + ) + const onSubmit = useCallback( async (values: CreateRun) => { try { - if (mode === "openrouter") { - values.settings = { ...(values.settings || {}), openRouterModelId: model } + // Validate jobToken for Roo Code Cloud provider + if (provider === "roo" && !values.jobToken?.trim()) { + toast.error("Roo Code Cloud Token is required") + return + } + + // Build experiments settings + const experimentsSettings = useMultipleNativeToolCalls + ? { experiments: { multipleNativeToolCalls: true } } + : {} + + if (provider === "openrouter") { + values.settings = { + ...(values.settings || {}), + apiProvider: "openrouter", + openRouterModelId: model, + toolProtocol: useNativeToolProtocol ? "native" : "xml", + commandExecutionTimeout, + terminalShellIntegrationTimeout: terminalShellIntegrationTimeout * 1000, // Convert to ms + ...experimentsSettings, + } + } else if (provider === "roo") { + values.settings = { + ...(values.settings || {}), + apiProvider: "roo", + apiModelId: model, + toolProtocol: useNativeToolProtocol ? "native" : "xml", + commandExecutionTimeout, + terminalShellIntegrationTimeout: terminalShellIntegrationTimeout * 1000, // Convert to ms + ...experimentsSettings, + ...(reasoningEffort + ? { + enableReasoningEffort: true, + reasoningEffort: reasoningEffort as ReasoningEffort, + } + : {}), + } + } else if (provider === "other" && values.settings) { + // For imported settings, merge in experiments and tool protocol + values.settings = { + ...values.settings, + toolProtocol: useNativeToolProtocol ? "native" : "xml", + commandExecutionTimeout, + terminalShellIntegrationTimeout: terminalShellIntegrationTimeout * 1000, // Convert to ms + ...experimentsSettings, + } } const { id } = await createRun(values) @@ -103,28 +307,16 @@ export function NewRun() { toast.error(e instanceof Error ? e.message : "An unknown error occurred.") } }, - [mode, model, router], - ) - - const onFilterModels = useCallback( - (value: string, search: string) => { - if (modelSearchValueRef.current !== search) { - modelSearchValueRef.current = search - modelSearchResultsRef.current.clear() - - for (const { - obj: { id }, - score, - } of fuzzysort.go(search, models.data || [], { - key: "name", - })) { - modelSearchResultsRef.current.set(id, score) - } - } - - return modelSearchResultsRef.current.get(value) ?? 0 - }, - [models.data], + [ + provider, + model, + router, + useNativeToolProtocol, + useMultipleNativeToolCalls, + reasoningEffort, + commandExecutionTimeout, + terminalShellIntegrationTimeout, + ], ) const onSelectModel = useCallback( @@ -132,7 +324,7 @@ export function NewRun() { setValue("model", model) setModelPopoverOpen(false) }, - [setValue], + [setValue, setModelPopoverOpen], ) const onImportSettings = useCallback( @@ -156,11 +348,21 @@ export function NewRun() { }) .parse(JSON.parse(await file.text())) - const providerSettings = providerProfiles.apiConfigs[providerProfiles.currentApiConfigName] ?? {} + // Store all imported configs for user selection + setImportedSettings({ + apiConfigs: providerProfiles.apiConfigs, + globalSettings, + currentApiConfigName: providerProfiles.currentApiConfigName, + }) + // Default to the current config + const defaultConfigName = providerProfiles.currentApiConfigName + setSelectedConfigName(defaultConfigName) + + // Apply the default config + const providerSettings = providerProfiles.apiConfigs[defaultConfigName] ?? {} setValue("model", getModelId(providerSettings) ?? "") setValue("settings", { ...EVALS_SETTINGS, ...providerSettings, ...globalSettings }) - setMode("settings") event.target.value = "" } catch (e) { @@ -171,19 +373,155 @@ export function NewRun() { [clearErrors, setValue], ) + const onSelectConfig = useCallback( + (configName: string) => { + if (!importedSettings) { + return + } + + setSelectedConfigName(configName) + setConfigPopoverOpen(false) + + const providerSettings = importedSettings.apiConfigs[configName] ?? {} + setValue("model", getModelId(providerSettings) ?? "") + setValue("settings", { ...EVALS_SETTINGS, ...providerSettings, ...importedSettings.globalSettings }) + }, + [importedSettings, setValue], + ) + return ( <>
-
- {mode === "openrouter" && ( - ( - + ( + + setModelSource(value as "roo" | "openrouter" | "other")}> + + Import + Roo Code Cloud + OpenRouter + + + + {provider === "other" ? ( +
+ + + + {importedSettings && Object.keys(importedSettings.apiConfigs).length > 1 && ( +
+ + + + + + + + + + No config found. + + {Object.keys(importedSettings.apiConfigs).map( + (configName) => ( + + {configName} + {configName === + importedSettings.currentApiConfigName && ( + + (default) + + )} + + + ), + )} + + + + + +
+ )} + +
+ +
+ + +
+
+ + {settings && ( + + )} +
+ ) : ( + <> - + No model found. - {models.data?.map(({ id, name }) => ( + {models?.map(({ id, name }) => ( - -
- )} - /> - )} - - - - {settings && ( - - <> -
- -
- Imported valid Roo Code settings. Showing differences from default - settings. +
+
+ +
+ + +
+ + {provider === "roo" && ( +
+ + +

+ When set, enableReasoningEffort will be automatically enabled +

+
+ )}
- - - + )} + + + + )} + /> + + {provider === "roo" && ( + ( + +
+ Roo Code Cloud Token + + + + + +

+ If you have access to the Roo Code Cloud repository and the + decryption key for the .env.* files, generate a token with: +

+ + pnpm --filter @roo-code-cloud/auth production:create-auth-token + [email] [org] [ttl] + +
+
+
+ + + + +
)} - - -
+ /> + )} ( Exercises - setValue("suite", value as "full" | "partial")}> - - All - Some - - +
+ { + setValue("suite", value as "full" | "partial") + localStorage.setItem("evals-suite", value) + if (value === "full") { + setSelectedExercises([]) + setValue("exercises", []) + localStorage.removeItem("evals-exercises") + } + }}> + + All + Some + + + {suite === "partial" && languages.length > 0 && ( +
+ {languages.map((lang) => ( + + ))} +
+ )} +
{suite === "partial" && ( ({ value: path, label: path })) || []} - onValueChange={(value) => setValue("exercises", value)} + value={selectedExercises} + onValueChange={(value) => { + setSelectedExercises(value) + setValue("exercises", value) + localStorage.setItem("evals-exercises", JSON.stringify(value)) + }} placeholder="Select" variant="inverted" maxCount={4} @@ -306,11 +741,14 @@ export function NewRun() {
field.onChange(value[0])} + onValueChange={(value) => { + field.onChange(value[0]) + localStorage.setItem("evals-concurrency", String(value[0])) + }} />
{field.value}
@@ -329,11 +767,14 @@ export function NewRun() {
field.onChange(value[0])} + onValueChange={(value) => { + field.onChange(value[0]) + localStorage.setItem("evals-timeout", String(value[0])) + }} />
{field.value}
@@ -343,6 +784,96 @@ export function NewRun() { )} /> + ( + + Iterations per Exercise + +
+ { + field.onChange(value[0]) + }} + /> +
{field.value}
+
+
+ Run each exercise multiple times to compare results + +
+ )} + /> + + +
+ + + + + + +

+ Maximum time in seconds to wait for terminal command execution to complete + before timing out. This applies to commands run via the execute_command tool. +

+
+
+
+
+ { + if (value !== undefined) { + setCommandExecutionTimeout(value) + localStorage.setItem("evals-command-execution-timeout", String(value)) + } + }} + /> +
{commandExecutionTimeout}
+
+
+ + +
+ + + + + + +

+ Maximum time in seconds to wait for shell integration to initialize when opening + a new terminal. +

+
+
+
+
+ { + if (value !== undefined) { + setTerminalShellIntegrationTimeout(value) + localStorage.setItem("evals-shell-integration-timeout", String(value)) + } + }} + /> +
{terminalShellIntegrationTimeout}
+
+
+ [] +export const ROO_CODE_SETTINGS_KEYS = [ + ...new Set([...GLOBAL_SETTINGS_KEYS, ...PROVIDER_SETTINGS_KEYS]), +] as Keys[] -type SettingsDiffProps = HTMLAttributes & { +type SettingsDiffProps = { defaultSettings: RooCodeSettings customSettings: RooCodeSettings } @@ -14,53 +14,45 @@ type SettingsDiffProps = HTMLAttributes & { export function SettingsDiff({ customSettings: { experiments: customExperiments, ...customSettings }, defaultSettings: { experiments: defaultExperiments, ...defaultSettings }, - className, - ...props }: SettingsDiffProps) { const defaults = { ...defaultSettings, ...defaultExperiments } const custom = { ...customSettings, ...customExperiments } return ( -
-
Setting
-
Default
-
Custom
- {ROO_CODE_SETTINGS_KEYS.map((key) => { - const defaultValue = defaults[key as keyof typeof defaults] - const customValue = custom[key as keyof typeof custom] - const isDefault = JSON.stringify(defaultValue) === JSON.stringify(customValue) +
+ + + + Setting + Default + Custom + + + + {ROO_CODE_SETTINGS_KEYS.map((key) => { + const defaultValue = JSON.stringify(defaults[key as keyof typeof defaults], null, 2) + const customValue = JSON.stringify(custom[key as keyof typeof custom], null, 2) - return isDefault ? null : ( - - ) - })} + return defaultValue === customValue || + (isEmpty(defaultValue) && isEmpty(customValue)) ? null : ( + + + {key} + + + {defaultValue} + + + {customValue} + + + ) + })} + +
) } -type SettingDiffProps = HTMLAttributes & { - name: string - defaultValue?: string - customValue?: string -} - -export function SettingDiff({ name, defaultValue, customValue, ...props }: SettingDiffProps) { - return ( - -
- {name} -
-
-				{defaultValue}
-			
-
-				{customValue}
-			
-
- ) -} +const isEmpty = (value: string | undefined) => + value === undefined || value === "" || value === "null" || value === '""' || value === "[]" || value === "{}" diff --git a/apps/web-evals/src/components/home/run.tsx b/apps/web-evals/src/components/home/run.tsx index c35673885c..4abbfc67b6 100644 --- a/apps/web-evals/src/components/home/run.tsx +++ b/apps/web-evals/src/components/home/run.tsx @@ -1,11 +1,20 @@ import { useCallback, useState, useRef } from "react" import Link from "next/link" -import { Ellipsis, ClipboardList, Copy, Check, LoaderCircle, Trash } from "lucide-react" +import { useRouter } from "next/navigation" +import { toast } from "sonner" +import { Ellipsis, ClipboardList, Copy, Check, LoaderCircle, Trash, Settings, FileDown } from "lucide-react" import type { Run as EvalsRun, TaskMetrics as EvalsTaskMetrics } from "@roo-code/evals" +import type { ToolName } from "@roo-code/types" import { deleteRun } from "@/actions/runs" -import { formatCurrency, formatDuration, formatTokens, formatToolUsageSuccessRate } from "@/lib/formatters" +import { + formatCurrency, + formatDateTime, + formatDuration, + formatTokens, + formatToolUsageSuccessRate, +} from "@/lib/formatters" import { useCopyRun } from "@/hooks/use-copy-run" import { Button, @@ -23,18 +32,63 @@ import { AlertDialogFooter, AlertDialogHeader, AlertDialogTitle, + Dialog, + DialogContent, + DialogHeader, + DialogTitle, + ScrollArea, } from "@/components/ui" type RunProps = { run: EvalsRun taskMetrics: EvalsTaskMetrics | null + toolColumns: ToolName[] } -export function Run({ run, taskMetrics }: RunProps) { +export function Run({ run, taskMetrics, toolColumns }: RunProps) { + const router = useRouter() const [deleteRunId, setDeleteRunId] = useState() + const [showSettings, setShowSettings] = useState(false) + const [isExportingLogs, setIsExportingLogs] = useState(false) const continueRef = useRef(null) const { isPending, copyRun, copied } = useCopyRun(run.id) + const onExportFailedLogs = useCallback(async () => { + if (run.failed === 0) { + toast.error("No failed tasks to export") + return + } + + setIsExportingLogs(true) + try { + const response = await fetch(`/api/runs/${run.id}/logs/failed`) + + if (!response.ok) { + const error = await response.json() + toast.error(error.error || "Failed to export logs") + return + } + + // Download the zip file + const blob = await response.blob() + const url = window.URL.createObjectURL(blob) + const a = document.createElement("a") + a.href = url + a.download = `run-${run.id}-failed-logs.zip` + document.body.appendChild(a) + a.click() + window.URL.revokeObjectURL(url) + document.body.removeChild(a) + + toast.success("Failed logs exported successfully") + } catch (error) { + console.error("Error exporting logs:", error) + toast.error("Failed to export logs") + } finally { + setIsExportingLogs(false) + } + }, [run.id, run.failed]) + const onConfirmDelete = useCallback(async () => { if (!deleteRunId) { return @@ -48,40 +102,73 @@ export function Run({ run, taskMetrics }: RunProps) { } }, [deleteRunId]) + const handleRowClick = useCallback( + (e: React.MouseEvent) => { + // Don't navigate if clicking on the dropdown menu + if ((e.target as HTMLElement).closest("[data-dropdown-trigger]")) { + return + } + router.push(`/runs/${run.id}`) + }, + [router, run.id], + ) + return ( <> - - {run.model} + + {run.model} + {run.settings?.apiProvider ?? "-"} + + {formatDateTime(run.createdAt)} + {run.passed} {run.failed} - {run.passed + run.failed > 0 && ( - {((run.passed / (run.passed + run.failed)) * 100).toFixed(1)}% - )} + {run.passed + run.failed > 0 && + (() => { + const percent = (run.passed / (run.passed + run.failed)) * 100 + const colorClass = + percent === 100 ? "text-green-500" : percent >= 80 ? "text-yellow-500" : "text-red-500" + return {percent.toFixed(1)}% + })()} {taskMetrics && ( -
-
{formatTokens(taskMetrics.tokensIn)}
/ -
{formatTokens(taskMetrics.tokensOut)}
-
- )} -
- - {taskMetrics?.toolUsage?.apply_diff && ( -
-
{taskMetrics.toolUsage.apply_diff.attempts}
-
/
-
{formatToolUsageSuccessRate(taskMetrics.toolUsage.apply_diff)}
+
+ {formatTokens(taskMetrics.tokensIn)}/ + {formatTokens(taskMetrics.tokensOut)}
)} + {toolColumns.map((toolName) => { + const usage = taskMetrics?.toolUsage?.[toolName] + const successRate = + usage && usage.attempts > 0 ? ((usage.attempts - usage.failures) / usage.attempts) * 100 : 100 + const rateColor = + successRate === 100 + ? "text-muted-foreground" + : successRate >= 80 + ? "text-yellow-500" + : "text-red-500" + return ( + + {usage ? ( +
+ {usage.attempts} + {formatToolUsageSuccessRate(usage)} +
+ ) : ( + - + )} +
+ ) + })} {taskMetrics && formatCurrency(taskMetrics.cost)} {taskMetrics && formatDuration(taskMetrics.duration)} - + e.stopPropagation()}> @@ -94,6 +181,14 @@ export function Run({ run, taskMetrics }: RunProps) {
+ {run.settings && ( + setShowSettings(true)}> +
+ +
View Settings
+
+
+ )} {run.taskMetricsId && ( copyRun()} disabled={isPending || copied}>
@@ -116,6 +211,23 @@ export function Run({ run, taskMetrics }: RunProps) {
)} + {run.failed > 0 && ( + +
+ {isExportingLogs ? ( + <> + + Exporting... + + ) : ( + <> + + Export Failed Logs + + )} +
+
+ )} { setDeleteRunId(run.id) @@ -144,6 +256,18 @@ export function Run({ run, taskMetrics }: RunProps) { + + + + Run Settings + + +
+							{JSON.stringify(run.settings, null, 2)}
+						
+
+
+
) } diff --git a/apps/web-evals/src/components/home/runs.tsx b/apps/web-evals/src/components/home/runs.tsx index 8bc8739b28..283cc07ad2 100644 --- a/apps/web-evals/src/components/home/runs.tsx +++ b/apps/web-evals/src/components/home/runs.tsx @@ -1,40 +1,224 @@ "use client" +import { useMemo, useState } from "react" import { useRouter } from "next/navigation" -import { Rocket } from "lucide-react" +import { ArrowDown, ArrowUp, ArrowUpDown, Rocket } from "lucide-react" import type { Run, TaskMetrics } from "@roo-code/evals" +import type { ToolName } from "@roo-code/types" -import { Button, Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui" +import { + Button, + Table, + TableBody, + TableCell, + TableHead, + TableHeader, + TableRow, + Tooltip, + TooltipContent, + TooltipTrigger, +} from "@/components/ui" import { Run as Row } from "@/components/home/run" type RunWithTaskMetrics = Run & { taskMetrics: TaskMetrics | null } +type SortColumn = "model" | "provider" | "passed" | "failed" | "percent" | "cost" | "duration" | "createdAt" +type SortDirection = "asc" | "desc" + +// Generate abbreviation from tool name (e.g., "read_file" -> "RF", "list_code_definition_names" -> "LCDN") +function getToolAbbreviation(toolName: string): string { + return toolName + .split("_") + .map((word) => word[0]?.toUpperCase() ?? "") + .join("") +} + +function SortIcon({ + column, + sortColumn, + sortDirection, +}: { + column: SortColumn + sortColumn: SortColumn | null + sortDirection: SortDirection +}) { + if (sortColumn !== column) { + return + } + return sortDirection === "asc" ? : +} + export function Runs({ runs }: { runs: RunWithTaskMetrics[] }) { const router = useRouter() + const [sortColumn, setSortColumn] = useState("createdAt") + const [sortDirection, setSortDirection] = useState("desc") + + const handleSort = (column: SortColumn) => { + if (sortColumn === column) { + setSortDirection(sortDirection === "asc" ? "desc" : "asc") + } else { + setSortColumn(column) + setSortDirection("desc") + } + } + + // Collect all unique tool names from all runs and sort by total attempts + const toolColumns = useMemo(() => { + const toolTotals = new Map() + + for (const run of runs) { + if (run.taskMetrics?.toolUsage) { + for (const [toolName, usage] of Object.entries(run.taskMetrics.toolUsage)) { + const tool = toolName as ToolName + const current = toolTotals.get(tool) ?? 0 + toolTotals.set(tool, current + usage.attempts) + } + } + } + + // Sort by total attempts descending + return Array.from(toolTotals.entries()) + .sort((a, b) => b[1] - a[1]) + .map(([name]): ToolName => name) + }, [runs]) + + // Sort runs based on current sort column and direction + const sortedRuns = useMemo(() => { + if (!sortColumn) return runs + + return [...runs].sort((a, b) => { + let aVal: string | number | Date | null = null + let bVal: string | number | Date | null = null + + switch (sortColumn) { + case "model": + aVal = a.model + bVal = b.model + break + case "provider": + aVal = a.settings?.apiProvider ?? "" + bVal = b.settings?.apiProvider ?? "" + break + case "passed": + aVal = a.passed + bVal = b.passed + break + case "failed": + aVal = a.failed + bVal = b.failed + break + case "percent": + aVal = a.passed + a.failed > 0 ? a.passed / (a.passed + a.failed) : 0 + bVal = b.passed + b.failed > 0 ? b.passed / (b.passed + b.failed) : 0 + break + case "cost": + aVal = a.taskMetrics?.cost ?? 0 + bVal = b.taskMetrics?.cost ?? 0 + break + case "duration": + aVal = a.taskMetrics?.duration ?? 0 + bVal = b.taskMetrics?.duration ?? 0 + break + case "createdAt": + aVal = a.createdAt + bVal = b.createdAt + break + } + + if (aVal === null || bVal === null) return 0 + + let comparison = 0 + if (typeof aVal === "string" && typeof bVal === "string") { + comparison = aVal.localeCompare(bVal) + } else if (aVal instanceof Date && bVal instanceof Date) { + comparison = aVal.getTime() - bVal.getTime() + } else { + comparison = (aVal as number) - (bVal as number) + } + + return sortDirection === "asc" ? comparison : -comparison + }) + }, [runs, sortColumn, sortDirection]) + + // Calculate colSpan for empty state (7 base columns + dynamic tools + 3 end columns) + const totalColumns = 7 + toolColumns.length + 3 return ( <> - Model - Passed - Failed - % Correct - Tokens In / Out - Diff Edits - Cost - Duration - + handleSort("model")}> +
+ Model + +
+
+ handleSort("provider")}> +
+ Provider + +
+
+ handleSort("createdAt")}> +
+ Created + +
+
+ handleSort("passed")}> +
+ Passed + +
+
+ handleSort("failed")}> +
+ Failed + +
+
+ handleSort("percent")}> +
+ % + +
+
+ Tokens + {toolColumns.map((toolName) => ( + + + {getToolAbbreviation(toolName)} + {toolName} + + + ))} + handleSort("cost")}> +
+ Cost + +
+
+ handleSort("duration")}> +
+ Duration + +
+
+
- {runs.length ? ( - runs.map(({ taskMetrics, ...run }) => ) + {sortedRuns.length ? ( + sortedRuns.map(({ taskMetrics, ...run }) => ( + + )) ) : ( - + No eval runs yet. - - - +
+
+
+
+
+

+ Your AI Software Engineering Team is here. +
+ Interactive in the IDE, autonomous in the cloud. +

+
+

+ Use the Roo Code Extension on your computer for + full control, or delegate work to your{" "} + Roo Code Cloud Agents from the web, Slack, Github + or wherever your team is. +

+
+
+
+ + Free and Open Source
-
-
- -
+ +
+ + No credit card needed
+ +
+ +
-
- -
-
- -
-
- -
- + + + + + + + ) } diff --git a/apps/web-roo-code/src/app/pr-fixer/PrFixerContent.tsx b/apps/web-roo-code/src/app/pr-fixer/PrFixerContent.tsx deleted file mode 100644 index 285a28e6f6..0000000000 --- a/apps/web-roo-code/src/app/pr-fixer/PrFixerContent.tsx +++ /dev/null @@ -1,240 +0,0 @@ -"use client" - -import { ArrowRight, GitPullRequest, History, Key, MessageSquareCode, Wrench, type LucideIcon } from "lucide-react" -import Image from "next/image" -import Link from "next/link" - -import { Button } from "@/components/ui" -import { AnimatedBackground } from "@/components/homepage" -import { EXTERNAL_LINKS } from "@/lib/constants" -import { trackGoogleAdsConversion } from "@/lib/analytics/google-ads" - -// Workaround for next/image choking on these for some reason -import hero from "/public/heroes/agent-pr-fixer.png" - -interface Feature { - icon: LucideIcon - title: string - description: string | React.ReactNode - logos?: string[] -} - -const workflowSteps: Feature[] = [ - { - icon: GitPullRequest, - title: "1. Connect your GitHub repositories", - description: "Pick which repos the PR Fixer can work on by pushing to ongoing branches.", - }, - { - icon: MessageSquareCode, - title: "2. Invoke from a comment", - description: - 'Ask the agent to fix issues directly from GitHub PR comments (e.g. "@roomote: fix these review comments"). It’s fully aware of the entire comment history and latest diffs and focuses on fixing them – not random changes to your code.', - }, - { - icon: Wrench, - title: "3. Get clean scoped commits", - description: ( - <> - The agent proposes targeted changes and pushes concise commits or patch suggestions you (or{" "} - PR Reviewer) can review and merge quickly. - - ), - }, -] - -const howItWorks: Feature[] = [ - { - icon: History, - title: "Comment-history aware", - description: - "Understands the entire conversation on the PR – previous reviews, your replies, follow-ups – and uses that context to produce accurate fixes.", - }, - { - icon: Key, - title: "Bring your own key", - description: - "Use your preferred models at full strength. We optimize prompts and execution without capping your model to protect our margins.", - }, - { - icon: GitPullRequest, - title: "Repository- and diff-aware", - description: - "Analyzes the full repo context and the latest diff to ensure fixes align with project conventions and pass checks.", - }, -] - -export function PrFixerContent() { - return ( - <> -
- -
-
-
-
-

- - State-of-the-art fixes for the comments on your PRs. -

- -
-

- Roo Code{"'"}s PR Fixer applies high-quality changes to your PRs, right from - GitHub. Invoke via a PR comment and it will read the entire comment history to - understand context, agreements, and tradeoffs — then implement the right fix. -

-

- As always, you bring the model key; we orchestrate smart, efficient workflows. -

-
- - {/* Cross-agent link */} -
- Works great with - - - PR Reviewer Agent - - -
-
- -
- - - (cancel anytime) - -
-
- -
-
-
- -
-
-
-
-
-
- - {/* How It Works Section */} -
-
-
-
-

How It Works

-
-
- -
-
    - {workflowSteps.map((step, index) => { - const Icon = step.icon - return ( -
  • - -

    - {step.title} -

    -
    - {step.description} -
    -
  • - ) - })} -
-
-
-
- -
-
-
-
-

- Why Roo Code{"'"}s PR Fixer is different -

-
-
- -
-
    - {howItWorks.map((feature, index) => { - const Icon = feature.icon - return ( -
  • - -

    - {feature.title} -

    -
    - {feature.description} -
    -
  • - ) - })} -
-
-
-
- - {/* CTA Section */} -
-
-
-

- Ship fixes, not follow-ups. -

-

- Let Roo Code{"'"}s PR Fixer turn your review feedback into clean, ready-to-merge commits. -

- -
-
-
- - ) -} diff --git a/apps/web-roo-code/src/app/pr-fixer/content-a.tsx b/apps/web-roo-code/src/app/pr-fixer/content-a.tsx new file mode 100644 index 0000000000..1935ca2774 --- /dev/null +++ b/apps/web-roo-code/src/app/pr-fixer/content-a.tsx @@ -0,0 +1,95 @@ +import { type AgentPageContent } from "@/app/shared/agent-page-content" +import Link from "next/link" + +// Workaround for next/image choking on these for some reason +import hero from "/public/heroes/agent-pr-fixer.png" + +// Re-export for convenience +export type { AgentPageContent } + +export const content: AgentPageContent = { + agentName: "PR Fixer", + hero: { + icon: "Wrench", + heading: "State-of-the-art fixes for the comments on your PRs.", + paragraphs: [ + "Roo Code's PR Fixer applies high-quality changes to your PRs, right from GitHub. Invoke via a PR comment and it will read the entire comment history to understand context, agreements, and tradeoffs — then implement the right fix.", + "As always, you bring the model key; we orchestrate smart, efficient workflows.", + ], + image: { + url: hero.src, + width: 800, + height: 711, + alt: "Example of a PR Fixer applying changes from review comments", + }, + crossAgentLink: { + text: "Works great with", + links: [ + { + text: "PR Reviewer Agent", + href: "/reviewer", + icon: "GitPullRequest", + }, + ], + }, + cta: { + buttonText: "Try now for free", + disclaimer: "", + tracking: "&agent=pr-fixer", + }, + }, + howItWorks: { + heading: "How It Works", + steps: [ + { + title: "1. Connect your GitHub repositories", + description: "Pick which repos the PR Fixer can work on by pushing to ongoing branches.", + icon: "GitPullRequest", + }, + { + title: "2. Invoke from a comment", + description: + 'Ask the agent to fix issues directly from GitHub PR comments (e.g. "@roomote: fix these review comments"). It\'s fully aware of the entire comment history and latest diffs and focuses on fixing them – not random changes to your code.', + icon: "MessageSquareCode", + }, + { + title: "3. Get clean scoped commits", + description: ( + <> + The agent proposes targeted changes and pushes concise commits or patch suggestions you (or{" "} + PR Reviewer) can review and merge quickly. + + ), + icon: "Wrench", + }, + ], + }, + whyBetter: { + heading: "Why Roo Code's PR Fixer is different", + features: [ + { + title: "Comment-history aware", + description: + "Understands the entire conversation on the PR – previous reviews, your replies, follow-ups – and uses that context to produce accurate fixes.", + icon: "History", + }, + { + title: "Bring your own key", + description: + "Use your preferred models at full strength. We optimize prompts and execution without capping your model to protect our margins.", + icon: "Key", + }, + { + title: "Repository- and diff-aware", + description: + "Analyzes the full repo context and the latest diff to ensure fixes align with project conventions and pass checks.", + icon: "GitPullRequest", + }, + ], + }, + cta: { + heading: "Ship fixes, not follow-ups.", + description: "Let Roo Code's PR Fixer turn your review feedback into clean, ready-to-merge commits.", + buttonText: "Try now for free", + }, +} diff --git a/apps/web-roo-code/src/app/pr-fixer/page.tsx b/apps/web-roo-code/src/app/pr-fixer/page.tsx index 3d6e1f865d..f2317161e0 100644 --- a/apps/web-roo-code/src/app/pr-fixer/page.tsx +++ b/apps/web-roo-code/src/app/pr-fixer/page.tsx @@ -2,7 +2,9 @@ import type { Metadata } from "next" import { SEO } from "@/lib/seo" import { ogImageUrl } from "@/lib/og" -import { PrFixerContent } from "./PrFixerContent" +import { AgentLandingContent } from "@/app/shared/AgentLandingContent" +import { getContentVariant } from "@/app/shared/getContentVariant" +import { content as contentA } from "./content-a" const TITLE = "PR Fixer" const DESCRIPTION = @@ -55,6 +57,11 @@ export const metadata: Metadata = { ], } -export default function AgentPrFixerPage() { - return +export default async function AgentPrFixerPage({ searchParams }: { searchParams: Promise<{ v?: string }> }) { + const params = await searchParams + const content = getContentVariant(params, { + A: contentA, + }) + + return } diff --git a/apps/web-roo-code/src/app/pricing/page.tsx b/apps/web-roo-code/src/app/pricing/page.tsx index 9985881e1d..361bc3f788 100644 --- a/apps/web-roo-code/src/app/pricing/page.tsx +++ b/apps/web-roo-code/src/app/pricing/page.tsx @@ -1,17 +1,16 @@ -import { Users, Building2, ArrowRight, Star, LucideIcon, Check, Cloud } from "lucide-react" +import { Users, ArrowRight, LucideIcon, Check, SquareTerminal, CornerRightDown, Cloud } from "lucide-react" import type { Metadata } from "next" import Link from "next/link" import { Button } from "@/components/ui" import { AnimatedBackground } from "@/components/homepage" -import { ContactForm } from "@/components/enterprise/contact-form" import { SEO } from "@/lib/seo" import { ogImageUrl } from "@/lib/og" import { EXTERNAL_LINKS } from "@/lib/constants" -const TITLE = "Roo Code Cloud Pricing" +const TITLE = "Roo Code Pricing" const DESCRIPTION = - "Simple, transparent pricing for Roo Code Cloud. The VS Code extension is free forever. Choose the cloud plan that fits your needs." + "Simple, transparent pricing for all Roo Code products. The VS Code extension is free forever. Choose the cloud plan that fits your needs." const OG_DESCRIPTION = "" const PATH = "/pricing" @@ -61,72 +60,67 @@ interface PricingTier { name: string icon: LucideIcon price: string + priceSuffix: string period?: string creditPrice?: string trial?: string - cancellation?: string description: string featuresIntro?: string features: string[] cta: { text: string href?: string - isContactForm?: boolean } } const pricingTiers: PricingTier[] = [ + { + name: "VS Code Extension", + icon: SquareTerminal, + price: "Free", + priceSuffix: "inference", + description: "The best local coding agent", + features: ["Unlimited local use", "Bring your own model", "Powerful, extensible modes", "Community support"], + cta: { + text: "Install Now", + href: EXTERNAL_LINKS.MARKETPLACE, + }, + }, { name: "Cloud Free", icon: Cloud, price: "$0", - cancellation: "Cancel anytime", - description: "For folks just getting started", + period: "/mo", + priceSuffix: "credits", + creditPrice: `$${PRICE_CREDITS}`, + description: "For AI-forward engineers", + featuresIntro: "Go beyond the extension with", features: [ - "Token usage analytics", + "Access to Cloud Agents: fully autonomous development you can call from Slack, Github and the web", + "Access to the Roo Code Cloud Provider", "Follow your tasks from anywhere", "Share tasks with friends and co-workers", - "Early access to free AI Models", - "Community support", + "Token usage analytics", + "Professional support", ], cta: { - text: "Get started", + text: "Sign up", href: EXTERNAL_LINKS.CLOUD_APP_SIGNUP, }, }, { - name: "Pro", - icon: Star, - price: "$20", - period: "/mo", - trial: "Free 14-day trial · ", - creditPrice: `$${PRICE_CREDITS}`, - cancellation: "Cancel anytime", - description: "For pro Roo coders", - featuresIntro: "Everything in Free +", - features: [ - "Cloud Agents: PR Reviewer and more", - "Roomote Control: Start, stop and control tasks from anywhere", - "Paid support", - ], - cta: { - text: "Get started", - href: EXTERNAL_LINKS.CLOUD_APP_SIGNUP + "?redirect_url=/billing", - }, - }, - { - name: "Team", + name: "Cloud Team", icon: Users, price: "$99", + priceSuffix: "credits", period: "/mo", creditPrice: `$${PRICE_CREDITS}`, - trial: "Free 14-day trial · ", - cancellation: "Cancel anytime", + trial: "Free for 14 days, then", description: "For AI-forward teams", - featuresIntro: "Everything in Pro +", + featuresIntro: "Everything in Free +", features: ["Unlimited users (no per-seat cost)", "Shared configuration & policies", "Centralized billing"], cta: { - text: "Get started", + text: "Sign up", href: EXTERNAL_LINKS.CLOUD_APP_SIGNUP + "?redirect_url=/billing", }, }, @@ -138,52 +132,43 @@ export default function PricingPage() { {/* Hero Section */} -
+
-

Roo Code Cloud Pricing

-

- Simple, transparent pricing that scales with your needs. -
- No inference markups. Free 14-day trials to kick the tires. +

Roo Code Pricing

+

+ For all of our products: the Roo Code VS Code Extension, Roo Code Cloud and the Roo Code + Cloud inference Provider.

- {/* Free Extension Notice */} -
-
-

- The Roo Code extension is free! - Roo Code Cloud is an optional service which takes it to the next level. -

-
-
- {/* Pricing Tiers */}
-
+
{pricingTiers.map((tier) => { const Icon = tier.icon return (
+ className="relative group p-6 flex flex-col justify-start bg-background rounded-2xl outline outline-2 outline-border/50 hover:outline-8 transition-all shadow-xl hover:shadow-2xl hover:outline-6">

{tier.name}

-
-

{tier.description}

+

{tier.description}

+
+
+

{tier.featuresIntro} 

-
    +
      {tier.features.map((feature) => (
    • @@ -193,52 +178,67 @@ export default function PricingPage() {
-

- {tier.price} - {tier.period} +

{tier.trial}

+ +

+ {tier.price} + {tier.period} + {tier.priceSuffix} +

- {tier.creditPrice && ( -

- + {tier.creditPrice}/hour for Cloud tasks -

- )} - -

- {tier.trial} - {tier.cancellation} +

+ {tier.creditPrice && ( + <> + Cloud Agents: {tier.creditPrice}/hour in credits +
+ + )} + Inference:{" "} + + Roo Provider + {" "} + credits or{" "} + + BYOM +

- {tier.cta.isContactForm ? ( - - ) : ( - - )} + + + {/*
*/} +
) })}
-
-
-

- - Need SAML, advanced security, custom integrations or terms? Enterprise is for you. - - Talk to Sales - - . -

+
+
+

Roo Code Provider

+
+

+ On any plan, you can use your own LLM provider API key or use the built-in Roo Code + Cloud provider – curated models to work with Roo with no markup, including the + latest Gemini, GPT and Claude. Paid with credits. + + See per model pricing. + +

+
+
+
+

Credits

+

+ Credits are pre-paid, in dollars, and are deducted with usage for inference and Cloud + Agent runs. You're always in control of your spend, no surprises. +

+
+
@@ -249,7 +249,7 @@ export default function PricingPage() {

Frequently Asked Questions

-
+

Wait, is Roo Code free or not?

Yes! The Roo Code VS Code extension is open source and free forever. The extension acts @@ -257,7 +257,7 @@ export default function PricingPage() { Code Cloud.

-
+

Is there a free trial?

Yes, all paid plans come with a 14-day free trial to try out functionality. @@ -266,12 +266,25 @@ export default function PricingPage() { To use Cloud Agents, you can buy credits.

-
-

How do Cloud Agent credits work?

+
+

How do credits work?

- Cloud Agents are a version of Roo running in the cloud without depending on your IDE. - You can run as many as you want, and bring your own inference provider key. + Roo Code Cloud credits can be used in two ways:

+
    +
  • To pay for Cloud Agents running time (${PRICE_CREDITS}/hour)
  • +
  • + To pay for AI model inference costs ( + + varies by model + + ) +
  • +

To cover our infrastructure costs, we charge ${PRICE_CREDITS}/hour while the agent is running (independent of inference costs). @@ -280,25 +293,38 @@ export default function PricingPage() { There are no markups, no tiers, no dumbing-down of models to increase our profit.

-
+

Do I need a credit card for the free trial?

Yes, but you won't be charged until your trial ends, except for credit purchases.

You can cancel anytime with one click.

-
+

What payment methods do you accept?

We accept all major credit cards, debit cards, and can arrange invoice billing for Enterprise customers.

-
-

Can I change plans anytime?

+
+

Can I cancel or change plans?

- Yes, you can upgrade or downgrade your plan at any time. Changes will be reflected in - your next billing cycle. + Yes, you can upgrade, downgrade or cancel your plan at any time. Changes will be + reflected in your next billing cycle. +

+
+
+

+ What if I have enterprise-level needs like SAML/SCIM, large-scale deployments, specific + integrations and custom terms? +

+

+ We have an Enterprise plan which can be a fit. Please{" "} + + reach out to our sales team + {" "} + to discuss it.

diff --git a/apps/web-roo-code/src/app/provider/pricing/components/model-card.tsx b/apps/web-roo-code/src/app/provider/pricing/components/model-card.tsx new file mode 100644 index 0000000000..26f3545791 --- /dev/null +++ b/apps/web-roo-code/src/app/provider/pricing/components/model-card.tsx @@ -0,0 +1,190 @@ +import { ModelWithTotalPrice } from "@/lib/types/models" +import { formatCurrency, formatTokens } from "@/lib/formatters" +import { + ArrowLeftToLine, + ArrowRightToLine, + Building2, + Check, + Expand, + Gift, + HardDriveDownload, + HardDriveUpload, + RulerDimensionLine, + ChevronDown, + ChevronUp, +} from "lucide-react" +import { useState } from "react" + +interface ModelCardProps { + model: ModelWithTotalPrice +} + +export function ModelCard({ model }: ModelCardProps) { + // Prices are per token, multiply by 1M to get price per million tokens + const inputPrice = parseFloat(model.pricing.input) * 1_000_000 + const outputPrice = parseFloat(model.pricing.output) * 1_000_000 + const cacheReadPrice = parseFloat(model.pricing.input_cache_read || "0") * 1_000_000 + const cacheWritePrice = parseFloat(model.pricing.input_cache_write || "0") * 1_000_000 + + const free = model.tags.includes("free") + // Filter tags to only show vision and reasoning + const displayTags = model.tags.filter((tag) => tag === "vision" || tag === "reasoning") + + // Mobile collapsed/expanded state + const [expanded, setExpanded] = useState(false) + + return ( +
+ {/* Header: always visible */} +
+

+ {model.name} + {free && ( + + + Free! + + )} +

+

+ {model.description} +

+
+ + {/* Content - pinned to bottom */} +
+
+ + {/* Provider: always visible if present */} + {model.owned_by && ( + + + + + )} + + {/* Context Window: always visible */} + + + + + + {/* Max Output Tokens: always visible on >=sm, expandable on mobile */} + + + + + + {/* Input Price: always visible */} + + + + + + {/* Output Price: always visible */} + + + + + + {/* Cache pricing: only visible on mobile when expanded, always visible on >=sm */} + {cacheReadPrice > 0 && ( + + + + + )} + + {cacheWritePrice > 0 && ( + + + + + )} + + {/* Tags row: only show if there are vision or reasoning tags */} + {displayTags.length > 0 && ( + + + + + )} + + {/* Mobile-only toggle row */} + + + + +
+ + Provider + {model.owned_by}
+ + Context Window + {formatTokens(model.context_window)}
+ + Max Output Tokens + {formatTokens(model.max_tokens)}
+ + Input Price + + {inputPrice === 0 ? "Free" : `${formatCurrency(inputPrice)}/1M tokens`} +
+ + Output Price + + {outputPrice === 0 ? "Free" : `${formatCurrency(outputPrice)}/1M tokens`} +
+ + Cache Read + {formatCurrency(cacheReadPrice)}/1M tokens
+ + Cache Write + {formatCurrency(cacheWritePrice)}/1M tokens
Features + {displayTags.map((tag) => ( + + + {tag} + + ))} +
+ +
+
+
+ ) +} diff --git a/apps/web-roo-code/src/app/provider/pricing/page.tsx b/apps/web-roo-code/src/app/provider/pricing/page.tsx new file mode 100644 index 0000000000..4558355d2c --- /dev/null +++ b/apps/web-roo-code/src/app/provider/pricing/page.tsx @@ -0,0 +1,253 @@ +"use client" + +import { useEffect, useMemo, useState } from "react" +import { ModelCard } from "./components/model-card" +import { Model, ModelWithTotalPrice, ModelsResponse, SortOption } from "@/lib/types/models" +import Link from "next/link" +import { ChevronDown, CircleX, Loader, LoaderCircle, Search } from "lucide-react" + +const API_URL = "https://api.roocode.com/proxy/v1/models?include_paid=true" + +const faqs = [ + { + question: "What are AI model providers?", + answer: "AI model providers offer various language models with different capabilities and pricing.", + }, + { + question: "How is pricing calculated?", + answer: "Pricing is based on token usage for input and output, measured per million tokens, like pretty much any other provider out there.", + }, + { + question: "What is the Roo Code Cloud Provider?", + answer: ( + <> +

This is our very own model provider, optimized to work seamlessly with Roo Code Cloud.

+

+ It offers a selection of state-of-the-art LLMs (both closed and open weight) we know work well with + Roo for you to choose, with no markup. +

+

+ We also often feature 100% free models which labs share with us for the community to use and provide + feedback. +

+ + ), + }, + { + question: "But how much does the Roo Code Cloud service cost?", + answer: ( + <> + Our{" "} + + service pricing is here. + + + ), + }, +] + +function calculateTotalPrice(model: Model): number { + return parseFloat(model.pricing.input) + parseFloat(model.pricing.output) +} + +function enrichModelWithTotalPrice(model: Model): ModelWithTotalPrice { + return { + ...model, + totalPrice: calculateTotalPrice(model), + } +} + +export default function ProviderPricingPage() { + const [models, setModels] = useState([]) + const [loading, setLoading] = useState(true) + const [error, setError] = useState(null) + const [searchQuery, setSearchQuery] = useState("") + const [sortOption, setSortOption] = useState("alphabetical") + + useEffect(() => { + async function fetchModels() { + try { + setLoading(true) + setError(null) + const response = await fetch(API_URL) + if (!response.ok) { + throw new Error(`Failed to fetch models: ${response.statusText}`) + } + const data: ModelsResponse = await response.json() + const enrichedModels = data.data.map(enrichModelWithTotalPrice) + setModels(enrichedModels) + } catch (err) { + setError(err instanceof Error ? err.message : "An error occurred while fetching models") + } finally { + setLoading(false) + } + } + + fetchModels() + }, []) + + const filteredAndSortedModels = useMemo(() => { + // Filter out deprecated models + let filtered = models.filter((model) => !model.deprecated) + + // Filter by search query + if (searchQuery.trim()) { + const query = searchQuery.toLowerCase() + filtered = filtered.filter((model) => { + return ( + model.name.toLowerCase().includes(query) || + model.owned_by?.toLowerCase().includes(query) || + model.description.toLowerCase().includes(query) + ) + }) + } + + // Sort filtered results + const sorted = [...filtered] + switch (sortOption) { + case "alphabetical": + sorted.sort((a, b) => a.name.localeCompare(b.name)) + break + case "price-asc": + sorted.sort((a, b) => a.totalPrice - b.totalPrice) + break + case "price-desc": + sorted.sort((a, b) => b.totalPrice - a.totalPrice) + break + case "context-window-asc": + sorted.sort((a, b) => a.context_window - b.context_window) + break + case "context-window-desc": + sorted.sort((a, b) => b.context_window - a.context_window) + break + } + + return sorted + }, [models, searchQuery, sortOption]) + + // Count non-deprecated models for the display + const nonDeprecatedCount = useMemo(() => models.filter((model) => !model.deprecated).length, [models]) + + return ( + <> +
+
+
+

+ Roo Code Cloud Provider Pricing +

+

+ See pricing and features for all models we offer in our selection. +
+ You can always bring your own key ( + + FAQ + + ). +

+
+
+
+ +
+
+
+
+
+
+
+ + setSearchQuery(e.target.value)} + className="w-full rounded-full border border-input bg-background px-10 py-2 text-base ring-offset-background placeholder:text-muted-foreground focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-2" + /> + +
+ {filteredAndSortedModels.length} of {nonDeprecatedCount} models +
+
+
+
+
+ + +
+
+
+
+
+ +
+
+ {loading && ( +
+ +

Loading model list...

+
+ )} + + {error && ( +
+ +

Oops, couldn't load the model list.

+

Try again in a bit please.

+
+ )} + + {!loading && !error && filteredAndSortedModels.length === 0 && ( +
+ +

No models match your search.

+

+ Keep in mind we don't have every model under the sun – only the ones we think + are worth using. +
+ You can always use a third-party provider to access a wider selection. +

+
+ )} + + {!loading && !error && filteredAndSortedModels.length > 0 && ( +
+ {filteredAndSortedModels.map((model) => ( + + ))} +
+ )} +
+
+
+ + {/* FAQ Section */} +
+ +
+
+

Frequently Asked Questions

+
+
+ {faqs.map((faq, index) => ( +
+

{faq.question}

+

{faq.answer}

+
+ ))} +
+
+
+ + ) +} diff --git a/apps/web-roo-code/src/app/reviewer/ReviewerContent.tsx b/apps/web-roo-code/src/app/reviewer/ReviewerContent.tsx deleted file mode 100644 index 3f5a1cf12a..0000000000 --- a/apps/web-roo-code/src/app/reviewer/ReviewerContent.tsx +++ /dev/null @@ -1,296 +0,0 @@ -"use client" - -import { - ArrowRight, - Blocks, - BookMarked, - ListChecks, - LucideIcon, - GitPullRequest, - Key, - MessageSquareCode, - Wrench, -} from "lucide-react" -import Image from "next/image" -import Link from "next/link" - -import { Button } from "@/components/ui" -import { AnimatedBackground } from "@/components/homepage" -import { AgentCarousel } from "@/components/reviewer/agent-carousel" -import { EXTERNAL_LINKS } from "@/lib/constants" -import { trackGoogleAdsConversion } from "@/lib/analytics/google-ads" - -interface Feature { - icon: LucideIcon - title: string - description: string | React.ReactNode - logos?: string[] -} - -const workflowSteps: Feature[] = [ - { - icon: GitPullRequest, - title: "1. Connect Your Repository", - description: "Link your GitHub repository and configure which branches and pull requests should be reviewed.", - }, - { - icon: Key, - title: "2. Add Your API Key", - description: - "Provide your AI provider API key and set your review preferences, custom rules, and quality standards.", - }, - { - icon: MessageSquareCode, - title: "3. Get Review Comments", - description: - "Every pull request gets detailed GitHub comments in minutes from a Roo Code agent highlighting issues and suggesting improvements.", - }, -] - -const howItWorks: Feature[] = [ - { - icon: Blocks, - title: "Our agents, your provider keys", - description: ( - <> -

- We orchestrate the review, optimize the hell out of the prompts, integrate with GitHub, keep you - properly posted. -

-

We're thoughtful about token usage, but not incentivized to skimp to grow our margins.

- - ), - }, - { - icon: ListChecks, - title: "Advanced reasoning and workflows", - description: - "We optimize for state-of-the-art reasoning models and leverage powerful workflows (Diff analysis → Context Gathering → Impact Mapping → Contract checks) to produce crisp, actionable comments at the right level.", - }, - { - icon: BookMarked, - title: "Fully repository-aware", - description: - "Reviews traverse code ownership, dependency graphs, and historical patterns to surface risk and deviations, not noise.", - }, -] - -// Workaround for next/image choking on these for some reason -import hero from "/public/heroes/agent-reviewer.png" - -export function ReviewerContent() { - return ( - <> -
- - -
- - {/* How It Works Section */} -
-
-
-
-

How It Works

-
-
- -
-
    - {workflowSteps.map((step, index) => { - const Icon = step.icon - return ( -
  • - -

    - {step.title} -

    -
    - {step.description} -
    -
  • - ) - })} -
-
-
-
- -
-
-
-
-

- Why Roo's PR Reviewer is so much better -

-
-
- -
-
    - {howItWorks.map((feature, index) => { - const Icon = feature.icon - return ( -
  • - -

    - {feature.title} -

    -
    - {feature.description} -
    - {feature.logos && ( -
    - {feature.logos.map((logo) => ( - {`${logo} - ))} -
    - )} -
  • - ) - })} -
-
-
-
- -
-
-
-
-

- The first member of a whole new team -

- -

- Architecture, coding, reviewing, testing, debugging, documenting, designing –{" "} - almost everything we do today is mostly through our agents. Now we're - bringing them to you. -

-

- Roo's PR Reviewer isn't yet another single-purpose tool to add to your already - complicated stack. -
- It's the first member of your AI-powered development team. More agents are shipping - soon. -

-
-
- -
- -
-
-
- - {/* CTA Section */} -
-
-
-

Stop wasting time.

-

- Give Roo Code's PR Reviewer your model key and turn painful reviews into a tangible - quality advantage. -

- -
-
-
- - ) -} diff --git a/apps/web-roo-code/src/app/reviewer/content-b.ts b/apps/web-roo-code/src/app/reviewer/content-b.ts new file mode 100644 index 0000000000..0c2f76a2f5 --- /dev/null +++ b/apps/web-roo-code/src/app/reviewer/content-b.ts @@ -0,0 +1,93 @@ +import { type AgentPageContent } from "@/app/shared/agent-page-content" + +// Workaround for next/image choking on these for some reason +import hero from "/public/heroes/agent-reviewer.png" + +// Re-export for convenience +export type { AgentPageContent } + +export const content: AgentPageContent = { + agentName: "PR Reviewer", + hero: { + icon: "GitPullRequest", + heading: "Code reviews that catch what other AI tools (and most humans) miss.", + paragraphs: [ + "Run-of-the-mill, token-saving AI code review tools will surely catch syntax errors and style issues, but they'll usually miss the bugs that actually matter: logic flaws, security vulnerabilities, and misunderstood requirements.", + "Roo Code's PR Reviewer uses advanced reasoning models and full repository context to find the issues that slip through—before they reach production.", + ], + image: { + url: hero.src, + width: 800, + height: 474, + alt: "Example of a code review generated by Roo Code PR Reviewer", + }, + crossAgentLink: { + text: "Works great with", + links: [ + { + text: "PR Fixer Agent", + href: "/pr-fixer", + icon: "Wrench", + }, + ], + }, + cta: { + buttonText: "Try now for free", + disclaimer: "", + tracking: "&agent=reviewer", + }, + }, + howItWorks: { + heading: "How It Works", + steps: [ + { + title: "1. Connect Your Repository", + description: + "Link your GitHub repository and configure which branches and pull requests should be reviewed.", + icon: "GitPullRequest", + }, + { + title: "2. Add Your API Key", + description: + "Provide your AI provider API key and set your review preferences, custom rules, and quality standards.", + icon: "Key", + }, + { + title: "3. Get Review Comments", + description: + "Every pull request gets detailed GitHub comments in minutes from a Roo Code agent highlighting issues and suggesting improvements.", + icon: "MessageSquareCode", + }, + ], + }, + whyBetter: { + heading: "Why Roo's PR Reviewer is different", + features: [ + { + title: "Bring your own key, get uncompromised reviews", + paragraphs: [ + "Most AI review tools use fixed pricing, which means they skimp on tokens to protect their margins. That leads to shallow analysis and missed issues.", + "With Roo, you bring your own API key. We optimize prompts for depth, not cost-cutting, so reviews focus on real problems like business logic, security vulnerabilities, and architectural issues.", + ], + icon: "Blocks", + }, + { + title: "Advanced reasoning that understands what matters", + description: + "We leverage state-of-the-art reasoning models with sophisticated workflows: diff analysis, context gathering, impact mapping, and contract validation. This catches the subtle bugs that surface-level tools miss—misunderstood requirements, edge cases, and integration risks.", + icon: "ListChecks", + }, + { + title: "Repository-aware, not snippet-aware", + description: + "Roo analyzes your entire codebase context—dependency graphs, code ownership, team conventions, and historical patterns. It understands how changes interact with existing systems, not just whether individual lines look correct.", + icon: "BookMarked", + }, + ], + }, + cta: { + heading: "Ready for better code reviews?", + description: "Start finding the issues that matter with AI-powered reviews built for depth, not cost-cutting.", + buttonText: "Try now for free", + }, +} diff --git a/apps/web-roo-code/src/app/reviewer/content.ts b/apps/web-roo-code/src/app/reviewer/content.ts new file mode 100644 index 0000000000..0c2f76a2f5 --- /dev/null +++ b/apps/web-roo-code/src/app/reviewer/content.ts @@ -0,0 +1,93 @@ +import { type AgentPageContent } from "@/app/shared/agent-page-content" + +// Workaround for next/image choking on these for some reason +import hero from "/public/heroes/agent-reviewer.png" + +// Re-export for convenience +export type { AgentPageContent } + +export const content: AgentPageContent = { + agentName: "PR Reviewer", + hero: { + icon: "GitPullRequest", + heading: "Code reviews that catch what other AI tools (and most humans) miss.", + paragraphs: [ + "Run-of-the-mill, token-saving AI code review tools will surely catch syntax errors and style issues, but they'll usually miss the bugs that actually matter: logic flaws, security vulnerabilities, and misunderstood requirements.", + "Roo Code's PR Reviewer uses advanced reasoning models and full repository context to find the issues that slip through—before they reach production.", + ], + image: { + url: hero.src, + width: 800, + height: 474, + alt: "Example of a code review generated by Roo Code PR Reviewer", + }, + crossAgentLink: { + text: "Works great with", + links: [ + { + text: "PR Fixer Agent", + href: "/pr-fixer", + icon: "Wrench", + }, + ], + }, + cta: { + buttonText: "Try now for free", + disclaimer: "", + tracking: "&agent=reviewer", + }, + }, + howItWorks: { + heading: "How It Works", + steps: [ + { + title: "1. Connect Your Repository", + description: + "Link your GitHub repository and configure which branches and pull requests should be reviewed.", + icon: "GitPullRequest", + }, + { + title: "2. Add Your API Key", + description: + "Provide your AI provider API key and set your review preferences, custom rules, and quality standards.", + icon: "Key", + }, + { + title: "3. Get Review Comments", + description: + "Every pull request gets detailed GitHub comments in minutes from a Roo Code agent highlighting issues and suggesting improvements.", + icon: "MessageSquareCode", + }, + ], + }, + whyBetter: { + heading: "Why Roo's PR Reviewer is different", + features: [ + { + title: "Bring your own key, get uncompromised reviews", + paragraphs: [ + "Most AI review tools use fixed pricing, which means they skimp on tokens to protect their margins. That leads to shallow analysis and missed issues.", + "With Roo, you bring your own API key. We optimize prompts for depth, not cost-cutting, so reviews focus on real problems like business logic, security vulnerabilities, and architectural issues.", + ], + icon: "Blocks", + }, + { + title: "Advanced reasoning that understands what matters", + description: + "We leverage state-of-the-art reasoning models with sophisticated workflows: diff analysis, context gathering, impact mapping, and contract validation. This catches the subtle bugs that surface-level tools miss—misunderstood requirements, edge cases, and integration risks.", + icon: "ListChecks", + }, + { + title: "Repository-aware, not snippet-aware", + description: + "Roo analyzes your entire codebase context—dependency graphs, code ownership, team conventions, and historical patterns. It understands how changes interact with existing systems, not just whether individual lines look correct.", + icon: "BookMarked", + }, + ], + }, + cta: { + heading: "Ready for better code reviews?", + description: "Start finding the issues that matter with AI-powered reviews built for depth, not cost-cutting.", + buttonText: "Try now for free", + }, +} diff --git a/apps/web-roo-code/src/app/reviewer/page.tsx b/apps/web-roo-code/src/app/reviewer/page.tsx index 7f7cce862a..776ded6847 100644 --- a/apps/web-roo-code/src/app/reviewer/page.tsx +++ b/apps/web-roo-code/src/app/reviewer/page.tsx @@ -2,7 +2,10 @@ import type { Metadata } from "next" import { SEO } from "@/lib/seo" import { ogImageUrl } from "@/lib/og" -import { ReviewerContent } from "./ReviewerContent" +import { AgentLandingContent } from "@/app/shared/AgentLandingContent" +import { getContentVariant } from "@/app/shared/getContentVariant" +import { content as contentA } from "./content" +import { content as contentB } from "./content-b" const TITLE = "PR Reviewer" const DESCRIPTION = @@ -56,6 +59,12 @@ export const metadata: Metadata = { ], } -export default function AgentReviewerPage() { - return +export default async function AgentReviewerPage({ searchParams }: { searchParams: Promise<{ v?: string }> }) { + const params = await searchParams + const content = getContentVariant(params, { + A: contentA, + B: contentB, + }) + + return } diff --git a/apps/web-roo-code/src/app/shared/AgentLandingContent.tsx b/apps/web-roo-code/src/app/shared/AgentLandingContent.tsx new file mode 100644 index 0000000000..4db166b919 --- /dev/null +++ b/apps/web-roo-code/src/app/shared/AgentLandingContent.tsx @@ -0,0 +1,235 @@ +"use client" + +import { + ArrowRight, + GitPullRequest, + Wrench, + Key, + MessageSquareCode, + Blocks, + ListChecks, + BookMarked, + History, + LucideIcon, +} from "lucide-react" +import Image from "next/image" +import Link from "next/link" + +import { Button } from "@/components/ui" +import { AnimatedBackground, UseExamplesSection } from "@/components/homepage" +import { EXTERNAL_LINKS } from "@/lib/constants" +import { type AgentPageContent, type IconName } from "./agent-page-content" + +/** + * Maps icon names to actual Lucide icon components + */ +const iconMap: Record = { + GitPullRequest, + Wrench, + Key, + MessageSquareCode, + Blocks, + ListChecks, + BookMarked, + History, +} + +/** + * Converts an icon name string to a Lucide icon component + */ +function getIcon(iconName?: IconName): LucideIcon | undefined { + return iconName ? iconMap[iconName] : undefined +} + +export function AgentLandingContent({ content }: { content: AgentPageContent }) { + return ( + <> + {/* Hero Section */} +
+ +
+
+
+
+

+ {content.hero.icon && + (() => { + const Icon = getIcon(content.hero.icon) + return Icon ? : null + })()} + {content.hero.heading} +

+ +
+ {content.hero.paragraphs.map((paragraph, index) => ( +

{paragraph}

+ ))} +
+ + {/* Cross-agent link */} +
+ {content.hero.crossAgentLink.text} + {content.hero.crossAgentLink.links.map((link, index) => { + const Icon = getIcon(link.icon) + return ( + + {Icon && } + {link.text} + + + ) + })} +
+
+ +
+ + + {content.hero.cta.disclaimer} + +
+
+ + {content.hero.image && ( +
+
+ {content.hero.image.alt +
+
+ )} +
+
+
+ + {/* How It Works Section */} +
+
+
+
+
+
+
+

+ {content.howItWorks.heading} +

+
+
+ +
+
    + {content.howItWorks.steps.map((step, index) => { + const Icon = getIcon(step.icon) + return ( +
  • + {Icon && } +

    + {step.title} +

    +
    + {step.description} +
    +
  • + ) + })} +
+
+
+
+ + {/* Why Better Section */} +
+
+
+
+
+
+
+

+ {content.whyBetter.heading} +

+
+
+ +
+
    + {content.whyBetter.features.map((feature, index) => { + const Icon = getIcon(feature.icon) + return ( +
  • + {Icon && } +

    + {feature.title} +

    +
    + {feature.description &&

    {feature.description}

    } + {feature.paragraphs && + feature.paragraphs.map((paragraph, pIndex) => ( +

    {paragraph}

    + ))} +
    +
  • + ) + })} +
+
+
+
+ + + + {/* CTA Section */} +
+
+
+

{content.cta.heading}

+

+ {content.cta.description} +

+ +
+
+
+ + ) +} diff --git a/apps/web-roo-code/src/app/shared/agent-page-content.ts b/apps/web-roo-code/src/app/shared/agent-page-content.ts new file mode 100644 index 0000000000..01a64e8547 --- /dev/null +++ b/apps/web-roo-code/src/app/shared/agent-page-content.ts @@ -0,0 +1,75 @@ +/** + * Supported icon names that can be used in agent page content. + * These strings are mapped to actual Lucide components in the client. + */ +export type IconName = + | "GitPullRequest" + | "Wrench" + | "Key" + | "MessageSquareCode" + | "Blocks" + | "ListChecks" + | "BookMarked" + | "History" + +/** + * Generic content structure for agent landing pages. + * This interface can be reused across different agent pages (PR Reviewer, PR Fixer, etc.) + * to maintain consistency and enable A/B testing capabilities. + * + * Note: Icons are referenced by string names (not components) to support + * serialization from Server Components to Client Components. + */ +export interface AgentPageContent { + agentName: string + hero: { + /** Optional icon name to display in the hero section */ + icon?: IconName + heading: string + paragraphs: string[] + image?: { + url: string + width: number + height: number + alt?: string + } + crossAgentLink: { + text: string + links: Array<{ + text: string + href: string + icon?: IconName + }> + } + cta: { + buttonText: string + disclaimer: string + tracking: string + } + } + howItWorks: { + heading: string + steps: Array<{ + title: string + /** Supports rich text content including React components */ + description: string | React.ReactNode + icon?: IconName + }> + } + whyBetter: { + heading: string + features: Array<{ + title: string + /** Supports rich text content including React components */ + description?: string | React.ReactNode + /** Supports rich text content including React components */ + paragraphs?: Array + icon?: IconName + }> + } + cta: { + heading: string + description: string + buttonText: string + } +} diff --git a/apps/web-roo-code/src/app/shared/getContentVariant.ts b/apps/web-roo-code/src/app/shared/getContentVariant.ts new file mode 100644 index 0000000000..0d8fccdde4 --- /dev/null +++ b/apps/web-roo-code/src/app/shared/getContentVariant.ts @@ -0,0 +1,36 @@ +import type { AgentPageContent } from "./agent-page-content" + +/** + * Selects the appropriate content variant based on the query parameter. + * + * @param searchParams - The search parameters from the page props + * @param variants - A record mapping variant letters to content objects + * @returns The selected content variant, defaulting to variant 'A' if not found or invalid + * + * @example + * ```tsx + * const content = getContentVariant(searchParams, { + * A: contentA, + * B: contentB, + * C: contentC, + * }) + * ``` + */ +export function getContentVariant( + searchParams: { v?: string }, + variants: Record, +): AgentPageContent { + const variant = searchParams.v?.toUpperCase() + + // Return the specified variant if it exists, otherwise default to 'A' + if (variant && variants[variant]) { + return variants[variant] + } + + // Ensure 'A' variant always exists as fallback + if (!variants.A) { + throw new Error("Content variants must include variant 'A' as the default") + } + + return variants.A +} diff --git a/apps/web-roo-code/src/components/chromes/nav-bar.tsx b/apps/web-roo-code/src/components/chromes/nav-bar.tsx index 3e34dc7f90..2442e3b759 100644 --- a/apps/web-roo-code/src/components/chromes/nav-bar.tsx +++ b/apps/web-roo-code/src/components/chromes/nav-bar.tsx @@ -13,7 +13,7 @@ import { EXTERNAL_LINKS } from "@/lib/constants" import { useLogoSrc } from "@/lib/hooks/use-logo-src" import { ScrollButton } from "@/components/ui" import ThemeToggle from "@/components/chromes/theme-toggle" -import { ChevronDown, Cloud, X } from "lucide-react" +import { ChevronDown, X } from "lucide-react" interface NavBarProps { stars: string | null @@ -93,35 +93,41 @@ export function NavBar({ stars, downloads }: NavBarProps) {
-
-
+
+
+ className="hidden items-center gap-1.5 text-sm font-medium text-muted-foreground hover:text-foreground md:flex whitespace-nowrap"> {stars !== null && {stars}}
+ + Log in + + + Sign Up + + className="hidden items-center gap-1.5 rounded-md bg-primary px-4 py-2 text-sm font-medium text-primary-foreground transition-all duration-200 hover:shadow-lg hover:scale-105 md:flex whitespace-nowrap"> Install · {downloads !== null && {downloads}} - - - Log in -
{/* Mobile Menu Button */} @@ -226,15 +232,24 @@ export function NavBar({ stars, downloads }: NavBarProps) { {downloads !== null && {downloads}}
- setIsMenuOpen(false)}> - - Log in - +
diff --git a/apps/web-roo-code/src/components/homepage/cloud-section.tsx b/apps/web-roo-code/src/components/homepage/cloud-section.tsx new file mode 100644 index 0000000000..9b2539c4b0 --- /dev/null +++ b/apps/web-roo-code/src/components/homepage/cloud-section.tsx @@ -0,0 +1,109 @@ +import { Bot, Settings2, ShieldCheck } from "lucide-react" + +export function CloudSection() { + return ( +
+
+
+

Asynchronous Engineering.

+

+ Stop watching the cursor. Deploy specialized agents to work while you sleep. +

+
+ + {/* Pipeline Diagram Visual */} +
+
Ticket
+
+
+ Planner Agent +
+
+
+ Coder Agent +
+
+
+ GitHub PR +
+
+ +
+
+
+
+ +
+

Purpose-Built Agents (Safety)

+
+

Zero Drift via Role Constraints

+

+ Fear of agents going haywire is solved by architecture, not prompt engineering. Cloud Agents + enforce the strict Modes you use locally. +

+
    +
  • + +
    + The Planner: + + Maps dependencies. Read-Only access. + +
    +
  • +
  • + +
    + The Builder: + + Writes code based on the plan. Scoped file access. + +
    +
  • +
  • + +
    + The Reviewer: + + Analyzes diffs. Cannot push to main. + +
    +
  • +
+
+ +
+
+
+ +
+

Orchestrated Configuration

+
+

Optimize Your AI Workforce

+

+ Just as you choose models locally, you configure them for the cloud to balance performance + vs. cost. +

+
+
Config Example:
+
+
+ Planner Agent + + o1-preview (Reasoning) + +
+
+ Unit Test Agent + + Haiku (Speed/Cost) + +
+
+
+
+
+
+
+ ) +} diff --git a/apps/web-roo-code/src/components/homepage/company-logos.tsx b/apps/web-roo-code/src/components/homepage/company-logos.tsx index a27e8bbc16..6aeb126ade 100644 --- a/apps/web-roo-code/src/components/homepage/company-logos.tsx +++ b/apps/web-roo-code/src/components/homepage/company-logos.tsx @@ -7,13 +7,13 @@ const logos = ["Apple", "Netflix", "Microsoft", "Amazon", "ByteDance", "Rakuten" export function CompanyLogos() { return ( -
+
- Making devs more productive at + className="text-xs text-muted-foreground text-center mb-2 "> + Helping teams ship more at
{logos.map((logo, index) => ( @@ -25,7 +25,7 @@ export function CompanyLogos() { {`${logo} diff --git a/apps/web-roo-code/src/components/homepage/cta-section.tsx b/apps/web-roo-code/src/components/homepage/cta-section.tsx new file mode 100644 index 0000000000..cd9a54487d --- /dev/null +++ b/apps/web-roo-code/src/components/homepage/cta-section.tsx @@ -0,0 +1,37 @@ +import { Button } from "@/components/ui" +import { ArrowRight, Download } from "lucide-react" +import { EXTERNAL_LINKS } from "@/lib/constants" + +export function CTASection() { + return ( +
+
+

Build faster. Solo or Together.

+ + +
+
+ ) +} diff --git a/apps/web-roo-code/src/components/homepage/ecosystem-section.tsx b/apps/web-roo-code/src/components/homepage/ecosystem-section.tsx new file mode 100644 index 0000000000..8058512847 --- /dev/null +++ b/apps/web-roo-code/src/components/homepage/ecosystem-section.tsx @@ -0,0 +1,82 @@ +import { GitMerge, Terminal, MessageSquare } from "lucide-react" + +export function EcosystemSection() { + return ( +
+
+

Integrated into your SDLC.

+ +
+ {/* Triangle Connection Lines - Absolute positioned */} +
+ + + + + +
+ +
+ {/* Step 1: Dispatch */} +
+
+ +
+
01. DISPATCH
+

Trigger Task

+

+ Trigger a task via @Roo in Slack or the VS Code terminal. +

+
+ + {/* Step 2: Execute */} +
+
+ +
+
02. EXECUTE
+

Run Agents

+

+ Agents run in isolated, ephemeral docker containers. +

+
+ + {/* Step 3: Merge */} +
+
+ +
+
03. MERGE
+

Review PR

+

+ The output is always a standard GitHub Pull Request. You review code, not chat logs. +

+
+
+
+
+
+ ) +} diff --git a/apps/web-roo-code/src/components/homepage/index.ts b/apps/web-roo-code/src/components/homepage/index.ts index 9d4427448e..faafc908cc 100644 --- a/apps/web-roo-code/src/components/homepage/index.ts +++ b/apps/web-roo-code/src/components/homepage/index.ts @@ -6,3 +6,9 @@ export * from "./features" export * from "./install-section" export * from "./testimonials" export * from "./whats-new-button" +export * from "./option-overview-section" +export * from "./pillars-section" +export * from "./cloud-section" +export * from "./ecosystem-section" +export * from "./cta-section" +export * from "./use-examples-section" diff --git a/apps/web-roo-code/src/components/homepage/option-overview-section.tsx b/apps/web-roo-code/src/components/homepage/option-overview-section.tsx new file mode 100644 index 0000000000..60e3f4bc76 --- /dev/null +++ b/apps/web-roo-code/src/components/homepage/option-overview-section.tsx @@ -0,0 +1,99 @@ +import { Laptop, Cloud, ArrowRight } from "lucide-react" +import { Button } from "../ui" +import { EXTERNAL_LINKS } from "@/lib/constants" + +export function OptionOverviewSection() { + return ( +
+
+
+

+ Different form factors for different ways of working. +

+

+ Roo's always there to help you get stuff done. +

+
+
+
+
+
+ +
+
+ +
+

Roo Code VS Code Extension

+

For Individual Work

+ +
+

+ Run Roo directly in VS Code (or any fork – even Cursor!), stay close to the code and + control everything: +

+
    +
  • Approve every action (or set it to auto-approve)
  • +
  • Manage the context window
  • +
  • Configure every detail
  • +
  • Preview changes live
  • +
  • Stick to your customized editor
  • +
  • Write code by hand (gasp!)
  • +
+

+ Ideal for real-time debugging or quick iteration where you need full, immediate control. +

+
+ + +
+ +
+
+ +
+

Roo Code Cloud

+
For Team Work with Agents
+ +
+

+ Create your agent team in the Cloud, give them access to Github and start giving them + tasks: +

+
    +
  • + Use agents like the Planner, Coder, Explainer, Reviewer and Fixer +
  • +
  • Choose your provider and model
  • +
  • + Create tasks from the Web and Slack (more integrations soon) +
  • +
  • Get PR Reviews (and fixes) directly on Github
  • +
  • Collaborate with co-workers
  • +
+

+ Ideal for kicking projects off, parallelizing execution and looping in the rest of your + team. +

+
+ + +
+
+
+
+ ) +} diff --git a/apps/web-roo-code/src/components/homepage/pillars-section.tsx b/apps/web-roo-code/src/components/homepage/pillars-section.tsx new file mode 100644 index 0000000000..def772390f --- /dev/null +++ b/apps/web-roo-code/src/components/homepage/pillars-section.tsx @@ -0,0 +1,202 @@ +import { Brain, Keyboard, Shield, Users2, Map, Code, MessageCircleQuestion, Bug, TestTube } from "lucide-react" +import Image from "next/image" +import { Link } from "../ui" + +const MODEL_LOGOS = [ + "OpenRouter", + "Anthropic", + "OpenAI", + "Gemini", + "Grok", + "Bedrock", + "Moonshot", + "Qwen", + "Kimi", + "Mistral", + "Ollama", +] +const MODE_EXAMPLES = [ + { + name: "Architect", + description: "Plans complex changes without making changes.", + icon: Map, + }, + { + name: "Code", + description: "Implements, refactors and optimizes code.", + icon: Code, + }, + { + name: "Ask", + description: "Explains functionality and program behavior.", + icon: MessageCircleQuestion, + }, + { + name: "Debug", + description: "Diagnoses issues, traces failures, and proposes targeted, reliable fixes.", + icon: Bug, + }, + { + name: "Test", + description: "Creates and improves performant tests without changing the actual functionality.", + icon: TestTube, + }, +] + +export function PillarsSection() { + return ( +
+
+
+
+
+
+

+ To trust an agent, you have to do it on your own terms. +

+

+ Roo is designed from the ground up to give you the confidence to do ever more with AI. +

+
+ +
+
+
+
+ +
+
+

Model-agnostic by design

+

Flexible and future-proof.

+
+

+ "The best model in the world" changes every other week. Providers + throttle models with no warning. 1st-party coding agents only work with their + own models. +

+

Roo doesn't care.

+

+ It works great with 10s of models, from frontier to open weight. Choose from{" "} + the curated selection we offer at-cost or + bring your own key. +

+
+
+ + Compatible with dozens of providers + +
+ {MODEL_LOGOS.map((logo, index) => ( + {`${logo} + ))} +
+
+
+
+
+ +
+
+
+ +
+
+

Role-specific Modes

+

On-task and under control.

+
+

+ As capable as they are, when let loose, LLMs hallucinate, cheat and can cause + real damage. +

+

+ Roo's Modes keep models focused on a given task and limit their access to + tools which are relevant to their role, keeping the context window clearer and + avoiding surprises. +

+

+ Modes are even smart enough to ask to switch to another when stepping outside + their responsibilities. +

+
+
+ Some examples +
    + {MODE_EXAMPLES.map((mode) => { + const Icon = mode.icon + return ( +
  • + +
    +

    {mode.name}

    +

    + {mode.description} +

    +
    +
  • + ) + })} +
+
+
+
+
+ +
+
+
+ +
+
+

Highly configurable

+

Make it fit your workflow.

+
+

+ Developer tools need to fit like gloves. Highly tweakable, + keyboard-shortcut-heavy gloves. +

+

We made Roo thoughtfully configurable to fit your workflow as best it can.

+
+
+
+
+ +
+
+
+ +
+
+

Secure and transparent

+

Open source from the get go.

+
+

+ The Roo Code Extension is{" "} + + open source + {" "} + so you can see for yourself exactly what it's doing and we don't use + your data for training. +

+

+ Plus we're fully SOC2 Type 2 compliant and follow industry-standard + security practices. +

+
+
+
+
+
+
+
+ ) +} diff --git a/apps/web-roo-code/src/components/homepage/testimonials.tsx b/apps/web-roo-code/src/components/homepage/testimonials.tsx index 84d9e6cf40..c37907ab4a 100644 --- a/apps/web-roo-code/src/components/homepage/testimonials.tsx +++ b/apps/web-roo-code/src/components/homepage/testimonials.tsx @@ -193,11 +193,9 @@ export function Testimonials() {

- Developers really shipping with AI are using Roo Code + More than 1 million people are shipping with Roo.

-

- Join more than 1M people revolutionizing their workflow worldwide -

+

And they have some great things to say.

- {testimonial.role} at {testimonial.origin} + {testimonial.role !== "Reviewer" && ( + <> + {testimonial.role} at {testimonial.origin} + + )} {testimonial.stars && ( {" "} diff --git a/apps/web-roo-code/src/components/homepage/use-examples-section.tsx b/apps/web-roo-code/src/components/homepage/use-examples-section.tsx new file mode 100644 index 0000000000..f1170321df --- /dev/null +++ b/apps/web-roo-code/src/components/homepage/use-examples-section.tsx @@ -0,0 +1,438 @@ +"use client" + +import { useMemo, useState } from "react" +import { motion, AnimatePresence } from "framer-motion" +import { + LucideIcon, + Pointer, + Slack, + Github, + Code, + GitPullRequest, + Wrench, + Map, + MessageCircleQuestionMark, + CornerDownRight, + ChevronDown, +} from "lucide-react" +import Image from "next/image" +import { Button } from "../ui" + +interface UseCase { + role: string + use: string + agent: UseCaseAgent + context: UseCaseSource +} + +interface UseCaseSource { + name: string + icon: LucideIcon +} + +interface UseCaseAgent { + name: string + icon: LucideIcon +} + +interface PositionedUseCase extends UseCase { + layer: 1 | 2 | 3 | 4 + position: { x: number; y: number } + scale: number + zIndex: number + avatar: string +} + +const SOURCES = { + slack: { + name: "Slack", + icon: Slack, + }, + web: { + name: "Web", + icon: Pointer, + }, + github: { + name: "Github", + icon: Github, + }, + extension: { + name: "Extension", + icon: Code, + }, +} + +const AGENTS = { + explainer: { + name: "Explainer", + icon: MessageCircleQuestionMark, + }, + planner: { + name: "Planner", + icon: Map, + }, + coder: { + name: "Coder", + icon: Code, + }, + reviewer: { + name: "Reviewer", + icon: GitPullRequest, + }, + fixer: { + name: "Fixer", + icon: Wrench, + }, +} + +const USE_CASES: UseCase[] = [ + { + role: "Frontend Developer", + use: "Take Lisa's feedback above and incorporate it into the landing page.", + agent: AGENTS.coder, + context: SOURCES.slack, + }, + { + role: "Customer Success", + use: "What could be causing this bug as described by the customer?", + agent: AGENTS.explainer, + context: SOURCES.web, + }, + { + role: "Backend Engineer", + use: "Create a migration denormalizing total_cost calculation and backfill the remainder.", + agent: AGENTS.coder, + context: SOURCES.extension, + }, + { + role: "Security Engineer", + use: "Do we use any of the libraries mentioned in the thread?", + agent: AGENTS.explainer, + context: SOURCES.slack, + }, + { + role: "Designer", + use: "Refactor the button component to use CSS variables", + agent: AGENTS.coder, + context: SOURCES.slack, + }, + { + role: "Product Manager", + use: "How big of a change would it be to turn this from a yes/no to have 4 options?", + agent: AGENTS.coder, + context: SOURCES.web, + }, + { + role: "QA Engineer", + use: "Write a Playwright test for the login flow failure case, extract existing mocks into shared.", + agent: AGENTS.coder, + context: SOURCES.github, + }, + { + role: "DevOps Engineer", + use: "Update the Dockerfile to use Node 20 Alpine.", + agent: AGENTS.fixer, + context: SOURCES.slack, + }, + { + role: "Mobile Developer", + use: "Copy what we did in PR #4253 and apply to this component.", + agent: AGENTS.coder, + context: SOURCES.slack, + }, + { + role: "Technical Writer", + use: "Generate JSDoc comments for the auth utility functions.", + agent: AGENTS.coder, + context: SOURCES.github, + }, + { + role: "Junior Developer", + use: "Review this pull request for potential performance improvements.", + agent: AGENTS.reviewer, + context: SOURCES.github, + }, + { + role: "Engineering Manager", + use: "Break down this user profile feature into technical tasks, grouped by skill.", + agent: AGENTS.planner, + context: SOURCES.web, + }, + { + role: "Support Engineer", + use: "What's causing this stack trace? The customer is on MacOS 26.1.", + agent: AGENTS.explainer, + context: SOURCES.web, + }, + { + role: "Frontend Developer", + use: "Make the navigation menu responsive on mobile devices.", + agent: AGENTS.coder, + context: SOURCES.web, + }, + { + role: "Backend Engineer", + use: "Give me two architecture options for the notification system in this PRD.", + agent: AGENTS.planner, + context: SOURCES.web, + }, + { + role: "Designer", + use: "Implement the loading spinner animation in CSS.", + agent: AGENTS.coder, + context: SOURCES.web, + }, + { + role: "Customer Success", + use: "Write a script to find patterns in these CPU load logs.", + agent: AGENTS.coder, + context: SOURCES.slack, + }, + { + role: "Full Stack Dev", + use: "Refactor user_preferences to use named columns instead of a single JSON blob", + agent: AGENTS.coder, + context: SOURCES.extension, + }, + { + role: "QA Engineer", + use: "Automate the regression suite for the checkout process.", + agent: AGENTS.coder, + context: SOURCES.extension, + }, + { + role: "DevOps Engineer", + use: "Understand why this build error only happens in prod and fix it.", + agent: AGENTS.coder, + context: SOURCES.extension, + }, + { + role: "Product Marketer", + use: "What were the 5 most significant PRs merged in the past week?", + agent: AGENTS.explainer, + context: SOURCES.slack, + }, + { + role: "Junior Developer", + use: "Explain how useEffect dependency arrays work here.", + agent: AGENTS.explainer, + context: SOURCES.extension, + }, + { + role: "Senior Engineer", + use: "Check if this implementation follows the Single Responsibility Principle.", + agent: AGENTS.reviewer, + context: SOURCES.github, + }, +] + +// Seeded random number generator for consistent layout +function seededRandom(seed: number) { + let value = seed + return () => { + value = (value * 9301 + 49297) % 233280 + return value / 233280 + } +} + +const LAYER_SCALES = { + 1: 0.7, + 2: 0.85, + 3: 1.0, + 4: 1.15, +} + +function distributeItems(items: UseCase[]): PositionedUseCase[] { + const rng = seededRandom(Math.random() * 12345) + const zones = { rows: 7, cols: 4 } + const zoneWidth = 100 / zones.cols + const zoneHeight = 100 / zones.rows + + // Create array of zone indices [0...19] and shuffle them + const zoneIndices = Array.from({ length: items.length }, (_, i) => i) + for (let i = zoneIndices.length - 1; i > 0; i--) { + const j = Math.floor(rng() * (i + 1)) + const temp = zoneIndices[i]! + zoneIndices[i] = zoneIndices[j]! + zoneIndices[j] = temp + } + + return items.map((item, index) => { + // Assign to a random unique zone + const zoneIndex = zoneIndices[index]! + const row = Math.floor(zoneIndex / zones.cols) + const col = zoneIndex % zones.cols + + // Distribute layers evenly + const layer = ((index % 4) + 1) as 1 | 2 | 3 | 4 + + // Calculate base position (center of zone) + const baseX = col * zoneWidth + zoneWidth / 2 + const baseY = row * zoneHeight + zoneHeight / 2 + + // Add jitter (±35% of zone size to keep somewhat contained but messy) + const jitterX = (rng() - 0.5) * zoneWidth * 0.7 + const jitterY = (rng() - 0.5) * zoneHeight * 0.7 + + return { + ...item, + avatar: `/illustrations/user-faces/${index + 1}.jpg`, + layer, + position: { + x: baseX + jitterX, + y: baseY + jitterY, + }, + scale: LAYER_SCALES[layer], + zIndex: layer, + } + }) +} + +function UseCaseCardContent({ + item, + opacity = 1, + className = "", +}: { + item: UseCase & { avatar: string } + opacity?: number + className?: string +}) { + const ContextIcon: LucideIcon = item.context.icon + return ( +

+
+ + {item.role} +
+ +
+ + To {item.agent.name} Agent +
+ +
+ {item.use} +
+ +
+ via {item.context.name} +
+
+ ) +} + +function DesktopUseCaseCard({ item }: { item: PositionedUseCase }) { + const opacity = Math.min(1, 0.5 + item.layer / 3) + + return ( + `translate(-50%, -50%) scale(${scale})`}> + + + ) +} + +export function UseExamplesSection({ agentTitle = false }: { agentTitle?: boolean }) { + const positionedItems = useMemo(() => distributeItems(USE_CASES), []) + const [showAllMobile, setShowAllMobile] = useState(false) + + return ( +
+
+
+
+
+
+

+ {agentTitle ? ( + <> + Part of the AI team to help your entire human team + + ) : ( + <> + The AI team to help your entire human team + + )} +

+

+ Developers, PMs, Designers, Customer Success: everyone moves faster and more independently with + Roo. +

+
+ + {/* Mobile: Vertical Staggered List */} +
+ + {positionedItems.slice(0, showAllMobile ? undefined : 8).map((item, index) => ( + + + + ))} + + + {!showAllMobile && ( +
+ +
+ )} +
+ + {/* Desktop: Positioned Items Container */} +
+ {positionedItems.map((item, index) => ( + + ))} +
+
+
+ ) +} diff --git a/apps/web-roo-code/src/components/providers/google-analytics-provider.tsx b/apps/web-roo-code/src/components/providers/google-analytics-provider.tsx deleted file mode 100644 index 3d274db509..0000000000 --- a/apps/web-roo-code/src/components/providers/google-analytics-provider.tsx +++ /dev/null @@ -1,153 +0,0 @@ -"use client" - -import { useEffect, useState } from "react" -import Script from "next/script" -import { hasConsent, onConsentChange } from "@/lib/analytics/consent-manager" - -// Google Tag Manager ID -const GTM_ID = "AW-17391954825" - -/** - * Google Analytics Provider with Consent Mode v2 - * Implements cookieless pings and advanced consent management - */ -export function GoogleAnalyticsProvider({ children }: { children: React.ReactNode }) { - const [shouldLoad, setShouldLoad] = useState(false) - - useEffect(() => { - // Initialize consent defaults BEFORE loading gtag.js (required for Consent Mode v2) - initializeConsentDefaults() - - // Check initial consent status - if (hasConsent()) { - setShouldLoad(true) - updateConsentGranted() - } - - // Listen for consent changes - const unsubscribe = onConsentChange((consented) => { - if (consented) { - if (!shouldLoad) { - setShouldLoad(true) - } - updateConsentGranted() - } else { - updateConsentDenied() - } - }) - - return unsubscribe - // eslint-disable-next-line react-hooks/exhaustive-deps -- shouldLoad intentionally omitted to prevent re-initialization loop - }, []) - - const initializeConsentDefaults = () => { - // Set up consent defaults before gtag loads (Consent Mode v2 requirement) - if (typeof window !== "undefined") { - window.dataLayer = window.dataLayer || [] - window.gtag = function (...args: GtagArgs) { - window.dataLayer.push(args) - } - - // Set default consent state to 'denied' with cookieless pings enabled - window.gtag("consent", "default", { - ad_storage: "denied", - ad_user_data: "denied", - ad_personalization: "denied", - analytics_storage: "denied", - functionality_storage: "denied", - personalization_storage: "denied", - security_storage: "granted", // Always granted for security - wait_for_update: 500, // Wait 500ms for consent before sending data - }) - - // Enable cookieless pings for Google Ads - window.gtag("set", "url_passthrough", true) - } - } - - const updateConsentGranted = () => { - // User accepted cookies - update consent to granted - if (typeof window !== "undefined" && window.gtag) { - window.gtag("consent", "update", { - ad_storage: "granted", - ad_user_data: "granted", - ad_personalization: "granted", - analytics_storage: "granted", - functionality_storage: "granted", - personalization_storage: "granted", - }) - } - } - - const updateConsentDenied = () => { - // User declined cookies - keep consent denied (cookieless pings still work) - if (typeof window !== "undefined" && window.gtag) { - window.gtag("consent", "update", { - ad_storage: "denied", - ad_user_data: "denied", - ad_personalization: "denied", - analytics_storage: "denied", - functionality_storage: "denied", - personalization_storage: "denied", - }) - } - } - - // Always render scripts (Consent Mode v2 needs gtag loaded even without consent) - // Cookieless pings will work with denied consent - - return ( - <> - {/* Google tag (gtag.js) - Loads immediately for Consent Mode v2 */} - + ` + + const csp = [ + "default-src 'none'", + `font-src ${webview.cspSource} data:`, + `style-src ${webview.cspSource} 'unsafe-inline' https://* http://${localServerUrl}`, + `img-src ${webview.cspSource} data:`, + `script-src 'unsafe-eval' ${webview.cspSource} http://${localServerUrl} 'nonce-${nonce}'`, + `connect-src ${webview.cspSource} ws://${localServerUrl} http://${localServerUrl}`, + ] + + return ` + + + + + + + + + Browser Session + + +
+ ${reactRefresh} + + + + ` + } + + private getHtmlContent(webview: vscode.Webview, extensionUri: vscode.Uri): string { + const stylesUri = getUri(webview, extensionUri, ["webview-ui", "build", "assets", "index.css"]) + const scriptUri = getUri(webview, extensionUri, ["webview-ui", "build", "assets", "browser-panel.js"]) + const codiconsUri = getUri(webview, extensionUri, ["assets", "codicons", "codicon.css"]) + + const nonce = getNonce() + + const csp = [ + "default-src 'none'", + `font-src ${webview.cspSource} data:`, + `style-src ${webview.cspSource} 'unsafe-inline'`, + `img-src ${webview.cspSource} data:`, + `script-src ${webview.cspSource} 'wasm-unsafe-eval' 'nonce-${nonce}'`, + `connect-src ${webview.cspSource}`, + ] + + return ` + + + + + + + + + Browser Session + + +
+ + + + ` + } +} diff --git a/src/core/webview/ClineProvider.ts b/src/core/webview/ClineProvider.ts index f97ca2577b..c05147e396 100644 --- a/src/core/webview/ClineProvider.ts +++ b/src/core/webview/ClineProvider.ts @@ -42,6 +42,7 @@ import { ORGANIZATION_ALLOW_ALL, DEFAULT_MODES, DEFAULT_CHECKPOINT_TIMEOUT_SECONDS, + getModelId, } from "@roo-code/types" import { TelemetryService } from "@roo-code/telemetry" import { CloudService, BridgeOrchestrator, getRooCodeApiUrl } from "@roo-code/cloud" @@ -50,7 +51,6 @@ import { Package } from "../../shared/package" import { findLast } from "../../shared/array" import { supportPrompt } from "../../shared/support-prompt" import { GlobalFileNames } from "../../shared/globalFileNames" -import { safeJsonParse } from "../../shared/safeJsonParse" import type { ExtensionMessage, ExtensionState, MarketplaceInstalledMetadata } from "../../shared/ExtensionMessage" import { Mode, defaultModeSlug, getModeBySlug } from "../../shared/modes" import { experimentDefault } from "../../shared/experiments" @@ -92,8 +92,9 @@ import { Task } from "../task/Task" import { getSystemPromptFilePath } from "../prompts/sections/custom-system-prompt" import { webviewMessageHandler } from "./webviewMessageHandler" -import type { ClineMessage } from "@roo-code/types" +import type { ClineMessage, TodoItem } from "@roo-code/types" import { readApiMessages, saveApiMessages, saveTaskMessages } from "../task-persistence" +import { readTaskMessages } from "../task-persistence/taskMessages" import { getNonce } from "./getNonce" import { getUri } from "./getUri" import { REQUESTY_BASE_URL } from "../../shared/utils/requesty" @@ -145,13 +146,13 @@ export class ClineProvider private pendingOperations: Map = new Map() private static readonly PENDING_OPERATION_TIMEOUT_MS = 30000 // 30 seconds - // Transactional state posting - private uiUpdatePaused: boolean = false - private pendingState: ExtensionState | null = null + private cloudOrganizationsCache: CloudOrganizationMembership[] | null = null + private cloudOrganizationsCacheTimestamp: number | null = null + private static readonly CLOUD_ORGANIZATIONS_CACHE_DURATION_MS = 5 * 1000 // 5 seconds public isViewLaunched = false public settingsImportedAt?: number - public readonly latestAnnouncementId = "nov-2025-v3.30.0-pr-fixer" // v3.30.0 PR Fixer announcement + public readonly latestAnnouncementId = "dec-2025-v3.36.0-context-rewind-roo-provider" // v3.36.0 Context Rewind & Roo Provider Improvements public readonly providerSettingsManager: ProviderSettingsManager public readonly customModesManager: CustomModesManager @@ -483,19 +484,6 @@ export class ClineProvider return this.clineStack.map((cline) => cline.taskId) } - // Remove the current task/cline instance (at the top of the stack), so this - // task is finished and resume the previous task/cline instance (if it - // exists). - // This is used when a subtask is finished and the parent task needs to be - // resumed. - async finishSubTask(lastMessage: string) { - // Remove the last cline instance from the stack (this is the finished - // subtask). - await this.removeClineFromStack() - // Resume the last cline instance in the stack (if it exists - this is - // the 'parent' calling task). - await this.getCurrentTask()?.completeSubtask(lastMessage) - } // Pending Edit Operations Management /** @@ -791,7 +779,7 @@ export class ClineProvider webviewView.webview.html = this.contextProxy.extensionMode === vscode.ExtensionMode.Development ? await this.getHMRHtmlContent(webviewView.webview) - : this.getHtmlContent(webviewView.webview) + : await this.getHtmlContent(webviewView.webview) // Sets up an event listener to listen for messages passed from the webview view context // and executes code based on the message that is received. @@ -862,8 +850,17 @@ export class ClineProvider await this.removeClineFromStack() } - public async createTaskWithHistoryItem(historyItem: HistoryItem & { rootTask?: Task; parentTask?: Task }) { - await this.removeClineFromStack() + public async createTaskWithHistoryItem( + historyItem: HistoryItem & { rootTask?: Task; parentTask?: Task }, + options?: { startTask?: boolean }, + ) { + // Check if we're rehydrating the current task to avoid flicker + const currentTask = this.getCurrentTask() + const isRehydratingCurrentTask = currentTask && currentTask.taskId === historyItem.id + + if (!isRehydratingCurrentTask) { + await this.removeClineFromStack() + } // If the history item has a saved mode, restore it and its associated API configuration. if (historyItem.mode) { @@ -934,14 +931,52 @@ export class ClineProvider taskNumber: historyItem.number, workspacePath: historyItem.workspace, onCreated: this.taskCreationCallback, + startTask: options?.startTask ?? true, enableBridge: BridgeOrchestrator.isEnabled(cloudUserInfo, taskSyncEnabled), + // Preserve the status from the history item to avoid overwriting it when the task saves messages + initialStatus: historyItem.status, }) - await this.addClineToStack(task) + if (isRehydratingCurrentTask) { + // Replace the current task in-place to avoid UI flicker + const stackIndex = this.clineStack.length - 1 - this.log( - `[createTaskWithHistoryItem] ${task.parentTask ? "child" : "parent"} task ${task.taskId}.${task.instanceId} instantiated`, - ) + // Properly dispose of the old task to ensure garbage collection + const oldTask = this.clineStack[stackIndex] + + // Abort the old task to stop running processes and mark as abandoned + try { + await oldTask.abortTask(true) + } catch (e) { + this.log( + `[createTaskWithHistoryItem] abortTask() failed for old task ${oldTask.taskId}.${oldTask.instanceId}: ${e.message}`, + ) + } + + // Remove event listeners from the old task + const cleanupFunctions = this.taskEventListeners.get(oldTask) + if (cleanupFunctions) { + cleanupFunctions.forEach((cleanup) => cleanup()) + this.taskEventListeners.delete(oldTask) + } + + // Replace the task in the stack + this.clineStack[stackIndex] = task + task.emit(RooCodeEventName.TaskFocused) + + // Perform preparation tasks and set up event listeners + await this.performPreparationTasks(task) + + this.log( + `[createTaskWithHistoryItem] rehydrated task ${task.taskId}.${task.instanceId} in-place (flicker-free)`, + ) + } else { + await this.addClineToStack(task) + + this.log( + `[createTaskWithHistoryItem] ${task.parentTask ? "child" : "parent"} task ${task.taskId}.${task.instanceId} instantiated`, + ) + } // Check if there's a pending edit after checkpoint restoration const operationId = `task-${task.taskId}` @@ -1025,6 +1060,12 @@ export class ClineProvider const nonce = getNonce() + // Get the OpenRouter base URL from configuration + const { apiConfiguration } = await this.getState() + const openRouterBaseUrl = apiConfiguration.openRouterBaseUrl || "https://openrouter.ai" + // Extract the domain for CSP + const openRouterDomain = openRouterBaseUrl.match(/^(https?:\/\/[^\/]+)/)?.[1] || "https://openrouter.ai" + const stylesUri = getUri(webview, this.contextProxy.extensionUri, [ "webview-ui", "build", @@ -1061,7 +1102,7 @@ export class ClineProvider `img-src ${webview.cspSource} https://storage.googleapis.com https://img.clerk.com data:`, `media-src ${webview.cspSource}`, `script-src 'unsafe-eval' ${webview.cspSource} https://* https://*.posthog.com http://${localServerUrl} http://0.0.0.0:${localPort} 'nonce-${nonce}'`, - `connect-src ${webview.cspSource} https://* https://*.posthog.com ws://${localServerUrl} ws://0.0.0.0:${localPort} http://${localServerUrl} http://0.0.0.0:${localPort}`, + `connect-src ${webview.cspSource} ${openRouterDomain} https://* https://*.posthog.com ws://${localServerUrl} ws://0.0.0.0:${localPort} http://${localServerUrl} http://0.0.0.0:${localPort}`, ] return /*html*/ ` @@ -1100,7 +1141,7 @@ export class ClineProvider * @returns A template string literal containing the HTML that should be * rendered within the webview panel */ - private getHtmlContent(webview: vscode.Webview): string { + private async getHtmlContent(webview: vscode.Webview): Promise { // Get the local path to main script run in the webview, // then convert it to a uri we can use in the webview. @@ -1135,6 +1176,12 @@ export class ClineProvider */ const nonce = getNonce() + // Get the OpenRouter base URL from configuration + const { apiConfiguration } = await this.getState() + const openRouterBaseUrl = apiConfiguration.openRouterBaseUrl || "https://openrouter.ai" + // Extract the domain for CSP + const openRouterDomain = openRouterBaseUrl.match(/^(https?:\/\/[^\/]+)/)?.[1] || "https://openrouter.ai" + // Tip: Install the es6-string-html VS Code extension to enable code highlighting below return /*html*/ ` @@ -1143,7 +1190,7 @@ export class ClineProvider - +