ReMe/benchmark/pibench/env.sh.example
imrewce 3924f89bb4
feat(bench): adding eval adapter for proactiveness on Pi-Bench (#439)
* feat(bench): adding eval adapter for proactiveness on Pi-Bench

* Revise README for π-Bench evaluation suite

Updated the README to reflect the new project name and description.

* fix(bench): refining pi-bench scripts according to cr comments

* fix(bench): restore agent builtin tools in prebuilt toolkit
2026-08-11 16:37:54 +08:00

57 lines
3.6 KiB
Bash

#!/bin/bash
# ═══════════════════════════════════════════════════════════════════════
# pibench evaluation suite - environment configuration template
# Usage: cp env.sh.example env.sh, then fill in the TODO items below.
# ⚠️ env.sh contains real API keys; never commit or share it
# (already excluded via .gitignore).
# ═══════════════════════════════════════════════════════════════════════
SUITE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# ─── TODO: π-Bench repository root ────────────────────────────────────
# Must contain src/, data/, scripts/test_server.py, third_party/appworld
# and .venv (see README setup).
export PI_BENCH_ROOT=""
# ─── ReMe repository ──────────────────────────────────────────────────
# Defaults to two levels above this directory (the layout this suite uses
# when placed at ReMe/benchmark/pibench); point it at the actual ReMe
# repository root if the suite lives elsewhere.
export REME_DIR="${REME_DIR:-$(cd "${SUITE_DIR}/../.." && pwd)}"
# ─── Base model of the agent under test (LLM used by the ReMe agent) ──
export REME_MODEL_NAME="${REME_MODEL_NAME:-qwen3.6-plus}"
# ─── LLM service endpoint (default: DashScope OpenAI-compatible; any
# OpenAI-compatible endpoint works) ────────────────────────────────
DASHSCOPE_BASE_URL="https://dashscope.aliyuncs.com/compatible-mode/v1"
export REME_LLM_BASE_URL="${REME_LLM_BASE_URL:-${DASHSCOPE_BASE_URL}}"
# ─── TODO: API keys ───────────────────────────────────────────────────
# USER_API_KEY : drives the simulated user LLM (run phase; judges whether
# hidden intents are satisfied and asks follow-ups)
# JUDGER_API_KEY: drives the judger LLM (eval phase; scores the checklist)
# The two may be identical; one strong model is recommended for both.
export USER_BASE_URL="${DASHSCOPE_BASE_URL}"
export USER_API_KEY="TODO-fill-in-user-agent-api-key"
export JUDGER_BASE_URL="${DASHSCOPE_BASE_URL}"
export JUDGER_API_KEY="TODO-fill-in-judger-api-key"
# The ReMe agent's key reuses USER_API_KEY by default (no need to repeat
# it when both use the same service and key).
export REME_LLM_API_KEY="${REME_LLM_API_KEY:-${USER_API_KEY}}"
# Brave Search (optional; used by the agent's web_search tool - use
# "dummy" when not needed).
export BRAVE_SEARCH_API_KEY="TODO-optional-brave-search-key-or-dummy"
# ─── Persistent memory workspaces (one subdirectory per persona,
# created automatically) ───────────────────────────────────────────
export REME_WORKSPACE_ROOT="${REME_WORKSPACE_ROOT:-${SUITE_DIR}/reme_workspace}"
# ─── Variables consumed by ReMe's default.yaml model config expansion;
# do not remove ────────────────────────────────────────────────────
export LLM_MODEL_NAME="${REME_MODEL_NAME}"
export LLM_BASE_URL="${REME_LLM_BASE_URL}"
export LLM_API_KEY="${REME_LLM_API_KEY}"