mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-08-28 05:25:04 +00:00
* feat(bench): adding eval adapter for proactiveness on Pi-Bench * Revise README for π-Bench evaluation suite Updated the README to reflect the new project name and description. * fix(bench): refining pi-bench scripts according to cr comments * fix(bench): restore agent builtin tools in prebuilt toolkit
57 lines
3.6 KiB
Bash
57 lines
3.6 KiB
Bash
#!/bin/bash
|
|
# ═══════════════════════════════════════════════════════════════════════
|
|
# pibench evaluation suite - environment configuration template
|
|
# Usage: cp env.sh.example env.sh, then fill in the TODO items below.
|
|
# ⚠️ env.sh contains real API keys; never commit or share it
|
|
# (already excluded via .gitignore).
|
|
# ═══════════════════════════════════════════════════════════════════════
|
|
|
|
SUITE_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
|
|
# ─── TODO: π-Bench repository root ────────────────────────────────────
|
|
# Must contain src/, data/, scripts/test_server.py, third_party/appworld
|
|
# and .venv (see README setup).
|
|
export PI_BENCH_ROOT=""
|
|
|
|
# ─── ReMe repository ──────────────────────────────────────────────────
|
|
# Defaults to two levels above this directory (the layout this suite uses
|
|
# when placed at ReMe/benchmark/pibench); point it at the actual ReMe
|
|
# repository root if the suite lives elsewhere.
|
|
export REME_DIR="${REME_DIR:-$(cd "${SUITE_DIR}/../.." && pwd)}"
|
|
|
|
# ─── Base model of the agent under test (LLM used by the ReMe agent) ──
|
|
export REME_MODEL_NAME="${REME_MODEL_NAME:-qwen3.6-plus}"
|
|
|
|
# ─── LLM service endpoint (default: DashScope OpenAI-compatible; any
|
|
# OpenAI-compatible endpoint works) ────────────────────────────────
|
|
DASHSCOPE_BASE_URL="https://dashscope.aliyuncs.com/compatible-mode/v1"
|
|
export REME_LLM_BASE_URL="${REME_LLM_BASE_URL:-${DASHSCOPE_BASE_URL}}"
|
|
|
|
# ─── TODO: API keys ───────────────────────────────────────────────────
|
|
# USER_API_KEY : drives the simulated user LLM (run phase; judges whether
|
|
# hidden intents are satisfied and asks follow-ups)
|
|
# JUDGER_API_KEY: drives the judger LLM (eval phase; scores the checklist)
|
|
# The two may be identical; one strong model is recommended for both.
|
|
export USER_BASE_URL="${DASHSCOPE_BASE_URL}"
|
|
export USER_API_KEY="TODO-fill-in-user-agent-api-key"
|
|
|
|
export JUDGER_BASE_URL="${DASHSCOPE_BASE_URL}"
|
|
export JUDGER_API_KEY="TODO-fill-in-judger-api-key"
|
|
|
|
# The ReMe agent's key reuses USER_API_KEY by default (no need to repeat
|
|
# it when both use the same service and key).
|
|
export REME_LLM_API_KEY="${REME_LLM_API_KEY:-${USER_API_KEY}}"
|
|
|
|
# Brave Search (optional; used by the agent's web_search tool - use
|
|
# "dummy" when not needed).
|
|
export BRAVE_SEARCH_API_KEY="TODO-optional-brave-search-key-or-dummy"
|
|
|
|
# ─── Persistent memory workspaces (one subdirectory per persona,
|
|
# created automatically) ───────────────────────────────────────────
|
|
export REME_WORKSPACE_ROOT="${REME_WORKSPACE_ROOT:-${SUITE_DIR}/reme_workspace}"
|
|
|
|
# ─── Variables consumed by ReMe's default.yaml model config expansion;
|
|
# do not remove ────────────────────────────────────────────────────
|
|
export LLM_MODEL_NAME="${REME_MODEL_NAME}"
|
|
export LLM_BASE_URL="${REME_LLM_BASE_URL}"
|
|
export LLM_API_KEY="${REME_LLM_API_KEY}"
|