GitNexus/eval/workflow_bench/free-model.litellm.yaml
Gergo Magyar 37415cb1ca feat(eval): route skill evolution through OpenAI
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-03 10:52:37 +00:00

43 lines
1.8 KiB
YAML

# LiteLLM proxy config for running the workflow benchmark on a FREE model.
#
# The proxy exposes an Anthropic-compatible /v1/messages endpoint that
# headless Claude Code can use via ANTHROPIC_BASE_URL, routing to a model
# that costs nothing:
#
# export LITELLM_MASTER_KEY="$(openssl rand -hex 16)"
# uv run --with 'litellm[proxy]' litellm --config workflow_bench/free-model.litellm.yaml --port 4000
#
# uv run python -m workflow_bench.runner \
# --tasks workflow_bench/tasks.scenarios.yaml \
# --base-url http://localhost:4000 --anthropic-api-key "$LITELLM_MASTER_KEY" \
# --model free-coder
#
# Keep the proxy on loopback (litellm's default host). Anyone who can reach
# the port with the master key can spend the configured backend's quota.
#
# Pick ONE of the model routes below (or add your own — anything litellm
# supports works):
model_list:
# Hosted free tier: needs only a free OpenRouter account key in
# OPENROUTER_API_KEY (eval/.env). ":free" variants are rate-limited
# (~50 requests/day on a fresh account, 1000/day with a $10 balance) —
# fine for a few benchmark tasks, not for large sweeps.
- model_name: free-coder
litellm_params:
model: openrouter/qwen/qwen3-coder:free
api_key: os.environ/OPENROUTER_API_KEY
# Fully local and free: any Ollama model, no API key, no rate limits.
# Needs `ollama serve` running and the model pulled
# (`ollama pull qwen2.5-coder:14b`). Prefer a coding-tuned model that
# handles tool calls; small models follow skills less reliably.
- model_name: local-coder
litellm_params:
model: ollama_chat/qwen2.5-coder:14b
api_base: http://localhost:11434
general_settings:
# No static default — export LITELLM_MASTER_KEY before starting the proxy
# and pass the same value as --anthropic-api-key (see header).
master_key: os.environ/LITELLM_MASTER_KEY