Files
SkillCompiler/scripts/evaluate/run-raw-task.sh
T
2026-09-04 14:58:42 +08:00

807 lines
32 KiB
Bash

#!/usr/bin/env bash
# Thin compatibility wrapper around the official SkillsBench/BenchFlow runner.
set -euo pipefail
PROJECT_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)
cd "$PROJECT_ROOT"
# Evaluation credentials are user-owned configuration. Load them once for
# BenchFlow so its isolated agent container receives only the registered
# provider variables, never a copied host config directory.
if [ -f "$PROJECT_ROOT/.env" ]; then
set -a
# shellcheck disable=SC1091
source "$PROJECT_ROOT/.env"
set +a
fi
# OpenCode's OpenAI-compatible provider expects a base URL, while the project
# .env records the concrete chat-completions endpoint used by other harnesses.
if [ -n "${SILICONFLOW_CHAT_COMPLETIONS_URL:-}" ]; then
SILICONFLOW_BASE_URL=${SILICONFLOW_CHAT_COMPLETIONS_URL%/chat/completions}
fi
: "${SILICONFLOW_BASE_URL:=https://api.siliconflow.cn/v1}"
export SILICONFLOW_BASE_URL
usage() {
cat <<'EOF'
Usage:
bash scripts/evaluate/run-raw-task.sh \
--harness opencode \
--model <provider/model> \
--task <SkillsBench task directory> \
--skill-source <skill directory or skills root> \
--output <result directory> \
[--require-skill] \
[--repeat N] [--max-parallel N] [--image <local-image-ref>]
This wrapper delegates every attempt to `bench eval run --sandbox docker`.
It does not inject verifier files or implement scoring.
Provider credentials are loaded from the project `.env`.
Each repeat is allocated before any workers start, at:
<output>/test-NNN/
The official BenchFlow job and summary remain inside that test directory.
Interactive terminals show one in-place progress row per attempt for both
single and concurrent runs. Worker output is kept in test-NNN/console.log and
only a concise result summary is printed after the progress display finishes.
External Skill sources are staged into each disposable task copy's
environment/skills directory. This keeps BenchFlow's task-bundled deployment
and subagent registration policy identical for source and compiled Skills.
Both <output>/skills/<name>/SKILL.md and <skills-root>/<name>/SKILL.md layouts
are accepted.
--image makes an ephemeral copy of the selected task and applies the official
SkillsBench prebuilt-image policy to that copy. The original task and Skills
remain untouched. This avoids a Docker build when the supplied image already
exists locally.
Without --image, the wrapper uses skillc/<task-name>:local. It builds and tags
that image only when needed, then adds the pinned Node/OpenCode runtime and
cleans installation caches. Later evaluations reuse the same single image
through the official prebuilt-image policy.
--require-skill makes at least one Skill invocation an evaluation invariant
rather than an agent choice. Every temporary task prompt requires the agent to
load a relevant Skill during the task, immediately before applying its guidance.
The completed ACP trajectory is checked for a successful Skill call. Without
this flag, Skill invocation remains the agent's choice. An attempt with no
successful Skill call exits with status 86 and writes required-skill.json.
EOF
}
HARNESS=""
MODEL_REF=""
TASK_DIR=""
SKILL_SOURCE=""
OUTPUT_DIR=""
REPEAT=1
MAX_PARALLEL=3
REPEAT_SEEN=false
MAX_PARALLEL_SEEN=false
PREBUILT_IMAGE=""
REQUIRE_SKILLS=false
while [ "$#" -gt 0 ]; do
case "$1" in
--harness) HARNESS=${2:?missing value for --harness}; shift 2 ;;
--model) MODEL_REF=${2:?missing value for --model}; shift 2 ;;
--task) TASK_DIR=${2:?missing value for --task}; shift 2 ;;
--skill-source) SKILL_SOURCE=${2:?missing value for --skill-source}; shift 2 ;;
--output) OUTPUT_DIR=${2:?missing value for --output}; shift 2 ;;
--require-skill) REQUIRE_SKILLS=true; shift ;;
--repeat)
[ "$REPEAT_SEEN" = false ] || { printf '%s\n' 'Duplicate --repeat option.' >&2; exit 2; }
REPEAT=${2:?missing value for --repeat}
REPEAT_SEEN=true
shift 2
;;
--max-parallel)
[ "$MAX_PARALLEL_SEEN" = false ] || { printf '%s\n' 'Duplicate --max-parallel option.' >&2; exit 2; }
MAX_PARALLEL=${2:?missing value for --max-parallel}
MAX_PARALLEL_SEEN=true
shift 2
;;
--image|--prebuilt-image) PREBUILT_IMAGE=${2:?missing value for --image}; shift 2 ;;
-h|--help) usage; exit 0 ;;
*) printf 'Unsupported option for the official BenchFlow wrapper: %s\n' "$1" >&2; usage >&2; exit 2 ;;
esac
done
[ -n "$HARNESS" ] || { printf '%s\n' '--harness is required.' >&2; exit 2; }
[ -n "$MODEL_REF" ] || { printf '%s\n' '--model is required.' >&2; exit 2; }
[ -n "$TASK_DIR" ] || { printf '%s\n' '--task is required.' >&2; exit 2; }
[ -n "$SKILL_SOURCE" ] || { printf '%s\n' '--skill-source is required.' >&2; exit 2; }
[ -n "$OUTPUT_DIR" ] || { printf '%s\n' '--output is required.' >&2; exit 2; }
[[ "$REPEAT" =~ ^[1-9][0-9]*$ ]] || { printf '%s\n' '--repeat must be a positive integer.' >&2; exit 2; }
[[ "$MAX_PARALLEL" =~ ^[1-9][0-9]*$ ]] || { printf '%s\n' '--max-parallel must be a positive integer.' >&2; exit 2; }
TASK_DIR=$(realpath "$TASK_DIR")
SKILL_SOURCE=$(realpath "$SKILL_SOURCE")
OUTPUT_DIR=$(realpath -m "$OUTPUT_DIR")
[ -f "$TASK_DIR/task.md" ] || { printf 'Not a native SkillsBench task: %s\n' "$TASK_DIR" >&2; exit 2; }
[ -d "$SKILL_SOURCE" ] || { printf 'Skill source does not exist: %s\n' "$SKILL_SOURCE" >&2; exit 2; }
# Compilers may emit either a Skills root directly:
# <output>/skill-a/SKILL.md
# or a package containing that root:
# <output>/skills/skill-a/SKILL.md
# Normalize both forms before staging the selected Skills into a task copy.
SKILL_PAYLOAD_DIR="$SKILL_SOURCE"
if [ ! -f "$SKILL_SOURCE/SKILL.md" ] && \
[ -d "$SKILL_SOURCE/skills" ] && \
[ -z "$(find "$SKILL_SOURCE" -mindepth 2 -maxdepth 2 -type f -name SKILL.md -print -quit)" ] && \
[ -n "$(find "$SKILL_SOURCE/skills" -type f -name SKILL.md -print -quit)" ]; then
SKILL_PAYLOAD_DIR=$(realpath "$SKILL_SOURCE/skills")
fi
if [ "$REQUIRE_SKILLS" = true ]; then
[ -n "$(find "$SKILL_SOURCE" -type f -name SKILL.md -print -quit)" ] || {
printf 'No SKILL.md files were found under --skill-source: %s\n' "$SKILL_SOURCE" >&2
exit 2
}
fi
TASK_BUNDLED_SKILLS="$TASK_DIR/environment/skills"
SKILL_SOURCE_IS_TASK_BUNDLED=false
if [ -d "$TASK_BUNDLED_SKILLS" ] && [ "$SKILL_SOURCE" = "$(realpath "$TASK_BUNDLED_SKILLS")" ]; then
SKILL_SOURCE_IS_TASK_BUNDLED=true
fi
case "$HARNESS" in
opencode)
AGENT=opencode
# This project also keeps OPENCODE_* variables for the legacy harness.
# They are OpenCode-hosted-service credentials, not SiliconFlow
# credentials, and current OpenCode releases may prefer them over a
# configured custom provider. The official BenchFlow container must use
# only the explicitly registered SiliconFlow provider below.
unset OPENCODE_API_KEY OPENCODE_CHAT_COMPLETIONS_URL
# BenchFlow intentionally inherits only its built-in provider variables.
# SiliconFlow is an OpenAI-compatible custom provider, so its two values
# are supplied through the official evaluation config below. They
# originate in .env; no task, Skill, or verifier file is changed.
[ -n "${SILICONFLOW_API_KEY:-}" ] || {
printf '%s\n' 'SILICONFLOW_API_KEY is required in .env for --harness opencode.' >&2
exit 2
}
# Values are placed in a mode-0600 temporary BenchFlow config per attempt.
# Credentials must not appear in CLI arguments visible through `ps`.
;;
claude-code|claude) AGENT=claude-agent-acp ;;
*)
printf 'Harness %q is not a registered BenchFlow ACP agent in the pinned SkillsBench runner.\n' "$HARNESS" >&2
printf '%s\n' 'Use an official BenchFlow agent name (for example: opencode or claude-agent-acp).' >&2
exit 2
;;
esac
SKILLSBENCH_ROOT=${SKILLSBENCH_ROOT:-"$PROJECT_ROOT/data/skills-bench"}
if command -v bench >/dev/null 2>&1; then
BENCH=(bench)
elif command -v benchflow >/dev/null 2>&1; then
BENCH=(benchflow)
elif [ -x "$SKILLSBENCH_ROOT/.venv/bin/bench" ]; then
# `uv sync` keeps the official CLI inside the dataset virtual environment.
BENCH=("$SKILLSBENCH_ROOT/.venv/bin/bench")
elif command -v uv >/dev/null 2>&1 && [ -f "$SKILLSBENCH_ROOT/pyproject.toml" ]; then
BENCH=(uv run --directory "$SKILLSBENCH_ROOT" bench)
else
printf '%s\n' 'BenchFlow is required but neither `bench` nor `benchflow` is on PATH.' >&2
printf 'Run: cd %s && uv sync --locked\n' "$SKILLSBENCH_ROOT" >&2
exit 127
fi
TASK_NAME=$(basename "$TASK_DIR")
IMAGE_LOCK_SCOPE=$(
printf '%s' "$HARNESS-$MODEL_REF" \
| tr '[:upper:]' '[:lower:]' \
| sed -E 's#[^a-z0-9]+#-#g; s#^-+##; s#-+$##'
)
TASK_IMAGE_LOCK_DIR="${TMPDIR:-/tmp}/skill-agent-evaluate-image-locks/$IMAGE_LOCK_SCOPE/$TASK_NAME"
TASK_RESULTS_DIR="$OUTPUT_DIR"
ensure_result_directory() {
local directory=$1 mkdir_error
[ -d "$directory" ] && return 0
if mkdir_error=$(mkdir -p "$directory" 2>&1); then
return 0
fi
# On a Windows-backed /mnt filesystem, deleting a directory while a Windows,
# WSL, or Docker process still has it open can leave a delete-pending name.
# The entry is invisible to stat/find, but NTFS rejects recreating the same
# name with "Already exists". Report that state explicitly; silently using
# a different category would put results under the wrong Skill provenance.
if [ ! -e "$directory" ] && [[ "$mkdir_error" == *"Already exists"* || "$mkdir_error" == *"File exists"* ]]; then
cat >&2 <<EOF
Cannot create the result category directory because NTFS still reserves its deleted name:
$directory
No evaluation was started and no existing result was changed.
Close Explorer/editors or other processes holding this path. If it remains blocked,
run "wsl --shutdown" in Windows PowerShell or Command Prompt, reopen WSL, and retry.
This releases the stale handle; it does not delete Docker images or result data.
EOF
exit 73
fi
printf 'Unable to create result directory %s: %s\n' "$directory" "$mkdir_error" >&2
exit 73
}
ensure_result_directory "$TASK_RESULTS_DIR"
SETUP_LOG="$TASK_RESULTS_DIR/setup.log"
touch "$SETUP_LOG"
# The wrapper owns the only terminal display. BenchFlow worker output always
# goes to per-attempt logs so single and concurrent runs behave identically.
DASHBOARD_ENABLED=false
if [ -t 1 ] && [ "${TERM:-dumb}" != dumb ]; then
DASHBOARD_ENABLED=true
fi
# Reserve all test numbers before starting any background workers. `flock`
# also keeps two separately launched wrapper processes from selecting the same
# test-NNN directory.
TEST_DIRS=()
allocate_test_dirs() {
local max_test=0 name number next_test candidate reserved=0 mkdir_error
exec 9>"$TASK_RESULTS_DIR/.test-number.lock"
flock 9
while IFS= read -r name; do
if [[ "$name" =~ ^test-([0-9]+)$ ]]; then
number=$((10#${BASH_REMATCH[1]}))
(( number > max_test )) && max_test=$number
fi
done < <(find "$TASK_RESULTS_DIR" -mindepth 1 -maxdepth 1 -type d -printf '%f\n')
next_test=$((max_test + 1))
while [ "$reserved" -lt "$REPEAT" ]; do
printf -v name 'test-%03d' "$next_test"
candidate="$TASK_RESULTS_DIR/$name"
mkdir_error=''
if mkdir_error=$(mkdir "$candidate" 2>&1); then
TEST_DIRS+=("$candidate")
reserved=$((reserved + 1))
elif [ -d "$candidate" ] || [[ "$mkdir_error" == *'Already exists'* ]] || [[ "$mkdir_error" == *'File exists'* ]]; then
# On drvfs (/mnt/c), a recently deleted Windows directory can remain in
# delete-pending state: stat/find cannot see it, but mkdir still reports
# Already exists. Treat that ghost name exactly like a live collision.
:
else
printf 'Unable to reserve result directory %s: %s\n' \
"$candidate" "$mkdir_error" >&2
flock -u 9
exec 9>&-
return 1
fi
# An existing directory is already reserved by another run (or was
# recreated by a still-finishing old run). Skip it atomically.
next_test=$((next_test + 1))
done
flock -u 9
exec 9>&-
}
allocate_test_dirs
if ! command -v docker >/dev/null 2>&1; then
printf '%s\n' 'Docker is required for the official BenchFlow Docker sandbox.' >&2
exit 127
fi
[ -x "$SKILLSBENCH_ROOT/.venv/bin/python" ] || {
printf 'The official SkillsBench Python environment is required: %s\n' "$SKILLSBENCH_ROOT/.venv/bin/python" >&2
exit 127
}
if [ -z "$PREBUILT_IMAGE" ]; then
PREBUILT_IMAGE="skillc/$TASK_NAME:local"
mkdir -p "$TASK_IMAGE_LOCK_DIR"
exec 8>"$TASK_IMAGE_LOCK_DIR/.image-build.lock"
flock 8
DOCKERFILE="$TASK_DIR/environment/Dockerfile"
[ -f "$DOCKERFILE" ] || {
printf 'Task Dockerfile is missing: %s\n' "$DOCKERFILE" >&2
exit 2
}
# Rebuild when any task environment input changes. Previously an existing
# tag was reused forever, which could preserve a stale task image even after
# its Dockerfile or fixtures were fixed.
TASK_ENVIRONMENT_FINGERPRINT=$(
find "$TASK_DIR/environment" -type f -print0 \
| sort -z \
| xargs -0 sha256sum \
| sha256sum \
| awk '{print $1}'
)
task_environment_label=$(docker image inspect --format \
'{{ index .Config.Labels "org.skillc.task-environment" }}' \
"$PREBUILT_IMAGE" 2>/dev/null || true)
if ! docker image inspect "$PREBUILT_IMAGE" >/dev/null 2>&1 || \
[ "$task_environment_label" != "$TASK_ENVIRONMENT_FINGERPRINT" ]; then
printf 'Building reusable task image: %s\n' "$PREBUILT_IMAGE" >>"$SETUP_LOG"
# BenchFlow builds every task Dockerfile with environment/ as its context.
# Input fixtures referenced by COPY therefore live in that directory.
docker build \
--file "$DOCKERFILE" \
--build-arg PIP_INDEX_URL=https://mirrors.aliyun.com/pypi/simple \
--label "org.skillc.task-environment=$TASK_ENVIRONMENT_FINGERPRINT" \
--tag "$PREBUILT_IMAGE" \
"$TASK_DIR/environment" >>"$SETUP_LOG" 2>&1
fi
OPENCODE_RUNTIME_VERSION=1.18.16
OPENCODE_RUNTIME_BUILD=2
NODE_RUNTIME_VERSION=22.20.0
RUNTIME_DOCKERFILE="$PROJECT_ROOT/scripts/evaluate/docker/opencode-runtime.Dockerfile"
RUNTIME_SOURCE_FINGERPRINT=$(sha256sum "$RUNTIME_DOCKERFILE" | awk '{print $1}')
runtime_label=$(docker image inspect --format \
'{{ index .Config.Labels "org.skillc.opencode-runtime" }}' \
"$PREBUILT_IMAGE" 2>/dev/null || true)
runtime_build_label=$(docker image inspect --format \
'{{ index .Config.Labels "org.skillc.opencode-runtime-build" }}' \
"$PREBUILT_IMAGE" 2>/dev/null || true)
runtime_source_label=$(docker image inspect --format \
'{{ index .Config.Labels "org.skillc.opencode-runtime-source" }}' \
"$PREBUILT_IMAGE" 2>/dev/null || true)
if [ "$runtime_label" != "$OPENCODE_RUNTIME_VERSION" ] || \
[ "$runtime_build_label" != "$OPENCODE_RUNTIME_BUILD" ] || \
[ "$runtime_source_label" != "$RUNTIME_SOURCE_FINGERPRINT" ]; then
printf 'Adding reusable Node %s + OpenCode %s runtime to: %s\n' \
"$NODE_RUNTIME_VERSION" "$OPENCODE_RUNTIME_VERSION" "$PREBUILT_IMAGE" >>"$SETUP_LOG"
docker build \
--file "$RUNTIME_DOCKERFILE" \
--build-arg "TASK_IMAGE=$PREBUILT_IMAGE" \
--build-arg "NODE_VERSION=$NODE_RUNTIME_VERSION" \
--build-arg "OPENCODE_VERSION=$OPENCODE_RUNTIME_VERSION" \
--build-arg "RUNTIME_BUILD_REVISION=$OPENCODE_RUNTIME_BUILD" \
--build-arg "RUNTIME_SOURCE_FINGERPRINT=$RUNTIME_SOURCE_FINGERPRINT" \
--tag "$PREBUILT_IMAGE" \
"$PROJECT_ROOT/scripts/evaluate/docker" >>"$SETUP_LOG" 2>&1
else
printf 'Reusing task image with preinstalled OpenCode: %s\n' "$PREBUILT_IMAGE" >>"$SETUP_LOG"
fi
flock -u 8
exec 8>&-
elif ! docker image inspect "$PREBUILT_IMAGE" >/dev/null 2>&1; then
printf 'The requested local prebuilt image is unavailable: %s\n' "$PREBUILT_IMAGE" >&2
printf '%s\n' 'Check it with: docker image inspect <image-ref>' >&2
exit 2
fi
run_official_attempt() {
# This function always runs as a background worker. Do not inherit the
# outer wrapper's EXIT guard: a normally finishing worker must never treat
# its concurrently running siblings as orphaned processes.
trap - EXIT
local attempt=$1
local jobs_dir="${TEST_DIRS[$((attempt - 1))]}"
local task_dir_for_attempt="$TASK_DIR"
local skills_dir_for_attempt="$SKILL_SOURCE"
local temporary_task_root=""
local eval_config=""
local console_log="$jobs_dir/console.log"
local bundled_skills_dir=""
local single_skill_dir=""
if [ -n "$PREBUILT_IMAGE" ]; then
temporary_task_root=$(mktemp -d "${TMPDIR:-/tmp}/skillsbench-prebuilt-task.XXXXXX")
trap '[ -z "$temporary_task_root" ] || rm -rf -- "$temporary_task_root"' RETURN
task_dir_for_attempt="$temporary_task_root/$(basename "$TASK_DIR")"
cp -a "$TASK_DIR" "$task_dir_for_attempt"
# This is the same helper used by the official SkillsBench AgentBeats worker.
PYTHONPATH="$SKILLSBENCH_ROOT" "$SKILLSBENCH_ROOT/.venv/bin/python" -c \
'import sys; from pathlib import Path; from skillsbench_agentbeats.worker import _write_task_md_prebuilt_image; _write_task_md_prebuilt_image(Path(sys.argv[1]), sys.argv[2])' \
"$task_dir_for_attempt/task.md" "$PREBUILT_IMAGE"
fi
# BenchFlow gives task-bundled Skills and external custom-runtime Skills
# different deployment policies. Some task Skills invoke a same-named
# subagent (for example enterprise-artifact-search); loading their SKILL.md
# externally exposes the instructions but does not register that agent type.
# Stage external compiler output into this disposable task copy so source and
# treatment runs differ only in Skill contents, not in deployment policy.
if [ "$SKILL_SOURCE_IS_TASK_BUNDLED" = false ]; then
if [ -z "$temporary_task_root" ]; then
printf '%s\n' 'External Skill staging requires a temporary task copy.' >&2
return 2
fi
bundled_skills_dir="$task_dir_for_attempt/environment/skills"
case "$bundled_skills_dir/" in
"$temporary_task_root/"*) ;;
*)
printf 'Refusing to stage external Skills outside the temporary task root: %s\n' \
"$bundled_skills_dir" >&2
return 2
;;
esac
mkdir -p "$bundled_skills_dir"
find "$bundled_skills_dir" -mindepth 1 -maxdepth 1 -exec rm -rf -- {} +
if [ -f "$SKILL_PAYLOAD_DIR/SKILL.md" ]; then
single_skill_dir="$bundled_skills_dir/$(basename "$SKILL_PAYLOAD_DIR")"
mkdir -p "$single_skill_dir"
cp -a "$SKILL_PAYLOAD_DIR/." "$single_skill_dir/"
else
cp -a "$SKILL_PAYLOAD_DIR/." "$bundled_skills_dir/"
fi
skills_dir_for_attempt="$bundled_skills_dir"
fi
if [ "$REQUIRE_SKILLS" = true ]; then
{
printf '\n\n## Mandatory Skill requirement\n\n'
printf 'You MUST invoke a relevant Skill tool before final verification. Do NOT load Skills at the beginning. First inspect the task and project, then invoke the Skill immediately before performing the work it covers and apply its guidance. Merely mentioning a Skill or reading files directly does not satisfy this requirement.\n'
} >> "$task_dir_for_attempt/task.md"
fi
# A source task Skill must also be resolved relative to the task copy.
if [ "$SKILL_SOURCE_IS_TASK_BUNDLED" = true ]; then
skills_dir_for_attempt="$task_dir_for_attempt/environment/skills"
fi
# JSON is valid YAML and keeps credentials out of the process command line
# while still using the official `bench eval run --config` entrypoint.
eval_config="$temporary_task_root/benchflow-eval.json"
umask 077
BF_CONFIG_PATH="$eval_config" BF_TASKS_DIR="$task_dir_for_attempt" \
BF_JOBS_DIR="$jobs_dir" BF_AGENT="$AGENT" BF_MODEL="$MODEL_REF" \
BF_SKILLS_DIR="$skills_dir_for_attempt" \
BF_SILICONFLOW_API_KEY="${SILICONFLOW_API_KEY:-}" \
BF_SILICONFLOW_BASE_URL="${SILICONFLOW_BASE_URL:-}" \
"$SKILLSBENCH_ROOT/.venv/bin/python" -c '
import json
import os
from pathlib import Path
agent_env = {}
if os.environ.get("BF_AGENT") == "opencode":
provider, model = os.environ["BF_MODEL"].split("/", 1)
opencode_config = {
"$schema": "https://opencode.ai/config.json",
"model": os.environ["BF_MODEL"],
"small_model": os.environ["BF_MODEL"],
"provider": {
provider: {
"npm": "@ai-sdk/openai-compatible",
"name": provider,
"options": {
"baseURL": os.environ["BF_SILICONFLOW_BASE_URL"],
"apiKey": "{env:SILICONFLOW_API_KEY}",
"timeout": 600000,
},
"models": {model: {"name": model}},
}
},
}
agent_env = {
"SILICONFLOW_API_KEY": os.environ["BF_SILICONFLOW_API_KEY"],
"SILICONFLOW_BASE_URL": os.environ["BF_SILICONFLOW_BASE_URL"],
"OPENCODE_CONFIG_CONTENT": json.dumps(opencode_config, separators=(",", ":")),
}
config = {
"tasks_dir": os.environ["BF_TASKS_DIR"],
"jobs_dir": os.environ["BF_JOBS_DIR"],
"agent": os.environ["BF_AGENT"],
"model": os.environ["BF_MODEL"],
"environment": "docker",
"skills_dir": os.environ["BF_SKILLS_DIR"],
"skill_mode": "with-skill",
"agent_env": agent_env,
# Do not let BenchFlow silently substitute its default non-root "agent"
# user. SkillsBench task images may intentionally provision task tools
# (for example the SDKMAN Maven installation) in /root; the agent must see
# the same
# task-provided toolchain as the verifier. JSON null is the BenchFlow
# documented root/no-lockdown sentinel. This changes only the evaluation
# process inside the task container, never the dataset image.
"sandbox_user": None,
# A verifier timeout occurs after the agent rollout has finished. Retrying
# it would sample a new agent trajectory and confound experiment results;
# preserve the timeout as a terminal evaluation-infrastructure outcome.
"retry": {"retry_on_verifier_infra": False},
}
Path(os.environ["BF_CONFIG_PATH"]).write_text(json.dumps(config), encoding="utf-8")
'
PYTHONPATH="$PROJECT_ROOT/scripts/evaluate${PYTHONPATH:+:$PYTHONPATH}" \
"${BENCH[@]}" eval run --config "$eval_config" >"$console_log" 2>&1
if [ "$REQUIRE_SKILLS" = true ]; then
if ! node "$PROJECT_ROOT/scripts/evaluate/verify-required-skill.mjs" \
"$jobs_dir" >>"$console_log" 2>&1; then
printf 'Required Skill validation failed for %s: no successful Skill invocation was found.\n' \
"$(basename "$jobs_dir")" >>"$console_log"
return 86
fi
fi
}
attempt=1
failures=0
WORKER_PIDS=()
FAILED_ATTEMPTS=()
FAILED_LOGS=()
declare -a ATTEMPT_STATUS ATTEMPT_STARTED ATTEMPT_ENDED ATTEMPT_REAPED
DASHBOARD_RENDERED=false
DASHBOARD_LINE_COUNT=$REPEAT
for ((dashboard_attempt = 1; dashboard_attempt <= REPEAT; dashboard_attempt++)); do
ATTEMPT_STATUS[$dashboard_attempt]=queued
ATTEMPT_STARTED[$dashboard_attempt]=0
ATTEMPT_ENDED[$dashboard_attempt]=0
ATTEMPT_REAPED[$dashboard_attempt]=false
done
format_elapsed() {
local seconds=$1
printf '%02d:%02d' "$((seconds / 60))" "$((seconds % 60))"
}
attempt_stage() {
local log=$1
if [ ! -s "$log" ]; then
printf '%s' preparing
elif rg -q 'Running verifier|Verifier running' "$log"; then
printf '%s' verifying
elif rg -q 'end_turn|Process terminated|Agent finished' "$log"; then
printf '%s' 'agent finalizing'
elif rg -q 'Prompt [0-9]+/[0-9]+' "$log"; then
printf '%s' 'agent running'
elif rg -q 'ACP agent:|Session:' "$log"; then
printf '%s' 'agent connecting'
elif rg -q 'Deploying skills|Skills deployed' "$log"; then
printf '%s' 'deploying skills'
elif rg -q 'Installing opencode' "$log"; then
printf '%s' 'installing agent'
elif rg -q 'Starting environment' "$log"; then
printf '%s' 'starting container'
else
printf '%s' preparing
fi
}
stage_progress() {
case "$1" in
preparing) printf '%d' 5 ;;
'starting container') printf '%d' 12 ;;
'installing agent') printf '%d' 20 ;;
'deploying skills') printf '%d' 30 ;;
'agent connecting') printf '%d' 40 ;;
'agent running') printf '%d' 70 ;;
'agent finalizing') printf '%d' 82 ;;
verifying) printf '%d' 92 ;;
finished) printf '%d' 100 ;;
*) printf '%d' 0 ;;
esac
}
attempt_outcome() {
local jobs_dir process_rc summary
jobs_dir=$1
process_rc=$2
summary="$jobs_dir/summary.json"
if [ "$process_rc" -ne 0 ]; then
printf '%s' ERROR
elif [ ! -f "$summary" ]; then
printf '%s' COMPLETE
else
"$SKILLSBENCH_ROOT/.venv/bin/python" - "$summary" <<'PY'
import json
import sys
data = json.load(open(sys.argv[1], encoding="utf-8"))
if int(data.get("errored", 0) or 0) or int(data.get("verifier_errored", 0) or 0):
print("ERROR", end="")
elif int(data.get("passed", data.get("pass", 0)) or 0):
print("PASS", end="")
else:
print("FAIL", end="")
PY
fi
}
render_dashboard() {
local now dashboard_attempt status elapsed stage test_name progress
local bar_done bar_left done_chars left_chars elapsed_end
local frame='' cursor_prefix=''
now=$(date +%s)
for ((dashboard_attempt = 1; dashboard_attempt <= REPEAT; dashboard_attempt++)); do
status=${ATTEMPT_STATUS[$dashboard_attempt]}
test_name="$(basename "${TEST_DIRS[$((dashboard_attempt - 1))]}") ($dashboard_attempt/$REPEAT)"
if [ "${ATTEMPT_STARTED[$dashboard_attempt]}" -gt 0 ]; then
elapsed_end=$now
[ "${ATTEMPT_ENDED[$dashboard_attempt]}" -eq 0 ] || elapsed_end=${ATTEMPT_ENDED[$dashboard_attempt]}
elapsed=$(format_elapsed "$((elapsed_end - ATTEMPT_STARTED[$dashboard_attempt]))")
else
elapsed='--:--'
fi
if [ "$status" = running ]; then
stage=$(attempt_stage "${TEST_DIRS[$((dashboard_attempt - 1))]}/console.log")
elif [ "$status" = queued ]; then
stage=waiting
else
stage=finished
fi
progress=$(stage_progress "$stage")
bar_done=$((progress * 20 / 100))
bar_left=$((20 - bar_done))
printf -v done_chars '%*s' "$bar_done" ''
printf -v left_chars '%*s' "$bar_left" ''
done_chars=${done_chars// /#}
left_chars=${left_chars// /-}
if [ "$stage" = finished ]; then
stage=$status
fi
printf -v frame '%s%-16s [%s%s] %3d%% %-16s elapsed=%s\033[K\n' \
"$frame" "$test_name" "$done_chars" "$left_chars" "$progress" "$stage" "$elapsed"
done
if [ "$DASHBOARD_RENDERED" = true ]; then
printf -v cursor_prefix '\033[%dA\r' "$DASHBOARD_LINE_COUNT"
fi
# DEC mode 2026 asks supporting terminals (including current xterm.js) to
# present the cursor move + complete frame atomically. Unsupported terminals
# ignore it and still receive one assembled write, without a blanking pass.
printf '\033[?2026h%s%s\033[?2026l' "$cursor_prefix" "$frame"
DASHBOARD_RENDERED=true
}
monitor_dashboard_batch() {
local remaining=${#pids[@]} index pid current_attempt current_log process_rc
printf '\033[?25l'
while [ "$remaining" -gt 0 ]; do
for index in "${!pids[@]}"; do
current_attempt=${batch_attempts[$index]}
[ "${ATTEMPT_REAPED[$current_attempt]}" = false ] || continue
pid=${pids[$index]}
if ! kill -0 "$pid" 2>/dev/null; then
process_rc=0
wait "$pid" || process_rc=$?
ATTEMPT_REAPED[$current_attempt]=true
ATTEMPT_ENDED[$current_attempt]=$(date +%s)
current_log=${batch_logs[$index]}
ATTEMPT_STATUS[$current_attempt]=$(attempt_outcome \
"${TEST_DIRS[$((current_attempt - 1))]}" "$process_rc")
if [ "$process_rc" -ne 0 ]; then
cleanup_owned_containers_for_jobs_dir "${TEST_DIRS[$((current_attempt - 1))]}"
failures=$((failures + 1))
FAILED_ATTEMPTS+=("$current_attempt")
FAILED_LOGS+=("$current_log")
fi
remaining=$((remaining - 1))
fi
done
render_dashboard
[ "$remaining" -eq 0 ] || sleep 1
done
printf '\033[?25h'
}
cleanup_owned_containers_for_jobs_dir() {
local jobs_dir=$1 container_id mount_source matched
while IFS= read -r container_id; do
[ -n "$container_id" ] || continue
matched=false
while IFS= read -r mount_source; do
case "$mount_source/" in
"$jobs_dir/"*) matched=true; break ;;
esac
done < <(docker inspect --format '{{range .Mounts}}{{println .Source}}{{end}}' "$container_id" 2>/dev/null || true)
if [ "$matched" = true ]; then
printf 'Cleaning orphaned BenchFlow container bound to %s: %s\n' \
"$jobs_dir" "$container_id" >>"$SETUP_LOG"
docker rm -f "$container_id" >/dev/null 2>&1 || true
fi
done < <(docker ps -aq --filter label=benchflow.owned=true)
}
terminate_workers() {
local signal=$1 pid jobs_dir
trap - EXIT INT TERM
for pid in "${WORKER_PIDS[@]:-}"; do
terminate_process_tree "$pid"
done
for pid in "${WORKER_PIDS[@]:-}"; do
wait "$pid" 2>/dev/null || true
done
for jobs_dir in "${TEST_DIRS[@]}"; do
cleanup_owned_containers_for_jobs_dir "$jobs_dir"
done
[ "$DASHBOARD_ENABLED" = false ] || printf '\033[?25h'
printf '\nStopped concurrent attempts after %s; completed artifacts remain in their test directories.\n' "$signal" >&2
exit 130
}
terminate_process_tree() {
local root_pid=$1 child_pid
while IFS= read -r child_pid; do
[ -n "$child_pid" ] && terminate_process_tree "$child_pid"
done < <(pgrep -P "$root_pid" 2>/dev/null || true)
kill -TERM "$root_pid" 2>/dev/null || true
}
cleanup_workers_on_exit() {
local exit_code=$? pid jobs_dir active_workers=0
trap - EXIT INT TERM
[ "$DASHBOARD_ENABLED" = false ] || printf '\033[?25h'
for pid in "${WORKER_PIDS[@]:-}"; do
if kill -0 "$pid" 2>/dev/null; then
active_workers=$((active_workers + 1))
terminate_process_tree "$pid"
fi
done
for pid in "${WORKER_PIDS[@]:-}"; do
wait "$pid" 2>/dev/null || true
done
if [ "$active_workers" -gt 0 ]; then
for jobs_dir in "${TEST_DIRS[@]}"; do
cleanup_owned_containers_for_jobs_dir "$jobs_dir"
done
fi
if [ "$active_workers" -gt 0 ]; then
printf '\nWrapper exited unexpectedly (code %d); stopped %d active worker(s) to prevent orphaned evaluations.\n' \
"$exit_code" "$active_workers" >&2
fi
exit "$exit_code"
}
trap cleanup_workers_on_exit EXIT
trap 'terminate_workers SIGINT' INT
trap 'terminate_workers SIGTERM' TERM
while [ "$attempt" -le "$REPEAT" ]; do
batch_last=$((attempt + MAX_PARALLEL - 1))
[ "$batch_last" -le "$REPEAT" ] || batch_last=$REPEAT
pids=()
batch_attempts=()
batch_logs=()
for current_attempt in $(seq "$attempt" "$batch_last"); do
current_jobs_dir="${TEST_DIRS[$((current_attempt - 1))]}"
current_log="$current_jobs_dir/console.log"
ATTEMPT_STATUS[$current_attempt]=running
ATTEMPT_STARTED[$current_attempt]=$(date +%s)
run_official_attempt "$current_attempt" &
pids+=("$!")
WORKER_PIDS+=("$!")
batch_attempts+=("$current_attempt")
batch_logs+=("$current_log")
done
if [ "$DASHBOARD_ENABLED" = true ]; then
monitor_dashboard_batch
else
for index in "${!pids[@]}"; do
pid="${pids[$index]}"
current_attempt="${batch_attempts[$index]}"
current_log="${batch_logs[$index]}"
process_rc=0
wait "$pid" || process_rc=$?
ATTEMPT_ENDED[$current_attempt]=$(date +%s)
ATTEMPT_STATUS[$current_attempt]=$(attempt_outcome \
"${TEST_DIRS[$((current_attempt - 1))]}" "$process_rc")
if [ "$process_rc" -ne 0 ]; then
cleanup_owned_containers_for_jobs_dir "${TEST_DIRS[$((current_attempt - 1))]}"
failures=$((failures + 1))
FAILED_ATTEMPTS+=("$current_attempt")
FAILED_LOGS+=("$current_log")
fi
done
fi
WORKER_PIDS=()
attempt=$((batch_last + 1))
done
trap - INT TERM
trap - EXIT
pass_count=0
fail_count=0
error_count=0
complete_count=0
for ((summary_attempt = 1; summary_attempt <= REPEAT; summary_attempt++)); do
case "${ATTEMPT_STATUS[$summary_attempt]}" in
PASS) pass_count=$((pass_count + 1)) ;;
FAIL) fail_count=$((fail_count + 1)) ;;
ERROR) error_count=$((error_count + 1)) ;;
*) complete_count=$((complete_count + 1)) ;;
esac
done
printf '\nResults:\n'
for ((summary_attempt = 1; summary_attempt <= REPEAT; summary_attempt++)); do
printf ' %-10s %-8s %s\n' \
"$(basename "${TEST_DIRS[$((summary_attempt - 1))]}")" \
"${ATTEMPT_STATUS[$summary_attempt]}" \
"${TEST_DIRS[$((summary_attempt - 1))]}"
done
printf 'Summary: %d total, %d passed, %d failed, %d errored' \
"$REPEAT" "$pass_count" "$fail_count" "$error_count"
[ "$complete_count" -eq 0 ] || printf ', %d completed without a readable summary' "$complete_count"
printf '.\n'
printf 'Artifacts root: %s\n' "$TASK_RESULTS_DIR"
[ "$failures" -eq 0 ] || exit 1