Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
@@ -0,0 +1,11 @@
ARG BASE_IMAGE
FROM ${BASE_IMAGE}
# BenchFlow's official OpenCode registry uses precisely these paths.
COPY node /opt/benchflow/node
COPY js-agents /opt/benchflow/js-agents
RUN mkdir -p /opt/benchflow/bin \
&& printf '%s\n' '#!/bin/sh' 'exec /opt/benchflow/node/bin/node /opt/benchflow/js-agents/bin/opencode "$@"' > /opt/benchflow/bin/opencode \
&& chmod +x /opt/benchflow/bin/opencode \
&& chmod -R a+rX /opt/benchflow
@@ -0,0 +1,91 @@
ARG TASK_IMAGE=ubuntu:24.04
FROM ${TASK_IMAGE}
ARG NODE_VERSION=22.20.0
ARG OPENCODE_VERSION=1.18.16
ARG RUNTIME_BUILD_REVISION=2
ARG RUNTIME_SOURCE_FINGERPRINT=unknown
# OpenCode's Skill, glob, and grep tools use ripgrep. Install it in the
# reusable image so an evaluation never has to download rg from GitHub at
# agent runtime.
RUN set -eux; \
if command -v rg >/dev/null 2>&1; then \
:; \
elif command -v apt-get >/dev/null 2>&1; then \
apt-get update; \
apt-get install -y --no-install-recommends ripgrep; \
rm -rf /var/lib/apt/lists/*; \
elif command -v dnf >/dev/null 2>&1; then \
dnf -y install ripgrep; \
dnf clean all; \
elif command -v apk >/dev/null 2>&1; then \
apk add --no-cache ripgrep; \
else \
echo 'OpenCode runtime requires a package manager to install ripgrep' >&2; \
exit 127; \
fi; \
rg --version
RUN set -eux; \
if [ -x /opt/benchflow/node/bin/node ]; then \
/opt/benchflow/node/bin/node --version; \
else \
if ! command -v curl >/dev/null 2>&1 || ! command -v tar >/dev/null 2>&1; then \
if command -v apt-get >/dev/null 2>&1; then \
apt-get update; \
apt-get install -y --no-install-recommends curl ca-certificates tar; \
rm -rf /var/lib/apt/lists/*; \
elif command -v dnf >/dev/null 2>&1; then \
dnf -y install curl ca-certificates tar; \
dnf clean all; \
elif command -v apk >/dev/null 2>&1; then \
apk add --no-cache curl ca-certificates tar; \
else \
echo 'Node/OpenCode bootstrap requires curl and tar' >&2; \
exit 127; \
fi; \
fi; \
arch="$(uname -m)"; \
case "$arch" in \
x86_64|amd64) node_arch=x64 ;; \
aarch64|arm64) node_arch=arm64 ;; \
*) echo "Unsupported architecture for Node.js: $arch" >&2; exit 1 ;; \
esac; \
temporary_dir="$(mktemp -d)"; \
curl -fL \
--retry 8 --retry-delay 2 \
--connect-timeout 20 --max-time 900 \
-o "$temporary_dir/node.tar.gz" \
"https://nodejs.org/dist/v${NODE_VERSION}/node-v${NODE_VERSION}-linux-${node_arch}.tar.gz"; \
mkdir -p /opt/benchflow/node /opt/benchflow/js-agents /opt/benchflow/bin; \
tar -xzf "$temporary_dir/node.tar.gz" \
-C /opt/benchflow/node --strip-components=1 --no-same-owner; \
rm -rf "$temporary_dir"; \
fi
ENV PATH="/opt/benchflow/bin:/opt/benchflow/js-agents/bin:/opt/benchflow/node/bin:${PATH}"
RUN set -eux; \
if [ ! -x /opt/benchflow/js-agents/bin/opencode ]; then \
npm install -g \
--fetch-retries=8 \
--fetch-retry-factor=2 \
--fetch-retry-mintimeout=2000 \
--fetch-retry-maxtimeout=60000 \
--prefix /opt/benchflow/js-agents \
"opencode-ai@${OPENCODE_VERSION}"; \
fi; \
npm cache clean --force; \
rm -rf /root/.npm; \
printf '%s\n' \
'#!/bin/sh' \
'exec /opt/benchflow/js-agents/bin/opencode "$@"' \
> /opt/benchflow/bin/opencode; \
chmod +x /opt/benchflow/bin/opencode; \
chmod -R a+rX /opt/benchflow
LABEL org.skillc.opencode-runtime="1.18.16" \
org.skillc.node-runtime="22.20.0" \
org.skillc.opencode-runtime-build="${RUNTIME_BUILD_REVISION}" \
org.skillc.opencode-runtime-source="${RUNTIME_SOURCE_FINGERPRINT}"
+117
View File
@@ -0,0 +1,117 @@
#!/usr/bin/env bash
# One fully isolated attempt: provenance, container, agent, artifacts, verifier.
run_attempt() (
local attempt_number=$1
local model_label task_label condition_label run_prefix stamp run_root workspace run_id
local started agent_started agent_ended ended agent_wall total_wall agent_exit
local prompt verifier_exit run_status description skill_list install_root cost_estimate_cny
local index proxy_var proxy_value
local -a container_env_args=()
model_label=$(normalize_run_component "${MODEL_ID##*/}")
task_label=$(run_task_label)
condition_label=$(run_condition_label)
run_prefix="$HARNESS-$model_label-$task_label-$condition_label"
stamp="$(date -u +%Y%m%dT%H%M%SZ)-$RANDOM"
run_root=$(reserve_run_root "$run_prefix")
workspace="$run_root/workspace"
run_id="$HARNESS-$MODE-$stamp"
mkdir -p "$workspace"
started=$(now_ms)
progress "$run_id" "Stage 1/5: preparing workspace and recording task/Skill provenance."
cleanup_attempt() { docker rm -f "$run_id" >/dev/null 2>&1 || true; }
trap cleanup_attempt EXIT
for proxy_var in HTTP_PROXY HTTPS_PROXY NO_PROXY http_proxy https_proxy no_proxy; do
proxy_value="${!proxy_var-}"
[ -z "$proxy_value" ] || container_env_args+=(-e "$proxy_var=$proxy_value")
done
progress "$run_id" "Stage 1/5: checking task image and input/Skill checksums."
docker image inspect --format '{{.Id}}' "$IMAGE" > "$run_root/task-image-id.txt"
find "$TASK_DIR/environment" -maxdepth 1 -type f -print0 | sort -z | xargs -0 -r sha256sum > "$run_root/input-sha256.txt"
: > "$run_root/skill-sha256.txt"
for index in "${!SKILL_DIRS[@]}"; do sha256sum "${SKILL_DIRS[$index]}/SKILL.md" >> "$run_root/skill-sha256.txt"; done
skill_list=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}")
case "$HARNESS" in
opencode) install_root="$workspace/.opencode/skills" ;;
hermes) install_root="$run_root/hermes-home/skills" ;;
*) install_root="$workspace/.claude/skills" ;;
esac
progress "$run_id" "Stage 2/5: writing manifest and preparing the isolated task container."
{
printf 'harness=%s\nharness_version=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\n' "$HARNESS" "$(harness_version)" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF"
printf 'task_slug=%s\ntask_dir=%s\ntask_image=%s\ntask_image_id=%s\n' "$TASK_SLUG" "$TASK_DIR" "$IMAGE" "$(tr -d '\n' < "$run_root/task-image-id.txt")"
printf 'skill_source=%s\nskill_count=%s\nskill_names=%s\nskill_install_root=%s\n' "$SOURCE_SKILL" "${#SKILL_NAMES[@]}" "$skill_list" "$install_root"
case "$HARNESS" in
opencode) printf 'tool_approval_policy=opencode_auto\nopencode_auto_approval=true\n' ;;
hermes) printf 'tool_approval_policy=hermes_yolo\n' ;;
*) printf 'tool_approval_policy=claude_dangerously_skip_permissions\nauthentication_scope=Claude Code global OAuth profile\nsetting_sources=Claude Code defaults (user,project,local; required by OAuth)\n' ;;
esac
printf 'network_policy=%s\ncpu_limit=%s\nmemory_limit=%s\nsampling_parameters=Harness defaults (not overridden)\n' "$CONTAINER_NETWORK" "$CPU_LIMIT" "$MEMORY_LIMIT"
printf 'timeout_seconds=%s\nverifier_timeout_seconds=%s\nmode=%s\nattempt_number=%s\n' "$TIMEOUT_SECONDS" "$VERIFIER_TIMEOUT_SECONDS" "$MODE" "$attempt_number"
} > "$run_root/target-manifest.env"
progress "$run_id" "Stage 2/5: starting the task container and mounting its workspace."
docker run -d --name "$run_id" --cpus="$CPU_LIMIT" --memory="$MEMORY_LIMIT" --network "$CONTAINER_NETWORK" "${container_env_args[@]}" --mount "type=bind,src=$workspace,dst=/workspace" "$IMAGE" sleep infinity >/dev/null
progress "$run_id" "Stage 3/5: container ready; building the agent prompt."
if [ "$MODE" = probe ]; then
prompt="The following Skills are installed and available: $skill_list.
Use bash commands only. Run these exact commands one at a time:
1. docker exec $run_id bash -lc 'test -d /root && test -w /root && ls -1 /root | head -n 20'
2. docker exec $run_id bash -lc 'probe_file=/root/.skill-agent-probe; printf probe-ok > \"\$probe_file\"; test -s \"\$probe_file\"; rm -f \"\$probe_file\"; printf HARNESS_CONTAINER_PROBE_OK'
Do not run any other command. If both commands succeed, finish with exactly: HARNESS_CONTAINER_PROBE_OK"
else
prompt="Use the installed Skills when relevant: $skill_list.
Complete this task:
$TASK_PROMPT
Execution environment:
- The fresh task container is named $run_id.
- Run every task inspection, analysis, edit, build, and test inside it with: docker exec $run_id ...
- Do not run ls, find, grep, cat, Maven, or any task command against host paths.
- Never inspect or access /mnt, the runner project, runs/, another attempt's workspace, task.md, oracle, verifier, or files outside the named container.
- The host working directory is only a transport mount at /workspace; use it only for a helper file you create, then execute that helper through /workspace inside the named container.
- Do not use host paths inside docker exec.
- Do not access task.md, oracle, verifier, or files outside the current workspace and installed Skills.
- Before finishing, inspect the result inside the task container and make sure the requested output or repository changes exist."
fi
agent_started=$(now_ms)
progress "$run_id" "Stage 3/5: agent running (model output is being saved to agent-trace.txt)."
set +e
case "$HARNESS" in opencode) run_opencode "$run_root" "$workspace" "$prompt" ;; hermes) run_hermes "$run_root" "$workspace" "$prompt" ;; *) run_claude_code "$run_root" "$workspace" "$prompt" ;; esac
agent_exit=$?
set -e
agent_ended=$(now_ms); ended=$(now_ms); agent_wall=$((agent_ended-agent_started)); total_wall=$((ended-started))
progress "$run_id" "Stage 4/5: agent finished (exit code $agent_exit); exporting metrics and collecting artifacts."
{
printf 'agent_exit_code=%s\nharness=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\nagent_wall_ms=%s\ntotal_wall_ms=%s\n' "$agent_exit" "$HARNESS" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF" "$agent_wall" "$total_wall"
[ "$agent_exit" -eq 124 ] && printf 'timed_out=true\n' || printf 'timed_out=false\n'
} > "$run_root/metrics.env"
export_agent_session "$run_root" "$agent_wall" "$total_wall"
cost_estimate_cny=unavailable
[ ! -s "$run_root/agent-metrics.json" ] || cost_estimate_cny=$(node -e 'const m=require(process.argv[1]); const c=m.official_cost_estimate?.amount_cny; process.stdout.write(Number.isFinite(c) ? c.toFixed(8) : "unavailable")' "$run_root/agent-metrics.json")
printf 'official_cost_estimate_cny=%s\n' "$cost_estimate_cny" >> "$run_root/metrics.env"
if [ "$MODE" = probe ]; then
if [ "$agent_exit" -eq 0 ] && rg -q HARNESS_CONTAINER_PROBE_OK "$run_root/agent-trace.txt"; then run_status=success; description=none; else run_status=probe_failed; description='The Harness probe did not complete successfully. Inspect agent-trace.txt.'; fi
printf '# Harness probe summary\n\nstatus=%s\nagent_exit_code=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md"
else
progress "$run_id" "Stage 4/5: capturing changed files and artifacts from the task container."
docker diff "$run_id" > "$run_root/container-diff.txt" 2>/dev/null || true
capture_agent_artifacts "$run_root" "$run_id"
progress "$run_id" "Stage 5/5: running the task verifier."
set +e; verify_output "$run_root" "$run_id"; verifier_exit=$?; set -e
progress "$run_id" "Stage 5/5: verifier finished (exit code $verifier_exit); writing run summary."
printf 'verifier_exit=%s\n' "$verifier_exit" >> "$run_root/metrics.env"
if [ "$agent_exit" -eq 0 ] && [ "$verifier_exit" = 0 ]; then run_status=success; description=none
elif [ "$agent_exit" -eq 124 ] && [ "$verifier_exit" != 0 ]; then run_status=agent_timed_out; description="The agent timed out after ${TIMEOUT_SECONDS} seconds and the task verifier failed."
elif [ "$verifier_exit" != 0 ]; then run_status=verifier_failed; description="The agent completed, but the task verifier failed (exit code ${verifier_exit})."
else run_status=agent_failed_output_verified; description="The agent exited with code ${agent_exit}, but its output passed verification."; fi
printf '# Raw run summary\n\nstatus=%s\nagent_exit_code=%s\nverifier_exit=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$verifier_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md"
fi
printf 'run_status=%s\nproblem_description=%s\n' "$run_status" "$description" >> "$run_root/metrics.env"
progress "$run_id" "Completed: $run_status. Details saved to $run_root."
[ "$run_status" = success ]
)
+65
View File
@@ -0,0 +1,65 @@
#!/usr/bin/env bash
# Shared presentation and run-directory helpers for run-raw-task.sh.
now_ms() { node -p 'Date.now()'; }
progress() {
local run_label=$1 message=$2
printf '[%s] [%s] %s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$run_label" "$message"
}
progress_bar() {
local label=$1 completed=$2 total=$3 width=24 filled empty percent bar
[ "$total" -gt 0 ] || return
filled=$((completed * width / total))
empty=$((width - filled))
percent=$((completed * 100 / total))
printf -v bar '%*s' "$filled" ''
bar=${bar// /#}
printf -v empty '%*s' "$empty" ''
empty=${empty// /-}
printf '[%s] [%s%s] %d/%d (%d%%)\n' "$label" "$bar" "$empty" "$completed" "$total" "$percent"
}
normalize_run_component() {
printf '%s' "$1" | tr '[:upper:]' '[:lower:]' | tr -cs 'a-z0-9' '-' | sed -E 's/^-+//; s/-+$//'
}
run_task_label() {
case "$TASK_SLUG" in
111-offer-letter-generator) printf '%s' 'task-1' ;;
222-software-dependency-audit) printf '%s' 'task-2' ;;
333-fix-build-google-auto) printf '%s' 'task-3' ;;
*) printf 'task-%s' "$(normalize_run_component "$TASK_SLUG")" ;;
esac
}
run_condition_label() {
case "$SOURCE_SKILL" in
*/results/model-compiled-skills/*|*/dist/*|*/conditions/*|/tmp/*) printf '%s' '编译后skill' ;;
*) printf '%s' '原始skill' ;;
esac
}
reserve_run_root() {
# Allocation and mkdir share one lock: concurrent attempts must never claim
# the same trace/workspace directory.
local prefix=$1 run_root suffix=1 lock_fd
local lock_path="/tmp/skill-agent-raw-name-index.lock"
mkdir -p "$PROJECT_ROOT/runs/$MODE"
exec {lock_fd}>"$lock_path"
flock "$lock_fd"
# Do the check and the mkdir while holding the same lock. Starting at 1
# also tolerates old, interrupted runs and avoids parsing a path whose
# prefix itself contains hyphens.
while :; do
run_root="$PROJECT_ROOT/runs/$MODE/$prefix-$suffix"
if mkdir "$run_root" 2>/dev/null; then
break
fi
suffix=$((suffix + 1))
done
flock -u "$lock_fd"
exec {lock_fd}>&-
printf '%s' "$run_root"
}
+67
View File
@@ -0,0 +1,67 @@
#!/usr/bin/env bash
# Container output verification and bounded artifact capture.
verify_output() {
local run_root=$1 run_id=$2 log_dir="$1/verifier"
local verifier_started verifier_ended verifier_wall docker_exit reward verification_status description
mkdir -p "$log_dir"
# BenchFlow's native task.md verifier contract: upload verifier/ to
# /verifier, then expose the legacy /tests path as a symlink only if the
# image has not already provided real /tests content. Do not overwrite that
# content; older verifier scripts may legitimately depend on it.
docker exec "$run_id" mkdir -p /verifier /logs/verifier
docker cp "$VERIFIER_SOURCE/." "$run_id:/verifier"
docker exec "$run_id" bash -lc '[ -e /tests ] || ln -s /verifier /tests'
docker exec "$run_id" chmod +x /verifier/test.sh
verifier_started=$(now_ms)
if timeout --foreground --signal=INT --kill-after=30s "${VERIFIER_TIMEOUT_SECONDS}s" \
docker exec "${VERIFIER_ENV_ARGS[@]}" "$run_id" /verifier/test.sh > "$log_dir/verifier.stdout.log" 2>&1; then
docker_exit=0
else
docker_exit=$?
fi
docker cp "$run_id:/logs/verifier/." "$log_dir" >/dev/null 2>&1 || true
verifier_ended=$(now_ms)
verifier_wall=$((verifier_ended - verifier_started))
reward=""
[ ! -f "$log_dir/reward.txt" ] || reward=$(tr -d '[:space:]' < "$log_dir/reward.txt")
if [ "$docker_exit" -eq 0 ] && [ "$reward" = 1 ]; then
verification_status=passed
description=none
else
verification_status=failed
description="Verifier exited with code ${docker_exit} and wrote reward=${reward:-missing}. See verifier.stdout.log for details."
fi
{
printf 'docker_exit_code=%s\nreward=%s\nverifier_wall_ms=%s\n' "$docker_exit" "$reward" "$verifier_wall"
printf 'verifier_log=%s\nverification_status=%s\nproblem_description=%s\n' "$log_dir/verifier.stdout.log" "$verification_status" "$description"
} > "$log_dir/summary.env"
[ "$verification_status" = passed ] && return 0
printf 'VERIFIER_ISSUE: %s\n' "$description" >&2
return 1
}
capture_agent_artifacts() {
local run_root=$1 run_id=$2 diff_path="$1/container-diff.txt"
local artifact_root="$1/artifacts" status container_path relative_path size destination
mkdir -p "$artifact_root"
while IFS=' ' read -r status container_path; do
[ "$status" = A ] || [ "$status" = C ] || continue
# Copy only modest, task-created outputs. Package caches and the mounted
# workspace are inputs/ephemera, never benchmark artifacts.
case "$container_path" in
/root/.cache/*|/root/.local/*|/root/.m2/*|/home/*/.cache/*|/home/*/.local/*|/home/*/.m2/*|*/.git/*|/etc/*|/opt/*|/tmp/*|/usr/*|/var/*|/workspace/*) continue ;;
esac
docker exec "$run_id" test -f "$container_path" >/dev/null 2>&1 || continue
size=$(docker exec "$run_id" stat -c '%s' "$container_path" 2>/dev/null || printf '0')
[[ "$size" =~ ^[0-9]+$ ]] || continue
[ "$size" -le 20971520 ] || continue
relative_path=${container_path#/}
destination="$artifact_root/$relative_path"
mkdir -p "$(dirname "$destination")"
docker cp "$run_id:$container_path" "$destination" >/dev/null
done < "$diff_path"
if find "$artifact_root" -type f -print -quit | grep -q .; then
find "$artifact_root" -type f -print0 | sort -z | xargs -0 sha256sum > "$run_root/output-sha256.txt"
fi
}
+68
View File
@@ -0,0 +1,68 @@
#!/usr/bin/env bash
# Harness-specific configuration and agent invocation.
install_skill_set() {
local destination=$1 index
mkdir -p "$destination"
for index in "${!SKILL_DIRS[@]}"; do
cp -a "${SKILL_DIRS[$index]}" "$destination/${SKILL_NAMES[$index]}"
done
}
write_opencode_config() {
local config_path=$1
node - "$config_path" "$PROVIDER_ID" "$MODEL_ID" "$PROVIDER_BASE_URL" "$PROVIDER_API_KEY" <<'NODE'
const fs = require("fs");
const [configPath, provider, model, baseURL, apiKey] = process.argv.slice(2);
const config = {$schema: "https://opencode.ai/config.json", model: `${provider}/${model}`,
provider: {[provider]: {npm: "@ai-sdk/openai-compatible", name: provider,
options: {baseURL, apiKey}, models: {[model]: {name: model}}}}};
fs.writeFileSync(configPath, `${JSON.stringify(config, null, 2)}\n`, {mode: 0o600});
NODE
}
write_hermes_profile() {
local profile_home=$1 workspace=$2
mkdir -p "$profile_home/skills"
touch "$profile_home/.no-bundled-skills"
install_skill_set "$profile_home/skills"
{
printf '%s\n' 'model:' " default: \"$MODEL_ID\"" ' provider: custom' ' base_url: "https://api.siliconflow.cn/v1"'
printf '%s\n' 'terminal:' ' backend: local' " cwd: \"$workspace\"" ' timeout: 180' ' home_mode: profile'
printf '%s\n' 'memory:' ' memory_enabled: false' ' user_profile_enabled: false'
printf '%s\n' 'skills:' ' external_dirs: []' ' inline_shell: false' ' write_approval: true'
printf '%s\n' 'curator:' ' enabled: false' 'fallback_providers: []'
printf '%s\n' 'delegation:' ' orchestrator_enabled: false' ' max_spawn_depth: 1' ' max_concurrent_children: 1' ' max_async_children: 1'
} > "$profile_home/config.yaml"
}
run_opencode() {
local run_root=$1 workspace=$2 prompt=$3
install_skill_set "$workspace/.opencode/skills"
write_opencode_config "$run_root/opencode.json"
(cd "$workspace"; OPENCODE_CONFIG="$run_root/opencode.json" OPENCODE_CONFIG_DIR="$workspace/.opencode" \
XDG_DATA_HOME="$run_root/opencode-data" XDG_STATE_HOME="$run_root/opencode-state" \
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
opencode --pure --auto --model "$PROVIDER_ID/$MODEL_ID" run "$prompt") > "$run_root/agent-trace.txt" 2>&1
}
run_hermes() {
local run_root=$1 workspace=$2 prompt=$3 skill_csv
skill_csv=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}")
write_hermes_profile "$run_root/hermes-home" "$workspace"
(cd "$workspace"; OPENAI_API_KEY="$SILICONFLOW_API_KEY" HERMES_HOME="$run_root/hermes-home" HERMES_OPTIONAL_SKILLS="" \
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
hermes --yolo --provider custom --model "$MODEL_ID" --toolsets terminal --skills "$skill_csv" --oneshot "$prompt") > "$run_root/agent-trace.txt" 2>&1
}
prepare_claude_workspace() { install_skill_set "$1/.claude/skills"; }
run_claude_code() {
local run_root=$1 workspace=$2 prompt=$3
prepare_claude_workspace "$workspace"
(cd "$workspace"; ANTHROPIC_BASE_URL="$PROVIDER_BASE_URL" ANTHROPIC_AUTH_TOKEN="$PROVIDER_API_KEY" \
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
claude --print --output-format json --model "$MODEL_ID" --dangerously-skip-permissions "$prompt") \
> "$run_root/claude-result.json" 2> "$run_root/claude-stderr.txt"
cat "$run_root/claude-result.json" "$run_root/claude-stderr.txt" > "$run_root/agent-trace.txt"
}
@@ -0,0 +1,65 @@
#!/usr/bin/env bash
# Create a derived task image that satisfies BenchFlow's official OpenCode
# bootstrap checks without requiring each evaluation container to download Node.
set -euo pipefail
PROJECT_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)
DOCKERFILE="$PROJECT_ROOT/scripts/evaluate/Dockerfile.benchflow-opencode"
NODE_VERSION=22.20.0
BASE_IMAGE=""
OUTPUT_IMAGE=""
usage() {
cat <<'EOF'
Usage:
bash scripts/evaluate/prepare-benchflow-opencode-image.sh \
--base-image <existing-task-image> \
--tag <new-derived-image-tag>
The base image is never changed. The resulting image contains the exact Node
runtime expected by BenchFlow plus opencode-ai@latest under /opt/benchflow.
EOF
}
while [ "$#" -gt 0 ]; do
case "$1" in
--base-image) BASE_IMAGE=${2:?missing value for --base-image}; shift 2 ;;
--tag) OUTPUT_IMAGE=${2:?missing value for --tag}; shift 2 ;;
--node-version) NODE_VERSION=${2:?missing value for --node-version}; shift 2 ;;
-h|--help) usage; exit 0 ;;
*) printf 'Unknown option: %s\n' "$1" >&2; usage >&2; exit 2 ;;
esac
done
[ -n "$BASE_IMAGE" ] || { printf '%s\n' '--base-image is required.' >&2; exit 2; }
[ -n "$OUTPUT_IMAGE" ] || { printf '%s\n' '--tag is required.' >&2; exit 2; }
command -v docker >/dev/null || { printf '%s\n' 'docker is required.' >&2; exit 127; }
command -v curl >/dev/null || { printf '%s\n' 'curl is required.' >&2; exit 127; }
[ -f "$DOCKERFILE" ] || { printf 'Missing Dockerfile: %s\n' "$DOCKERFILE" >&2; exit 1; }
docker image inspect "$BASE_IMAGE" >/dev/null
BUILD_CONTEXT=$(mktemp -d "${TMPDIR:-/tmp}/benchflow-opencode-image.XXXXXX")
cleanup() { rm -rf -- "$BUILD_CONTEXT"; }
trap cleanup EXIT
NODE_ARCH=$(uname -m)
case "$NODE_ARCH" in
x86_64|amd64) NODE_ARCH=x64 ;;
aarch64|arm64) NODE_ARCH=arm64 ;;
*) printf 'Unsupported architecture: %s\n' "$NODE_ARCH" >&2; exit 2 ;;
esac
NODE_ARCHIVE="node-v${NODE_VERSION}-linux-${NODE_ARCH}.tar.xz"
printf 'Downloading Node.js %s for the BenchFlow runtime cache...\n' "$NODE_VERSION"
curl --fail --location --retry 3 --output "$BUILD_CONTEXT/node.tar.xz" \
"https://nodejs.org/dist/v${NODE_VERSION}/${NODE_ARCHIVE}"
mkdir -p "$BUILD_CONTEXT/node"
tar -xJf "$BUILD_CONTEXT/node.tar.xz" -C "$BUILD_CONTEXT/node" --strip-components=1 --no-same-owner
printf '%s\n' 'Installing the official opencode-ai package into the runtime cache...'
"$BUILD_CONTEXT/node/bin/npm" install --global --prefix "$BUILD_CONTEXT/js-agents" opencode-ai@latest
[ -x "$BUILD_CONTEXT/js-agents/bin/opencode" ] || { printf '%s\n' 'opencode-ai installation did not create its executable.' >&2; exit 1; }
printf 'Building derived image: %s\n' "$OUTPUT_IMAGE"
docker build --build-arg "BASE_IMAGE=$BASE_IMAGE" --tag "$OUTPUT_IMAGE" --file "$DOCKERFILE" "$BUILD_CONTEXT"
printf 'Ready: %s\n' "$OUTPUT_IMAGE"
+806
View File
@@ -0,0 +1,806 @@
#!/usr/bin/env bash
# Thin compatibility wrapper around the official SkillsBench/BenchFlow runner.
set -euo pipefail
PROJECT_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)
cd "$PROJECT_ROOT"
# Evaluation credentials are user-owned configuration. Load them once for
# BenchFlow so its isolated agent container receives only the registered
# provider variables, never a copied host config directory.
if [ -f "$PROJECT_ROOT/.env" ]; then
set -a
# shellcheck disable=SC1091
source "$PROJECT_ROOT/.env"
set +a
fi
# OpenCode's OpenAI-compatible provider expects a base URL, while the project
# .env records the concrete chat-completions endpoint used by other harnesses.
if [ -n "${SILICONFLOW_CHAT_COMPLETIONS_URL:-}" ]; then
SILICONFLOW_BASE_URL=${SILICONFLOW_CHAT_COMPLETIONS_URL%/chat/completions}
fi
: "${SILICONFLOW_BASE_URL:=https://api.siliconflow.cn/v1}"
export SILICONFLOW_BASE_URL
usage() {
cat <<'EOF'
Usage:
bash scripts/evaluate/run-raw-task.sh \
--harness opencode \
--model <provider/model> \
--task <SkillsBench task directory> \
--skill-source <skill directory or skills root> \
--output <result directory> \
[--require-skill] \
[--repeat N] [--max-parallel N] [--image <local-image-ref>]
This wrapper delegates every attempt to `bench eval run --sandbox docker`.
It does not inject verifier files or implement scoring.
Provider credentials are loaded from the project `.env`.
Each repeat is allocated before any workers start, at:
<output>/test-NNN/
The official BenchFlow job and summary remain inside that test directory.
Interactive terminals show one in-place progress row per attempt for both
single and concurrent runs. Worker output is kept in test-NNN/console.log and
only a concise result summary is printed after the progress display finishes.
External Skill sources are staged into each disposable task copy's
environment/skills directory. This keeps BenchFlow's task-bundled deployment
and subagent registration policy identical for source and compiled Skills.
Both <output>/skills/<name>/SKILL.md and <skills-root>/<name>/SKILL.md layouts
are accepted.
--image makes an ephemeral copy of the selected task and applies the official
SkillsBench prebuilt-image policy to that copy. The original task and Skills
remain untouched. This avoids a Docker build when the supplied image already
exists locally.
Without --image, the wrapper uses skillc/<task-name>:local. It builds and tags
that image only when needed, then adds the pinned Node/OpenCode runtime and
cleans installation caches. Later evaluations reuse the same single image
through the official prebuilt-image policy.
--require-skill makes at least one Skill invocation an evaluation invariant
rather than an agent choice. Every temporary task prompt requires the agent to
load a relevant Skill during the task, immediately before applying its guidance.
The completed ACP trajectory is checked for a successful Skill call. Without
this flag, Skill invocation remains the agent's choice. An attempt with no
successful Skill call exits with status 86 and writes required-skill.json.
EOF
}
HARNESS=""
MODEL_REF=""
TASK_DIR=""
SKILL_SOURCE=""
OUTPUT_DIR=""
REPEAT=1
MAX_PARALLEL=3
REPEAT_SEEN=false
MAX_PARALLEL_SEEN=false
PREBUILT_IMAGE=""
REQUIRE_SKILLS=false
while [ "$#" -gt 0 ]; do
case "$1" in
--harness) HARNESS=${2:?missing value for --harness}; shift 2 ;;
--model) MODEL_REF=${2:?missing value for --model}; shift 2 ;;
--task) TASK_DIR=${2:?missing value for --task}; shift 2 ;;
--skill-source) SKILL_SOURCE=${2:?missing value for --skill-source}; shift 2 ;;
--output) OUTPUT_DIR=${2:?missing value for --output}; shift 2 ;;
--require-skill) REQUIRE_SKILLS=true; shift ;;
--repeat)
[ "$REPEAT_SEEN" = false ] || { printf '%s\n' 'Duplicate --repeat option.' >&2; exit 2; }
REPEAT=${2:?missing value for --repeat}
REPEAT_SEEN=true
shift 2
;;
--max-parallel)
[ "$MAX_PARALLEL_SEEN" = false ] || { printf '%s\n' 'Duplicate --max-parallel option.' >&2; exit 2; }
MAX_PARALLEL=${2:?missing value for --max-parallel}
MAX_PARALLEL_SEEN=true
shift 2
;;
--image|--prebuilt-image) PREBUILT_IMAGE=${2:?missing value for --image}; shift 2 ;;
-h|--help) usage; exit 0 ;;
*) printf 'Unsupported option for the official BenchFlow wrapper: %s\n' "$1" >&2; usage >&2; exit 2 ;;
esac
done
[ -n "$HARNESS" ] || { printf '%s\n' '--harness is required.' >&2; exit 2; }
[ -n "$MODEL_REF" ] || { printf '%s\n' '--model is required.' >&2; exit 2; }
[ -n "$TASK_DIR" ] || { printf '%s\n' '--task is required.' >&2; exit 2; }
[ -n "$SKILL_SOURCE" ] || { printf '%s\n' '--skill-source is required.' >&2; exit 2; }
[ -n "$OUTPUT_DIR" ] || { printf '%s\n' '--output is required.' >&2; exit 2; }
[[ "$REPEAT" =~ ^[1-9][0-9]*$ ]] || { printf '%s\n' '--repeat must be a positive integer.' >&2; exit 2; }
[[ "$MAX_PARALLEL" =~ ^[1-9][0-9]*$ ]] || { printf '%s\n' '--max-parallel must be a positive integer.' >&2; exit 2; }
TASK_DIR=$(realpath "$TASK_DIR")
SKILL_SOURCE=$(realpath "$SKILL_SOURCE")
OUTPUT_DIR=$(realpath -m "$OUTPUT_DIR")
[ -f "$TASK_DIR/task.md" ] || { printf 'Not a native SkillsBench task: %s\n' "$TASK_DIR" >&2; exit 2; }
[ -d "$SKILL_SOURCE" ] || { printf 'Skill source does not exist: %s\n' "$SKILL_SOURCE" >&2; exit 2; }
# Compilers may emit either a Skills root directly:
# <output>/skill-a/SKILL.md
# or a package containing that root:
# <output>/skills/skill-a/SKILL.md
# Normalize both forms before staging the selected Skills into a task copy.
SKILL_PAYLOAD_DIR="$SKILL_SOURCE"
if [ ! -f "$SKILL_SOURCE/SKILL.md" ] && \
[ -d "$SKILL_SOURCE/skills" ] && \
[ -z "$(find "$SKILL_SOURCE" -mindepth 2 -maxdepth 2 -type f -name SKILL.md -print -quit)" ] && \
[ -n "$(find "$SKILL_SOURCE/skills" -type f -name SKILL.md -print -quit)" ]; then
SKILL_PAYLOAD_DIR=$(realpath "$SKILL_SOURCE/skills")
fi
if [ "$REQUIRE_SKILLS" = true ]; then
[ -n "$(find "$SKILL_SOURCE" -type f -name SKILL.md -print -quit)" ] || {
printf 'No SKILL.md files were found under --skill-source: %s\n' "$SKILL_SOURCE" >&2
exit 2
}
fi
TASK_BUNDLED_SKILLS="$TASK_DIR/environment/skills"
SKILL_SOURCE_IS_TASK_BUNDLED=false
if [ -d "$TASK_BUNDLED_SKILLS" ] && [ "$SKILL_SOURCE" = "$(realpath "$TASK_BUNDLED_SKILLS")" ]; then
SKILL_SOURCE_IS_TASK_BUNDLED=true
fi
case "$HARNESS" in
opencode)
AGENT=opencode
# This project also keeps OPENCODE_* variables for the legacy harness.
# They are OpenCode-hosted-service credentials, not SiliconFlow
# credentials, and current OpenCode releases may prefer them over a
# configured custom provider. The official BenchFlow container must use
# only the explicitly registered SiliconFlow provider below.
unset OPENCODE_API_KEY OPENCODE_CHAT_COMPLETIONS_URL
# BenchFlow intentionally inherits only its built-in provider variables.
# SiliconFlow is an OpenAI-compatible custom provider, so its two values
# are supplied through the official evaluation config below. They
# originate in .env; no task, Skill, or verifier file is changed.
[ -n "${SILICONFLOW_API_KEY:-}" ] || {
printf '%s\n' 'SILICONFLOW_API_KEY is required in .env for --harness opencode.' >&2
exit 2
}
# Values are placed in a mode-0600 temporary BenchFlow config per attempt.
# Credentials must not appear in CLI arguments visible through `ps`.
;;
claude-code|claude) AGENT=claude-agent-acp ;;
*)
printf 'Harness %q is not a registered BenchFlow ACP agent in the pinned SkillsBench runner.\n' "$HARNESS" >&2
printf '%s\n' 'Use an official BenchFlow agent name (for example: opencode or claude-agent-acp).' >&2
exit 2
;;
esac
SKILLSBENCH_ROOT=${SKILLSBENCH_ROOT:-"$PROJECT_ROOT/data/skills-bench"}
if command -v bench >/dev/null 2>&1; then
BENCH=(bench)
elif command -v benchflow >/dev/null 2>&1; then
BENCH=(benchflow)
elif [ -x "$SKILLSBENCH_ROOT/.venv/bin/bench" ]; then
# `uv sync` keeps the official CLI inside the dataset virtual environment.
BENCH=("$SKILLSBENCH_ROOT/.venv/bin/bench")
elif command -v uv >/dev/null 2>&1 && [ -f "$SKILLSBENCH_ROOT/pyproject.toml" ]; then
BENCH=(uv run --directory "$SKILLSBENCH_ROOT" bench)
else
printf '%s\n' 'BenchFlow is required but neither `bench` nor `benchflow` is on PATH.' >&2
printf 'Run: cd %s && uv sync --locked\n' "$SKILLSBENCH_ROOT" >&2
exit 127
fi
TASK_NAME=$(basename "$TASK_DIR")
IMAGE_LOCK_SCOPE=$(
printf '%s' "$HARNESS-$MODEL_REF" \
| tr '[:upper:]' '[:lower:]' \
| sed -E 's#[^a-z0-9]+#-#g; s#^-+##; s#-+$##'
)
TASK_IMAGE_LOCK_DIR="${TMPDIR:-/tmp}/skill-agent-evaluate-image-locks/$IMAGE_LOCK_SCOPE/$TASK_NAME"
TASK_RESULTS_DIR="$OUTPUT_DIR"
ensure_result_directory() {
local directory=$1 mkdir_error
[ -d "$directory" ] && return 0
if mkdir_error=$(mkdir -p "$directory" 2>&1); then
return 0
fi
# On a Windows-backed /mnt filesystem, deleting a directory while a Windows,
# WSL, or Docker process still has it open can leave a delete-pending name.
# The entry is invisible to stat/find, but NTFS rejects recreating the same
# name with "Already exists". Report that state explicitly; silently using
# a different category would put results under the wrong Skill provenance.
if [ ! -e "$directory" ] && [[ "$mkdir_error" == *"Already exists"* || "$mkdir_error" == *"File exists"* ]]; then
cat >&2 <<EOF
Cannot create the result category directory because NTFS still reserves its deleted name:
$directory
No evaluation was started and no existing result was changed.
Close Explorer/editors or other processes holding this path. If it remains blocked,
run "wsl --shutdown" in Windows PowerShell or Command Prompt, reopen WSL, and retry.
This releases the stale handle; it does not delete Docker images or result data.
EOF
exit 73
fi
printf 'Unable to create result directory %s: %s\n' "$directory" "$mkdir_error" >&2
exit 73
}
ensure_result_directory "$TASK_RESULTS_DIR"
SETUP_LOG="$TASK_RESULTS_DIR/setup.log"
touch "$SETUP_LOG"
# The wrapper owns the only terminal display. BenchFlow worker output always
# goes to per-attempt logs so single and concurrent runs behave identically.
DASHBOARD_ENABLED=false
if [ -t 1 ] && [ "${TERM:-dumb}" != dumb ]; then
DASHBOARD_ENABLED=true
fi
# Reserve all test numbers before starting any background workers. `flock`
# also keeps two separately launched wrapper processes from selecting the same
# test-NNN directory.
TEST_DIRS=()
allocate_test_dirs() {
local max_test=0 name number next_test candidate reserved=0 mkdir_error
exec 9>"$TASK_RESULTS_DIR/.test-number.lock"
flock 9
while IFS= read -r name; do
if [[ "$name" =~ ^test-([0-9]+)$ ]]; then
number=$((10#${BASH_REMATCH[1]}))
(( number > max_test )) && max_test=$number
fi
done < <(find "$TASK_RESULTS_DIR" -mindepth 1 -maxdepth 1 -type d -printf '%f\n')
next_test=$((max_test + 1))
while [ "$reserved" -lt "$REPEAT" ]; do
printf -v name 'test-%03d' "$next_test"
candidate="$TASK_RESULTS_DIR/$name"
mkdir_error=''
if mkdir_error=$(mkdir "$candidate" 2>&1); then
TEST_DIRS+=("$candidate")
reserved=$((reserved + 1))
elif [ -d "$candidate" ] || [[ "$mkdir_error" == *'Already exists'* ]] || [[ "$mkdir_error" == *'File exists'* ]]; then
# On drvfs (/mnt/c), a recently deleted Windows directory can remain in
# delete-pending state: stat/find cannot see it, but mkdir still reports
# Already exists. Treat that ghost name exactly like a live collision.
:
else
printf 'Unable to reserve result directory %s: %s\n' \
"$candidate" "$mkdir_error" >&2
flock -u 9
exec 9>&-
return 1
fi
# An existing directory is already reserved by another run (or was
# recreated by a still-finishing old run). Skip it atomically.
next_test=$((next_test + 1))
done
flock -u 9
exec 9>&-
}
allocate_test_dirs
if ! command -v docker >/dev/null 2>&1; then
printf '%s\n' 'Docker is required for the official BenchFlow Docker sandbox.' >&2
exit 127
fi
[ -x "$SKILLSBENCH_ROOT/.venv/bin/python" ] || {
printf 'The official SkillsBench Python environment is required: %s\n' "$SKILLSBENCH_ROOT/.venv/bin/python" >&2
exit 127
}
if [ -z "$PREBUILT_IMAGE" ]; then
PREBUILT_IMAGE="skillc/$TASK_NAME:local"
mkdir -p "$TASK_IMAGE_LOCK_DIR"
exec 8>"$TASK_IMAGE_LOCK_DIR/.image-build.lock"
flock 8
DOCKERFILE="$TASK_DIR/environment/Dockerfile"
[ -f "$DOCKERFILE" ] || {
printf 'Task Dockerfile is missing: %s\n' "$DOCKERFILE" >&2
exit 2
}
# Rebuild when any task environment input changes. Previously an existing
# tag was reused forever, which could preserve a stale task image even after
# its Dockerfile or fixtures were fixed.
TASK_ENVIRONMENT_FINGERPRINT=$(
find "$TASK_DIR/environment" -type f -print0 \
| sort -z \
| xargs -0 sha256sum \
| sha256sum \
| awk '{print $1}'
)
task_environment_label=$(docker image inspect --format \
'{{ index .Config.Labels "org.skillc.task-environment" }}' \
"$PREBUILT_IMAGE" 2>/dev/null || true)
if ! docker image inspect "$PREBUILT_IMAGE" >/dev/null 2>&1 || \
[ "$task_environment_label" != "$TASK_ENVIRONMENT_FINGERPRINT" ]; then
printf 'Building reusable task image: %s\n' "$PREBUILT_IMAGE" >>"$SETUP_LOG"
# BenchFlow builds every task Dockerfile with environment/ as its context.
# Input fixtures referenced by COPY therefore live in that directory.
docker build \
--file "$DOCKERFILE" \
--build-arg PIP_INDEX_URL=https://mirrors.aliyun.com/pypi/simple \
--label "org.skillc.task-environment=$TASK_ENVIRONMENT_FINGERPRINT" \
--tag "$PREBUILT_IMAGE" \
"$TASK_DIR/environment" >>"$SETUP_LOG" 2>&1
fi
OPENCODE_RUNTIME_VERSION=1.18.16
OPENCODE_RUNTIME_BUILD=2
NODE_RUNTIME_VERSION=22.20.0
RUNTIME_DOCKERFILE="$PROJECT_ROOT/scripts/evaluate/docker/opencode-runtime.Dockerfile"
RUNTIME_SOURCE_FINGERPRINT=$(sha256sum "$RUNTIME_DOCKERFILE" | awk '{print $1}')
runtime_label=$(docker image inspect --format \
'{{ index .Config.Labels "org.skillc.opencode-runtime" }}' \
"$PREBUILT_IMAGE" 2>/dev/null || true)
runtime_build_label=$(docker image inspect --format \
'{{ index .Config.Labels "org.skillc.opencode-runtime-build" }}' \
"$PREBUILT_IMAGE" 2>/dev/null || true)
runtime_source_label=$(docker image inspect --format \
'{{ index .Config.Labels "org.skillc.opencode-runtime-source" }}' \
"$PREBUILT_IMAGE" 2>/dev/null || true)
if [ "$runtime_label" != "$OPENCODE_RUNTIME_VERSION" ] || \
[ "$runtime_build_label" != "$OPENCODE_RUNTIME_BUILD" ] || \
[ "$runtime_source_label" != "$RUNTIME_SOURCE_FINGERPRINT" ]; then
printf 'Adding reusable Node %s + OpenCode %s runtime to: %s\n' \
"$NODE_RUNTIME_VERSION" "$OPENCODE_RUNTIME_VERSION" "$PREBUILT_IMAGE" >>"$SETUP_LOG"
docker build \
--file "$RUNTIME_DOCKERFILE" \
--build-arg "TASK_IMAGE=$PREBUILT_IMAGE" \
--build-arg "NODE_VERSION=$NODE_RUNTIME_VERSION" \
--build-arg "OPENCODE_VERSION=$OPENCODE_RUNTIME_VERSION" \
--build-arg "RUNTIME_BUILD_REVISION=$OPENCODE_RUNTIME_BUILD" \
--build-arg "RUNTIME_SOURCE_FINGERPRINT=$RUNTIME_SOURCE_FINGERPRINT" \
--tag "$PREBUILT_IMAGE" \
"$PROJECT_ROOT/scripts/evaluate/docker" >>"$SETUP_LOG" 2>&1
else
printf 'Reusing task image with preinstalled OpenCode: %s\n' "$PREBUILT_IMAGE" >>"$SETUP_LOG"
fi
flock -u 8
exec 8>&-
elif ! docker image inspect "$PREBUILT_IMAGE" >/dev/null 2>&1; then
printf 'The requested local prebuilt image is unavailable: %s\n' "$PREBUILT_IMAGE" >&2
printf '%s\n' 'Check it with: docker image inspect <image-ref>' >&2
exit 2
fi
run_official_attempt() {
# This function always runs as a background worker. Do not inherit the
# outer wrapper's EXIT guard: a normally finishing worker must never treat
# its concurrently running siblings as orphaned processes.
trap - EXIT
local attempt=$1
local jobs_dir="${TEST_DIRS[$((attempt - 1))]}"
local task_dir_for_attempt="$TASK_DIR"
local skills_dir_for_attempt="$SKILL_SOURCE"
local temporary_task_root=""
local eval_config=""
local console_log="$jobs_dir/console.log"
local bundled_skills_dir=""
local single_skill_dir=""
if [ -n "$PREBUILT_IMAGE" ]; then
temporary_task_root=$(mktemp -d "${TMPDIR:-/tmp}/skillsbench-prebuilt-task.XXXXXX")
trap '[ -z "$temporary_task_root" ] || rm -rf -- "$temporary_task_root"' RETURN
task_dir_for_attempt="$temporary_task_root/$(basename "$TASK_DIR")"
cp -a "$TASK_DIR" "$task_dir_for_attempt"
# This is the same helper used by the official SkillsBench AgentBeats worker.
PYTHONPATH="$SKILLSBENCH_ROOT" "$SKILLSBENCH_ROOT/.venv/bin/python" -c \
'import sys; from pathlib import Path; from skillsbench_agentbeats.worker import _write_task_md_prebuilt_image; _write_task_md_prebuilt_image(Path(sys.argv[1]), sys.argv[2])' \
"$task_dir_for_attempt/task.md" "$PREBUILT_IMAGE"
fi
# BenchFlow gives task-bundled Skills and external custom-runtime Skills
# different deployment policies. Some task Skills invoke a same-named
# subagent (for example enterprise-artifact-search); loading their SKILL.md
# externally exposes the instructions but does not register that agent type.
# Stage external compiler output into this disposable task copy so source and
# treatment runs differ only in Skill contents, not in deployment policy.
if [ "$SKILL_SOURCE_IS_TASK_BUNDLED" = false ]; then
if [ -z "$temporary_task_root" ]; then
printf '%s\n' 'External Skill staging requires a temporary task copy.' >&2
return 2
fi
bundled_skills_dir="$task_dir_for_attempt/environment/skills"
case "$bundled_skills_dir/" in
"$temporary_task_root/"*) ;;
*)
printf 'Refusing to stage external Skills outside the temporary task root: %s\n' \
"$bundled_skills_dir" >&2
return 2
;;
esac
mkdir -p "$bundled_skills_dir"
find "$bundled_skills_dir" -mindepth 1 -maxdepth 1 -exec rm -rf -- {} +
if [ -f "$SKILL_PAYLOAD_DIR/SKILL.md" ]; then
single_skill_dir="$bundled_skills_dir/$(basename "$SKILL_PAYLOAD_DIR")"
mkdir -p "$single_skill_dir"
cp -a "$SKILL_PAYLOAD_DIR/." "$single_skill_dir/"
else
cp -a "$SKILL_PAYLOAD_DIR/." "$bundled_skills_dir/"
fi
skills_dir_for_attempt="$bundled_skills_dir"
fi
if [ "$REQUIRE_SKILLS" = true ]; then
{
printf '\n\n## Mandatory Skill requirement\n\n'
printf 'You MUST invoke a relevant Skill tool before final verification. Do NOT load Skills at the beginning. First inspect the task and project, then invoke the Skill immediately before performing the work it covers and apply its guidance. Merely mentioning a Skill or reading files directly does not satisfy this requirement.\n'
} >> "$task_dir_for_attempt/task.md"
fi
# A source task Skill must also be resolved relative to the task copy.
if [ "$SKILL_SOURCE_IS_TASK_BUNDLED" = true ]; then
skills_dir_for_attempt="$task_dir_for_attempt/environment/skills"
fi
# JSON is valid YAML and keeps credentials out of the process command line
# while still using the official `bench eval run --config` entrypoint.
eval_config="$temporary_task_root/benchflow-eval.json"
umask 077
BF_CONFIG_PATH="$eval_config" BF_TASKS_DIR="$task_dir_for_attempt" \
BF_JOBS_DIR="$jobs_dir" BF_AGENT="$AGENT" BF_MODEL="$MODEL_REF" \
BF_SKILLS_DIR="$skills_dir_for_attempt" \
BF_SILICONFLOW_API_KEY="${SILICONFLOW_API_KEY:-}" \
BF_SILICONFLOW_BASE_URL="${SILICONFLOW_BASE_URL:-}" \
"$SKILLSBENCH_ROOT/.venv/bin/python" -c '
import json
import os
from pathlib import Path
agent_env = {}
if os.environ.get("BF_AGENT") == "opencode":
provider, model = os.environ["BF_MODEL"].split("/", 1)
opencode_config = {
"$schema": "https://opencode.ai/config.json",
"model": os.environ["BF_MODEL"],
"small_model": os.environ["BF_MODEL"],
"provider": {
provider: {
"npm": "@ai-sdk/openai-compatible",
"name": provider,
"options": {
"baseURL": os.environ["BF_SILICONFLOW_BASE_URL"],
"apiKey": "{env:SILICONFLOW_API_KEY}",
"timeout": 600000,
},
"models": {model: {"name": model}},
}
},
}
agent_env = {
"SILICONFLOW_API_KEY": os.environ["BF_SILICONFLOW_API_KEY"],
"SILICONFLOW_BASE_URL": os.environ["BF_SILICONFLOW_BASE_URL"],
"OPENCODE_CONFIG_CONTENT": json.dumps(opencode_config, separators=(",", ":")),
}
config = {
"tasks_dir": os.environ["BF_TASKS_DIR"],
"jobs_dir": os.environ["BF_JOBS_DIR"],
"agent": os.environ["BF_AGENT"],
"model": os.environ["BF_MODEL"],
"environment": "docker",
"skills_dir": os.environ["BF_SKILLS_DIR"],
"skill_mode": "with-skill",
"agent_env": agent_env,
# Do not let BenchFlow silently substitute its default non-root "agent"
# user. SkillsBench task images may intentionally provision task tools
# (for example the SDKMAN Maven installation) in /root; the agent must see
# the same
# task-provided toolchain as the verifier. JSON null is the BenchFlow
# documented root/no-lockdown sentinel. This changes only the evaluation
# process inside the task container, never the dataset image.
"sandbox_user": None,
# A verifier timeout occurs after the agent rollout has finished. Retrying
# it would sample a new agent trajectory and confound experiment results;
# preserve the timeout as a terminal evaluation-infrastructure outcome.
"retry": {"retry_on_verifier_infra": False},
}
Path(os.environ["BF_CONFIG_PATH"]).write_text(json.dumps(config), encoding="utf-8")
'
PYTHONPATH="$PROJECT_ROOT/scripts/evaluate${PYTHONPATH:+:$PYTHONPATH}" \
"${BENCH[@]}" eval run --config "$eval_config" >"$console_log" 2>&1
if [ "$REQUIRE_SKILLS" = true ]; then
if ! node "$PROJECT_ROOT/scripts/evaluate/verify-required-skill.mjs" \
"$jobs_dir" >>"$console_log" 2>&1; then
printf 'Required Skill validation failed for %s: no successful Skill invocation was found.\n' \
"$(basename "$jobs_dir")" >>"$console_log"
return 86
fi
fi
}
attempt=1
failures=0
WORKER_PIDS=()
FAILED_ATTEMPTS=()
FAILED_LOGS=()
declare -a ATTEMPT_STATUS ATTEMPT_STARTED ATTEMPT_ENDED ATTEMPT_REAPED
DASHBOARD_RENDERED=false
DASHBOARD_LINE_COUNT=$REPEAT
for ((dashboard_attempt = 1; dashboard_attempt <= REPEAT; dashboard_attempt++)); do
ATTEMPT_STATUS[$dashboard_attempt]=queued
ATTEMPT_STARTED[$dashboard_attempt]=0
ATTEMPT_ENDED[$dashboard_attempt]=0
ATTEMPT_REAPED[$dashboard_attempt]=false
done
format_elapsed() {
local seconds=$1
printf '%02d:%02d' "$((seconds / 60))" "$((seconds % 60))"
}
attempt_stage() {
local log=$1
if [ ! -s "$log" ]; then
printf '%s' preparing
elif rg -q 'Running verifier|Verifier running' "$log"; then
printf '%s' verifying
elif rg -q 'end_turn|Process terminated|Agent finished' "$log"; then
printf '%s' 'agent finalizing'
elif rg -q 'Prompt [0-9]+/[0-9]+' "$log"; then
printf '%s' 'agent running'
elif rg -q 'ACP agent:|Session:' "$log"; then
printf '%s' 'agent connecting'
elif rg -q 'Deploying skills|Skills deployed' "$log"; then
printf '%s' 'deploying skills'
elif rg -q 'Installing opencode' "$log"; then
printf '%s' 'installing agent'
elif rg -q 'Starting environment' "$log"; then
printf '%s' 'starting container'
else
printf '%s' preparing
fi
}
stage_progress() {
case "$1" in
preparing) printf '%d' 5 ;;
'starting container') printf '%d' 12 ;;
'installing agent') printf '%d' 20 ;;
'deploying skills') printf '%d' 30 ;;
'agent connecting') printf '%d' 40 ;;
'agent running') printf '%d' 70 ;;
'agent finalizing') printf '%d' 82 ;;
verifying) printf '%d' 92 ;;
finished) printf '%d' 100 ;;
*) printf '%d' 0 ;;
esac
}
attempt_outcome() {
local jobs_dir process_rc summary
jobs_dir=$1
process_rc=$2
summary="$jobs_dir/summary.json"
if [ "$process_rc" -ne 0 ]; then
printf '%s' ERROR
elif [ ! -f "$summary" ]; then
printf '%s' COMPLETE
else
"$SKILLSBENCH_ROOT/.venv/bin/python" - "$summary" <<'PY'
import json
import sys
data = json.load(open(sys.argv[1], encoding="utf-8"))
if int(data.get("errored", 0) or 0) or int(data.get("verifier_errored", 0) or 0):
print("ERROR", end="")
elif int(data.get("passed", data.get("pass", 0)) or 0):
print("PASS", end="")
else:
print("FAIL", end="")
PY
fi
}
render_dashboard() {
local now dashboard_attempt status elapsed stage test_name progress
local bar_done bar_left done_chars left_chars elapsed_end
local frame='' cursor_prefix=''
now=$(date +%s)
for ((dashboard_attempt = 1; dashboard_attempt <= REPEAT; dashboard_attempt++)); do
status=${ATTEMPT_STATUS[$dashboard_attempt]}
test_name="$(basename "${TEST_DIRS[$((dashboard_attempt - 1))]}") ($dashboard_attempt/$REPEAT)"
if [ "${ATTEMPT_STARTED[$dashboard_attempt]}" -gt 0 ]; then
elapsed_end=$now
[ "${ATTEMPT_ENDED[$dashboard_attempt]}" -eq 0 ] || elapsed_end=${ATTEMPT_ENDED[$dashboard_attempt]}
elapsed=$(format_elapsed "$((elapsed_end - ATTEMPT_STARTED[$dashboard_attempt]))")
else
elapsed='--:--'
fi
if [ "$status" = running ]; then
stage=$(attempt_stage "${TEST_DIRS[$((dashboard_attempt - 1))]}/console.log")
elif [ "$status" = queued ]; then
stage=waiting
else
stage=finished
fi
progress=$(stage_progress "$stage")
bar_done=$((progress * 20 / 100))
bar_left=$((20 - bar_done))
printf -v done_chars '%*s' "$bar_done" ''
printf -v left_chars '%*s' "$bar_left" ''
done_chars=${done_chars// /#}
left_chars=${left_chars// /-}
if [ "$stage" = finished ]; then
stage=$status
fi
printf -v frame '%s%-16s [%s%s] %3d%% %-16s elapsed=%s\033[K\n' \
"$frame" "$test_name" "$done_chars" "$left_chars" "$progress" "$stage" "$elapsed"
done
if [ "$DASHBOARD_RENDERED" = true ]; then
printf -v cursor_prefix '\033[%dA\r' "$DASHBOARD_LINE_COUNT"
fi
# DEC mode 2026 asks supporting terminals (including current xterm.js) to
# present the cursor move + complete frame atomically. Unsupported terminals
# ignore it and still receive one assembled write, without a blanking pass.
printf '\033[?2026h%s%s\033[?2026l' "$cursor_prefix" "$frame"
DASHBOARD_RENDERED=true
}
monitor_dashboard_batch() {
local remaining=${#pids[@]} index pid current_attempt current_log process_rc
printf '\033[?25l'
while [ "$remaining" -gt 0 ]; do
for index in "${!pids[@]}"; do
current_attempt=${batch_attempts[$index]}
[ "${ATTEMPT_REAPED[$current_attempt]}" = false ] || continue
pid=${pids[$index]}
if ! kill -0 "$pid" 2>/dev/null; then
process_rc=0
wait "$pid" || process_rc=$?
ATTEMPT_REAPED[$current_attempt]=true
ATTEMPT_ENDED[$current_attempt]=$(date +%s)
current_log=${batch_logs[$index]}
ATTEMPT_STATUS[$current_attempt]=$(attempt_outcome \
"${TEST_DIRS[$((current_attempt - 1))]}" "$process_rc")
if [ "$process_rc" -ne 0 ]; then
cleanup_owned_containers_for_jobs_dir "${TEST_DIRS[$((current_attempt - 1))]}"
failures=$((failures + 1))
FAILED_ATTEMPTS+=("$current_attempt")
FAILED_LOGS+=("$current_log")
fi
remaining=$((remaining - 1))
fi
done
render_dashboard
[ "$remaining" -eq 0 ] || sleep 1
done
printf '\033[?25h'
}
cleanup_owned_containers_for_jobs_dir() {
local jobs_dir=$1 container_id mount_source matched
while IFS= read -r container_id; do
[ -n "$container_id" ] || continue
matched=false
while IFS= read -r mount_source; do
case "$mount_source/" in
"$jobs_dir/"*) matched=true; break ;;
esac
done < <(docker inspect --format '{{range .Mounts}}{{println .Source}}{{end}}' "$container_id" 2>/dev/null || true)
if [ "$matched" = true ]; then
printf 'Cleaning orphaned BenchFlow container bound to %s: %s\n' \
"$jobs_dir" "$container_id" >>"$SETUP_LOG"
docker rm -f "$container_id" >/dev/null 2>&1 || true
fi
done < <(docker ps -aq --filter label=benchflow.owned=true)
}
terminate_workers() {
local signal=$1 pid jobs_dir
trap - EXIT INT TERM
for pid in "${WORKER_PIDS[@]:-}"; do
terminate_process_tree "$pid"
done
for pid in "${WORKER_PIDS[@]:-}"; do
wait "$pid" 2>/dev/null || true
done
for jobs_dir in "${TEST_DIRS[@]}"; do
cleanup_owned_containers_for_jobs_dir "$jobs_dir"
done
[ "$DASHBOARD_ENABLED" = false ] || printf '\033[?25h'
printf '\nStopped concurrent attempts after %s; completed artifacts remain in their test directories.\n' "$signal" >&2
exit 130
}
terminate_process_tree() {
local root_pid=$1 child_pid
while IFS= read -r child_pid; do
[ -n "$child_pid" ] && terminate_process_tree "$child_pid"
done < <(pgrep -P "$root_pid" 2>/dev/null || true)
kill -TERM "$root_pid" 2>/dev/null || true
}
cleanup_workers_on_exit() {
local exit_code=$? pid jobs_dir active_workers=0
trap - EXIT INT TERM
[ "$DASHBOARD_ENABLED" = false ] || printf '\033[?25h'
for pid in "${WORKER_PIDS[@]:-}"; do
if kill -0 "$pid" 2>/dev/null; then
active_workers=$((active_workers + 1))
terminate_process_tree "$pid"
fi
done
for pid in "${WORKER_PIDS[@]:-}"; do
wait "$pid" 2>/dev/null || true
done
if [ "$active_workers" -gt 0 ]; then
for jobs_dir in "${TEST_DIRS[@]}"; do
cleanup_owned_containers_for_jobs_dir "$jobs_dir"
done
fi
if [ "$active_workers" -gt 0 ]; then
printf '\nWrapper exited unexpectedly (code %d); stopped %d active worker(s) to prevent orphaned evaluations.\n' \
"$exit_code" "$active_workers" >&2
fi
exit "$exit_code"
}
trap cleanup_workers_on_exit EXIT
trap 'terminate_workers SIGINT' INT
trap 'terminate_workers SIGTERM' TERM
while [ "$attempt" -le "$REPEAT" ]; do
batch_last=$((attempt + MAX_PARALLEL - 1))
[ "$batch_last" -le "$REPEAT" ] || batch_last=$REPEAT
pids=()
batch_attempts=()
batch_logs=()
for current_attempt in $(seq "$attempt" "$batch_last"); do
current_jobs_dir="${TEST_DIRS[$((current_attempt - 1))]}"
current_log="$current_jobs_dir/console.log"
ATTEMPT_STATUS[$current_attempt]=running
ATTEMPT_STARTED[$current_attempt]=$(date +%s)
run_official_attempt "$current_attempt" &
pids+=("$!")
WORKER_PIDS+=("$!")
batch_attempts+=("$current_attempt")
batch_logs+=("$current_log")
done
if [ "$DASHBOARD_ENABLED" = true ]; then
monitor_dashboard_batch
else
for index in "${!pids[@]}"; do
pid="${pids[$index]}"
current_attempt="${batch_attempts[$index]}"
current_log="${batch_logs[$index]}"
process_rc=0
wait "$pid" || process_rc=$?
ATTEMPT_ENDED[$current_attempt]=$(date +%s)
ATTEMPT_STATUS[$current_attempt]=$(attempt_outcome \
"${TEST_DIRS[$((current_attempt - 1))]}" "$process_rc")
if [ "$process_rc" -ne 0 ]; then
cleanup_owned_containers_for_jobs_dir "${TEST_DIRS[$((current_attempt - 1))]}"
failures=$((failures + 1))
FAILED_ATTEMPTS+=("$current_attempt")
FAILED_LOGS+=("$current_log")
fi
done
fi
WORKER_PIDS=()
attempt=$((batch_last + 1))
done
trap - INT TERM
trap - EXIT
pass_count=0
fail_count=0
error_count=0
complete_count=0
for ((summary_attempt = 1; summary_attempt <= REPEAT; summary_attempt++)); do
case "${ATTEMPT_STATUS[$summary_attempt]}" in
PASS) pass_count=$((pass_count + 1)) ;;
FAIL) fail_count=$((fail_count + 1)) ;;
ERROR) error_count=$((error_count + 1)) ;;
*) complete_count=$((complete_count + 1)) ;;
esac
done
printf '\nResults:\n'
for ((summary_attempt = 1; summary_attempt <= REPEAT; summary_attempt++)); do
printf ' %-10s %-8s %s\n' \
"$(basename "${TEST_DIRS[$((summary_attempt - 1))]}")" \
"${ATTEMPT_STATUS[$summary_attempt]}" \
"${TEST_DIRS[$((summary_attempt - 1))]}"
done
printf 'Summary: %d total, %d passed, %d failed, %d errored' \
"$REPEAT" "$pass_count" "$fail_count" "$error_count"
[ "$complete_count" -eq 0 ] || printf ', %d completed without a readable summary' "$complete_count"
printf '.\n'
printf 'Artifacts root: %s\n' "$TASK_RESULTS_DIR"
[ "$failures" -eq 0 ] || exit 1
+15
View File
@@ -0,0 +1,15 @@
"""Restore the OpenCode launcher used by this project's verified runs."""
try:
from benchflow.agents.registry import AGENT_INSTALLERS
command = AGENT_INSTALLERS.get("opencode")
if command and "js-agents/bin/opencode \"$@\"' > /opt/benchflow/bin/opencode" in command:
AGENT_INSTALLERS["opencode"] = (
command
+ " && printf '%s\\n' '#!/bin/sh' "
"'exec /opt/benchflow/js-agents/bin/opencode \"$@\"' "
"> /opt/benchflow/bin/opencode && chmod +x /opt/benchflow/bin/opencode"
)
except ImportError:
pass
@@ -0,0 +1,94 @@
#!/usr/bin/env node
import fs from "node:fs";
import path from "node:path";
const [jobsDir] = process.argv.slice(2);
if (!jobsDir) {
console.error("Usage: verify-required-skill.mjs <jobs-dir>");
process.exit(2);
}
function collectTrajectories(directory) {
const trajectories = [];
for (const entry of fs.readdirSync(directory, {withFileTypes: true})) {
const entryPath = path.join(directory, entry.name);
if (entry.isDirectory()) {
trajectories.push(...collectTrajectories(entryPath));
} else if (entry.isFile() && entry.name === "acp_trajectory.jsonl") {
trajectories.push(entryPath);
}
}
return trajectories;
}
function contentText(event) {
return (event.content ?? [])
.map((item) => item?.content?.text)
.filter((text) => typeof text === "string")
.join("\n");
}
const invokedSkills = new Set();
const attemptedSkillCalls = [];
const parseErrors = [];
const trajectories = collectTrajectories(jobsDir);
for (const trajectory of trajectories) {
const lines = fs.readFileSync(trajectory, "utf8").split(/\r?\n/);
for (let index = 0; index < lines.length; index += 1) {
if (!lines[index].trim()) continue;
let event;
try {
event = JSON.parse(lines[index]);
} catch (error) {
parseErrors.push(`${trajectory}:${index + 1}: ${error.message}`);
continue;
}
if (event.type !== "tool_call") continue;
const text = contentText(event);
if (event.title === "skill") {
attemptedSkillCalls.push({
status: event.status ?? "unknown",
message: text.slice(0, 500),
trajectory,
line: index + 1,
});
}
if (event.status !== "completed") continue;
for (const match of text.matchAll(/<skill_content\s+name="([^"]+)"/g)) {
invokedSkills.add(match[1]);
}
}
}
const invoked = attemptedSkillCalls.some((attempt) => attempt.status === "completed");
const report = {
invoked,
invoked_skills: [...invokedSkills].sort(),
attempted_skill_calls: attemptedSkillCalls,
trajectory_files: trajectories.length,
parse_errors: parseErrors,
};
fs.writeFileSync(
path.join(jobsDir, "required-skill.json"),
`${JSON.stringify(report, null, 2)}\n`,
"utf8",
);
if (!invoked) {
const failedAttempts = attemptedSkillCalls.filter(
(attempt) => attempt.status !== "completed",
);
const failureDetail = failedAttempts.length
? `; ${failedAttempts.length} Skill call(s) failed: ${failedAttempts.map((attempt) => attempt.message || attempt.status).join(" | ")}`
: "";
console.error(
`No successful Skill invocation was observed${failureDetail}`,
);
process.exit(1);
}
console.log("Required Skill validation passed.");