Initial commit
This commit is contained in:
@@ -0,0 +1,11 @@
|
||||
ARG BASE_IMAGE
|
||||
FROM ${BASE_IMAGE}
|
||||
|
||||
# BenchFlow's official OpenCode registry uses precisely these paths.
|
||||
COPY node /opt/benchflow/node
|
||||
COPY js-agents /opt/benchflow/js-agents
|
||||
|
||||
RUN mkdir -p /opt/benchflow/bin \
|
||||
&& printf '%s\n' '#!/bin/sh' 'exec /opt/benchflow/node/bin/node /opt/benchflow/js-agents/bin/opencode "$@"' > /opt/benchflow/bin/opencode \
|
||||
&& chmod +x /opt/benchflow/bin/opencode \
|
||||
&& chmod -R a+rX /opt/benchflow
|
||||
@@ -0,0 +1,91 @@
|
||||
ARG TASK_IMAGE=ubuntu:24.04
|
||||
FROM ${TASK_IMAGE}
|
||||
|
||||
ARG NODE_VERSION=22.20.0
|
||||
ARG OPENCODE_VERSION=1.18.16
|
||||
ARG RUNTIME_BUILD_REVISION=2
|
||||
ARG RUNTIME_SOURCE_FINGERPRINT=unknown
|
||||
|
||||
# OpenCode's Skill, glob, and grep tools use ripgrep. Install it in the
|
||||
# reusable image so an evaluation never has to download rg from GitHub at
|
||||
# agent runtime.
|
||||
RUN set -eux; \
|
||||
if command -v rg >/dev/null 2>&1; then \
|
||||
:; \
|
||||
elif command -v apt-get >/dev/null 2>&1; then \
|
||||
apt-get update; \
|
||||
apt-get install -y --no-install-recommends ripgrep; \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
elif command -v dnf >/dev/null 2>&1; then \
|
||||
dnf -y install ripgrep; \
|
||||
dnf clean all; \
|
||||
elif command -v apk >/dev/null 2>&1; then \
|
||||
apk add --no-cache ripgrep; \
|
||||
else \
|
||||
echo 'OpenCode runtime requires a package manager to install ripgrep' >&2; \
|
||||
exit 127; \
|
||||
fi; \
|
||||
rg --version
|
||||
|
||||
RUN set -eux; \
|
||||
if [ -x /opt/benchflow/node/bin/node ]; then \
|
||||
/opt/benchflow/node/bin/node --version; \
|
||||
else \
|
||||
if ! command -v curl >/dev/null 2>&1 || ! command -v tar >/dev/null 2>&1; then \
|
||||
if command -v apt-get >/dev/null 2>&1; then \
|
||||
apt-get update; \
|
||||
apt-get install -y --no-install-recommends curl ca-certificates tar; \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
elif command -v dnf >/dev/null 2>&1; then \
|
||||
dnf -y install curl ca-certificates tar; \
|
||||
dnf clean all; \
|
||||
elif command -v apk >/dev/null 2>&1; then \
|
||||
apk add --no-cache curl ca-certificates tar; \
|
||||
else \
|
||||
echo 'Node/OpenCode bootstrap requires curl and tar' >&2; \
|
||||
exit 127; \
|
||||
fi; \
|
||||
fi; \
|
||||
arch="$(uname -m)"; \
|
||||
case "$arch" in \
|
||||
x86_64|amd64) node_arch=x64 ;; \
|
||||
aarch64|arm64) node_arch=arm64 ;; \
|
||||
*) echo "Unsupported architecture for Node.js: $arch" >&2; exit 1 ;; \
|
||||
esac; \
|
||||
temporary_dir="$(mktemp -d)"; \
|
||||
curl -fL \
|
||||
--retry 8 --retry-delay 2 \
|
||||
--connect-timeout 20 --max-time 900 \
|
||||
-o "$temporary_dir/node.tar.gz" \
|
||||
"https://nodejs.org/dist/v${NODE_VERSION}/node-v${NODE_VERSION}-linux-${node_arch}.tar.gz"; \
|
||||
mkdir -p /opt/benchflow/node /opt/benchflow/js-agents /opt/benchflow/bin; \
|
||||
tar -xzf "$temporary_dir/node.tar.gz" \
|
||||
-C /opt/benchflow/node --strip-components=1 --no-same-owner; \
|
||||
rm -rf "$temporary_dir"; \
|
||||
fi
|
||||
|
||||
ENV PATH="/opt/benchflow/bin:/opt/benchflow/js-agents/bin:/opt/benchflow/node/bin:${PATH}"
|
||||
|
||||
RUN set -eux; \
|
||||
if [ ! -x /opt/benchflow/js-agents/bin/opencode ]; then \
|
||||
npm install -g \
|
||||
--fetch-retries=8 \
|
||||
--fetch-retry-factor=2 \
|
||||
--fetch-retry-mintimeout=2000 \
|
||||
--fetch-retry-maxtimeout=60000 \
|
||||
--prefix /opt/benchflow/js-agents \
|
||||
"opencode-ai@${OPENCODE_VERSION}"; \
|
||||
fi; \
|
||||
npm cache clean --force; \
|
||||
rm -rf /root/.npm; \
|
||||
printf '%s\n' \
|
||||
'#!/bin/sh' \
|
||||
'exec /opt/benchflow/js-agents/bin/opencode "$@"' \
|
||||
> /opt/benchflow/bin/opencode; \
|
||||
chmod +x /opt/benchflow/bin/opencode; \
|
||||
chmod -R a+rX /opt/benchflow
|
||||
|
||||
LABEL org.skillc.opencode-runtime="1.18.16" \
|
||||
org.skillc.node-runtime="22.20.0" \
|
||||
org.skillc.opencode-runtime-build="${RUNTIME_BUILD_REVISION}" \
|
||||
org.skillc.opencode-runtime-source="${RUNTIME_SOURCE_FINGERPRINT}"
|
||||
@@ -0,0 +1,117 @@
|
||||
#!/usr/bin/env bash
|
||||
# One fully isolated attempt: provenance, container, agent, artifacts, verifier.
|
||||
|
||||
run_attempt() (
|
||||
local attempt_number=$1
|
||||
local model_label task_label condition_label run_prefix stamp run_root workspace run_id
|
||||
local started agent_started agent_ended ended agent_wall total_wall agent_exit
|
||||
local prompt verifier_exit run_status description skill_list install_root cost_estimate_cny
|
||||
local index proxy_var proxy_value
|
||||
local -a container_env_args=()
|
||||
model_label=$(normalize_run_component "${MODEL_ID##*/}")
|
||||
task_label=$(run_task_label)
|
||||
condition_label=$(run_condition_label)
|
||||
run_prefix="$HARNESS-$model_label-$task_label-$condition_label"
|
||||
stamp="$(date -u +%Y%m%dT%H%M%SZ)-$RANDOM"
|
||||
run_root=$(reserve_run_root "$run_prefix")
|
||||
workspace="$run_root/workspace"
|
||||
run_id="$HARNESS-$MODE-$stamp"
|
||||
mkdir -p "$workspace"
|
||||
started=$(now_ms)
|
||||
progress "$run_id" "Stage 1/5: preparing workspace and recording task/Skill provenance."
|
||||
cleanup_attempt() { docker rm -f "$run_id" >/dev/null 2>&1 || true; }
|
||||
trap cleanup_attempt EXIT
|
||||
|
||||
for proxy_var in HTTP_PROXY HTTPS_PROXY NO_PROXY http_proxy https_proxy no_proxy; do
|
||||
proxy_value="${!proxy_var-}"
|
||||
[ -z "$proxy_value" ] || container_env_args+=(-e "$proxy_var=$proxy_value")
|
||||
done
|
||||
progress "$run_id" "Stage 1/5: checking task image and input/Skill checksums."
|
||||
docker image inspect --format '{{.Id}}' "$IMAGE" > "$run_root/task-image-id.txt"
|
||||
find "$TASK_DIR/environment" -maxdepth 1 -type f -print0 | sort -z | xargs -0 -r sha256sum > "$run_root/input-sha256.txt"
|
||||
: > "$run_root/skill-sha256.txt"
|
||||
for index in "${!SKILL_DIRS[@]}"; do sha256sum "${SKILL_DIRS[$index]}/SKILL.md" >> "$run_root/skill-sha256.txt"; done
|
||||
skill_list=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}")
|
||||
case "$HARNESS" in
|
||||
opencode) install_root="$workspace/.opencode/skills" ;;
|
||||
hermes) install_root="$run_root/hermes-home/skills" ;;
|
||||
*) install_root="$workspace/.claude/skills" ;;
|
||||
esac
|
||||
progress "$run_id" "Stage 2/5: writing manifest and preparing the isolated task container."
|
||||
{
|
||||
printf 'harness=%s\nharness_version=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\n' "$HARNESS" "$(harness_version)" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF"
|
||||
printf 'task_slug=%s\ntask_dir=%s\ntask_image=%s\ntask_image_id=%s\n' "$TASK_SLUG" "$TASK_DIR" "$IMAGE" "$(tr -d '\n' < "$run_root/task-image-id.txt")"
|
||||
printf 'skill_source=%s\nskill_count=%s\nskill_names=%s\nskill_install_root=%s\n' "$SOURCE_SKILL" "${#SKILL_NAMES[@]}" "$skill_list" "$install_root"
|
||||
case "$HARNESS" in
|
||||
opencode) printf 'tool_approval_policy=opencode_auto\nopencode_auto_approval=true\n' ;;
|
||||
hermes) printf 'tool_approval_policy=hermes_yolo\n' ;;
|
||||
*) printf 'tool_approval_policy=claude_dangerously_skip_permissions\nauthentication_scope=Claude Code global OAuth profile\nsetting_sources=Claude Code defaults (user,project,local; required by OAuth)\n' ;;
|
||||
esac
|
||||
printf 'network_policy=%s\ncpu_limit=%s\nmemory_limit=%s\nsampling_parameters=Harness defaults (not overridden)\n' "$CONTAINER_NETWORK" "$CPU_LIMIT" "$MEMORY_LIMIT"
|
||||
printf 'timeout_seconds=%s\nverifier_timeout_seconds=%s\nmode=%s\nattempt_number=%s\n' "$TIMEOUT_SECONDS" "$VERIFIER_TIMEOUT_SECONDS" "$MODE" "$attempt_number"
|
||||
} > "$run_root/target-manifest.env"
|
||||
progress "$run_id" "Stage 2/5: starting the task container and mounting its workspace."
|
||||
docker run -d --name "$run_id" --cpus="$CPU_LIMIT" --memory="$MEMORY_LIMIT" --network "$CONTAINER_NETWORK" "${container_env_args[@]}" --mount "type=bind,src=$workspace,dst=/workspace" "$IMAGE" sleep infinity >/dev/null
|
||||
progress "$run_id" "Stage 3/5: container ready; building the agent prompt."
|
||||
if [ "$MODE" = probe ]; then
|
||||
prompt="The following Skills are installed and available: $skill_list.
|
||||
|
||||
Use bash commands only. Run these exact commands one at a time:
|
||||
1. docker exec $run_id bash -lc 'test -d /root && test -w /root && ls -1 /root | head -n 20'
|
||||
2. docker exec $run_id bash -lc 'probe_file=/root/.skill-agent-probe; printf probe-ok > \"\$probe_file\"; test -s \"\$probe_file\"; rm -f \"\$probe_file\"; printf HARNESS_CONTAINER_PROBE_OK'
|
||||
|
||||
Do not run any other command. If both commands succeed, finish with exactly: HARNESS_CONTAINER_PROBE_OK"
|
||||
else
|
||||
prompt="Use the installed Skills when relevant: $skill_list.
|
||||
|
||||
Complete this task:
|
||||
|
||||
$TASK_PROMPT
|
||||
|
||||
Execution environment:
|
||||
- The fresh task container is named $run_id.
|
||||
- Run every task inspection, analysis, edit, build, and test inside it with: docker exec $run_id ...
|
||||
- Do not run ls, find, grep, cat, Maven, or any task command against host paths.
|
||||
- Never inspect or access /mnt, the runner project, runs/, another attempt's workspace, task.md, oracle, verifier, or files outside the named container.
|
||||
- The host working directory is only a transport mount at /workspace; use it only for a helper file you create, then execute that helper through /workspace inside the named container.
|
||||
- Do not use host paths inside docker exec.
|
||||
- Do not access task.md, oracle, verifier, or files outside the current workspace and installed Skills.
|
||||
- Before finishing, inspect the result inside the task container and make sure the requested output or repository changes exist."
|
||||
fi
|
||||
agent_started=$(now_ms)
|
||||
progress "$run_id" "Stage 3/5: agent running (model output is being saved to agent-trace.txt)."
|
||||
set +e
|
||||
case "$HARNESS" in opencode) run_opencode "$run_root" "$workspace" "$prompt" ;; hermes) run_hermes "$run_root" "$workspace" "$prompt" ;; *) run_claude_code "$run_root" "$workspace" "$prompt" ;; esac
|
||||
agent_exit=$?
|
||||
set -e
|
||||
agent_ended=$(now_ms); ended=$(now_ms); agent_wall=$((agent_ended-agent_started)); total_wall=$((ended-started))
|
||||
progress "$run_id" "Stage 4/5: agent finished (exit code $agent_exit); exporting metrics and collecting artifacts."
|
||||
{
|
||||
printf 'agent_exit_code=%s\nharness=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\nagent_wall_ms=%s\ntotal_wall_ms=%s\n' "$agent_exit" "$HARNESS" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF" "$agent_wall" "$total_wall"
|
||||
[ "$agent_exit" -eq 124 ] && printf 'timed_out=true\n' || printf 'timed_out=false\n'
|
||||
} > "$run_root/metrics.env"
|
||||
export_agent_session "$run_root" "$agent_wall" "$total_wall"
|
||||
cost_estimate_cny=unavailable
|
||||
[ ! -s "$run_root/agent-metrics.json" ] || cost_estimate_cny=$(node -e 'const m=require(process.argv[1]); const c=m.official_cost_estimate?.amount_cny; process.stdout.write(Number.isFinite(c) ? c.toFixed(8) : "unavailable")' "$run_root/agent-metrics.json")
|
||||
printf 'official_cost_estimate_cny=%s\n' "$cost_estimate_cny" >> "$run_root/metrics.env"
|
||||
if [ "$MODE" = probe ]; then
|
||||
if [ "$agent_exit" -eq 0 ] && rg -q HARNESS_CONTAINER_PROBE_OK "$run_root/agent-trace.txt"; then run_status=success; description=none; else run_status=probe_failed; description='The Harness probe did not complete successfully. Inspect agent-trace.txt.'; fi
|
||||
printf '# Harness probe summary\n\nstatus=%s\nagent_exit_code=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md"
|
||||
else
|
||||
progress "$run_id" "Stage 4/5: capturing changed files and artifacts from the task container."
|
||||
docker diff "$run_id" > "$run_root/container-diff.txt" 2>/dev/null || true
|
||||
capture_agent_artifacts "$run_root" "$run_id"
|
||||
progress "$run_id" "Stage 5/5: running the task verifier."
|
||||
set +e; verify_output "$run_root" "$run_id"; verifier_exit=$?; set -e
|
||||
progress "$run_id" "Stage 5/5: verifier finished (exit code $verifier_exit); writing run summary."
|
||||
printf 'verifier_exit=%s\n' "$verifier_exit" >> "$run_root/metrics.env"
|
||||
if [ "$agent_exit" -eq 0 ] && [ "$verifier_exit" = 0 ]; then run_status=success; description=none
|
||||
elif [ "$agent_exit" -eq 124 ] && [ "$verifier_exit" != 0 ]; then run_status=agent_timed_out; description="The agent timed out after ${TIMEOUT_SECONDS} seconds and the task verifier failed."
|
||||
elif [ "$verifier_exit" != 0 ]; then run_status=verifier_failed; description="The agent completed, but the task verifier failed (exit code ${verifier_exit})."
|
||||
else run_status=agent_failed_output_verified; description="The agent exited with code ${agent_exit}, but its output passed verification."; fi
|
||||
printf '# Raw run summary\n\nstatus=%s\nagent_exit_code=%s\nverifier_exit=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$verifier_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md"
|
||||
fi
|
||||
printf 'run_status=%s\nproblem_description=%s\n' "$run_status" "$description" >> "$run_root/metrics.env"
|
||||
progress "$run_id" "Completed: $run_status. Details saved to $run_root."
|
||||
[ "$run_status" = success ]
|
||||
)
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared presentation and run-directory helpers for run-raw-task.sh.
|
||||
|
||||
now_ms() { node -p 'Date.now()'; }
|
||||
|
||||
progress() {
|
||||
local run_label=$1 message=$2
|
||||
printf '[%s] [%s] %s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$run_label" "$message"
|
||||
}
|
||||
|
||||
progress_bar() {
|
||||
local label=$1 completed=$2 total=$3 width=24 filled empty percent bar
|
||||
[ "$total" -gt 0 ] || return
|
||||
filled=$((completed * width / total))
|
||||
empty=$((width - filled))
|
||||
percent=$((completed * 100 / total))
|
||||
printf -v bar '%*s' "$filled" ''
|
||||
bar=${bar// /#}
|
||||
printf -v empty '%*s' "$empty" ''
|
||||
empty=${empty// /-}
|
||||
printf '[%s] [%s%s] %d/%d (%d%%)\n' "$label" "$bar" "$empty" "$completed" "$total" "$percent"
|
||||
}
|
||||
|
||||
normalize_run_component() {
|
||||
printf '%s' "$1" | tr '[:upper:]' '[:lower:]' | tr -cs 'a-z0-9' '-' | sed -E 's/^-+//; s/-+$//'
|
||||
}
|
||||
|
||||
run_task_label() {
|
||||
case "$TASK_SLUG" in
|
||||
111-offer-letter-generator) printf '%s' 'task-1' ;;
|
||||
222-software-dependency-audit) printf '%s' 'task-2' ;;
|
||||
333-fix-build-google-auto) printf '%s' 'task-3' ;;
|
||||
*) printf 'task-%s' "$(normalize_run_component "$TASK_SLUG")" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
run_condition_label() {
|
||||
case "$SOURCE_SKILL" in
|
||||
*/results/model-compiled-skills/*|*/dist/*|*/conditions/*|/tmp/*) printf '%s' '编译后skill' ;;
|
||||
*) printf '%s' '原始skill' ;;
|
||||
esac
|
||||
}
|
||||
|
||||
reserve_run_root() {
|
||||
# Allocation and mkdir share one lock: concurrent attempts must never claim
|
||||
# the same trace/workspace directory.
|
||||
local prefix=$1 run_root suffix=1 lock_fd
|
||||
local lock_path="/tmp/skill-agent-raw-name-index.lock"
|
||||
mkdir -p "$PROJECT_ROOT/runs/$MODE"
|
||||
exec {lock_fd}>"$lock_path"
|
||||
flock "$lock_fd"
|
||||
# Do the check and the mkdir while holding the same lock. Starting at 1
|
||||
# also tolerates old, interrupted runs and avoids parsing a path whose
|
||||
# prefix itself contains hyphens.
|
||||
while :; do
|
||||
run_root="$PROJECT_ROOT/runs/$MODE/$prefix-$suffix"
|
||||
if mkdir "$run_root" 2>/dev/null; then
|
||||
break
|
||||
fi
|
||||
suffix=$((suffix + 1))
|
||||
done
|
||||
flock -u "$lock_fd"
|
||||
exec {lock_fd}>&-
|
||||
printf '%s' "$run_root"
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
#!/usr/bin/env bash
|
||||
# Container output verification and bounded artifact capture.
|
||||
|
||||
verify_output() {
|
||||
local run_root=$1 run_id=$2 log_dir="$1/verifier"
|
||||
local verifier_started verifier_ended verifier_wall docker_exit reward verification_status description
|
||||
mkdir -p "$log_dir"
|
||||
# BenchFlow's native task.md verifier contract: upload verifier/ to
|
||||
# /verifier, then expose the legacy /tests path as a symlink only if the
|
||||
# image has not already provided real /tests content. Do not overwrite that
|
||||
# content; older verifier scripts may legitimately depend on it.
|
||||
docker exec "$run_id" mkdir -p /verifier /logs/verifier
|
||||
docker cp "$VERIFIER_SOURCE/." "$run_id:/verifier"
|
||||
docker exec "$run_id" bash -lc '[ -e /tests ] || ln -s /verifier /tests'
|
||||
docker exec "$run_id" chmod +x /verifier/test.sh
|
||||
verifier_started=$(now_ms)
|
||||
if timeout --foreground --signal=INT --kill-after=30s "${VERIFIER_TIMEOUT_SECONDS}s" \
|
||||
docker exec "${VERIFIER_ENV_ARGS[@]}" "$run_id" /verifier/test.sh > "$log_dir/verifier.stdout.log" 2>&1; then
|
||||
docker_exit=0
|
||||
else
|
||||
docker_exit=$?
|
||||
fi
|
||||
docker cp "$run_id:/logs/verifier/." "$log_dir" >/dev/null 2>&1 || true
|
||||
verifier_ended=$(now_ms)
|
||||
verifier_wall=$((verifier_ended - verifier_started))
|
||||
reward=""
|
||||
[ ! -f "$log_dir/reward.txt" ] || reward=$(tr -d '[:space:]' < "$log_dir/reward.txt")
|
||||
if [ "$docker_exit" -eq 0 ] && [ "$reward" = 1 ]; then
|
||||
verification_status=passed
|
||||
description=none
|
||||
else
|
||||
verification_status=failed
|
||||
description="Verifier exited with code ${docker_exit} and wrote reward=${reward:-missing}. See verifier.stdout.log for details."
|
||||
fi
|
||||
{
|
||||
printf 'docker_exit_code=%s\nreward=%s\nverifier_wall_ms=%s\n' "$docker_exit" "$reward" "$verifier_wall"
|
||||
printf 'verifier_log=%s\nverification_status=%s\nproblem_description=%s\n' "$log_dir/verifier.stdout.log" "$verification_status" "$description"
|
||||
} > "$log_dir/summary.env"
|
||||
[ "$verification_status" = passed ] && return 0
|
||||
printf 'VERIFIER_ISSUE: %s\n' "$description" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
capture_agent_artifacts() {
|
||||
local run_root=$1 run_id=$2 diff_path="$1/container-diff.txt"
|
||||
local artifact_root="$1/artifacts" status container_path relative_path size destination
|
||||
mkdir -p "$artifact_root"
|
||||
while IFS=' ' read -r status container_path; do
|
||||
[ "$status" = A ] || [ "$status" = C ] || continue
|
||||
# Copy only modest, task-created outputs. Package caches and the mounted
|
||||
# workspace are inputs/ephemera, never benchmark artifacts.
|
||||
case "$container_path" in
|
||||
/root/.cache/*|/root/.local/*|/root/.m2/*|/home/*/.cache/*|/home/*/.local/*|/home/*/.m2/*|*/.git/*|/etc/*|/opt/*|/tmp/*|/usr/*|/var/*|/workspace/*) continue ;;
|
||||
esac
|
||||
docker exec "$run_id" test -f "$container_path" >/dev/null 2>&1 || continue
|
||||
size=$(docker exec "$run_id" stat -c '%s' "$container_path" 2>/dev/null || printf '0')
|
||||
[[ "$size" =~ ^[0-9]+$ ]] || continue
|
||||
[ "$size" -le 20971520 ] || continue
|
||||
relative_path=${container_path#/}
|
||||
destination="$artifact_root/$relative_path"
|
||||
mkdir -p "$(dirname "$destination")"
|
||||
docker cp "$run_id:$container_path" "$destination" >/dev/null
|
||||
done < "$diff_path"
|
||||
if find "$artifact_root" -type f -print -quit | grep -q .; then
|
||||
find "$artifact_root" -type f -print0 | sort -z | xargs -0 sha256sum > "$run_root/output-sha256.txt"
|
||||
fi
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
#!/usr/bin/env bash
|
||||
# Harness-specific configuration and agent invocation.
|
||||
|
||||
install_skill_set() {
|
||||
local destination=$1 index
|
||||
mkdir -p "$destination"
|
||||
for index in "${!SKILL_DIRS[@]}"; do
|
||||
cp -a "${SKILL_DIRS[$index]}" "$destination/${SKILL_NAMES[$index]}"
|
||||
done
|
||||
}
|
||||
|
||||
write_opencode_config() {
|
||||
local config_path=$1
|
||||
node - "$config_path" "$PROVIDER_ID" "$MODEL_ID" "$PROVIDER_BASE_URL" "$PROVIDER_API_KEY" <<'NODE'
|
||||
const fs = require("fs");
|
||||
const [configPath, provider, model, baseURL, apiKey] = process.argv.slice(2);
|
||||
const config = {$schema: "https://opencode.ai/config.json", model: `${provider}/${model}`,
|
||||
provider: {[provider]: {npm: "@ai-sdk/openai-compatible", name: provider,
|
||||
options: {baseURL, apiKey}, models: {[model]: {name: model}}}}};
|
||||
fs.writeFileSync(configPath, `${JSON.stringify(config, null, 2)}\n`, {mode: 0o600});
|
||||
NODE
|
||||
}
|
||||
|
||||
write_hermes_profile() {
|
||||
local profile_home=$1 workspace=$2
|
||||
mkdir -p "$profile_home/skills"
|
||||
touch "$profile_home/.no-bundled-skills"
|
||||
install_skill_set "$profile_home/skills"
|
||||
{
|
||||
printf '%s\n' 'model:' " default: \"$MODEL_ID\"" ' provider: custom' ' base_url: "https://api.siliconflow.cn/v1"'
|
||||
printf '%s\n' 'terminal:' ' backend: local' " cwd: \"$workspace\"" ' timeout: 180' ' home_mode: profile'
|
||||
printf '%s\n' 'memory:' ' memory_enabled: false' ' user_profile_enabled: false'
|
||||
printf '%s\n' 'skills:' ' external_dirs: []' ' inline_shell: false' ' write_approval: true'
|
||||
printf '%s\n' 'curator:' ' enabled: false' 'fallback_providers: []'
|
||||
printf '%s\n' 'delegation:' ' orchestrator_enabled: false' ' max_spawn_depth: 1' ' max_concurrent_children: 1' ' max_async_children: 1'
|
||||
} > "$profile_home/config.yaml"
|
||||
}
|
||||
|
||||
run_opencode() {
|
||||
local run_root=$1 workspace=$2 prompt=$3
|
||||
install_skill_set "$workspace/.opencode/skills"
|
||||
write_opencode_config "$run_root/opencode.json"
|
||||
(cd "$workspace"; OPENCODE_CONFIG="$run_root/opencode.json" OPENCODE_CONFIG_DIR="$workspace/.opencode" \
|
||||
XDG_DATA_HOME="$run_root/opencode-data" XDG_STATE_HOME="$run_root/opencode-state" \
|
||||
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
|
||||
opencode --pure --auto --model "$PROVIDER_ID/$MODEL_ID" run "$prompt") > "$run_root/agent-trace.txt" 2>&1
|
||||
}
|
||||
|
||||
run_hermes() {
|
||||
local run_root=$1 workspace=$2 prompt=$3 skill_csv
|
||||
skill_csv=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}")
|
||||
write_hermes_profile "$run_root/hermes-home" "$workspace"
|
||||
(cd "$workspace"; OPENAI_API_KEY="$SILICONFLOW_API_KEY" HERMES_HOME="$run_root/hermes-home" HERMES_OPTIONAL_SKILLS="" \
|
||||
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
|
||||
hermes --yolo --provider custom --model "$MODEL_ID" --toolsets terminal --skills "$skill_csv" --oneshot "$prompt") > "$run_root/agent-trace.txt" 2>&1
|
||||
}
|
||||
|
||||
prepare_claude_workspace() { install_skill_set "$1/.claude/skills"; }
|
||||
|
||||
run_claude_code() {
|
||||
local run_root=$1 workspace=$2 prompt=$3
|
||||
prepare_claude_workspace "$workspace"
|
||||
(cd "$workspace"; ANTHROPIC_BASE_URL="$PROVIDER_BASE_URL" ANTHROPIC_AUTH_TOKEN="$PROVIDER_API_KEY" \
|
||||
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
|
||||
claude --print --output-format json --model "$MODEL_ID" --dangerously-skip-permissions "$prompt") \
|
||||
> "$run_root/claude-result.json" 2> "$run_root/claude-stderr.txt"
|
||||
cat "$run_root/claude-result.json" "$run_root/claude-stderr.txt" > "$run_root/agent-trace.txt"
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env bash
|
||||
# Create a derived task image that satisfies BenchFlow's official OpenCode
|
||||
# bootstrap checks without requiring each evaluation container to download Node.
|
||||
set -euo pipefail
|
||||
|
||||
PROJECT_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)
|
||||
DOCKERFILE="$PROJECT_ROOT/scripts/evaluate/Dockerfile.benchflow-opencode"
|
||||
NODE_VERSION=22.20.0
|
||||
BASE_IMAGE=""
|
||||
OUTPUT_IMAGE=""
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage:
|
||||
bash scripts/evaluate/prepare-benchflow-opencode-image.sh \
|
||||
--base-image <existing-task-image> \
|
||||
--tag <new-derived-image-tag>
|
||||
|
||||
The base image is never changed. The resulting image contains the exact Node
|
||||
runtime expected by BenchFlow plus opencode-ai@latest under /opt/benchflow.
|
||||
EOF
|
||||
}
|
||||
|
||||
while [ "$#" -gt 0 ]; do
|
||||
case "$1" in
|
||||
--base-image) BASE_IMAGE=${2:?missing value for --base-image}; shift 2 ;;
|
||||
--tag) OUTPUT_IMAGE=${2:?missing value for --tag}; shift 2 ;;
|
||||
--node-version) NODE_VERSION=${2:?missing value for --node-version}; shift 2 ;;
|
||||
-h|--help) usage; exit 0 ;;
|
||||
*) printf 'Unknown option: %s\n' "$1" >&2; usage >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
[ -n "$BASE_IMAGE" ] || { printf '%s\n' '--base-image is required.' >&2; exit 2; }
|
||||
[ -n "$OUTPUT_IMAGE" ] || { printf '%s\n' '--tag is required.' >&2; exit 2; }
|
||||
command -v docker >/dev/null || { printf '%s\n' 'docker is required.' >&2; exit 127; }
|
||||
command -v curl >/dev/null || { printf '%s\n' 'curl is required.' >&2; exit 127; }
|
||||
[ -f "$DOCKERFILE" ] || { printf 'Missing Dockerfile: %s\n' "$DOCKERFILE" >&2; exit 1; }
|
||||
docker image inspect "$BASE_IMAGE" >/dev/null
|
||||
|
||||
BUILD_CONTEXT=$(mktemp -d "${TMPDIR:-/tmp}/benchflow-opencode-image.XXXXXX")
|
||||
cleanup() { rm -rf -- "$BUILD_CONTEXT"; }
|
||||
trap cleanup EXIT
|
||||
|
||||
NODE_ARCH=$(uname -m)
|
||||
case "$NODE_ARCH" in
|
||||
x86_64|amd64) NODE_ARCH=x64 ;;
|
||||
aarch64|arm64) NODE_ARCH=arm64 ;;
|
||||
*) printf 'Unsupported architecture: %s\n' "$NODE_ARCH" >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
NODE_ARCHIVE="node-v${NODE_VERSION}-linux-${NODE_ARCH}.tar.xz"
|
||||
printf 'Downloading Node.js %s for the BenchFlow runtime cache...\n' "$NODE_VERSION"
|
||||
curl --fail --location --retry 3 --output "$BUILD_CONTEXT/node.tar.xz" \
|
||||
"https://nodejs.org/dist/v${NODE_VERSION}/${NODE_ARCHIVE}"
|
||||
mkdir -p "$BUILD_CONTEXT/node"
|
||||
tar -xJf "$BUILD_CONTEXT/node.tar.xz" -C "$BUILD_CONTEXT/node" --strip-components=1 --no-same-owner
|
||||
|
||||
printf '%s\n' 'Installing the official opencode-ai package into the runtime cache...'
|
||||
"$BUILD_CONTEXT/node/bin/npm" install --global --prefix "$BUILD_CONTEXT/js-agents" opencode-ai@latest
|
||||
[ -x "$BUILD_CONTEXT/js-agents/bin/opencode" ] || { printf '%s\n' 'opencode-ai installation did not create its executable.' >&2; exit 1; }
|
||||
|
||||
printf 'Building derived image: %s\n' "$OUTPUT_IMAGE"
|
||||
docker build --build-arg "BASE_IMAGE=$BASE_IMAGE" --tag "$OUTPUT_IMAGE" --file "$DOCKERFILE" "$BUILD_CONTEXT"
|
||||
printf 'Ready: %s\n' "$OUTPUT_IMAGE"
|
||||
@@ -0,0 +1,806 @@
|
||||
#!/usr/bin/env bash
|
||||
# Thin compatibility wrapper around the official SkillsBench/BenchFlow runner.
|
||||
set -euo pipefail
|
||||
|
||||
PROJECT_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)
|
||||
cd "$PROJECT_ROOT"
|
||||
|
||||
# Evaluation credentials are user-owned configuration. Load them once for
|
||||
# BenchFlow so its isolated agent container receives only the registered
|
||||
# provider variables, never a copied host config directory.
|
||||
if [ -f "$PROJECT_ROOT/.env" ]; then
|
||||
set -a
|
||||
# shellcheck disable=SC1091
|
||||
source "$PROJECT_ROOT/.env"
|
||||
set +a
|
||||
fi
|
||||
|
||||
# OpenCode's OpenAI-compatible provider expects a base URL, while the project
|
||||
# .env records the concrete chat-completions endpoint used by other harnesses.
|
||||
if [ -n "${SILICONFLOW_CHAT_COMPLETIONS_URL:-}" ]; then
|
||||
SILICONFLOW_BASE_URL=${SILICONFLOW_CHAT_COMPLETIONS_URL%/chat/completions}
|
||||
fi
|
||||
: "${SILICONFLOW_BASE_URL:=https://api.siliconflow.cn/v1}"
|
||||
export SILICONFLOW_BASE_URL
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage:
|
||||
bash scripts/evaluate/run-raw-task.sh \
|
||||
--harness opencode \
|
||||
--model <provider/model> \
|
||||
--task <SkillsBench task directory> \
|
||||
--skill-source <skill directory or skills root> \
|
||||
--output <result directory> \
|
||||
[--require-skill] \
|
||||
[--repeat N] [--max-parallel N] [--image <local-image-ref>]
|
||||
|
||||
This wrapper delegates every attempt to `bench eval run --sandbox docker`.
|
||||
It does not inject verifier files or implement scoring.
|
||||
Provider credentials are loaded from the project `.env`.
|
||||
|
||||
Each repeat is allocated before any workers start, at:
|
||||
<output>/test-NNN/
|
||||
The official BenchFlow job and summary remain inside that test directory.
|
||||
Interactive terminals show one in-place progress row per attempt for both
|
||||
single and concurrent runs. Worker output is kept in test-NNN/console.log and
|
||||
only a concise result summary is printed after the progress display finishes.
|
||||
|
||||
External Skill sources are staged into each disposable task copy's
|
||||
environment/skills directory. This keeps BenchFlow's task-bundled deployment
|
||||
and subagent registration policy identical for source and compiled Skills.
|
||||
Both <output>/skills/<name>/SKILL.md and <skills-root>/<name>/SKILL.md layouts
|
||||
are accepted.
|
||||
|
||||
--image makes an ephemeral copy of the selected task and applies the official
|
||||
SkillsBench prebuilt-image policy to that copy. The original task and Skills
|
||||
remain untouched. This avoids a Docker build when the supplied image already
|
||||
exists locally.
|
||||
|
||||
Without --image, the wrapper uses skillc/<task-name>:local. It builds and tags
|
||||
that image only when needed, then adds the pinned Node/OpenCode runtime and
|
||||
cleans installation caches. Later evaluations reuse the same single image
|
||||
through the official prebuilt-image policy.
|
||||
|
||||
--require-skill makes at least one Skill invocation an evaluation invariant
|
||||
rather than an agent choice. Every temporary task prompt requires the agent to
|
||||
load a relevant Skill during the task, immediately before applying its guidance.
|
||||
The completed ACP trajectory is checked for a successful Skill call. Without
|
||||
this flag, Skill invocation remains the agent's choice. An attempt with no
|
||||
successful Skill call exits with status 86 and writes required-skill.json.
|
||||
EOF
|
||||
}
|
||||
|
||||
HARNESS=""
|
||||
MODEL_REF=""
|
||||
TASK_DIR=""
|
||||
SKILL_SOURCE=""
|
||||
OUTPUT_DIR=""
|
||||
REPEAT=1
|
||||
MAX_PARALLEL=3
|
||||
REPEAT_SEEN=false
|
||||
MAX_PARALLEL_SEEN=false
|
||||
PREBUILT_IMAGE=""
|
||||
REQUIRE_SKILLS=false
|
||||
|
||||
while [ "$#" -gt 0 ]; do
|
||||
case "$1" in
|
||||
--harness) HARNESS=${2:?missing value for --harness}; shift 2 ;;
|
||||
--model) MODEL_REF=${2:?missing value for --model}; shift 2 ;;
|
||||
--task) TASK_DIR=${2:?missing value for --task}; shift 2 ;;
|
||||
--skill-source) SKILL_SOURCE=${2:?missing value for --skill-source}; shift 2 ;;
|
||||
--output) OUTPUT_DIR=${2:?missing value for --output}; shift 2 ;;
|
||||
--require-skill) REQUIRE_SKILLS=true; shift ;;
|
||||
--repeat)
|
||||
[ "$REPEAT_SEEN" = false ] || { printf '%s\n' 'Duplicate --repeat option.' >&2; exit 2; }
|
||||
REPEAT=${2:?missing value for --repeat}
|
||||
REPEAT_SEEN=true
|
||||
shift 2
|
||||
;;
|
||||
--max-parallel)
|
||||
[ "$MAX_PARALLEL_SEEN" = false ] || { printf '%s\n' 'Duplicate --max-parallel option.' >&2; exit 2; }
|
||||
MAX_PARALLEL=${2:?missing value for --max-parallel}
|
||||
MAX_PARALLEL_SEEN=true
|
||||
shift 2
|
||||
;;
|
||||
--image|--prebuilt-image) PREBUILT_IMAGE=${2:?missing value for --image}; shift 2 ;;
|
||||
-h|--help) usage; exit 0 ;;
|
||||
*) printf 'Unsupported option for the official BenchFlow wrapper: %s\n' "$1" >&2; usage >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
[ -n "$HARNESS" ] || { printf '%s\n' '--harness is required.' >&2; exit 2; }
|
||||
[ -n "$MODEL_REF" ] || { printf '%s\n' '--model is required.' >&2; exit 2; }
|
||||
[ -n "$TASK_DIR" ] || { printf '%s\n' '--task is required.' >&2; exit 2; }
|
||||
[ -n "$SKILL_SOURCE" ] || { printf '%s\n' '--skill-source is required.' >&2; exit 2; }
|
||||
[ -n "$OUTPUT_DIR" ] || { printf '%s\n' '--output is required.' >&2; exit 2; }
|
||||
[[ "$REPEAT" =~ ^[1-9][0-9]*$ ]] || { printf '%s\n' '--repeat must be a positive integer.' >&2; exit 2; }
|
||||
[[ "$MAX_PARALLEL" =~ ^[1-9][0-9]*$ ]] || { printf '%s\n' '--max-parallel must be a positive integer.' >&2; exit 2; }
|
||||
|
||||
TASK_DIR=$(realpath "$TASK_DIR")
|
||||
SKILL_SOURCE=$(realpath "$SKILL_SOURCE")
|
||||
OUTPUT_DIR=$(realpath -m "$OUTPUT_DIR")
|
||||
[ -f "$TASK_DIR/task.md" ] || { printf 'Not a native SkillsBench task: %s\n' "$TASK_DIR" >&2; exit 2; }
|
||||
[ -d "$SKILL_SOURCE" ] || { printf 'Skill source does not exist: %s\n' "$SKILL_SOURCE" >&2; exit 2; }
|
||||
|
||||
# Compilers may emit either a Skills root directly:
|
||||
# <output>/skill-a/SKILL.md
|
||||
# or a package containing that root:
|
||||
# <output>/skills/skill-a/SKILL.md
|
||||
# Normalize both forms before staging the selected Skills into a task copy.
|
||||
SKILL_PAYLOAD_DIR="$SKILL_SOURCE"
|
||||
if [ ! -f "$SKILL_SOURCE/SKILL.md" ] && \
|
||||
[ -d "$SKILL_SOURCE/skills" ] && \
|
||||
[ -z "$(find "$SKILL_SOURCE" -mindepth 2 -maxdepth 2 -type f -name SKILL.md -print -quit)" ] && \
|
||||
[ -n "$(find "$SKILL_SOURCE/skills" -type f -name SKILL.md -print -quit)" ]; then
|
||||
SKILL_PAYLOAD_DIR=$(realpath "$SKILL_SOURCE/skills")
|
||||
fi
|
||||
|
||||
if [ "$REQUIRE_SKILLS" = true ]; then
|
||||
[ -n "$(find "$SKILL_SOURCE" -type f -name SKILL.md -print -quit)" ] || {
|
||||
printf 'No SKILL.md files were found under --skill-source: %s\n' "$SKILL_SOURCE" >&2
|
||||
exit 2
|
||||
}
|
||||
fi
|
||||
TASK_BUNDLED_SKILLS="$TASK_DIR/environment/skills"
|
||||
SKILL_SOURCE_IS_TASK_BUNDLED=false
|
||||
if [ -d "$TASK_BUNDLED_SKILLS" ] && [ "$SKILL_SOURCE" = "$(realpath "$TASK_BUNDLED_SKILLS")" ]; then
|
||||
SKILL_SOURCE_IS_TASK_BUNDLED=true
|
||||
fi
|
||||
|
||||
case "$HARNESS" in
|
||||
opencode)
|
||||
AGENT=opencode
|
||||
# This project also keeps OPENCODE_* variables for the legacy harness.
|
||||
# They are OpenCode-hosted-service credentials, not SiliconFlow
|
||||
# credentials, and current OpenCode releases may prefer them over a
|
||||
# configured custom provider. The official BenchFlow container must use
|
||||
# only the explicitly registered SiliconFlow provider below.
|
||||
unset OPENCODE_API_KEY OPENCODE_CHAT_COMPLETIONS_URL
|
||||
# BenchFlow intentionally inherits only its built-in provider variables.
|
||||
# SiliconFlow is an OpenAI-compatible custom provider, so its two values
|
||||
# are supplied through the official evaluation config below. They
|
||||
# originate in .env; no task, Skill, or verifier file is changed.
|
||||
[ -n "${SILICONFLOW_API_KEY:-}" ] || {
|
||||
printf '%s\n' 'SILICONFLOW_API_KEY is required in .env for --harness opencode.' >&2
|
||||
exit 2
|
||||
}
|
||||
# Values are placed in a mode-0600 temporary BenchFlow config per attempt.
|
||||
# Credentials must not appear in CLI arguments visible through `ps`.
|
||||
;;
|
||||
claude-code|claude) AGENT=claude-agent-acp ;;
|
||||
*)
|
||||
printf 'Harness %q is not a registered BenchFlow ACP agent in the pinned SkillsBench runner.\n' "$HARNESS" >&2
|
||||
printf '%s\n' 'Use an official BenchFlow agent name (for example: opencode or claude-agent-acp).' >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
SKILLSBENCH_ROOT=${SKILLSBENCH_ROOT:-"$PROJECT_ROOT/data/skills-bench"}
|
||||
if command -v bench >/dev/null 2>&1; then
|
||||
BENCH=(bench)
|
||||
elif command -v benchflow >/dev/null 2>&1; then
|
||||
BENCH=(benchflow)
|
||||
elif [ -x "$SKILLSBENCH_ROOT/.venv/bin/bench" ]; then
|
||||
# `uv sync` keeps the official CLI inside the dataset virtual environment.
|
||||
BENCH=("$SKILLSBENCH_ROOT/.venv/bin/bench")
|
||||
elif command -v uv >/dev/null 2>&1 && [ -f "$SKILLSBENCH_ROOT/pyproject.toml" ]; then
|
||||
BENCH=(uv run --directory "$SKILLSBENCH_ROOT" bench)
|
||||
else
|
||||
printf '%s\n' 'BenchFlow is required but neither `bench` nor `benchflow` is on PATH.' >&2
|
||||
printf 'Run: cd %s && uv sync --locked\n' "$SKILLSBENCH_ROOT" >&2
|
||||
exit 127
|
||||
fi
|
||||
|
||||
TASK_NAME=$(basename "$TASK_DIR")
|
||||
IMAGE_LOCK_SCOPE=$(
|
||||
printf '%s' "$HARNESS-$MODEL_REF" \
|
||||
| tr '[:upper:]' '[:lower:]' \
|
||||
| sed -E 's#[^a-z0-9]+#-#g; s#^-+##; s#-+$##'
|
||||
)
|
||||
TASK_IMAGE_LOCK_DIR="${TMPDIR:-/tmp}/skill-agent-evaluate-image-locks/$IMAGE_LOCK_SCOPE/$TASK_NAME"
|
||||
TASK_RESULTS_DIR="$OUTPUT_DIR"
|
||||
ensure_result_directory() {
|
||||
local directory=$1 mkdir_error
|
||||
|
||||
[ -d "$directory" ] && return 0
|
||||
if mkdir_error=$(mkdir -p "$directory" 2>&1); then
|
||||
return 0
|
||||
fi
|
||||
|
||||
# On a Windows-backed /mnt filesystem, deleting a directory while a Windows,
|
||||
# WSL, or Docker process still has it open can leave a delete-pending name.
|
||||
# The entry is invisible to stat/find, but NTFS rejects recreating the same
|
||||
# name with "Already exists". Report that state explicitly; silently using
|
||||
# a different category would put results under the wrong Skill provenance.
|
||||
if [ ! -e "$directory" ] && [[ "$mkdir_error" == *"Already exists"* || "$mkdir_error" == *"File exists"* ]]; then
|
||||
cat >&2 <<EOF
|
||||
Cannot create the result category directory because NTFS still reserves its deleted name:
|
||||
$directory
|
||||
|
||||
No evaluation was started and no existing result was changed.
|
||||
Close Explorer/editors or other processes holding this path. If it remains blocked,
|
||||
run "wsl --shutdown" in Windows PowerShell or Command Prompt, reopen WSL, and retry.
|
||||
This releases the stale handle; it does not delete Docker images or result data.
|
||||
EOF
|
||||
exit 73
|
||||
fi
|
||||
|
||||
printf 'Unable to create result directory %s: %s\n' "$directory" "$mkdir_error" >&2
|
||||
exit 73
|
||||
}
|
||||
|
||||
ensure_result_directory "$TASK_RESULTS_DIR"
|
||||
SETUP_LOG="$TASK_RESULTS_DIR/setup.log"
|
||||
touch "$SETUP_LOG"
|
||||
|
||||
# The wrapper owns the only terminal display. BenchFlow worker output always
|
||||
# goes to per-attempt logs so single and concurrent runs behave identically.
|
||||
DASHBOARD_ENABLED=false
|
||||
if [ -t 1 ] && [ "${TERM:-dumb}" != dumb ]; then
|
||||
DASHBOARD_ENABLED=true
|
||||
fi
|
||||
|
||||
# Reserve all test numbers before starting any background workers. `flock`
|
||||
# also keeps two separately launched wrapper processes from selecting the same
|
||||
# test-NNN directory.
|
||||
TEST_DIRS=()
|
||||
allocate_test_dirs() {
|
||||
local max_test=0 name number next_test candidate reserved=0 mkdir_error
|
||||
exec 9>"$TASK_RESULTS_DIR/.test-number.lock"
|
||||
flock 9
|
||||
while IFS= read -r name; do
|
||||
if [[ "$name" =~ ^test-([0-9]+)$ ]]; then
|
||||
number=$((10#${BASH_REMATCH[1]}))
|
||||
(( number > max_test )) && max_test=$number
|
||||
fi
|
||||
done < <(find "$TASK_RESULTS_DIR" -mindepth 1 -maxdepth 1 -type d -printf '%f\n')
|
||||
next_test=$((max_test + 1))
|
||||
while [ "$reserved" -lt "$REPEAT" ]; do
|
||||
printf -v name 'test-%03d' "$next_test"
|
||||
candidate="$TASK_RESULTS_DIR/$name"
|
||||
mkdir_error=''
|
||||
if mkdir_error=$(mkdir "$candidate" 2>&1); then
|
||||
TEST_DIRS+=("$candidate")
|
||||
reserved=$((reserved + 1))
|
||||
elif [ -d "$candidate" ] || [[ "$mkdir_error" == *'Already exists'* ]] || [[ "$mkdir_error" == *'File exists'* ]]; then
|
||||
# On drvfs (/mnt/c), a recently deleted Windows directory can remain in
|
||||
# delete-pending state: stat/find cannot see it, but mkdir still reports
|
||||
# Already exists. Treat that ghost name exactly like a live collision.
|
||||
:
|
||||
else
|
||||
printf 'Unable to reserve result directory %s: %s\n' \
|
||||
"$candidate" "$mkdir_error" >&2
|
||||
flock -u 9
|
||||
exec 9>&-
|
||||
return 1
|
||||
fi
|
||||
# An existing directory is already reserved by another run (or was
|
||||
# recreated by a still-finishing old run). Skip it atomically.
|
||||
next_test=$((next_test + 1))
|
||||
done
|
||||
flock -u 9
|
||||
exec 9>&-
|
||||
}
|
||||
allocate_test_dirs
|
||||
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
printf '%s\n' 'Docker is required for the official BenchFlow Docker sandbox.' >&2
|
||||
exit 127
|
||||
fi
|
||||
[ -x "$SKILLSBENCH_ROOT/.venv/bin/python" ] || {
|
||||
printf 'The official SkillsBench Python environment is required: %s\n' "$SKILLSBENCH_ROOT/.venv/bin/python" >&2
|
||||
exit 127
|
||||
}
|
||||
|
||||
if [ -z "$PREBUILT_IMAGE" ]; then
|
||||
PREBUILT_IMAGE="skillc/$TASK_NAME:local"
|
||||
mkdir -p "$TASK_IMAGE_LOCK_DIR"
|
||||
exec 8>"$TASK_IMAGE_LOCK_DIR/.image-build.lock"
|
||||
flock 8
|
||||
DOCKERFILE="$TASK_DIR/environment/Dockerfile"
|
||||
[ -f "$DOCKERFILE" ] || {
|
||||
printf 'Task Dockerfile is missing: %s\n' "$DOCKERFILE" >&2
|
||||
exit 2
|
||||
}
|
||||
# Rebuild when any task environment input changes. Previously an existing
|
||||
# tag was reused forever, which could preserve a stale task image even after
|
||||
# its Dockerfile or fixtures were fixed.
|
||||
TASK_ENVIRONMENT_FINGERPRINT=$(
|
||||
find "$TASK_DIR/environment" -type f -print0 \
|
||||
| sort -z \
|
||||
| xargs -0 sha256sum \
|
||||
| sha256sum \
|
||||
| awk '{print $1}'
|
||||
)
|
||||
task_environment_label=$(docker image inspect --format \
|
||||
'{{ index .Config.Labels "org.skillc.task-environment" }}' \
|
||||
"$PREBUILT_IMAGE" 2>/dev/null || true)
|
||||
if ! docker image inspect "$PREBUILT_IMAGE" >/dev/null 2>&1 || \
|
||||
[ "$task_environment_label" != "$TASK_ENVIRONMENT_FINGERPRINT" ]; then
|
||||
printf 'Building reusable task image: %s\n' "$PREBUILT_IMAGE" >>"$SETUP_LOG"
|
||||
# BenchFlow builds every task Dockerfile with environment/ as its context.
|
||||
# Input fixtures referenced by COPY therefore live in that directory.
|
||||
docker build \
|
||||
--file "$DOCKERFILE" \
|
||||
--build-arg PIP_INDEX_URL=https://mirrors.aliyun.com/pypi/simple \
|
||||
--label "org.skillc.task-environment=$TASK_ENVIRONMENT_FINGERPRINT" \
|
||||
--tag "$PREBUILT_IMAGE" \
|
||||
"$TASK_DIR/environment" >>"$SETUP_LOG" 2>&1
|
||||
fi
|
||||
OPENCODE_RUNTIME_VERSION=1.18.16
|
||||
OPENCODE_RUNTIME_BUILD=2
|
||||
NODE_RUNTIME_VERSION=22.20.0
|
||||
RUNTIME_DOCKERFILE="$PROJECT_ROOT/scripts/evaluate/docker/opencode-runtime.Dockerfile"
|
||||
RUNTIME_SOURCE_FINGERPRINT=$(sha256sum "$RUNTIME_DOCKERFILE" | awk '{print $1}')
|
||||
runtime_label=$(docker image inspect --format \
|
||||
'{{ index .Config.Labels "org.skillc.opencode-runtime" }}' \
|
||||
"$PREBUILT_IMAGE" 2>/dev/null || true)
|
||||
runtime_build_label=$(docker image inspect --format \
|
||||
'{{ index .Config.Labels "org.skillc.opencode-runtime-build" }}' \
|
||||
"$PREBUILT_IMAGE" 2>/dev/null || true)
|
||||
runtime_source_label=$(docker image inspect --format \
|
||||
'{{ index .Config.Labels "org.skillc.opencode-runtime-source" }}' \
|
||||
"$PREBUILT_IMAGE" 2>/dev/null || true)
|
||||
if [ "$runtime_label" != "$OPENCODE_RUNTIME_VERSION" ] || \
|
||||
[ "$runtime_build_label" != "$OPENCODE_RUNTIME_BUILD" ] || \
|
||||
[ "$runtime_source_label" != "$RUNTIME_SOURCE_FINGERPRINT" ]; then
|
||||
printf 'Adding reusable Node %s + OpenCode %s runtime to: %s\n' \
|
||||
"$NODE_RUNTIME_VERSION" "$OPENCODE_RUNTIME_VERSION" "$PREBUILT_IMAGE" >>"$SETUP_LOG"
|
||||
docker build \
|
||||
--file "$RUNTIME_DOCKERFILE" \
|
||||
--build-arg "TASK_IMAGE=$PREBUILT_IMAGE" \
|
||||
--build-arg "NODE_VERSION=$NODE_RUNTIME_VERSION" \
|
||||
--build-arg "OPENCODE_VERSION=$OPENCODE_RUNTIME_VERSION" \
|
||||
--build-arg "RUNTIME_BUILD_REVISION=$OPENCODE_RUNTIME_BUILD" \
|
||||
--build-arg "RUNTIME_SOURCE_FINGERPRINT=$RUNTIME_SOURCE_FINGERPRINT" \
|
||||
--tag "$PREBUILT_IMAGE" \
|
||||
"$PROJECT_ROOT/scripts/evaluate/docker" >>"$SETUP_LOG" 2>&1
|
||||
else
|
||||
printf 'Reusing task image with preinstalled OpenCode: %s\n' "$PREBUILT_IMAGE" >>"$SETUP_LOG"
|
||||
fi
|
||||
flock -u 8
|
||||
exec 8>&-
|
||||
elif ! docker image inspect "$PREBUILT_IMAGE" >/dev/null 2>&1; then
|
||||
printf 'The requested local prebuilt image is unavailable: %s\n' "$PREBUILT_IMAGE" >&2
|
||||
printf '%s\n' 'Check it with: docker image inspect <image-ref>' >&2
|
||||
exit 2
|
||||
fi
|
||||
run_official_attempt() {
|
||||
# This function always runs as a background worker. Do not inherit the
|
||||
# outer wrapper's EXIT guard: a normally finishing worker must never treat
|
||||
# its concurrently running siblings as orphaned processes.
|
||||
trap - EXIT
|
||||
local attempt=$1
|
||||
local jobs_dir="${TEST_DIRS[$((attempt - 1))]}"
|
||||
local task_dir_for_attempt="$TASK_DIR"
|
||||
local skills_dir_for_attempt="$SKILL_SOURCE"
|
||||
local temporary_task_root=""
|
||||
local eval_config=""
|
||||
local console_log="$jobs_dir/console.log"
|
||||
local bundled_skills_dir=""
|
||||
local single_skill_dir=""
|
||||
if [ -n "$PREBUILT_IMAGE" ]; then
|
||||
temporary_task_root=$(mktemp -d "${TMPDIR:-/tmp}/skillsbench-prebuilt-task.XXXXXX")
|
||||
trap '[ -z "$temporary_task_root" ] || rm -rf -- "$temporary_task_root"' RETURN
|
||||
task_dir_for_attempt="$temporary_task_root/$(basename "$TASK_DIR")"
|
||||
cp -a "$TASK_DIR" "$task_dir_for_attempt"
|
||||
# This is the same helper used by the official SkillsBench AgentBeats worker.
|
||||
PYTHONPATH="$SKILLSBENCH_ROOT" "$SKILLSBENCH_ROOT/.venv/bin/python" -c \
|
||||
'import sys; from pathlib import Path; from skillsbench_agentbeats.worker import _write_task_md_prebuilt_image; _write_task_md_prebuilt_image(Path(sys.argv[1]), sys.argv[2])' \
|
||||
"$task_dir_for_attempt/task.md" "$PREBUILT_IMAGE"
|
||||
fi
|
||||
|
||||
# BenchFlow gives task-bundled Skills and external custom-runtime Skills
|
||||
# different deployment policies. Some task Skills invoke a same-named
|
||||
# subagent (for example enterprise-artifact-search); loading their SKILL.md
|
||||
# externally exposes the instructions but does not register that agent type.
|
||||
# Stage external compiler output into this disposable task copy so source and
|
||||
# treatment runs differ only in Skill contents, not in deployment policy.
|
||||
if [ "$SKILL_SOURCE_IS_TASK_BUNDLED" = false ]; then
|
||||
if [ -z "$temporary_task_root" ]; then
|
||||
printf '%s\n' 'External Skill staging requires a temporary task copy.' >&2
|
||||
return 2
|
||||
fi
|
||||
bundled_skills_dir="$task_dir_for_attempt/environment/skills"
|
||||
case "$bundled_skills_dir/" in
|
||||
"$temporary_task_root/"*) ;;
|
||||
*)
|
||||
printf 'Refusing to stage external Skills outside the temporary task root: %s\n' \
|
||||
"$bundled_skills_dir" >&2
|
||||
return 2
|
||||
;;
|
||||
esac
|
||||
mkdir -p "$bundled_skills_dir"
|
||||
find "$bundled_skills_dir" -mindepth 1 -maxdepth 1 -exec rm -rf -- {} +
|
||||
if [ -f "$SKILL_PAYLOAD_DIR/SKILL.md" ]; then
|
||||
single_skill_dir="$bundled_skills_dir/$(basename "$SKILL_PAYLOAD_DIR")"
|
||||
mkdir -p "$single_skill_dir"
|
||||
cp -a "$SKILL_PAYLOAD_DIR/." "$single_skill_dir/"
|
||||
else
|
||||
cp -a "$SKILL_PAYLOAD_DIR/." "$bundled_skills_dir/"
|
||||
fi
|
||||
skills_dir_for_attempt="$bundled_skills_dir"
|
||||
fi
|
||||
if [ "$REQUIRE_SKILLS" = true ]; then
|
||||
{
|
||||
printf '\n\n## Mandatory Skill requirement\n\n'
|
||||
printf 'You MUST invoke a relevant Skill tool before final verification. Do NOT load Skills at the beginning. First inspect the task and project, then invoke the Skill immediately before performing the work it covers and apply its guidance. Merely mentioning a Skill or reading files directly does not satisfy this requirement.\n'
|
||||
} >> "$task_dir_for_attempt/task.md"
|
||||
fi
|
||||
# A source task Skill must also be resolved relative to the task copy.
|
||||
if [ "$SKILL_SOURCE_IS_TASK_BUNDLED" = true ]; then
|
||||
skills_dir_for_attempt="$task_dir_for_attempt/environment/skills"
|
||||
fi
|
||||
# JSON is valid YAML and keeps credentials out of the process command line
|
||||
# while still using the official `bench eval run --config` entrypoint.
|
||||
eval_config="$temporary_task_root/benchflow-eval.json"
|
||||
umask 077
|
||||
BF_CONFIG_PATH="$eval_config" BF_TASKS_DIR="$task_dir_for_attempt" \
|
||||
BF_JOBS_DIR="$jobs_dir" BF_AGENT="$AGENT" BF_MODEL="$MODEL_REF" \
|
||||
BF_SKILLS_DIR="$skills_dir_for_attempt" \
|
||||
BF_SILICONFLOW_API_KEY="${SILICONFLOW_API_KEY:-}" \
|
||||
BF_SILICONFLOW_BASE_URL="${SILICONFLOW_BASE_URL:-}" \
|
||||
"$SKILLSBENCH_ROOT/.venv/bin/python" -c '
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
agent_env = {}
|
||||
if os.environ.get("BF_AGENT") == "opencode":
|
||||
provider, model = os.environ["BF_MODEL"].split("/", 1)
|
||||
opencode_config = {
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"model": os.environ["BF_MODEL"],
|
||||
"small_model": os.environ["BF_MODEL"],
|
||||
"provider": {
|
||||
provider: {
|
||||
"npm": "@ai-sdk/openai-compatible",
|
||||
"name": provider,
|
||||
"options": {
|
||||
"baseURL": os.environ["BF_SILICONFLOW_BASE_URL"],
|
||||
"apiKey": "{env:SILICONFLOW_API_KEY}",
|
||||
"timeout": 600000,
|
||||
},
|
||||
"models": {model: {"name": model}},
|
||||
}
|
||||
},
|
||||
}
|
||||
agent_env = {
|
||||
"SILICONFLOW_API_KEY": os.environ["BF_SILICONFLOW_API_KEY"],
|
||||
"SILICONFLOW_BASE_URL": os.environ["BF_SILICONFLOW_BASE_URL"],
|
||||
"OPENCODE_CONFIG_CONTENT": json.dumps(opencode_config, separators=(",", ":")),
|
||||
}
|
||||
config = {
|
||||
"tasks_dir": os.environ["BF_TASKS_DIR"],
|
||||
"jobs_dir": os.environ["BF_JOBS_DIR"],
|
||||
"agent": os.environ["BF_AGENT"],
|
||||
"model": os.environ["BF_MODEL"],
|
||||
"environment": "docker",
|
||||
"skills_dir": os.environ["BF_SKILLS_DIR"],
|
||||
"skill_mode": "with-skill",
|
||||
"agent_env": agent_env,
|
||||
# Do not let BenchFlow silently substitute its default non-root "agent"
|
||||
# user. SkillsBench task images may intentionally provision task tools
|
||||
# (for example the SDKMAN Maven installation) in /root; the agent must see
|
||||
# the same
|
||||
# task-provided toolchain as the verifier. JSON null is the BenchFlow
|
||||
# documented root/no-lockdown sentinel. This changes only the evaluation
|
||||
# process inside the task container, never the dataset image.
|
||||
"sandbox_user": None,
|
||||
# A verifier timeout occurs after the agent rollout has finished. Retrying
|
||||
# it would sample a new agent trajectory and confound experiment results;
|
||||
# preserve the timeout as a terminal evaluation-infrastructure outcome.
|
||||
"retry": {"retry_on_verifier_infra": False},
|
||||
}
|
||||
Path(os.environ["BF_CONFIG_PATH"]).write_text(json.dumps(config), encoding="utf-8")
|
||||
'
|
||||
PYTHONPATH="$PROJECT_ROOT/scripts/evaluate${PYTHONPATH:+:$PYTHONPATH}" \
|
||||
"${BENCH[@]}" eval run --config "$eval_config" >"$console_log" 2>&1
|
||||
if [ "$REQUIRE_SKILLS" = true ]; then
|
||||
if ! node "$PROJECT_ROOT/scripts/evaluate/verify-required-skill.mjs" \
|
||||
"$jobs_dir" >>"$console_log" 2>&1; then
|
||||
printf 'Required Skill validation failed for %s: no successful Skill invocation was found.\n' \
|
||||
"$(basename "$jobs_dir")" >>"$console_log"
|
||||
return 86
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
attempt=1
|
||||
failures=0
|
||||
WORKER_PIDS=()
|
||||
FAILED_ATTEMPTS=()
|
||||
FAILED_LOGS=()
|
||||
declare -a ATTEMPT_STATUS ATTEMPT_STARTED ATTEMPT_ENDED ATTEMPT_REAPED
|
||||
DASHBOARD_RENDERED=false
|
||||
DASHBOARD_LINE_COUNT=$REPEAT
|
||||
for ((dashboard_attempt = 1; dashboard_attempt <= REPEAT; dashboard_attempt++)); do
|
||||
ATTEMPT_STATUS[$dashboard_attempt]=queued
|
||||
ATTEMPT_STARTED[$dashboard_attempt]=0
|
||||
ATTEMPT_ENDED[$dashboard_attempt]=0
|
||||
ATTEMPT_REAPED[$dashboard_attempt]=false
|
||||
done
|
||||
|
||||
format_elapsed() {
|
||||
local seconds=$1
|
||||
printf '%02d:%02d' "$((seconds / 60))" "$((seconds % 60))"
|
||||
}
|
||||
|
||||
attempt_stage() {
|
||||
local log=$1
|
||||
if [ ! -s "$log" ]; then
|
||||
printf '%s' preparing
|
||||
elif rg -q 'Running verifier|Verifier running' "$log"; then
|
||||
printf '%s' verifying
|
||||
elif rg -q 'end_turn|Process terminated|Agent finished' "$log"; then
|
||||
printf '%s' 'agent finalizing'
|
||||
elif rg -q 'Prompt [0-9]+/[0-9]+' "$log"; then
|
||||
printf '%s' 'agent running'
|
||||
elif rg -q 'ACP agent:|Session:' "$log"; then
|
||||
printf '%s' 'agent connecting'
|
||||
elif rg -q 'Deploying skills|Skills deployed' "$log"; then
|
||||
printf '%s' 'deploying skills'
|
||||
elif rg -q 'Installing opencode' "$log"; then
|
||||
printf '%s' 'installing agent'
|
||||
elif rg -q 'Starting environment' "$log"; then
|
||||
printf '%s' 'starting container'
|
||||
else
|
||||
printf '%s' preparing
|
||||
fi
|
||||
}
|
||||
|
||||
stage_progress() {
|
||||
case "$1" in
|
||||
preparing) printf '%d' 5 ;;
|
||||
'starting container') printf '%d' 12 ;;
|
||||
'installing agent') printf '%d' 20 ;;
|
||||
'deploying skills') printf '%d' 30 ;;
|
||||
'agent connecting') printf '%d' 40 ;;
|
||||
'agent running') printf '%d' 70 ;;
|
||||
'agent finalizing') printf '%d' 82 ;;
|
||||
verifying) printf '%d' 92 ;;
|
||||
finished) printf '%d' 100 ;;
|
||||
*) printf '%d' 0 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
attempt_outcome() {
|
||||
local jobs_dir process_rc summary
|
||||
jobs_dir=$1
|
||||
process_rc=$2
|
||||
summary="$jobs_dir/summary.json"
|
||||
if [ "$process_rc" -ne 0 ]; then
|
||||
printf '%s' ERROR
|
||||
elif [ ! -f "$summary" ]; then
|
||||
printf '%s' COMPLETE
|
||||
else
|
||||
"$SKILLSBENCH_ROOT/.venv/bin/python" - "$summary" <<'PY'
|
||||
import json
|
||||
import sys
|
||||
|
||||
data = json.load(open(sys.argv[1], encoding="utf-8"))
|
||||
if int(data.get("errored", 0) or 0) or int(data.get("verifier_errored", 0) or 0):
|
||||
print("ERROR", end="")
|
||||
elif int(data.get("passed", data.get("pass", 0)) or 0):
|
||||
print("PASS", end="")
|
||||
else:
|
||||
print("FAIL", end="")
|
||||
PY
|
||||
fi
|
||||
}
|
||||
|
||||
render_dashboard() {
|
||||
local now dashboard_attempt status elapsed stage test_name progress
|
||||
local bar_done bar_left done_chars left_chars elapsed_end
|
||||
local frame='' cursor_prefix=''
|
||||
now=$(date +%s)
|
||||
for ((dashboard_attempt = 1; dashboard_attempt <= REPEAT; dashboard_attempt++)); do
|
||||
status=${ATTEMPT_STATUS[$dashboard_attempt]}
|
||||
test_name="$(basename "${TEST_DIRS[$((dashboard_attempt - 1))]}") ($dashboard_attempt/$REPEAT)"
|
||||
if [ "${ATTEMPT_STARTED[$dashboard_attempt]}" -gt 0 ]; then
|
||||
elapsed_end=$now
|
||||
[ "${ATTEMPT_ENDED[$dashboard_attempt]}" -eq 0 ] || elapsed_end=${ATTEMPT_ENDED[$dashboard_attempt]}
|
||||
elapsed=$(format_elapsed "$((elapsed_end - ATTEMPT_STARTED[$dashboard_attempt]))")
|
||||
else
|
||||
elapsed='--:--'
|
||||
fi
|
||||
if [ "$status" = running ]; then
|
||||
stage=$(attempt_stage "${TEST_DIRS[$((dashboard_attempt - 1))]}/console.log")
|
||||
elif [ "$status" = queued ]; then
|
||||
stage=waiting
|
||||
else
|
||||
stage=finished
|
||||
fi
|
||||
progress=$(stage_progress "$stage")
|
||||
bar_done=$((progress * 20 / 100))
|
||||
bar_left=$((20 - bar_done))
|
||||
printf -v done_chars '%*s' "$bar_done" ''
|
||||
printf -v left_chars '%*s' "$bar_left" ''
|
||||
done_chars=${done_chars// /#}
|
||||
left_chars=${left_chars// /-}
|
||||
if [ "$stage" = finished ]; then
|
||||
stage=$status
|
||||
fi
|
||||
printf -v frame '%s%-16s [%s%s] %3d%% %-16s elapsed=%s\033[K\n' \
|
||||
"$frame" "$test_name" "$done_chars" "$left_chars" "$progress" "$stage" "$elapsed"
|
||||
done
|
||||
if [ "$DASHBOARD_RENDERED" = true ]; then
|
||||
printf -v cursor_prefix '\033[%dA\r' "$DASHBOARD_LINE_COUNT"
|
||||
fi
|
||||
# DEC mode 2026 asks supporting terminals (including current xterm.js) to
|
||||
# present the cursor move + complete frame atomically. Unsupported terminals
|
||||
# ignore it and still receive one assembled write, without a blanking pass.
|
||||
printf '\033[?2026h%s%s\033[?2026l' "$cursor_prefix" "$frame"
|
||||
DASHBOARD_RENDERED=true
|
||||
}
|
||||
|
||||
monitor_dashboard_batch() {
|
||||
local remaining=${#pids[@]} index pid current_attempt current_log process_rc
|
||||
printf '\033[?25l'
|
||||
while [ "$remaining" -gt 0 ]; do
|
||||
for index in "${!pids[@]}"; do
|
||||
current_attempt=${batch_attempts[$index]}
|
||||
[ "${ATTEMPT_REAPED[$current_attempt]}" = false ] || continue
|
||||
pid=${pids[$index]}
|
||||
if ! kill -0 "$pid" 2>/dev/null; then
|
||||
process_rc=0
|
||||
wait "$pid" || process_rc=$?
|
||||
ATTEMPT_REAPED[$current_attempt]=true
|
||||
ATTEMPT_ENDED[$current_attempt]=$(date +%s)
|
||||
current_log=${batch_logs[$index]}
|
||||
ATTEMPT_STATUS[$current_attempt]=$(attempt_outcome \
|
||||
"${TEST_DIRS[$((current_attempt - 1))]}" "$process_rc")
|
||||
if [ "$process_rc" -ne 0 ]; then
|
||||
cleanup_owned_containers_for_jobs_dir "${TEST_DIRS[$((current_attempt - 1))]}"
|
||||
failures=$((failures + 1))
|
||||
FAILED_ATTEMPTS+=("$current_attempt")
|
||||
FAILED_LOGS+=("$current_log")
|
||||
fi
|
||||
remaining=$((remaining - 1))
|
||||
fi
|
||||
done
|
||||
render_dashboard
|
||||
[ "$remaining" -eq 0 ] || sleep 1
|
||||
done
|
||||
printf '\033[?25h'
|
||||
}
|
||||
|
||||
cleanup_owned_containers_for_jobs_dir() {
|
||||
local jobs_dir=$1 container_id mount_source matched
|
||||
while IFS= read -r container_id; do
|
||||
[ -n "$container_id" ] || continue
|
||||
matched=false
|
||||
while IFS= read -r mount_source; do
|
||||
case "$mount_source/" in
|
||||
"$jobs_dir/"*) matched=true; break ;;
|
||||
esac
|
||||
done < <(docker inspect --format '{{range .Mounts}}{{println .Source}}{{end}}' "$container_id" 2>/dev/null || true)
|
||||
if [ "$matched" = true ]; then
|
||||
printf 'Cleaning orphaned BenchFlow container bound to %s: %s\n' \
|
||||
"$jobs_dir" "$container_id" >>"$SETUP_LOG"
|
||||
docker rm -f "$container_id" >/dev/null 2>&1 || true
|
||||
fi
|
||||
done < <(docker ps -aq --filter label=benchflow.owned=true)
|
||||
}
|
||||
|
||||
terminate_workers() {
|
||||
local signal=$1 pid jobs_dir
|
||||
trap - EXIT INT TERM
|
||||
for pid in "${WORKER_PIDS[@]:-}"; do
|
||||
terminate_process_tree "$pid"
|
||||
done
|
||||
for pid in "${WORKER_PIDS[@]:-}"; do
|
||||
wait "$pid" 2>/dev/null || true
|
||||
done
|
||||
for jobs_dir in "${TEST_DIRS[@]}"; do
|
||||
cleanup_owned_containers_for_jobs_dir "$jobs_dir"
|
||||
done
|
||||
[ "$DASHBOARD_ENABLED" = false ] || printf '\033[?25h'
|
||||
printf '\nStopped concurrent attempts after %s; completed artifacts remain in their test directories.\n' "$signal" >&2
|
||||
exit 130
|
||||
}
|
||||
|
||||
terminate_process_tree() {
|
||||
local root_pid=$1 child_pid
|
||||
while IFS= read -r child_pid; do
|
||||
[ -n "$child_pid" ] && terminate_process_tree "$child_pid"
|
||||
done < <(pgrep -P "$root_pid" 2>/dev/null || true)
|
||||
kill -TERM "$root_pid" 2>/dev/null || true
|
||||
}
|
||||
|
||||
cleanup_workers_on_exit() {
|
||||
local exit_code=$? pid jobs_dir active_workers=0
|
||||
trap - EXIT INT TERM
|
||||
[ "$DASHBOARD_ENABLED" = false ] || printf '\033[?25h'
|
||||
for pid in "${WORKER_PIDS[@]:-}"; do
|
||||
if kill -0 "$pid" 2>/dev/null; then
|
||||
active_workers=$((active_workers + 1))
|
||||
terminate_process_tree "$pid"
|
||||
fi
|
||||
done
|
||||
for pid in "${WORKER_PIDS[@]:-}"; do
|
||||
wait "$pid" 2>/dev/null || true
|
||||
done
|
||||
if [ "$active_workers" -gt 0 ]; then
|
||||
for jobs_dir in "${TEST_DIRS[@]}"; do
|
||||
cleanup_owned_containers_for_jobs_dir "$jobs_dir"
|
||||
done
|
||||
fi
|
||||
if [ "$active_workers" -gt 0 ]; then
|
||||
printf '\nWrapper exited unexpectedly (code %d); stopped %d active worker(s) to prevent orphaned evaluations.\n' \
|
||||
"$exit_code" "$active_workers" >&2
|
||||
fi
|
||||
exit "$exit_code"
|
||||
}
|
||||
trap cleanup_workers_on_exit EXIT
|
||||
trap 'terminate_workers SIGINT' INT
|
||||
trap 'terminate_workers SIGTERM' TERM
|
||||
|
||||
while [ "$attempt" -le "$REPEAT" ]; do
|
||||
batch_last=$((attempt + MAX_PARALLEL - 1))
|
||||
[ "$batch_last" -le "$REPEAT" ] || batch_last=$REPEAT
|
||||
pids=()
|
||||
batch_attempts=()
|
||||
batch_logs=()
|
||||
for current_attempt in $(seq "$attempt" "$batch_last"); do
|
||||
current_jobs_dir="${TEST_DIRS[$((current_attempt - 1))]}"
|
||||
current_log="$current_jobs_dir/console.log"
|
||||
ATTEMPT_STATUS[$current_attempt]=running
|
||||
ATTEMPT_STARTED[$current_attempt]=$(date +%s)
|
||||
run_official_attempt "$current_attempt" &
|
||||
pids+=("$!")
|
||||
WORKER_PIDS+=("$!")
|
||||
batch_attempts+=("$current_attempt")
|
||||
batch_logs+=("$current_log")
|
||||
done
|
||||
if [ "$DASHBOARD_ENABLED" = true ]; then
|
||||
monitor_dashboard_batch
|
||||
else
|
||||
for index in "${!pids[@]}"; do
|
||||
pid="${pids[$index]}"
|
||||
current_attempt="${batch_attempts[$index]}"
|
||||
current_log="${batch_logs[$index]}"
|
||||
process_rc=0
|
||||
wait "$pid" || process_rc=$?
|
||||
ATTEMPT_ENDED[$current_attempt]=$(date +%s)
|
||||
ATTEMPT_STATUS[$current_attempt]=$(attempt_outcome \
|
||||
"${TEST_DIRS[$((current_attempt - 1))]}" "$process_rc")
|
||||
if [ "$process_rc" -ne 0 ]; then
|
||||
cleanup_owned_containers_for_jobs_dir "${TEST_DIRS[$((current_attempt - 1))]}"
|
||||
failures=$((failures + 1))
|
||||
FAILED_ATTEMPTS+=("$current_attempt")
|
||||
FAILED_LOGS+=("$current_log")
|
||||
fi
|
||||
done
|
||||
fi
|
||||
WORKER_PIDS=()
|
||||
attempt=$((batch_last + 1))
|
||||
done
|
||||
trap - INT TERM
|
||||
trap - EXIT
|
||||
pass_count=0
|
||||
fail_count=0
|
||||
error_count=0
|
||||
complete_count=0
|
||||
for ((summary_attempt = 1; summary_attempt <= REPEAT; summary_attempt++)); do
|
||||
case "${ATTEMPT_STATUS[$summary_attempt]}" in
|
||||
PASS) pass_count=$((pass_count + 1)) ;;
|
||||
FAIL) fail_count=$((fail_count + 1)) ;;
|
||||
ERROR) error_count=$((error_count + 1)) ;;
|
||||
*) complete_count=$((complete_count + 1)) ;;
|
||||
esac
|
||||
done
|
||||
printf '\nResults:\n'
|
||||
for ((summary_attempt = 1; summary_attempt <= REPEAT; summary_attempt++)); do
|
||||
printf ' %-10s %-8s %s\n' \
|
||||
"$(basename "${TEST_DIRS[$((summary_attempt - 1))]}")" \
|
||||
"${ATTEMPT_STATUS[$summary_attempt]}" \
|
||||
"${TEST_DIRS[$((summary_attempt - 1))]}"
|
||||
done
|
||||
printf 'Summary: %d total, %d passed, %d failed, %d errored' \
|
||||
"$REPEAT" "$pass_count" "$fail_count" "$error_count"
|
||||
[ "$complete_count" -eq 0 ] || printf ', %d completed without a readable summary' "$complete_count"
|
||||
printf '.\n'
|
||||
printf 'Artifacts root: %s\n' "$TASK_RESULTS_DIR"
|
||||
[ "$failures" -eq 0 ] || exit 1
|
||||
@@ -0,0 +1,15 @@
|
||||
"""Restore the OpenCode launcher used by this project's verified runs."""
|
||||
|
||||
try:
|
||||
from benchflow.agents.registry import AGENT_INSTALLERS
|
||||
|
||||
command = AGENT_INSTALLERS.get("opencode")
|
||||
if command and "js-agents/bin/opencode \"$@\"' > /opt/benchflow/bin/opencode" in command:
|
||||
AGENT_INSTALLERS["opencode"] = (
|
||||
command
|
||||
+ " && printf '%s\\n' '#!/bin/sh' "
|
||||
"'exec /opt/benchflow/js-agents/bin/opencode \"$@\"' "
|
||||
"> /opt/benchflow/bin/opencode && chmod +x /opt/benchflow/bin/opencode"
|
||||
)
|
||||
except ImportError:
|
||||
pass
|
||||
@@ -0,0 +1,94 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
import fs from "node:fs";
|
||||
import path from "node:path";
|
||||
|
||||
const [jobsDir] = process.argv.slice(2);
|
||||
|
||||
if (!jobsDir) {
|
||||
console.error("Usage: verify-required-skill.mjs <jobs-dir>");
|
||||
process.exit(2);
|
||||
}
|
||||
|
||||
function collectTrajectories(directory) {
|
||||
const trajectories = [];
|
||||
for (const entry of fs.readdirSync(directory, {withFileTypes: true})) {
|
||||
const entryPath = path.join(directory, entry.name);
|
||||
if (entry.isDirectory()) {
|
||||
trajectories.push(...collectTrajectories(entryPath));
|
||||
} else if (entry.isFile() && entry.name === "acp_trajectory.jsonl") {
|
||||
trajectories.push(entryPath);
|
||||
}
|
||||
}
|
||||
return trajectories;
|
||||
}
|
||||
|
||||
function contentText(event) {
|
||||
return (event.content ?? [])
|
||||
.map((item) => item?.content?.text)
|
||||
.filter((text) => typeof text === "string")
|
||||
.join("\n");
|
||||
}
|
||||
|
||||
const invokedSkills = new Set();
|
||||
const attemptedSkillCalls = [];
|
||||
const parseErrors = [];
|
||||
const trajectories = collectTrajectories(jobsDir);
|
||||
|
||||
for (const trajectory of trajectories) {
|
||||
const lines = fs.readFileSync(trajectory, "utf8").split(/\r?\n/);
|
||||
for (let index = 0; index < lines.length; index += 1) {
|
||||
if (!lines[index].trim()) continue;
|
||||
let event;
|
||||
try {
|
||||
event = JSON.parse(lines[index]);
|
||||
} catch (error) {
|
||||
parseErrors.push(`${trajectory}:${index + 1}: ${error.message}`);
|
||||
continue;
|
||||
}
|
||||
if (event.type !== "tool_call") continue;
|
||||
const text = contentText(event);
|
||||
if (event.title === "skill") {
|
||||
attemptedSkillCalls.push({
|
||||
status: event.status ?? "unknown",
|
||||
message: text.slice(0, 500),
|
||||
trajectory,
|
||||
line: index + 1,
|
||||
});
|
||||
}
|
||||
if (event.status !== "completed") continue;
|
||||
for (const match of text.matchAll(/<skill_content\s+name="([^"]+)"/g)) {
|
||||
invokedSkills.add(match[1]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const invoked = attemptedSkillCalls.some((attempt) => attempt.status === "completed");
|
||||
const report = {
|
||||
invoked,
|
||||
invoked_skills: [...invokedSkills].sort(),
|
||||
attempted_skill_calls: attemptedSkillCalls,
|
||||
trajectory_files: trajectories.length,
|
||||
parse_errors: parseErrors,
|
||||
};
|
||||
|
||||
fs.writeFileSync(
|
||||
path.join(jobsDir, "required-skill.json"),
|
||||
`${JSON.stringify(report, null, 2)}\n`,
|
||||
"utf8",
|
||||
);
|
||||
|
||||
if (!invoked) {
|
||||
const failedAttempts = attemptedSkillCalls.filter(
|
||||
(attempt) => attempt.status !== "completed",
|
||||
);
|
||||
const failureDetail = failedAttempts.length
|
||||
? `; ${failedAttempts.length} Skill call(s) failed: ${failedAttempts.map((attempt) => attempt.message || attempt.status).join(" | ")}`
|
||||
: "";
|
||||
console.error(
|
||||
`No successful Skill invocation was observed${failureDetail}`,
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log("Required Skill validation passed.");
|
||||
Reference in New Issue
Block a user