Initial commit
This commit is contained in:
@@ -0,0 +1,117 @@
|
||||
#!/usr/bin/env bash
|
||||
# One fully isolated attempt: provenance, container, agent, artifacts, verifier.
|
||||
|
||||
run_attempt() (
|
||||
local attempt_number=$1
|
||||
local model_label task_label condition_label run_prefix stamp run_root workspace run_id
|
||||
local started agent_started agent_ended ended agent_wall total_wall agent_exit
|
||||
local prompt verifier_exit run_status description skill_list install_root cost_estimate_cny
|
||||
local index proxy_var proxy_value
|
||||
local -a container_env_args=()
|
||||
model_label=$(normalize_run_component "${MODEL_ID##*/}")
|
||||
task_label=$(run_task_label)
|
||||
condition_label=$(run_condition_label)
|
||||
run_prefix="$HARNESS-$model_label-$task_label-$condition_label"
|
||||
stamp="$(date -u +%Y%m%dT%H%M%SZ)-$RANDOM"
|
||||
run_root=$(reserve_run_root "$run_prefix")
|
||||
workspace="$run_root/workspace"
|
||||
run_id="$HARNESS-$MODE-$stamp"
|
||||
mkdir -p "$workspace"
|
||||
started=$(now_ms)
|
||||
progress "$run_id" "Stage 1/5: preparing workspace and recording task/Skill provenance."
|
||||
cleanup_attempt() { docker rm -f "$run_id" >/dev/null 2>&1 || true; }
|
||||
trap cleanup_attempt EXIT
|
||||
|
||||
for proxy_var in HTTP_PROXY HTTPS_PROXY NO_PROXY http_proxy https_proxy no_proxy; do
|
||||
proxy_value="${!proxy_var-}"
|
||||
[ -z "$proxy_value" ] || container_env_args+=(-e "$proxy_var=$proxy_value")
|
||||
done
|
||||
progress "$run_id" "Stage 1/5: checking task image and input/Skill checksums."
|
||||
docker image inspect --format '{{.Id}}' "$IMAGE" > "$run_root/task-image-id.txt"
|
||||
find "$TASK_DIR/environment" -maxdepth 1 -type f -print0 | sort -z | xargs -0 -r sha256sum > "$run_root/input-sha256.txt"
|
||||
: > "$run_root/skill-sha256.txt"
|
||||
for index in "${!SKILL_DIRS[@]}"; do sha256sum "${SKILL_DIRS[$index]}/SKILL.md" >> "$run_root/skill-sha256.txt"; done
|
||||
skill_list=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}")
|
||||
case "$HARNESS" in
|
||||
opencode) install_root="$workspace/.opencode/skills" ;;
|
||||
hermes) install_root="$run_root/hermes-home/skills" ;;
|
||||
*) install_root="$workspace/.claude/skills" ;;
|
||||
esac
|
||||
progress "$run_id" "Stage 2/5: writing manifest and preparing the isolated task container."
|
||||
{
|
||||
printf 'harness=%s\nharness_version=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\n' "$HARNESS" "$(harness_version)" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF"
|
||||
printf 'task_slug=%s\ntask_dir=%s\ntask_image=%s\ntask_image_id=%s\n' "$TASK_SLUG" "$TASK_DIR" "$IMAGE" "$(tr -d '\n' < "$run_root/task-image-id.txt")"
|
||||
printf 'skill_source=%s\nskill_count=%s\nskill_names=%s\nskill_install_root=%s\n' "$SOURCE_SKILL" "${#SKILL_NAMES[@]}" "$skill_list" "$install_root"
|
||||
case "$HARNESS" in
|
||||
opencode) printf 'tool_approval_policy=opencode_auto\nopencode_auto_approval=true\n' ;;
|
||||
hermes) printf 'tool_approval_policy=hermes_yolo\n' ;;
|
||||
*) printf 'tool_approval_policy=claude_dangerously_skip_permissions\nauthentication_scope=Claude Code global OAuth profile\nsetting_sources=Claude Code defaults (user,project,local; required by OAuth)\n' ;;
|
||||
esac
|
||||
printf 'network_policy=%s\ncpu_limit=%s\nmemory_limit=%s\nsampling_parameters=Harness defaults (not overridden)\n' "$CONTAINER_NETWORK" "$CPU_LIMIT" "$MEMORY_LIMIT"
|
||||
printf 'timeout_seconds=%s\nverifier_timeout_seconds=%s\nmode=%s\nattempt_number=%s\n' "$TIMEOUT_SECONDS" "$VERIFIER_TIMEOUT_SECONDS" "$MODE" "$attempt_number"
|
||||
} > "$run_root/target-manifest.env"
|
||||
progress "$run_id" "Stage 2/5: starting the task container and mounting its workspace."
|
||||
docker run -d --name "$run_id" --cpus="$CPU_LIMIT" --memory="$MEMORY_LIMIT" --network "$CONTAINER_NETWORK" "${container_env_args[@]}" --mount "type=bind,src=$workspace,dst=/workspace" "$IMAGE" sleep infinity >/dev/null
|
||||
progress "$run_id" "Stage 3/5: container ready; building the agent prompt."
|
||||
if [ "$MODE" = probe ]; then
|
||||
prompt="The following Skills are installed and available: $skill_list.
|
||||
|
||||
Use bash commands only. Run these exact commands one at a time:
|
||||
1. docker exec $run_id bash -lc 'test -d /root && test -w /root && ls -1 /root | head -n 20'
|
||||
2. docker exec $run_id bash -lc 'probe_file=/root/.skill-agent-probe; printf probe-ok > \"\$probe_file\"; test -s \"\$probe_file\"; rm -f \"\$probe_file\"; printf HARNESS_CONTAINER_PROBE_OK'
|
||||
|
||||
Do not run any other command. If both commands succeed, finish with exactly: HARNESS_CONTAINER_PROBE_OK"
|
||||
else
|
||||
prompt="Use the installed Skills when relevant: $skill_list.
|
||||
|
||||
Complete this task:
|
||||
|
||||
$TASK_PROMPT
|
||||
|
||||
Execution environment:
|
||||
- The fresh task container is named $run_id.
|
||||
- Run every task inspection, analysis, edit, build, and test inside it with: docker exec $run_id ...
|
||||
- Do not run ls, find, grep, cat, Maven, or any task command against host paths.
|
||||
- Never inspect or access /mnt, the runner project, runs/, another attempt's workspace, task.md, oracle, verifier, or files outside the named container.
|
||||
- The host working directory is only a transport mount at /workspace; use it only for a helper file you create, then execute that helper through /workspace inside the named container.
|
||||
- Do not use host paths inside docker exec.
|
||||
- Do not access task.md, oracle, verifier, or files outside the current workspace and installed Skills.
|
||||
- Before finishing, inspect the result inside the task container and make sure the requested output or repository changes exist."
|
||||
fi
|
||||
agent_started=$(now_ms)
|
||||
progress "$run_id" "Stage 3/5: agent running (model output is being saved to agent-trace.txt)."
|
||||
set +e
|
||||
case "$HARNESS" in opencode) run_opencode "$run_root" "$workspace" "$prompt" ;; hermes) run_hermes "$run_root" "$workspace" "$prompt" ;; *) run_claude_code "$run_root" "$workspace" "$prompt" ;; esac
|
||||
agent_exit=$?
|
||||
set -e
|
||||
agent_ended=$(now_ms); ended=$(now_ms); agent_wall=$((agent_ended-agent_started)); total_wall=$((ended-started))
|
||||
progress "$run_id" "Stage 4/5: agent finished (exit code $agent_exit); exporting metrics and collecting artifacts."
|
||||
{
|
||||
printf 'agent_exit_code=%s\nharness=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\nagent_wall_ms=%s\ntotal_wall_ms=%s\n' "$agent_exit" "$HARNESS" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF" "$agent_wall" "$total_wall"
|
||||
[ "$agent_exit" -eq 124 ] && printf 'timed_out=true\n' || printf 'timed_out=false\n'
|
||||
} > "$run_root/metrics.env"
|
||||
export_agent_session "$run_root" "$agent_wall" "$total_wall"
|
||||
cost_estimate_cny=unavailable
|
||||
[ ! -s "$run_root/agent-metrics.json" ] || cost_estimate_cny=$(node -e 'const m=require(process.argv[1]); const c=m.official_cost_estimate?.amount_cny; process.stdout.write(Number.isFinite(c) ? c.toFixed(8) : "unavailable")' "$run_root/agent-metrics.json")
|
||||
printf 'official_cost_estimate_cny=%s\n' "$cost_estimate_cny" >> "$run_root/metrics.env"
|
||||
if [ "$MODE" = probe ]; then
|
||||
if [ "$agent_exit" -eq 0 ] && rg -q HARNESS_CONTAINER_PROBE_OK "$run_root/agent-trace.txt"; then run_status=success; description=none; else run_status=probe_failed; description='The Harness probe did not complete successfully. Inspect agent-trace.txt.'; fi
|
||||
printf '# Harness probe summary\n\nstatus=%s\nagent_exit_code=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md"
|
||||
else
|
||||
progress "$run_id" "Stage 4/5: capturing changed files and artifacts from the task container."
|
||||
docker diff "$run_id" > "$run_root/container-diff.txt" 2>/dev/null || true
|
||||
capture_agent_artifacts "$run_root" "$run_id"
|
||||
progress "$run_id" "Stage 5/5: running the task verifier."
|
||||
set +e; verify_output "$run_root" "$run_id"; verifier_exit=$?; set -e
|
||||
progress "$run_id" "Stage 5/5: verifier finished (exit code $verifier_exit); writing run summary."
|
||||
printf 'verifier_exit=%s\n' "$verifier_exit" >> "$run_root/metrics.env"
|
||||
if [ "$agent_exit" -eq 0 ] && [ "$verifier_exit" = 0 ]; then run_status=success; description=none
|
||||
elif [ "$agent_exit" -eq 124 ] && [ "$verifier_exit" != 0 ]; then run_status=agent_timed_out; description="The agent timed out after ${TIMEOUT_SECONDS} seconds and the task verifier failed."
|
||||
elif [ "$verifier_exit" != 0 ]; then run_status=verifier_failed; description="The agent completed, but the task verifier failed (exit code ${verifier_exit})."
|
||||
else run_status=agent_failed_output_verified; description="The agent exited with code ${agent_exit}, but its output passed verification."; fi
|
||||
printf '# Raw run summary\n\nstatus=%s\nagent_exit_code=%s\nverifier_exit=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$verifier_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md"
|
||||
fi
|
||||
printf 'run_status=%s\nproblem_description=%s\n' "$run_status" "$description" >> "$run_root/metrics.env"
|
||||
progress "$run_id" "Completed: $run_status. Details saved to $run_root."
|
||||
[ "$run_status" = success ]
|
||||
)
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared presentation and run-directory helpers for run-raw-task.sh.
|
||||
|
||||
now_ms() { node -p 'Date.now()'; }
|
||||
|
||||
progress() {
|
||||
local run_label=$1 message=$2
|
||||
printf '[%s] [%s] %s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$run_label" "$message"
|
||||
}
|
||||
|
||||
progress_bar() {
|
||||
local label=$1 completed=$2 total=$3 width=24 filled empty percent bar
|
||||
[ "$total" -gt 0 ] || return
|
||||
filled=$((completed * width / total))
|
||||
empty=$((width - filled))
|
||||
percent=$((completed * 100 / total))
|
||||
printf -v bar '%*s' "$filled" ''
|
||||
bar=${bar// /#}
|
||||
printf -v empty '%*s' "$empty" ''
|
||||
empty=${empty// /-}
|
||||
printf '[%s] [%s%s] %d/%d (%d%%)\n' "$label" "$bar" "$empty" "$completed" "$total" "$percent"
|
||||
}
|
||||
|
||||
normalize_run_component() {
|
||||
printf '%s' "$1" | tr '[:upper:]' '[:lower:]' | tr -cs 'a-z0-9' '-' | sed -E 's/^-+//; s/-+$//'
|
||||
}
|
||||
|
||||
run_task_label() {
|
||||
case "$TASK_SLUG" in
|
||||
111-offer-letter-generator) printf '%s' 'task-1' ;;
|
||||
222-software-dependency-audit) printf '%s' 'task-2' ;;
|
||||
333-fix-build-google-auto) printf '%s' 'task-3' ;;
|
||||
*) printf 'task-%s' "$(normalize_run_component "$TASK_SLUG")" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
run_condition_label() {
|
||||
case "$SOURCE_SKILL" in
|
||||
*/results/model-compiled-skills/*|*/dist/*|*/conditions/*|/tmp/*) printf '%s' '编译后skill' ;;
|
||||
*) printf '%s' '原始skill' ;;
|
||||
esac
|
||||
}
|
||||
|
||||
reserve_run_root() {
|
||||
# Allocation and mkdir share one lock: concurrent attempts must never claim
|
||||
# the same trace/workspace directory.
|
||||
local prefix=$1 run_root suffix=1 lock_fd
|
||||
local lock_path="/tmp/skill-agent-raw-name-index.lock"
|
||||
mkdir -p "$PROJECT_ROOT/runs/$MODE"
|
||||
exec {lock_fd}>"$lock_path"
|
||||
flock "$lock_fd"
|
||||
# Do the check and the mkdir while holding the same lock. Starting at 1
|
||||
# also tolerates old, interrupted runs and avoids parsing a path whose
|
||||
# prefix itself contains hyphens.
|
||||
while :; do
|
||||
run_root="$PROJECT_ROOT/runs/$MODE/$prefix-$suffix"
|
||||
if mkdir "$run_root" 2>/dev/null; then
|
||||
break
|
||||
fi
|
||||
suffix=$((suffix + 1))
|
||||
done
|
||||
flock -u "$lock_fd"
|
||||
exec {lock_fd}>&-
|
||||
printf '%s' "$run_root"
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
#!/usr/bin/env bash
|
||||
# Container output verification and bounded artifact capture.
|
||||
|
||||
verify_output() {
|
||||
local run_root=$1 run_id=$2 log_dir="$1/verifier"
|
||||
local verifier_started verifier_ended verifier_wall docker_exit reward verification_status description
|
||||
mkdir -p "$log_dir"
|
||||
# BenchFlow's native task.md verifier contract: upload verifier/ to
|
||||
# /verifier, then expose the legacy /tests path as a symlink only if the
|
||||
# image has not already provided real /tests content. Do not overwrite that
|
||||
# content; older verifier scripts may legitimately depend on it.
|
||||
docker exec "$run_id" mkdir -p /verifier /logs/verifier
|
||||
docker cp "$VERIFIER_SOURCE/." "$run_id:/verifier"
|
||||
docker exec "$run_id" bash -lc '[ -e /tests ] || ln -s /verifier /tests'
|
||||
docker exec "$run_id" chmod +x /verifier/test.sh
|
||||
verifier_started=$(now_ms)
|
||||
if timeout --foreground --signal=INT --kill-after=30s "${VERIFIER_TIMEOUT_SECONDS}s" \
|
||||
docker exec "${VERIFIER_ENV_ARGS[@]}" "$run_id" /verifier/test.sh > "$log_dir/verifier.stdout.log" 2>&1; then
|
||||
docker_exit=0
|
||||
else
|
||||
docker_exit=$?
|
||||
fi
|
||||
docker cp "$run_id:/logs/verifier/." "$log_dir" >/dev/null 2>&1 || true
|
||||
verifier_ended=$(now_ms)
|
||||
verifier_wall=$((verifier_ended - verifier_started))
|
||||
reward=""
|
||||
[ ! -f "$log_dir/reward.txt" ] || reward=$(tr -d '[:space:]' < "$log_dir/reward.txt")
|
||||
if [ "$docker_exit" -eq 0 ] && [ "$reward" = 1 ]; then
|
||||
verification_status=passed
|
||||
description=none
|
||||
else
|
||||
verification_status=failed
|
||||
description="Verifier exited with code ${docker_exit} and wrote reward=${reward:-missing}. See verifier.stdout.log for details."
|
||||
fi
|
||||
{
|
||||
printf 'docker_exit_code=%s\nreward=%s\nverifier_wall_ms=%s\n' "$docker_exit" "$reward" "$verifier_wall"
|
||||
printf 'verifier_log=%s\nverification_status=%s\nproblem_description=%s\n' "$log_dir/verifier.stdout.log" "$verification_status" "$description"
|
||||
} > "$log_dir/summary.env"
|
||||
[ "$verification_status" = passed ] && return 0
|
||||
printf 'VERIFIER_ISSUE: %s\n' "$description" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
capture_agent_artifacts() {
|
||||
local run_root=$1 run_id=$2 diff_path="$1/container-diff.txt"
|
||||
local artifact_root="$1/artifacts" status container_path relative_path size destination
|
||||
mkdir -p "$artifact_root"
|
||||
while IFS=' ' read -r status container_path; do
|
||||
[ "$status" = A ] || [ "$status" = C ] || continue
|
||||
# Copy only modest, task-created outputs. Package caches and the mounted
|
||||
# workspace are inputs/ephemera, never benchmark artifacts.
|
||||
case "$container_path" in
|
||||
/root/.cache/*|/root/.local/*|/root/.m2/*|/home/*/.cache/*|/home/*/.local/*|/home/*/.m2/*|*/.git/*|/etc/*|/opt/*|/tmp/*|/usr/*|/var/*|/workspace/*) continue ;;
|
||||
esac
|
||||
docker exec "$run_id" test -f "$container_path" >/dev/null 2>&1 || continue
|
||||
size=$(docker exec "$run_id" stat -c '%s' "$container_path" 2>/dev/null || printf '0')
|
||||
[[ "$size" =~ ^[0-9]+$ ]] || continue
|
||||
[ "$size" -le 20971520 ] || continue
|
||||
relative_path=${container_path#/}
|
||||
destination="$artifact_root/$relative_path"
|
||||
mkdir -p "$(dirname "$destination")"
|
||||
docker cp "$run_id:$container_path" "$destination" >/dev/null
|
||||
done < "$diff_path"
|
||||
if find "$artifact_root" -type f -print -quit | grep -q .; then
|
||||
find "$artifact_root" -type f -print0 | sort -z | xargs -0 sha256sum > "$run_root/output-sha256.txt"
|
||||
fi
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
#!/usr/bin/env bash
|
||||
# Harness-specific configuration and agent invocation.
|
||||
|
||||
install_skill_set() {
|
||||
local destination=$1 index
|
||||
mkdir -p "$destination"
|
||||
for index in "${!SKILL_DIRS[@]}"; do
|
||||
cp -a "${SKILL_DIRS[$index]}" "$destination/${SKILL_NAMES[$index]}"
|
||||
done
|
||||
}
|
||||
|
||||
write_opencode_config() {
|
||||
local config_path=$1
|
||||
node - "$config_path" "$PROVIDER_ID" "$MODEL_ID" "$PROVIDER_BASE_URL" "$PROVIDER_API_KEY" <<'NODE'
|
||||
const fs = require("fs");
|
||||
const [configPath, provider, model, baseURL, apiKey] = process.argv.slice(2);
|
||||
const config = {$schema: "https://opencode.ai/config.json", model: `${provider}/${model}`,
|
||||
provider: {[provider]: {npm: "@ai-sdk/openai-compatible", name: provider,
|
||||
options: {baseURL, apiKey}, models: {[model]: {name: model}}}}};
|
||||
fs.writeFileSync(configPath, `${JSON.stringify(config, null, 2)}\n`, {mode: 0o600});
|
||||
NODE
|
||||
}
|
||||
|
||||
write_hermes_profile() {
|
||||
local profile_home=$1 workspace=$2
|
||||
mkdir -p "$profile_home/skills"
|
||||
touch "$profile_home/.no-bundled-skills"
|
||||
install_skill_set "$profile_home/skills"
|
||||
{
|
||||
printf '%s\n' 'model:' " default: \"$MODEL_ID\"" ' provider: custom' ' base_url: "https://api.siliconflow.cn/v1"'
|
||||
printf '%s\n' 'terminal:' ' backend: local' " cwd: \"$workspace\"" ' timeout: 180' ' home_mode: profile'
|
||||
printf '%s\n' 'memory:' ' memory_enabled: false' ' user_profile_enabled: false'
|
||||
printf '%s\n' 'skills:' ' external_dirs: []' ' inline_shell: false' ' write_approval: true'
|
||||
printf '%s\n' 'curator:' ' enabled: false' 'fallback_providers: []'
|
||||
printf '%s\n' 'delegation:' ' orchestrator_enabled: false' ' max_spawn_depth: 1' ' max_concurrent_children: 1' ' max_async_children: 1'
|
||||
} > "$profile_home/config.yaml"
|
||||
}
|
||||
|
||||
run_opencode() {
|
||||
local run_root=$1 workspace=$2 prompt=$3
|
||||
install_skill_set "$workspace/.opencode/skills"
|
||||
write_opencode_config "$run_root/opencode.json"
|
||||
(cd "$workspace"; OPENCODE_CONFIG="$run_root/opencode.json" OPENCODE_CONFIG_DIR="$workspace/.opencode" \
|
||||
XDG_DATA_HOME="$run_root/opencode-data" XDG_STATE_HOME="$run_root/opencode-state" \
|
||||
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
|
||||
opencode --pure --auto --model "$PROVIDER_ID/$MODEL_ID" run "$prompt") > "$run_root/agent-trace.txt" 2>&1
|
||||
}
|
||||
|
||||
run_hermes() {
|
||||
local run_root=$1 workspace=$2 prompt=$3 skill_csv
|
||||
skill_csv=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}")
|
||||
write_hermes_profile "$run_root/hermes-home" "$workspace"
|
||||
(cd "$workspace"; OPENAI_API_KEY="$SILICONFLOW_API_KEY" HERMES_HOME="$run_root/hermes-home" HERMES_OPTIONAL_SKILLS="" \
|
||||
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
|
||||
hermes --yolo --provider custom --model "$MODEL_ID" --toolsets terminal --skills "$skill_csv" --oneshot "$prompt") > "$run_root/agent-trace.txt" 2>&1
|
||||
}
|
||||
|
||||
prepare_claude_workspace() { install_skill_set "$1/.claude/skills"; }
|
||||
|
||||
run_claude_code() {
|
||||
local run_root=$1 workspace=$2 prompt=$3
|
||||
prepare_claude_workspace "$workspace"
|
||||
(cd "$workspace"; ANTHROPIC_BASE_URL="$PROVIDER_BASE_URL" ANTHROPIC_AUTH_TOKEN="$PROVIDER_API_KEY" \
|
||||
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
|
||||
claude --print --output-format json --model "$MODEL_ID" --dangerously-skip-permissions "$prompt") \
|
||||
> "$run_root/claude-result.json" 2> "$run_root/claude-stderr.txt"
|
||||
cat "$run_root/claude-result.json" "$run_root/claude-stderr.txt" > "$run_root/agent-trace.txt"
|
||||
}
|
||||
Reference in New Issue
Block a user