#!/usr/bin/env bash # One fully isolated attempt: provenance, container, agent, artifacts, verifier. run_attempt() ( local attempt_number=$1 local model_label task_label condition_label run_prefix stamp run_root workspace run_id local started agent_started agent_ended ended agent_wall total_wall agent_exit local prompt verifier_exit run_status description skill_list install_root cost_estimate_cny local index proxy_var proxy_value local -a container_env_args=() model_label=$(normalize_run_component "${MODEL_ID##*/}") task_label=$(run_task_label) condition_label=$(run_condition_label) run_prefix="$HARNESS-$model_label-$task_label-$condition_label" stamp="$(date -u +%Y%m%dT%H%M%SZ)-$RANDOM" run_root=$(reserve_run_root "$run_prefix") workspace="$run_root/workspace" run_id="$HARNESS-$MODE-$stamp" mkdir -p "$workspace" started=$(now_ms) progress "$run_id" "Stage 1/5: preparing workspace and recording task/Skill provenance." cleanup_attempt() { docker rm -f "$run_id" >/dev/null 2>&1 || true; } trap cleanup_attempt EXIT for proxy_var in HTTP_PROXY HTTPS_PROXY NO_PROXY http_proxy https_proxy no_proxy; do proxy_value="${!proxy_var-}" [ -z "$proxy_value" ] || container_env_args+=(-e "$proxy_var=$proxy_value") done progress "$run_id" "Stage 1/5: checking task image and input/Skill checksums." docker image inspect --format '{{.Id}}' "$IMAGE" > "$run_root/task-image-id.txt" find "$TASK_DIR/environment" -maxdepth 1 -type f -print0 | sort -z | xargs -0 -r sha256sum > "$run_root/input-sha256.txt" : > "$run_root/skill-sha256.txt" for index in "${!SKILL_DIRS[@]}"; do sha256sum "${SKILL_DIRS[$index]}/SKILL.md" >> "$run_root/skill-sha256.txt"; done skill_list=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}") case "$HARNESS" in opencode) install_root="$workspace/.opencode/skills" ;; hermes) install_root="$run_root/hermes-home/skills" ;; *) install_root="$workspace/.claude/skills" ;; esac progress "$run_id" "Stage 2/5: writing manifest and preparing the isolated task container." { printf 'harness=%s\nharness_version=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\n' "$HARNESS" "$(harness_version)" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF" printf 'task_slug=%s\ntask_dir=%s\ntask_image=%s\ntask_image_id=%s\n' "$TASK_SLUG" "$TASK_DIR" "$IMAGE" "$(tr -d '\n' < "$run_root/task-image-id.txt")" printf 'skill_source=%s\nskill_count=%s\nskill_names=%s\nskill_install_root=%s\n' "$SOURCE_SKILL" "${#SKILL_NAMES[@]}" "$skill_list" "$install_root" case "$HARNESS" in opencode) printf 'tool_approval_policy=opencode_auto\nopencode_auto_approval=true\n' ;; hermes) printf 'tool_approval_policy=hermes_yolo\n' ;; *) printf 'tool_approval_policy=claude_dangerously_skip_permissions\nauthentication_scope=Claude Code global OAuth profile\nsetting_sources=Claude Code defaults (user,project,local; required by OAuth)\n' ;; esac printf 'network_policy=%s\ncpu_limit=%s\nmemory_limit=%s\nsampling_parameters=Harness defaults (not overridden)\n' "$CONTAINER_NETWORK" "$CPU_LIMIT" "$MEMORY_LIMIT" printf 'timeout_seconds=%s\nverifier_timeout_seconds=%s\nmode=%s\nattempt_number=%s\n' "$TIMEOUT_SECONDS" "$VERIFIER_TIMEOUT_SECONDS" "$MODE" "$attempt_number" } > "$run_root/target-manifest.env" progress "$run_id" "Stage 2/5: starting the task container and mounting its workspace." docker run -d --name "$run_id" --cpus="$CPU_LIMIT" --memory="$MEMORY_LIMIT" --network "$CONTAINER_NETWORK" "${container_env_args[@]}" --mount "type=bind,src=$workspace,dst=/workspace" "$IMAGE" sleep infinity >/dev/null progress "$run_id" "Stage 3/5: container ready; building the agent prompt." if [ "$MODE" = probe ]; then prompt="The following Skills are installed and available: $skill_list. Use bash commands only. Run these exact commands one at a time: 1. docker exec $run_id bash -lc 'test -d /root && test -w /root && ls -1 /root | head -n 20' 2. docker exec $run_id bash -lc 'probe_file=/root/.skill-agent-probe; printf probe-ok > \"\$probe_file\"; test -s \"\$probe_file\"; rm -f \"\$probe_file\"; printf HARNESS_CONTAINER_PROBE_OK' Do not run any other command. If both commands succeed, finish with exactly: HARNESS_CONTAINER_PROBE_OK" else prompt="Use the installed Skills when relevant: $skill_list. Complete this task: $TASK_PROMPT Execution environment: - The fresh task container is named $run_id. - Run every task inspection, analysis, edit, build, and test inside it with: docker exec $run_id ... - Do not run ls, find, grep, cat, Maven, or any task command against host paths. - Never inspect or access /mnt, the runner project, runs/, another attempt's workspace, task.md, oracle, verifier, or files outside the named container. - The host working directory is only a transport mount at /workspace; use it only for a helper file you create, then execute that helper through /workspace inside the named container. - Do not use host paths inside docker exec. - Do not access task.md, oracle, verifier, or files outside the current workspace and installed Skills. - Before finishing, inspect the result inside the task container and make sure the requested output or repository changes exist." fi agent_started=$(now_ms) progress "$run_id" "Stage 3/5: agent running (model output is being saved to agent-trace.txt)." set +e case "$HARNESS" in opencode) run_opencode "$run_root" "$workspace" "$prompt" ;; hermes) run_hermes "$run_root" "$workspace" "$prompt" ;; *) run_claude_code "$run_root" "$workspace" "$prompt" ;; esac agent_exit=$? set -e agent_ended=$(now_ms); ended=$(now_ms); agent_wall=$((agent_ended-agent_started)); total_wall=$((ended-started)) progress "$run_id" "Stage 4/5: agent finished (exit code $agent_exit); exporting metrics and collecting artifacts." { printf 'agent_exit_code=%s\nharness=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\nagent_wall_ms=%s\ntotal_wall_ms=%s\n' "$agent_exit" "$HARNESS" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF" "$agent_wall" "$total_wall" [ "$agent_exit" -eq 124 ] && printf 'timed_out=true\n' || printf 'timed_out=false\n' } > "$run_root/metrics.env" export_agent_session "$run_root" "$agent_wall" "$total_wall" cost_estimate_cny=unavailable [ ! -s "$run_root/agent-metrics.json" ] || cost_estimate_cny=$(node -e 'const m=require(process.argv[1]); const c=m.official_cost_estimate?.amount_cny; process.stdout.write(Number.isFinite(c) ? c.toFixed(8) : "unavailable")' "$run_root/agent-metrics.json") printf 'official_cost_estimate_cny=%s\n' "$cost_estimate_cny" >> "$run_root/metrics.env" if [ "$MODE" = probe ]; then if [ "$agent_exit" -eq 0 ] && rg -q HARNESS_CONTAINER_PROBE_OK "$run_root/agent-trace.txt"; then run_status=success; description=none; else run_status=probe_failed; description='The Harness probe did not complete successfully. Inspect agent-trace.txt.'; fi printf '# Harness probe summary\n\nstatus=%s\nagent_exit_code=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md" else progress "$run_id" "Stage 4/5: capturing changed files and artifacts from the task container." docker diff "$run_id" > "$run_root/container-diff.txt" 2>/dev/null || true capture_agent_artifacts "$run_root" "$run_id" progress "$run_id" "Stage 5/5: running the task verifier." set +e; verify_output "$run_root" "$run_id"; verifier_exit=$?; set -e progress "$run_id" "Stage 5/5: verifier finished (exit code $verifier_exit); writing run summary." printf 'verifier_exit=%s\n' "$verifier_exit" >> "$run_root/metrics.env" if [ "$agent_exit" -eq 0 ] && [ "$verifier_exit" = 0 ]; then run_status=success; description=none elif [ "$agent_exit" -eq 124 ] && [ "$verifier_exit" != 0 ]; then run_status=agent_timed_out; description="The agent timed out after ${TIMEOUT_SECONDS} seconds and the task verifier failed." elif [ "$verifier_exit" != 0 ]; then run_status=verifier_failed; description="The agent completed, but the task verifier failed (exit code ${verifier_exit})." else run_status=agent_failed_output_verified; description="The agent exited with code ${agent_exit}, but its output passed verification."; fi printf '# Raw run summary\n\nstatus=%s\nagent_exit_code=%s\nverifier_exit=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$verifier_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md" fi printf 'run_status=%s\nproblem_description=%s\n' "$run_status" "$description" >> "$run_root/metrics.env" progress "$run_id" "Completed: $run_status. Details saved to $run_root." [ "$run_status" = success ] )