Initial commit

This commit is contained in:
2026-09-04 14:58:42 +08:00
commit 439cad87d9
4601 changed files with 29440 additions and 0 deletions
+117
View File
@@ -0,0 +1,117 @@
#!/usr/bin/env bash
# One fully isolated attempt: provenance, container, agent, artifacts, verifier.
run_attempt() (
local attempt_number=$1
local model_label task_label condition_label run_prefix stamp run_root workspace run_id
local started agent_started agent_ended ended agent_wall total_wall agent_exit
local prompt verifier_exit run_status description skill_list install_root cost_estimate_cny
local index proxy_var proxy_value
local -a container_env_args=()
model_label=$(normalize_run_component "${MODEL_ID##*/}")
task_label=$(run_task_label)
condition_label=$(run_condition_label)
run_prefix="$HARNESS-$model_label-$task_label-$condition_label"
stamp="$(date -u +%Y%m%dT%H%M%SZ)-$RANDOM"
run_root=$(reserve_run_root "$run_prefix")
workspace="$run_root/workspace"
run_id="$HARNESS-$MODE-$stamp"
mkdir -p "$workspace"
started=$(now_ms)
progress "$run_id" "Stage 1/5: preparing workspace and recording task/Skill provenance."
cleanup_attempt() { docker rm -f "$run_id" >/dev/null 2>&1 || true; }
trap cleanup_attempt EXIT
for proxy_var in HTTP_PROXY HTTPS_PROXY NO_PROXY http_proxy https_proxy no_proxy; do
proxy_value="${!proxy_var-}"
[ -z "$proxy_value" ] || container_env_args+=(-e "$proxy_var=$proxy_value")
done
progress "$run_id" "Stage 1/5: checking task image and input/Skill checksums."
docker image inspect --format '{{.Id}}' "$IMAGE" > "$run_root/task-image-id.txt"
find "$TASK_DIR/environment" -maxdepth 1 -type f -print0 | sort -z | xargs -0 -r sha256sum > "$run_root/input-sha256.txt"
: > "$run_root/skill-sha256.txt"
for index in "${!SKILL_DIRS[@]}"; do sha256sum "${SKILL_DIRS[$index]}/SKILL.md" >> "$run_root/skill-sha256.txt"; done
skill_list=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}")
case "$HARNESS" in
opencode) install_root="$workspace/.opencode/skills" ;;
hermes) install_root="$run_root/hermes-home/skills" ;;
*) install_root="$workspace/.claude/skills" ;;
esac
progress "$run_id" "Stage 2/5: writing manifest and preparing the isolated task container."
{
printf 'harness=%s\nharness_version=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\n' "$HARNESS" "$(harness_version)" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF"
printf 'task_slug=%s\ntask_dir=%s\ntask_image=%s\ntask_image_id=%s\n' "$TASK_SLUG" "$TASK_DIR" "$IMAGE" "$(tr -d '\n' < "$run_root/task-image-id.txt")"
printf 'skill_source=%s\nskill_count=%s\nskill_names=%s\nskill_install_root=%s\n' "$SOURCE_SKILL" "${#SKILL_NAMES[@]}" "$skill_list" "$install_root"
case "$HARNESS" in
opencode) printf 'tool_approval_policy=opencode_auto\nopencode_auto_approval=true\n' ;;
hermes) printf 'tool_approval_policy=hermes_yolo\n' ;;
*) printf 'tool_approval_policy=claude_dangerously_skip_permissions\nauthentication_scope=Claude Code global OAuth profile\nsetting_sources=Claude Code defaults (user,project,local; required by OAuth)\n' ;;
esac
printf 'network_policy=%s\ncpu_limit=%s\nmemory_limit=%s\nsampling_parameters=Harness defaults (not overridden)\n' "$CONTAINER_NETWORK" "$CPU_LIMIT" "$MEMORY_LIMIT"
printf 'timeout_seconds=%s\nverifier_timeout_seconds=%s\nmode=%s\nattempt_number=%s\n' "$TIMEOUT_SECONDS" "$VERIFIER_TIMEOUT_SECONDS" "$MODE" "$attempt_number"
} > "$run_root/target-manifest.env"
progress "$run_id" "Stage 2/5: starting the task container and mounting its workspace."
docker run -d --name "$run_id" --cpus="$CPU_LIMIT" --memory="$MEMORY_LIMIT" --network "$CONTAINER_NETWORK" "${container_env_args[@]}" --mount "type=bind,src=$workspace,dst=/workspace" "$IMAGE" sleep infinity >/dev/null
progress "$run_id" "Stage 3/5: container ready; building the agent prompt."
if [ "$MODE" = probe ]; then
prompt="The following Skills are installed and available: $skill_list.
Use bash commands only. Run these exact commands one at a time:
1. docker exec $run_id bash -lc 'test -d /root && test -w /root && ls -1 /root | head -n 20'
2. docker exec $run_id bash -lc 'probe_file=/root/.skill-agent-probe; printf probe-ok > \"\$probe_file\"; test -s \"\$probe_file\"; rm -f \"\$probe_file\"; printf HARNESS_CONTAINER_PROBE_OK'
Do not run any other command. If both commands succeed, finish with exactly: HARNESS_CONTAINER_PROBE_OK"
else
prompt="Use the installed Skills when relevant: $skill_list.
Complete this task:
$TASK_PROMPT
Execution environment:
- The fresh task container is named $run_id.
- Run every task inspection, analysis, edit, build, and test inside it with: docker exec $run_id ...
- Do not run ls, find, grep, cat, Maven, or any task command against host paths.
- Never inspect or access /mnt, the runner project, runs/, another attempt's workspace, task.md, oracle, verifier, or files outside the named container.
- The host working directory is only a transport mount at /workspace; use it only for a helper file you create, then execute that helper through /workspace inside the named container.
- Do not use host paths inside docker exec.
- Do not access task.md, oracle, verifier, or files outside the current workspace and installed Skills.
- Before finishing, inspect the result inside the task container and make sure the requested output or repository changes exist."
fi
agent_started=$(now_ms)
progress "$run_id" "Stage 3/5: agent running (model output is being saved to agent-trace.txt)."
set +e
case "$HARNESS" in opencode) run_opencode "$run_root" "$workspace" "$prompt" ;; hermes) run_hermes "$run_root" "$workspace" "$prompt" ;; *) run_claude_code "$run_root" "$workspace" "$prompt" ;; esac
agent_exit=$?
set -e
agent_ended=$(now_ms); ended=$(now_ms); agent_wall=$((agent_ended-agent_started)); total_wall=$((ended-started))
progress "$run_id" "Stage 4/5: agent finished (exit code $agent_exit); exporting metrics and collecting artifacts."
{
printf 'agent_exit_code=%s\nharness=%s\nprovider_id=%s\nmodel_id=%s\nmodel_ref=%s\nagent_wall_ms=%s\ntotal_wall_ms=%s\n' "$agent_exit" "$HARNESS" "$PROVIDER_ID" "$MODEL_ID" "$MODEL_REF" "$agent_wall" "$total_wall"
[ "$agent_exit" -eq 124 ] && printf 'timed_out=true\n' || printf 'timed_out=false\n'
} > "$run_root/metrics.env"
export_agent_session "$run_root" "$agent_wall" "$total_wall"
cost_estimate_cny=unavailable
[ ! -s "$run_root/agent-metrics.json" ] || cost_estimate_cny=$(node -e 'const m=require(process.argv[1]); const c=m.official_cost_estimate?.amount_cny; process.stdout.write(Number.isFinite(c) ? c.toFixed(8) : "unavailable")' "$run_root/agent-metrics.json")
printf 'official_cost_estimate_cny=%s\n' "$cost_estimate_cny" >> "$run_root/metrics.env"
if [ "$MODE" = probe ]; then
if [ "$agent_exit" -eq 0 ] && rg -q HARNESS_CONTAINER_PROBE_OK "$run_root/agent-trace.txt"; then run_status=success; description=none; else run_status=probe_failed; description='The Harness probe did not complete successfully. Inspect agent-trace.txt.'; fi
printf '# Harness probe summary\n\nstatus=%s\nagent_exit_code=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md"
else
progress "$run_id" "Stage 4/5: capturing changed files and artifacts from the task container."
docker diff "$run_id" > "$run_root/container-diff.txt" 2>/dev/null || true
capture_agent_artifacts "$run_root" "$run_id"
progress "$run_id" "Stage 5/5: running the task verifier."
set +e; verify_output "$run_root" "$run_id"; verifier_exit=$?; set -e
progress "$run_id" "Stage 5/5: verifier finished (exit code $verifier_exit); writing run summary."
printf 'verifier_exit=%s\n' "$verifier_exit" >> "$run_root/metrics.env"
if [ "$agent_exit" -eq 0 ] && [ "$verifier_exit" = 0 ]; then run_status=success; description=none
elif [ "$agent_exit" -eq 124 ] && [ "$verifier_exit" != 0 ]; then run_status=agent_timed_out; description="The agent timed out after ${TIMEOUT_SECONDS} seconds and the task verifier failed."
elif [ "$verifier_exit" != 0 ]; then run_status=verifier_failed; description="The agent completed, but the task verifier failed (exit code ${verifier_exit})."
else run_status=agent_failed_output_verified; description="The agent exited with code ${agent_exit}, but its output passed verification."; fi
printf '# Raw run summary\n\nstatus=%s\nagent_exit_code=%s\nverifier_exit=%s\nofficial_cost_estimate_cny=%s\ndescription=%s\n' "$run_status" "$agent_exit" "$verifier_exit" "$cost_estimate_cny" "$description" > "$run_root/run-summary.md"
fi
printf 'run_status=%s\nproblem_description=%s\n' "$run_status" "$description" >> "$run_root/metrics.env"
progress "$run_id" "Completed: $run_status. Details saved to $run_root."
[ "$run_status" = success ]
)
+65
View File
@@ -0,0 +1,65 @@
#!/usr/bin/env bash
# Shared presentation and run-directory helpers for run-raw-task.sh.
now_ms() { node -p 'Date.now()'; }
progress() {
local run_label=$1 message=$2
printf '[%s] [%s] %s\n' "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$run_label" "$message"
}
progress_bar() {
local label=$1 completed=$2 total=$3 width=24 filled empty percent bar
[ "$total" -gt 0 ] || return
filled=$((completed * width / total))
empty=$((width - filled))
percent=$((completed * 100 / total))
printf -v bar '%*s' "$filled" ''
bar=${bar// /#}
printf -v empty '%*s' "$empty" ''
empty=${empty// /-}
printf '[%s] [%s%s] %d/%d (%d%%)\n' "$label" "$bar" "$empty" "$completed" "$total" "$percent"
}
normalize_run_component() {
printf '%s' "$1" | tr '[:upper:]' '[:lower:]' | tr -cs 'a-z0-9' '-' | sed -E 's/^-+//; s/-+$//'
}
run_task_label() {
case "$TASK_SLUG" in
111-offer-letter-generator) printf '%s' 'task-1' ;;
222-software-dependency-audit) printf '%s' 'task-2' ;;
333-fix-build-google-auto) printf '%s' 'task-3' ;;
*) printf 'task-%s' "$(normalize_run_component "$TASK_SLUG")" ;;
esac
}
run_condition_label() {
case "$SOURCE_SKILL" in
*/results/model-compiled-skills/*|*/dist/*|*/conditions/*|/tmp/*) printf '%s' '编译后skill' ;;
*) printf '%s' '原始skill' ;;
esac
}
reserve_run_root() {
# Allocation and mkdir share one lock: concurrent attempts must never claim
# the same trace/workspace directory.
local prefix=$1 run_root suffix=1 lock_fd
local lock_path="/tmp/skill-agent-raw-name-index.lock"
mkdir -p "$PROJECT_ROOT/runs/$MODE"
exec {lock_fd}>"$lock_path"
flock "$lock_fd"
# Do the check and the mkdir while holding the same lock. Starting at 1
# also tolerates old, interrupted runs and avoids parsing a path whose
# prefix itself contains hyphens.
while :; do
run_root="$PROJECT_ROOT/runs/$MODE/$prefix-$suffix"
if mkdir "$run_root" 2>/dev/null; then
break
fi
suffix=$((suffix + 1))
done
flock -u "$lock_fd"
exec {lock_fd}>&-
printf '%s' "$run_root"
}
+67
View File
@@ -0,0 +1,67 @@
#!/usr/bin/env bash
# Container output verification and bounded artifact capture.
verify_output() {
local run_root=$1 run_id=$2 log_dir="$1/verifier"
local verifier_started verifier_ended verifier_wall docker_exit reward verification_status description
mkdir -p "$log_dir"
# BenchFlow's native task.md verifier contract: upload verifier/ to
# /verifier, then expose the legacy /tests path as a symlink only if the
# image has not already provided real /tests content. Do not overwrite that
# content; older verifier scripts may legitimately depend on it.
docker exec "$run_id" mkdir -p /verifier /logs/verifier
docker cp "$VERIFIER_SOURCE/." "$run_id:/verifier"
docker exec "$run_id" bash -lc '[ -e /tests ] || ln -s /verifier /tests'
docker exec "$run_id" chmod +x /verifier/test.sh
verifier_started=$(now_ms)
if timeout --foreground --signal=INT --kill-after=30s "${VERIFIER_TIMEOUT_SECONDS}s" \
docker exec "${VERIFIER_ENV_ARGS[@]}" "$run_id" /verifier/test.sh > "$log_dir/verifier.stdout.log" 2>&1; then
docker_exit=0
else
docker_exit=$?
fi
docker cp "$run_id:/logs/verifier/." "$log_dir" >/dev/null 2>&1 || true
verifier_ended=$(now_ms)
verifier_wall=$((verifier_ended - verifier_started))
reward=""
[ ! -f "$log_dir/reward.txt" ] || reward=$(tr -d '[:space:]' < "$log_dir/reward.txt")
if [ "$docker_exit" -eq 0 ] && [ "$reward" = 1 ]; then
verification_status=passed
description=none
else
verification_status=failed
description="Verifier exited with code ${docker_exit} and wrote reward=${reward:-missing}. See verifier.stdout.log for details."
fi
{
printf 'docker_exit_code=%s\nreward=%s\nverifier_wall_ms=%s\n' "$docker_exit" "$reward" "$verifier_wall"
printf 'verifier_log=%s\nverification_status=%s\nproblem_description=%s\n' "$log_dir/verifier.stdout.log" "$verification_status" "$description"
} > "$log_dir/summary.env"
[ "$verification_status" = passed ] && return 0
printf 'VERIFIER_ISSUE: %s\n' "$description" >&2
return 1
}
capture_agent_artifacts() {
local run_root=$1 run_id=$2 diff_path="$1/container-diff.txt"
local artifact_root="$1/artifacts" status container_path relative_path size destination
mkdir -p "$artifact_root"
while IFS=' ' read -r status container_path; do
[ "$status" = A ] || [ "$status" = C ] || continue
# Copy only modest, task-created outputs. Package caches and the mounted
# workspace are inputs/ephemera, never benchmark artifacts.
case "$container_path" in
/root/.cache/*|/root/.local/*|/root/.m2/*|/home/*/.cache/*|/home/*/.local/*|/home/*/.m2/*|*/.git/*|/etc/*|/opt/*|/tmp/*|/usr/*|/var/*|/workspace/*) continue ;;
esac
docker exec "$run_id" test -f "$container_path" >/dev/null 2>&1 || continue
size=$(docker exec "$run_id" stat -c '%s' "$container_path" 2>/dev/null || printf '0')
[[ "$size" =~ ^[0-9]+$ ]] || continue
[ "$size" -le 20971520 ] || continue
relative_path=${container_path#/}
destination="$artifact_root/$relative_path"
mkdir -p "$(dirname "$destination")"
docker cp "$run_id:$container_path" "$destination" >/dev/null
done < "$diff_path"
if find "$artifact_root" -type f -print -quit | grep -q .; then
find "$artifact_root" -type f -print0 | sort -z | xargs -0 sha256sum > "$run_root/output-sha256.txt"
fi
}
+68
View File
@@ -0,0 +1,68 @@
#!/usr/bin/env bash
# Harness-specific configuration and agent invocation.
install_skill_set() {
local destination=$1 index
mkdir -p "$destination"
for index in "${!SKILL_DIRS[@]}"; do
cp -a "${SKILL_DIRS[$index]}" "$destination/${SKILL_NAMES[$index]}"
done
}
write_opencode_config() {
local config_path=$1
node - "$config_path" "$PROVIDER_ID" "$MODEL_ID" "$PROVIDER_BASE_URL" "$PROVIDER_API_KEY" <<'NODE'
const fs = require("fs");
const [configPath, provider, model, baseURL, apiKey] = process.argv.slice(2);
const config = {$schema: "https://opencode.ai/config.json", model: `${provider}/${model}`,
provider: {[provider]: {npm: "@ai-sdk/openai-compatible", name: provider,
options: {baseURL, apiKey}, models: {[model]: {name: model}}}}};
fs.writeFileSync(configPath, `${JSON.stringify(config, null, 2)}\n`, {mode: 0o600});
NODE
}
write_hermes_profile() {
local profile_home=$1 workspace=$2
mkdir -p "$profile_home/skills"
touch "$profile_home/.no-bundled-skills"
install_skill_set "$profile_home/skills"
{
printf '%s\n' 'model:' " default: \"$MODEL_ID\"" ' provider: custom' ' base_url: "https://api.siliconflow.cn/v1"'
printf '%s\n' 'terminal:' ' backend: local' " cwd: \"$workspace\"" ' timeout: 180' ' home_mode: profile'
printf '%s\n' 'memory:' ' memory_enabled: false' ' user_profile_enabled: false'
printf '%s\n' 'skills:' ' external_dirs: []' ' inline_shell: false' ' write_approval: true'
printf '%s\n' 'curator:' ' enabled: false' 'fallback_providers: []'
printf '%s\n' 'delegation:' ' orchestrator_enabled: false' ' max_spawn_depth: 1' ' max_concurrent_children: 1' ' max_async_children: 1'
} > "$profile_home/config.yaml"
}
run_opencode() {
local run_root=$1 workspace=$2 prompt=$3
install_skill_set "$workspace/.opencode/skills"
write_opencode_config "$run_root/opencode.json"
(cd "$workspace"; OPENCODE_CONFIG="$run_root/opencode.json" OPENCODE_CONFIG_DIR="$workspace/.opencode" \
XDG_DATA_HOME="$run_root/opencode-data" XDG_STATE_HOME="$run_root/opencode-state" \
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
opencode --pure --auto --model "$PROVIDER_ID/$MODEL_ID" run "$prompt") > "$run_root/agent-trace.txt" 2>&1
}
run_hermes() {
local run_root=$1 workspace=$2 prompt=$3 skill_csv
skill_csv=$(IFS=,; printf '%s' "${SKILL_NAMES[*]}")
write_hermes_profile "$run_root/hermes-home" "$workspace"
(cd "$workspace"; OPENAI_API_KEY="$SILICONFLOW_API_KEY" HERMES_HOME="$run_root/hermes-home" HERMES_OPTIONAL_SKILLS="" \
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
hermes --yolo --provider custom --model "$MODEL_ID" --toolsets terminal --skills "$skill_csv" --oneshot "$prompt") > "$run_root/agent-trace.txt" 2>&1
}
prepare_claude_workspace() { install_skill_set "$1/.claude/skills"; }
run_claude_code() {
local run_root=$1 workspace=$2 prompt=$3
prepare_claude_workspace "$workspace"
(cd "$workspace"; ANTHROPIC_BASE_URL="$PROVIDER_BASE_URL" ANTHROPIC_AUTH_TOKEN="$PROVIDER_API_KEY" \
timeout --foreground --signal=INT --kill-after=30s "${TIMEOUT_SECONDS}s" \
claude --print --output-format json --model "$MODEL_ID" --dangerously-skip-permissions "$prompt") \
> "$run_root/claude-result.json" 2> "$run_root/claude-stderr.txt"
cat "$run_root/claude-result.json" "$run_root/claude-stderr.txt" > "$run_root/agent-trace.txt"
}