#!/bin/bash # Run raccoon tasks via Harbor with standard defaults. # # Automatically detects snapshot-based tasks (those with environment/session.jsonl) # and uses the snapshot agent adapter for session resume. # # Usage: scripts/harbor-run [extra harbor args...] # Example: scripts/harbor-run harbor-tasks/my-task-slug # Example: scripts/harbor-run harbor-tasks/my-task-slug -k 4 --force-build # # To change the model, use --model (consumed here). Passing harbor's own -m does NOT # override: harbor's -m is repeatable and builds one agent per value, so `-m X` runs the # registry default AND X — two trials. # # Needs python 3.11+ on PATH (tomllib, to read the harness registry). Containers have it; # a macOS host running HARBOR_ENV=docker may not — set RACCOON_PYTHON if so. set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" # Source API key if [ -f "$REPO_ROOT/.env" ]; then set -a source "$REPO_ROOT/.env" set +a fi if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then . "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env fi # Per-harness credentials (OPENAI_API_KEY and friends) are DERIVED from the proxy root in # ANTHROPIC_BASE_URL — they are not in .env. The container-create derivation exported them # into a process that has long since exited, and only the auth FILES it wrote survive, so a # fresh shell has the key on disk but not in its environment. resolve_harness checks the # environment, and the trial passes it through to the sandbox, so re-derive here. # Quiet on purpose: if it does not work, resolve_harness refuses by name a second later. if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then # shellcheck disable=SC1091 HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true harness_setup_credentials >/dev/null 2>&1 || true fi export ANTHROPIC_BASE_URL="${ANTHROPIC_BASE_URL:-}" # When running inside a devcontainer, harbor computes absolute paths for # Docker bind mounts. These paths must be HOST paths because docker compose # talks to the host daemon via the shared socket. Switching CWD to the # host-equivalent workspace path makes harbor resolve paths correctly. if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then cd "$HOST_WORKSPACE" fi TASK_DIR="$1" shift # Harness selection. `--harness` / `--model` / `--check-model` are consumed here; # everything else passes through to harbor untouched, so existing invocations keep # working. Parsed with a loop rather than getopts because the remaining args are an # opaque harbor passthrough that getopts would try to interpret. HARNESS_ARGS=() PASSTHROUGH=() while [ $# -gt 0 ]; do case "$1" in --harness) HARNESS_ARGS+=(--harness "$2"); shift 2 ;; --harness=*) HARNESS_ARGS+=(--harness "${1#*=}"); shift ;; --model) HARNESS_ARGS+=(--model "$2"); shift 2 ;; --model=*) HARNESS_ARGS+=(--model "${1#*=}"); shift ;; --check-model) HARNESS_ARGS+=(--check-model); shift ;; *) PASSTHROUGH+=("$1"); shift ;; esac done set -- "${PASSTHROUGH[@]+"${PASSTHROUGH[@]}"}" # Preflight: workspace must be populated before harbor tries to docker-build it. # Without this, the Dockerfile's `COPY workspace/ .` fails with an opaque # "failed to calculate checksum of ref ...: \"/workspace\": not found" buried # several frames deep in harbor's asyncio + docker-compose traceback. Surface # the real fix here instead. WORKSPACE_DIR="$TASK_DIR/environment/workspace" if [ ! -d "$WORKSPACE_DIR" ] || [ -z "$(ls -A "$WORKSPACE_DIR" 2>/dev/null)" ]; then SLUG="$(basename "$TASK_DIR")" echo "Error: $WORKSPACE_DIR is missing or empty." >&2 echo "Build it first: bash scripts/build-workspace.sh $SLUG" >&2 echo "(reads commit from $TASK_DIR/task.toml; applies environment/workspace.patch if present.)" >&2 exit 1 fi # Preflight: recompute the browser marker from task.toml. # # `browser = true` decides whether the image installs Playwright, and a Dockerfile can only # learn it from its build context. build-workspace.sh writes the marker — but a task.toml # edited afterwards leaves it stale, and flipping the flag off would otherwise still build a # browser into a `browser = false` task. The file is derived, so there is nothing to preserve # by leaving it alone. BROWSER_OPTIN=0 if [ -f "$TASK_DIR/task.toml" ] && grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then BROWSER_OPTIN=1 fi if [ -d "$TASK_DIR/environment" ]; then printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin" fi # Preflight: report — never block — on edits to toolkit-managed files. # # environment/Dockerfile, tests/test.sh and tests/grader-system-prompt.md ship from # task-shared/ and decide how the trial runs and how the grade is produced, so an edit # makes a task's runs hard to compare with the rest. Surface that here, before a trial # burns agent time. It is advisory on purpose: an author who edited one did it to get # unstuck, and refusing to run their trial punishes a misunderstanding. `|| true` also # means a checker that can't run (a fresh unzip with no node_modules) never reads as an # edit. The checker only exists in the worker toolkit; here the file is absent. CHECK_INFRA="$REPO_ROOT/scripts/check-task-infra.ts" if [ -f "$CHECK_INFRA" ]; then (cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true fi # Preflight: warn — loudly, but never block — when the live workspace has # changes that a rebuild from the pinned commit + workspace.patch would lose. # Trials run against the live workspace, but the finalized task keeps only the # rebuild inputs (the workspace/ dir is gitignored) and every downstream # consumer rebuilds from them, so anything uncaptured silently vanishes after # packaging. # The helper sits next to this script in a packed toolkit and under the # toolkit's static scripts in the internal repo layout. for WS_SYNC in "$SCRIPT_DIR/check-workspace-sync.sh" \ "$REPO_ROOT/raccoon-worker-toolkit/static/scripts/check-workspace-sync.sh"; do if [ -f "$WS_SYNC" ]; then bash "$WS_SYNC" "$TASK_DIR" || true break fi done # Agent + model selection, from scripts/harness-registry.toml via resolve_harness. # The agent classes come from scripts/{snapshot,codex,gemini}_agent.py or # harness_agents.py (hence the PYTHONPATH). The Claude variants reuse the claude # binary baked into the task image instead of re-downloading it at agent-setup — # stock claude-code's runtime download (~240 MB) races the 360s agent-setup timeout # and loses on slow-egress hosts (AgentSetupTimeoutError). Tasks that ship a # non-empty environment/session.jsonl additionally resume the staged session; # single-turn tasks get the non-resuming class. # # resolve_harness exits non-zero (and prints why) when the selection could not # produce a usable grade — an unknown/disabled harness, one that writes no ATIF # trajectory, a missing credential, or a task needing resume on a harness that # can't. Failing here costs a second; failing later costs the whole trial, and the # resume case wouldn't fail at all, it would silently grade the wrong thing. # _raccoon_python comes from lib/harness-credentials.sh, sourced above. Define a fallback # only if that file was missing, so the error below is about the interpreter rather than an # unbound function. command -v _raccoon_python >/dev/null 2>&1 || _raccoon_python() { return 1; } export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}" RACCOON_PY=$(_raccoon_python) || { echo "harbor-run: ERROR — no python3.11+ with tomllib on PATH, so the harness registry" >&2 echo "harbor-run: cannot be read and the agent class cannot be resolved. Set" >&2 echo "harbor-run: RACCOON_PYTHON to an interpreter that has tomllib (3.11+)." >&2 exit 1 } RESOLVED="$("$RACCOON_PY" "$SCRIPT_DIR/resolve_harness.py" \ --task-dir "$TASK_DIR" \ ${HARNESS_ARGS[@]+"${HARNESS_ARGS[@]}"})" || exit 1 eval "$RESOLVED" # --agent-import-path, not --agent: harbor 0.20 deprecates it but still accepts it, and it # is the flag whose recorded shape (`agents[0].import_path`) every run on disk and every # reader expects. --agent leaves import_path null and puts the class in `name`, which # silently empties the agent field in published benchmark rows. Revisit if the pin moves. AGENT_FLAGS="--agent-import-path $AGENT_IMPORT_PATH" # Empty EFFORT_KWARG means "run the harness's native default config" — pass no # effort kwarg at all rather than an empty one, which harbor would reject. EFFORT_FLAGS="" [ -n "$EFFORT_KWARG" ] && EFFORT_FLAGS="--ak $EFFORT_KWARG=$EFFORT_VALUE" # Environment backend. An explicit HARBOR_ENV always wins (either direction). # Otherwise the default is context-dependent: # - daytona for the internal repo: runs the trial in a cloud sandbox over # HTTP, so it needs no local docker daemon and works *inside* the primary # devcontainer. (HARBOR_ENV=docker uses the host docker daemon instead — # free and offline, but host-only; the devcontainer has no docker.sock.) # - docker for the worker toolkit: it's provisioned only for the local docker # backend (docker CLI + bind-mounted docker.sock, no DAYTONA_API_KEY), so a # daytona default would just error out. We detect a toolkit checkout by # toolkit.json at the repo root — a file the packaging step writes that the # internal repo never has. It lives in the bind-mounted workspace, not a # baked image layer, so this holds even when the container's HARBOR_ENV pin # is missing (e.g. a stale, pre-pin image). if [ -n "${HARBOR_ENV:-}" ]; then ENV_TYPE="$HARBOR_ENV" elif [ -f "$REPO_ROOT/toolkit.json" ]; then ENV_TYPE="docker" else ENV_TYPE="daytona" fi # --no-delete keeps the environment around after the trial for inspection. # That's free for a local docker container, but a Daytona sandbox is *billed* # while it exists — keeping it would leak a paid sandbox on every run. Harbor # downloads the trial logs into harbor-jobs before teardown either way, so for # daytona we let it delete the sandbox; for docker we keep the container. DELETE_FLAGS="--no-delete" [ "$ENV_TYPE" = "daytona" ] && DELETE_FLAGS="" # Orphan resilience (daytona): harbor tears sandboxes down per-trial + via an # atexit that only closes the client — neither runs on SIGTERM/SIGKILL/crash, so # a killed run leaks STARTED sandboxes that hog the shared pool until (if ever) # an account default reaps them. Tell Daytona to auto-stop an IDLE sandbox after # 20 min (auto-delete on stop), so orphans self-clean however the process dies. # Safe for live trials: a running agent/grader keeps the sandbox active. AUTOSTOP_FLAGS="" [ "$ENV_TYPE" = "daytona" ] && AUTOSTOP_FLAGS="--ek auto_stop_interval_mins=20" if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2 echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2 exit 1 fi # Launch-time input-checksum capture. Snapshot the task inputs BEFORE harbor # starts, so the recorded hashes are what the agent actually ran against — # an input edited between this run and copy-reference-run no longer records # post-edit state and masks staleness. The capture is stamped into the trial # dirs after the run (below); copy-reference-run prefers it over its weaker # capture-at-copy fallback. Advisory end to end: a failure here never blocks # the run. # # The post-run stamping watches the default `harbor-jobs` output dir. If the # caller overrides the output dir via extra args, we can't know where the # trials will land — skip stamping and say so, rather than silently stamping # nothing (runs then fall back to copy-time capture in copy-reference-run). STAMP_FILE="" for arg in "$@"; do case "$arg" in -o|--output*) echo "Note: custom harbor output dir passed ($arg) — skipping launch-time input-checksum stamping; reference runs will fall back to copy-time capture." >&2 STAMP_FILE="skip" break ;; esac done if [ "$STAMP_FILE" != "skip" ]; then STAMP_FILE="$(mktemp)" if ! npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" capture "$TASK_DIR" --out "$STAMP_FILE"; then echo "Warning: could not capture launch-time input checksums (staleness will be judged from copy-time capture instead)" >&2 rm -f "$STAMP_FILE" STAMP_FILE="" fi else STAMP_FILE="" fi JOBS_BEFORE="$(ls -1 harbor-jobs 2>/dev/null || true)" # The model comes from the registry (already resolved above) and is always a # CONCRETE id, never a shorthand alias. On the manual (non-snapshot) path the agent # passes the model via ANTHROPIC_MODEL, where a shorthand is NOT alias-resolved, so # the configured base-URL endpoint rejects it (400 "Invalid model: ") and # trials die on turn 1. Bump `default_model` in scripts/harness-registry.toml when a # newer model ships. # # harbor runs as a child (this script used to `exec` it, but the post-run # stamping needs to run after harbor exits), so forward TERM/INT: a `kill` # aimed at this wrapper's PID must take harbor down with it, not orphan a # running job (this repo has been bitten by zombie harbor coordinators # before). HARBOR_EXIT=0 HARBOR_SIGNALLED="" harbor run \ -p "$TASK_DIR" \ $AGENT_FLAGS \ -m "$MODEL" \ -e "$ENV_TYPE" \ $DELETE_FLAGS \ $AUTOSTOP_FLAGS \ --yes \ -o harbor-jobs \ $EFFORT_FLAGS \ "$@" & HARBOR_PID=$! trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT wait "$HARBOR_PID" || HARBOR_EXIT=$? if [ -n "$HARBOR_SIGNALLED" ]; then # The first wait was interrupted by the trap; wait again so harbor's real # exit status (not the shell's signal status) is what we propagate. wait "$HARBOR_PID" || HARBOR_EXIT=$? fi trap - TERM INT # Stamp the launch-time capture into this task's trial dirs in the job dir(s) # this run created (the harbor-jobs entries that didn't exist before the # run). Harbor names job dirs with a timestamp, so new entries are this run's # output — plus, when several harbor-runs share a cwd, possibly a concurrent # run's; `apply` is slug-scoped so another task's trials are never stamped # with this task's inputs. if [ -n "$STAMP_FILE" ]; then NEW_JOBS="$(comm -13 <(printf '%s\n' "$JOBS_BEFORE" | sort) <(ls -1 harbor-jobs 2>/dev/null | sort) | sed 's|^|harbor-jobs/|')" if [ -n "$NEW_JOBS" ]; then # shellcheck disable=SC2086 # job-dir names are timestamps, never spaced npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" apply "$STAMP_FILE" $NEW_JOBS || \ echo "Warning: could not stamp trial dirs with launch-time input checksums" >&2 fi rm -f "$STAMP_FILE" fi # Repeat the toolkit-managed-file notice AFTER the trial. The preflight copy is # minutes of harbor output up the scrollback by now, which for a notice nothing # enforces means nobody reads it. This one lands where the author is looking. if [ -f "$CHECK_INFRA" ]; then (cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true fi exit "$HARBOR_EXIT"