#!/bin/bash # Verifier: grades the agent's work with Claude Code. # GRADER_MODE: "agentic" (default) explores the workspace with tools; "one-shot" # grades from the transcript alone (no tools); "rubric-trinary" / "rubric-scalar" # grade agentically against the task's atomic rubric criteria (staged as # tests/grader-context.md + tests/rubric-criteria.{md,json}) instead of the # seven behavioral dimensions — per-criterion pass/partial/fail verdicts or # 0.00-1.00 scores, rendered by tests/render-rubric-grade.py. # GRADING_STANDARD: "consolidated" (default) grades the agentic and one-shot # modes under the eight-criterion Consolidated Grading Standard # (tests/grader-system-prompt-consolidated.md + tests/grader-guidance-consolidated.md, # rendered by tests/render-grade-consolidated.py); "legacy" grades under the # seven behavioral dimensions plus a separate correctness score. Rubric modes # always grade against their staged assets and ignore the selector. TESTS_DIR="$(dirname "$0")" GRADER_MODE="${GRADER_MODE:-agentic}" RUBRIC_FORM="" case "$GRADER_MODE" in agentic|one-shot) ;; rubric-trinary) RUBRIC_FORM="trinary" ;; rubric-scalar) RUBRIC_FORM="scalar" ;; *) echo "ERROR: unknown GRADER_MODE '$GRADER_MODE' (expected agentic, one-shot, rubric-trinary, or rubric-scalar)" >&2; exit 1 ;; esac GRADING_STANDARD="${GRADING_STANDARD:-consolidated}" case "$GRADING_STANDARD" in consolidated|legacy) ;; *) echo "ERROR: unknown GRADING_STANDARD '$GRADING_STANDARD' (expected consolidated or legacy)" >&2; exit 1 ;; esac # Grading assets for the selected standard. The legacy names are the defaults; # the consolidated standard swaps all three together — grading half under each # standard (say, the consolidated prompt against the legacy guidance) is never # valid, so a tests/ dir missing any consolidated asset falls back to legacy as # a whole, loudly rather than fatally: task dirs created before the # consolidated assets shipped stay gradeable until they re-sync. GRADER_PROMPT_FILE="$TESTS_DIR/grader-system-prompt.md" GRADER_GUIDANCE_FILE="$TESTS_DIR/grader-guidance.md" RENDER_GRADE="$TESTS_DIR/render-grade.py" if [ -n "$RUBRIC_FORM" ]; then # Rubric criteria are authored in the legacy dimension vocabulary and carry # their own output protocol — the standard selector does not apply to them. GRADING_STANDARD="legacy" elif [ "$GRADING_STANDARD" = "consolidated" ]; then if [ -f "$TESTS_DIR/grader-system-prompt-consolidated.md" ] \ && [ -f "$TESTS_DIR/grader-guidance-consolidated.md" ] \ && [ -f "$TESTS_DIR/render-grade-consolidated.py" ]; then GRADER_PROMPT_FILE="$TESTS_DIR/grader-system-prompt-consolidated.md" GRADER_GUIDANCE_FILE="$TESTS_DIR/grader-guidance-consolidated.md" RENDER_GRADE="$TESTS_DIR/render-grade-consolidated.py" else for f in grader-system-prompt-consolidated.md grader-guidance-consolidated.md render-grade-consolidated.py; do [ -f "$TESTS_DIR/$f" ] || echo "WARN: GRADING_STANDARD=consolidated needs $TESTS_DIR/$f" >&2 done echo "WARN: this task's tests/ directory predates the consolidated grading assets — grading under the LEGACY standard instead." >&2 echo " Re-sync tests/ with the shared grader assets to grade consolidated (or set GRADING_STANDARD=legacy to silence this)." >&2 GRADING_STANDARD="legacy" fi fi # Grader model + number of samples (graded GRADER_SAMPLES times and averaged to # reduce noise). Override with GRADER_MODEL=... / GRADER_SAMPLES=... GRADER_MODEL="${GRADER_MODEL:-claude-fable-5}" GRADER_SAMPLES="${GRADER_SAMPLES:-3}" mkdir -p /logs/verifier # Expose the transcript and the agent's tree at the paths the grader prompt names. # These are the same paths the delivery runtime provides, so the shared grader # prompt needs no per-runtime branching: /tmp/admin-task/task_transcript.txt is the # harness-written session record, /tmp/agent-workspace is the agent's final tree. # `rm -rf` first because `ln -sfn` pointed at an existing DIRECTORY silently # creates the link INSIDE it — which would hand the grader an empty tree. mkdir -p /tmp/outputs /tmp/admin-task rm -rf /tmp/files /tmp/agent-workspace ln -sfn /workspace /tmp/files ln -sfn /workspace /tmp/agent-workspace ln -sfn /logs/agent/trajectory.json /tmp/admin-task/task_transcript.txt # Legacy path, kept until its remaining consumers (scripts/replay_agent.py, # scripts/grader-bench, scripts/earhart-grade, a few task guidance files) move to # /tmp/admin-task. Note the delivery runtime's /tmp/outputs is agent-writable, so # the grader prompt names only /tmp/admin-task as the session record. ln -sfn /logs/agent/trajectory.json /tmp/outputs/task_transcript.txt # Mark the workspace safe so git works regardless of file ownership. git config --global --add safe.directory '*' 2>/dev/null || true git config --global --add safe.directory /workspace 2>/dev/null || true # Capture any files the worker agent created or modified mkdir -p /logs/verifier/agent-output cd /tmp/files # Drop macOS metadata files (._*, .DS_Store) so they aren't captured or linted. find . \( -name '._*' -o -name '.DS_Store' \) -not -path './node_modules/*' -delete 2>/dev/null || true # Base = the pre-agent commit, so we also capture work the agent committed (the # queries below otherwise see only uncommitted changes). Falls back to the repo's # root commit; empty (helper becomes a no-op) if it can't be resolved. HARBOR_BASE=$(git rev-parse -q --verify _harbor_base 2>/dev/null || git rev-list --max-parents=0 HEAD 2>/dev/null | tail -1) committed_paths() { [ -n "$HARBOR_BASE" ] && git diff --name-only --diff-filter="$1" "$HARBOR_BASE" HEAD 2>/dev/null; } # The grader prompt calls the task's starting state the `base` commit (the delivery # overlay tags it at that name). Point the same name at it here so `git diff base` # and `git show base:` mean the same thing in both runtimes. A tag adds no # files and changes no content, so it cannot affect the agent's graded diff. # (`_harbor_base` above is never created by anything today — the fallback is what # actually resolves; tagging gives both runtimes one name that always exists.) [ -n "$HARBOR_BASE" ] && git tag -f base "$HARBOR_BASE" >/dev/null 2>&1 || true # Capture files the agent created or modified (unstaged, staged, and committed). # --diff-filter=d excludes deletions (handled below); sort -u dedups. { git ls-files --others --exclude-standard git diff --name-only --diff-filter=d git diff --cached --name-only --diff-filter=d committed_paths d } | sort -u | while IFS= read -r f; do mkdir -p "/logs/verifier/agent-output/$(dirname "$f")" cp "$f" "/logs/verifier/agent-output/$f" 2>/dev/null || true done echo "Captured $(find /logs/verifier/agent-output -type f | wc -l | tr -d ' ') agent output files" # Record files the agent deleted (unstaged, staged, or committed), so they aren't # silently restored before the checks run. { git ls-files --deleted git diff --cached --name-only --diff-filter=D committed_paths D } | sort -u > /logs/verifier/agent-output/_HARBOR_DELETIONS.txt # Did the agent change the workspace? If not (a no-code response — pushback, # clarifying question, or prose), skip the checks below and grade from the response. AGENT_OUTPUT_FILES=$(find /logs/verifier/agent-output -type f ! -name _HARBOR_DELETIONS.txt | wc -l | tr -d ' ') if [ -s /logs/verifier/agent-output/_HARBOR_DELETIONS.txt ]; then AGENT_DELETIONS=$(grep -c . /logs/verifier/agent-output/_HARBOR_DELETIONS.txt) else AGENT_DELETIONS=0 fi if [ "$AGENT_OUTPUT_FILES" -gt 0 ] || [ "$AGENT_DELETIONS" -gt 0 ]; then AGENT_CHANGED=1; else AGENT_CHANGED=0; fi [ "$AGENT_CHANGED" = 1 ] || echo "No agent workspace changes — skipping deterministic signals (grader scores correctness from the response / N/A)." # --- Deterministic signals (test / lint / typecheck) ------------------------- # If this task ships a tests/test-commands.sh, run the repo's checks against the # agent's workspace and hand their output to the grader to inform correctness. # Each check is a run_signal call (defined below) and carries its own baseline of # pre-existing failures for the grader to discount. No file => no signals. DETERMINISTIC_SIGNALS="" if [ -f "$TESTS_DIR/test-commands.sh" ] && [ "$AGENT_CHANGED" = 1 ]; then # Node's heap ceiling, sized to the container. A constant equal to the cgroup # limit leaves V8 no reason to collect before the kernel kills the process. _mem_bytes=$(cat /sys/fs/cgroup/memory.max 2>/dev/null \ || cat /sys/fs/cgroup/memory/memory.limit_in_bytes 2>/dev/null \ || echo max) case "$_mem_bytes" in ''|max|*[!0-9]*) _mem_mb=4096 ;; # unconstrained: assume the task's ask *) _mem_mb=$((_mem_bytes / 1024 / 1024)) ;; esac [ "$_mem_mb" -gt 65536 ] && _mem_mb=4096 # some runtimes report ~8 EiB _heap_mb=$(( _mem_mb * 75 / 100 )) # rest: off-heap, DB, page cache [ "$_heap_mb" -lt 512 ] && _heap_mb=512 export NODE_OPTIONS="${NODE_OPTIONS:-} --max-old-space-size=$_heap_mb" # Worker-pool size, from the cpu quota. Node runners read os.cpus(), which # reports the HOST's cores, so they over-fork here; Ruby reads the cgroup and # needs no bound. _cpu_max=$(cat /sys/fs/cgroup/cpu.max 2>/dev/null || echo max) case "$_cpu_max" in max*|'') RACCOON_CHECK_WORKERS=$(nproc 2>/dev/null || echo 2) ;; *) RACCOON_CHECK_WORKERS=$(( ${_cpu_max%% *} / ${_cpu_max##* } )) ;; esac [ "${RACCOON_CHECK_WORKERS:-0}" -lt 1 ] && RACCOON_CHECK_WORKERS=1 export RACCOON_CHECK_WORKERS # vitest needs MIN set too — its default is the host cpu count, and a min > max # pool is a hard error that runs no tests. export VITEST_MIN_THREADS=1 VITEST_MAX_THREADS="$RACCOON_CHECK_WORKERS" export VITEST_MIN_FORKS=1 VITEST_MAX_FORKS="$RACCOON_CHECK_WORKERS" # jest has no env lever; a jest command must pass --maxWorkers itself. echo "check budget: container ${_mem_mb}MB -> ${_heap_mb}MB node heap, ${RACCOON_CHECK_WORKERS} workers" >&2 # Make the source tree owner-writable so the checks can run (skip node_modules). find /tmp/files -type d -name node_modules -prune -o -print0 2>/dev/null \ | xargs -0 chmod u+rwX 2>/dev/null || true _SIG=$(mktemp) # run_signal