#!/bin/bash # Verifier: grades the agent's work with Claude Code. # GRADER_MODE: "agentic" (default) explores the workspace with tools; "one-shot" # grades from the transcript alone (no tools); "rubric-trinary" / "rubric-scalar" # grade agentically against the task's atomic rubric criteria (staged as # tests/grader-context.md + tests/rubric-criteria.{md,json}) with per-criterion # pass/partial/fail verdicts or 0.00-1.00 scores, rendered by # tests/render-rubric-grade.py. # # The agentic and one-shot modes grade under the Grading Standard: the eight # criteria defined in tests/grader-system-prompt-consolidated.md, informed by # the task's holistic rubric (tests/holistic-rubric.md; earlier tasks carry the # same document as tests/grader-guidance-consolidated.md or legacy # tests/grader-guidance.md), with each sample's grade.json rendered by # tests/render-grade-consolidated.py. Rubric modes grade against their staged # assets instead. TESTS_DIR="$(dirname "$0")" GRADER_MODE="${GRADER_MODE:-agentic}" RUBRIC_FORM="" case "$GRADER_MODE" in agentic|one-shot) ;; rubric-trinary) RUBRIC_FORM="trinary" ;; rubric-scalar) RUBRIC_FORM="scalar" ;; *) echo "ERROR: unknown GRADER_MODE '$GRADER_MODE' (expected agentic, one-shot, rubric-trinary, or rubric-scalar)" >&2; exit 1 ;; esac GRADER_PROMPT_FILE="$TESTS_DIR/grader-system-prompt-consolidated.md" # The task's holistic rubric, under whichever filename this task carries. # Renames are forward-only: new tasks ship tests/holistic-rubric.md, and every # earlier task keeps the name it was created with, so all three generations # resolve here indefinitely. The variable keeps its GRADER_GUIDANCE_FILE name # because provenance tooling and test presets reference it. GRADER_GUIDANCE_FILE="$TESTS_DIR/holistic-rubric.md" if [ ! -f "$GRADER_GUIDANCE_FILE" ]; then if [ -f "$TESTS_DIR/grader-guidance-consolidated.md" ]; then GRADER_GUIDANCE_FILE="$TESTS_DIR/grader-guidance-consolidated.md" elif [ -f "$TESTS_DIR/grader-guidance.md" ]; then GRADER_GUIDANCE_FILE="$TESTS_DIR/grader-guidance.md" fi elif [ -f "$TESTS_DIR/grader-guidance-consolidated.md" ]; then # Both names present: grading with the wrong one would be silent, so # refuse to guess unless the two are byte-identical. _hr_sha=$(sha256sum "$GRADER_GUIDANCE_FILE" 2>/dev/null | cut -d' ' -f1) _ggc_sha=$(sha256sum "$TESTS_DIR/grader-guidance-consolidated.md" 2>/dev/null | cut -d' ' -f1) if [ "$_hr_sha" != "$_ggc_sha" ]; then echo "ERROR: $TESTS_DIR carries both holistic-rubric.md and grader-guidance-consolidated.md with different content — keep exactly one (tests/holistic-rubric.md is the current name)" >&2 exit 1 fi fi RENDER_GRADE="$TESTS_DIR/render-grade-consolidated.py" # Grader model + number of samples (graded GRADER_SAMPLES times and averaged to # reduce noise). Override with GRADER_MODEL=... / GRADER_SAMPLES=... GRADER_MODEL="${GRADER_MODEL:-claude-fable-5-1}" GRADER_SAMPLES="${GRADER_SAMPLES:-3}" # The grader model's Claude Code floor. A task image installs Claude Code when it is first built # and Docker reuses that layer on every rebuild, so an image can carry a CLI the API refuses for # the current model ("does not support this model"). That used to surface only as a missing # reward file. Fail here with the reason instead. The check applies to the default grader # model; set GRADER_CLI_MIN to enforce a floor for another model. GRADER_CLI_MIN="${GRADER_CLI_MIN:-2.1.251}" if [ "$GRADER_MODEL" = "claude-fable-5-1" ] || [ -n "${GRADER_CLI_MIN_ENFORCE:-}" ]; then _cli_ver="$(claude --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)" if [ -n "$_cli_ver" ] && [ "$(printf '%s\n%s\n' "$GRADER_CLI_MIN" "$_cli_ver" | sort -V | head -1)" != "$GRADER_CLI_MIN" ]; then echo "ERROR: this task image carries Claude Code $_cli_ver, but the grader model $GRADER_MODEL needs $GRADER_CLI_MIN or newer." >&2 echo "Rebuild the task image with a current Claude Code: run scripts/harbor-run again (it removes the stale image), or remove it yourself with docker image rm." >&2 exit 1 fi fi # GRADER_FAST_MODE=true grades in claude's fast serving mode (harbor-run / # harbor-regrade --fast). Empty otherwise, leaving the grader command line unchanged. GRADER_FAST_FLAGS=() case "${GRADER_FAST_MODE:-}" in ""|false|0) ;; true|1|yes) GRADER_FAST_FLAGS=(--settings '{"fastMode":true}') ;; *) echo "ERROR: unknown GRADER_FAST_MODE '$GRADER_FAST_MODE' (expected true, 1, yes, false, 0, or empty)" >&2; exit 1 ;; esac mkdir -p /logs/verifier # GRADER-REGIME:BEGIN # What actually governed this grade. Nothing can recover a grading regime after # the fact, so it is captured here or not at all — and every step below tolerates # failure, because a provenance record must never be able to fail a grade. # Keep byte-identical to the copy in _smoke-test/_smoke-deletion-capture (asserted # by scripts/lib/grader-regime.test.ts, exercised by CI's regrade-smoke). _regime_sha() { [ -f "${1:-}" ] && sha256sum "$1" 2>/dev/null | cut -d' ' -f1; } _regime_json() { if [ -z "${1:-}" ]; then printf 'null'; else printf '"%s"' "$(printf '%s' "$1" | sed -e 's/\\/\\\\/g' -e 's/"/\\"/g')" fi } # What ran, not what was requested: one-shot grades exactly once whatever # GRADER_SAMPLES says, and the standard is read off the resolved assets below. _regime_samples="${GRADER_SAMPLES:-}" [ "${GRADER_MODE:-}" = "one-shot" ] && _regime_samples=1 _regime_standard="unknown" case "$(basename "${GRADER_PROMPT_FILE:-}")" in grader-system-prompt-consolidated.md) _regime_standard="consolidated" ;; grader-system-prompt.md) _regime_standard="legacy" ;; esac _regime_guidance="${GRADER_GUIDANCE_FILE:-}" _regime_render="${RENDER_GRADE:-}" if [ -n "${RUBRIC_FORM:-}" ]; then # Rubric modes grade against the staged rubric assets, not the guidance file. _regime_standard="rubric-${RUBRIC_FORM}" _regime_guidance="$TESTS_DIR/rubric-criteria.md" _regime_render="$TESTS_DIR/render-rubric-grade.py" fi # stderr is redirected FIRST so a missing target dir fails silently. cat 2>/dev/null > /logs/verifier/grader-regime.json </dev/null || true git config --global --add safe.directory /workspace 2>/dev/null || true # Capture any files the worker agent created or modified mkdir -p /logs/verifier/agent-output cd /tmp/files # Drop macOS metadata files (._*, .DS_Store) so they aren't captured or linted. find . \( -name '._*' -o -name '.DS_Store' \) -not -path './node_modules/*' -delete 2>/dev/null || true # Base = the pre-agent commit, so we also capture work the agent committed (the # queries below otherwise see only uncommitted changes). Falls back to the repo's # root commit; empty (helper becomes a no-op) if it can't be resolved. HARBOR_BASE=$(git rev-parse -q --verify _harbor_base 2>/dev/null || git rev-list --max-parents=0 HEAD 2>/dev/null | tail -1) # git C-quotes non-ASCII paths ("models/\360\237\245\207/x.sql"), and a quoted name is # not a path `cp` can open. `-c` outranks any repo-local override the agent set. # diff.renames=false: with detection on, a `git mv` is ONE R record, so the old path # never appears under --diff-filter=D and its deletion is silently lost. git_paths() { git -c core.quotePath=false -c diff.renames=false "$@"; } committed_paths() { [ -n "$HARBOR_BASE" ] && git_paths diff --name-only --diff-filter="$1" "$HARBOR_BASE" HEAD 2>/dev/null; } # The grader prompt calls the task's starting state the `base` commit (the delivery # overlay tags it at that name). Point the same name at it here so `git diff base` # and `git show base:` mean the same thing in both runtimes. A tag adds no # files and changes no content, so it cannot affect the agent's graded diff. # (`_harbor_base` above is never created by anything today — the fallback is what # actually resolves; tagging gives both runtimes one name that always exists.) [ -n "$HARBOR_BASE" ] && git tag -f base "$HARBOR_BASE" >/dev/null 2>&1 || true # Capture files the agent created or modified (unstaged, staged, and committed). # --diff-filter=d excludes deletions (handled below); sort -u dedups. { git_paths ls-files --others --exclude-standard git_paths diff --name-only --diff-filter=d git_paths diff --cached --name-only --diff-filter=d committed_paths d } | sort -u | while IFS= read -r f; do mkdir -p "/logs/verifier/agent-output/$(dirname "$f")" cp "$f" "/logs/verifier/agent-output/$f" 2>/dev/null || true done echo "Captured $(find /logs/verifier/agent-output -type f | wc -l | tr -d ' ') agent output files" # Record files the agent deleted (unstaged, staged, or committed), so they aren't # silently restored before the checks run. { git_paths ls-files --deleted git_paths diff --cached --name-only --diff-filter=D committed_paths D } | sort -u | while IFS= read -r f; do # Still on disk = not a deletion: the agent moved the old file aside (or removed # it) and wrote a new one there. Replay deletes AFTER the overlay, so listing it # would delete that new file. [ -e "$f" ] || [ -L "$f" ] || printf '%s\n' "$f" done > /logs/verifier/agent-output/_HARBOR_DELETIONS.txt # Did the agent change the workspace? If not (a no-code response — pushback, # clarifying question, or prose), skip the checks below and grade from the response. AGENT_OUTPUT_FILES=$(find /logs/verifier/agent-output -type f ! -name _HARBOR_DELETIONS.txt | wc -l | tr -d ' ') if [ -s /logs/verifier/agent-output/_HARBOR_DELETIONS.txt ]; then AGENT_DELETIONS=$(grep -c . /logs/verifier/agent-output/_HARBOR_DELETIONS.txt) else AGENT_DELETIONS=0 fi if [ "$AGENT_OUTPUT_FILES" -gt 0 ] || [ "$AGENT_DELETIONS" -gt 0 ]; then AGENT_CHANGED=1; else AGENT_CHANGED=0; fi [ "$AGENT_CHANGED" = 1 ] || echo "No agent workspace changes — skipping deterministic signals (grader scores correctness from the response / N/A)." # --- Deterministic signals (test / lint / typecheck) ------------------------- # If this task ships a tests/test-commands.sh, run the repo's checks against the # agent's workspace and hand their output to the grader to inform correctness. # Each check is a run_signal call (defined below) and carries its own baseline of # pre-existing failures for the grader to discount. No file => no signals. DETERMINISTIC_SIGNALS="" if [ -f "$TESTS_DIR/test-commands.sh" ] && [ "$AGENT_CHANGED" = 1 ]; then # Node's heap ceiling, sized to the container. A constant equal to the cgroup # limit leaves V8 no reason to collect before the kernel kills the process. _mem_bytes=$(cat /sys/fs/cgroup/memory.max 2>/dev/null \ || cat /sys/fs/cgroup/memory/memory.limit_in_bytes 2>/dev/null \ || echo max) case "$_mem_bytes" in ''|max|*[!0-9]*) _mem_mb=4096 ;; # unconstrained: assume the task's ask *) _mem_mb=$((_mem_bytes / 1024 / 1024)) ;; esac [ "$_mem_mb" -gt 65536 ] && _mem_mb=4096 # some runtimes report ~8 EiB _heap_mb=$(( _mem_mb * 75 / 100 )) # rest: off-heap, DB, page cache [ "$_heap_mb" -lt 512 ] && _heap_mb=512 export NODE_OPTIONS="${NODE_OPTIONS:-} --max-old-space-size=$_heap_mb" # Worker-pool size, from the cpu quota. Node runners read os.cpus(), which # reports the HOST's cores, so they over-fork here; Ruby reads the cgroup and # needs no bound. _cpu_max=$(cat /sys/fs/cgroup/cpu.max 2>/dev/null || echo max) case "$_cpu_max" in max*|'') RACCOON_CHECK_WORKERS=$(nproc 2>/dev/null || echo 2) ;; *) RACCOON_CHECK_WORKERS=$(( ${_cpu_max%% *} / ${_cpu_max##* } )) ;; esac [ "${RACCOON_CHECK_WORKERS:-0}" -lt 1 ] && RACCOON_CHECK_WORKERS=1 export RACCOON_CHECK_WORKERS # vitest needs MIN set too — its default is the host cpu count, and a min > max # pool is a hard error that runs no tests. export VITEST_MIN_THREADS=1 VITEST_MAX_THREADS="$RACCOON_CHECK_WORKERS" export VITEST_MIN_FORKS=1 VITEST_MAX_FORKS="$RACCOON_CHECK_WORKERS" # jest has no env lever; a jest command must pass --maxWorkers itself. echo "check budget: container ${_mem_mb}MB -> ${_heap_mb}MB node heap, ${RACCOON_CHECK_WORKERS} workers" >&2 # Make the source tree owner-writable so the checks can run (skip node_modules). find /tmp/files -type d -name node_modules -prune -o -print0 2>/dev/null \ | xargs -0 chmod u+rwX 2>/dev/null || true _SIG=$(mktemp) # run_signal