Files
project-work/worker-toolkit-stocks-in-the-future/scripts/harbor-regrade
Eric Bell 392781f7aa chore: init commit
in worker.../repo/GITFOLDER.zip is the .git folder.
2026-08-11 14:44:09 -04:00

198 lines
8.5 KiB
Bash
Executable File

#!/bin/bash
# Re-grade an existing reference run without re-invoking the agent.
#
# Spins up a normal harbor trial, but plugs in scripts/replay_agent.py
# instead of a real agent. The replay agent overlays the captured
# agent-output into /workspace, applies any captured deletions, drops
# the captured trajectory at /logs/agent/trajectory.json so the grader
# reads the same transcript it would for the original run, then exits.
# The verifier (real test.sh, real LLM grader if present) runs as it
# would for any other trial.
#
# Usage:
# scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
#
# Examples:
# # Single regrade
# scripts/harbor-regrade \
# harbor-tasks/<slug> \
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg
#
# # Ten regrades of the same reference run (independent grader trials)
# scripts/harbor-regrade \
# harbor-tasks/<slug> \
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg \
# -k 10
#
# See scripts/replay_agent.py for what the agent actually does, and the
# `verifier: capture tracked-file deletions in agent-output` PR for the
# capture half of this flow (_HARBOR_DELETIONS.txt).
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
# Source API key + any verifier env from the repo's .env
if [ -f "$REPO_ROOT/.env" ]; then
set -a
source "$REPO_ROOT/.env"
set +a
fi
if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
. "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
fi
# When running inside a devcontainer, harbor needs HOST paths for docker
# bind mounts (the docker daemon is on the host).
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
cd "$HOST_WORKSPACE"
fi
usage() {
cat >&2 <<EOF
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
Required arguments:
<task-dir> harbor-tasks/<slug> — same dir you'd pass to scripts/harbor-run.
<reference-run-dir> harbor-tasks/<slug>/reference-runs/<run-id> — must contain
agent-output/ (and ideally agent/trajectory.json).
EOF
exit 1
}
[ $# -lt 2 ] && usage
TASK_DIR="$1"
REF_RUN_DIR="$2"
shift 2
# Resolve to absolute paths — harbor cd's around internally; the replay
# agent receives the path as an --agent-kwarg and won't know our cwd.
TASK_DIR_ABS=$(cd "$TASK_DIR" 2>/dev/null && pwd) || {
echo "Error: task-dir does not exist: $TASK_DIR" >&2
exit 1
}
REF_RUN_DIR_ABS=$(cd "$REF_RUN_DIR" 2>/dev/null && pwd) || {
echo "Error: reference-run-dir does not exist: $REF_RUN_DIR" >&2
exit 1
}
# NOTE: agent-output/ is intentionally NOT required here. Advisory tasks (the
# agent only reads + answers in chat) make no workspace edits, so a faithful
# capture has an empty/absent agent-output/ — the deliverable lives in the
# captured transcript (agent/trajectory.json) that the grader reads. ReplayAgent
# overlays agent-output/ when present and otherwise grades base-workspace +
# transcript, but FAILS LOUDLY if the transcript shows file-mutating tool calls
# with no agent-output/ (genuine lost edits). So we let it make that call.
if [ ! -d "$REF_RUN_DIR_ABS/agent-output" ]; then
echo "Note: $REF_RUN_DIR_ABS has no agent-output/ — replaying as an" >&2
echo " advisory run (base workspace + captured transcript). See" >&2
echo " scripts/replay_agent.py for the lost-edits safety guard." >&2
fi
# Make scripts/ importable so harbor can find replay_agent:ReplayAgent.
# ${PYTHONPATH:+...} so an unset PYTHONPATH doesn't leave a trailing colon —
# python treats the resulting empty entry as the CWD, silently putting
# whatever directory the user ran this from on harbor's sys.path.
export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
# Environment backend. Explicit HARBOR_ENV wins; otherwise default to docker in
# a worker-toolkit checkout (detected by toolkit.json at the repo root) and
# daytona in the internal repo. See scripts/harbor-run for the full rationale
# (why the toolkit needs docker, why the marker is a workspace file not an image
# env, and why daytona must NOT pass --no-delete — billed sandbox).
if [ -n "${HARBOR_ENV:-}" ]; then
ENV_TYPE="$HARBOR_ENV"
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
ENV_TYPE="docker"
else
ENV_TYPE="daytona"
fi
DELETE_FLAGS="--no-delete"
case "$ENV_TYPE" in daytona|modal) DELETE_FLAGS="" ;; esac
if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
exit 1
fi
# Output dir. harbor names the job subdir by second-granularity timestamp, so
# many regrades launched in the same second under one -o collide
# ("Job directory ... already exists and cannot be resumed"). Set
# HARBOR_REGRADE_OUT to a per-run unique dir when running a parallel sweep.
OUT_DIR="${HARBOR_REGRADE_OUT:-harbor-jobs}"
# Optional grader mode: HARBOR_GRADER_MODE=one-shot flips the task's test.sh into
# the no-tools one-shot grader (vs the default agentic grader) via verifier env —
# lets us A/B the agenticity gap without forking the task. See raccoon-shared/test.sh.
GRADER_MODE_FLAG=()
[ -n "${HARBOR_GRADER_MODE:-}" ] && GRADER_MODE_FLAG=(--verifier-env "GRADER_MODE=$HARBOR_GRADER_MODE")
# Optional grader model: HARBOR_GRADER_MODEL=claude-fable-5 overrides the grader's
# model (default: the `opus` alias) via verifier env — lets us A/B the grader model
# inside the unchanged agentic harness. See raccoon-shared/test.sh GRADER_MODEL.
[ -n "${HARBOR_GRADER_MODEL:-}" ] && GRADER_MODE_FLAG+=(--verifier-env "GRADER_MODEL=$HARBOR_GRADER_MODEL")
# Optional sample count: HARBOR_GRADER_SAMPLES=1 overrides the default 3 samples.
# For measuring per-sample properties of the grader (e.g. how often it emits a
# schema-valid grade.json), 1 sample across N distinct trajectories is a better
# estimator than 3 samples across N/3 — it decorrelates per-task effects for the
# same token spend. See raccoon-shared/test.sh GRADER_SAMPLES.
[ -n "${HARBOR_GRADER_SAMPLES:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$HARBOR_GRADER_SAMPLES")
# Carry the SOURCE run's agent identity + model into this replay's own record.
#
# A replay reports `replay_agent:ReplayAgent` with model_name null, because no model
# ran — the behaviour being graded came from the source run. Recording only
# reference_run_dir makes that a pointer, and pointers dangle: a regrade is normally
# copied back over the run it regraded, so the source usually no longer exists (501 of
# 643 on-disk replays already point at a missing dir, none of them in the published
# manifest either). Stamping the values here makes the replay self-describing, so the
# originating harness and model survive the source's deletion.
#
# Read with python3 rather than jq — jq is not guaranteed on a worker's box, and a
# missing source result.json must degrade to "unknown", never abort the regrade.
SOURCE_PROV_FLAGS=()
if [ -f "$REF_RUN_DIR_ABS/result.json" ]; then
SOURCE_AGENT=$(python3 -c '
import json, sys
try:
a = (json.load(open(sys.argv[1])).get("config") or {}).get("agent") or {}
except Exception:
sys.exit(0)
print(a.get("import_path") or a.get("name") or "")
' "$REF_RUN_DIR_ABS/result.json" 2>/dev/null || true)
SOURCE_MODEL=$(python3 -c '
import json, sys
try:
a = (json.load(open(sys.argv[1])).get("config") or {}).get("agent") or {}
except Exception:
sys.exit(0)
print(a.get("model_name") or "")
' "$REF_RUN_DIR_ABS/result.json" 2>/dev/null || true)
[ -n "$SOURCE_AGENT" ] && SOURCE_PROV_FLAGS+=(--ak "source_agent_import_path=$SOURCE_AGENT")
[ -n "$SOURCE_MODEL" ] && SOURCE_PROV_FLAGS+=(--ak "source_model_name=$SOURCE_MODEL")
fi
# Optional grading standard: HARBOR_GRADING_STANDARD=legacy grades under the
# seven-dimension-plus-correctness flow instead of the default eight-criterion
# consolidated standard — lets us A/B the standards on the same trajectory
# without forking the task. See raccoon-shared/test.sh GRADING_STANDARD.
[ -n "${HARBOR_GRADING_STANDARD:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "GRADING_STANDARD=$HARBOR_GRADING_STANDARD")
exec harbor run \
-p "$TASK_DIR_ABS" \
--agent-import-path replay_agent:ReplayAgent \
--ak "reference_run_dir=$REF_RUN_DIR_ABS" \
${SOURCE_PROV_FLAGS[@]+"${SOURCE_PROV_FLAGS[@]}"} \
-e "$ENV_TYPE" \
$DELETE_FLAGS \
--yes \
-o "$OUT_DIR" \
${GRADER_MODE_FLAG[@]+"${GRADER_MODE_FLAG[@]}"} \
"$@"