ren worker folder adding orig, mv new one into root
This commit is contained in:
@@ -18,6 +18,9 @@
|
||||
# harbor-tasks/<slug> \
|
||||
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg
|
||||
#
|
||||
# # Every captured run for a task, 2 at a time (after a rubric edit)
|
||||
# scripts/harbor-regrade harbor-tasks/<slug> --all
|
||||
#
|
||||
# # Ten regrades of the same reference run (independent grader trials)
|
||||
# scripts/harbor-regrade \
|
||||
# harbor-tasks/<slug> \
|
||||
@@ -28,6 +31,11 @@
|
||||
# token rate). The replay agent runs no model, so the grader is the only model in
|
||||
# this path. Serving speed and cost change; the grade itself is not steered.
|
||||
#
|
||||
# A finished regrade files itself when nothing is filed for that run yet — which is
|
||||
# only ever the atomic case, since a run always carries a holistic grade already.
|
||||
# Otherwise it grades and leaves the result in its job dir, so the tune-and-diff loop
|
||||
# is untouched; --replace adopts it, superseding what was there.
|
||||
#
|
||||
# See scripts/replay_agent.py for what the agent actually does, and the
|
||||
# `verifier: capture tracked-file deletions in agent-output` PR for the
|
||||
# capture half of this flow (_HARBOR_DELETIONS.txt).
|
||||
@@ -37,6 +45,11 @@ set -euo pipefail
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
|
||||
# A sample count from the caller wins over .env, so a prefix can override a standing
|
||||
# setting for one regrade. Captured as one value so a caller's spelling beats .env's,
|
||||
# whichever each used (test.sh reads GRADER_SAMPLES; this script has long taken both).
|
||||
_CALLER_SAMPLES="${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}"
|
||||
|
||||
# Source API key + any verifier env from the repo's .env
|
||||
if [ -f "$REPO_ROOT/.env" ]; then
|
||||
set -a
|
||||
@@ -64,16 +77,28 @@ fi
|
||||
|
||||
usage() {
|
||||
cat >&2 <<EOF
|
||||
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
|
||||
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir>... [extra harbor args]
|
||||
scripts/harbor-regrade <task-dir> --all [extra harbor args]
|
||||
|
||||
Required arguments:
|
||||
<task-dir> harbor-tasks/<slug> — same dir you'd pass to scripts/harbor-run.
|
||||
<reference-run-dir> harbor-tasks/<slug>/reference-runs/<run-id> — must contain
|
||||
agent-output/ (and ideally agent/trajectory.json).
|
||||
agent-output/ (and ideally agent/trajectory.json). Name several
|
||||
to re-grade them all; each gets its own harbor job.
|
||||
|
||||
Optional arguments:
|
||||
--all re-grade every run under <task-dir>/reference-runs/ — the usual
|
||||
thing to do after editing the rubric.
|
||||
--jobs N how many runs to re-grade at once (default 2). Each one is a
|
||||
container, so raise it only as far as your machine allows.
|
||||
--fast grade in claude's fast serving mode (higher token rate,
|
||||
faster output). Anything else is passed through to harbor.
|
||||
--replace adopt the result, replacing the grade already filed for this
|
||||
run. Without it, a regrade that would overwrite an existing
|
||||
grade is left in its job dir for you to compare first.
|
||||
|
||||
Note: -k re-grades the SAME run N times (N independent grades of one trajectory).
|
||||
To re-grade DIFFERENT runs, name them all, or use --all.
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
@@ -81,19 +106,48 @@ EOF
|
||||
# `--fast` is consumed here (harbor has no such flag); everything else stays an opaque
|
||||
# passthrough. Stripped first so it may appear anywhere, not just after the positionals.
|
||||
FAST_REQUESTED=""
|
||||
ALL_RUNS=""
|
||||
REPLACE=""
|
||||
JOBS=2
|
||||
REGRADE_ARGS=()
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--fast) FAST_REQUESTED=1; shift ;;
|
||||
--replace) REPLACE=1; shift ;;
|
||||
--all) ALL_RUNS=1; shift ;;
|
||||
--jobs)
|
||||
[ $# -ge 2 ] || { echo "Error: --jobs needs a number." >&2; exit 1; }
|
||||
JOBS="$2"; shift 2 ;;
|
||||
--jobs=*) JOBS="${1#--jobs=}"; shift ;;
|
||||
*) REGRADE_ARGS+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
set -- "${REGRADE_ARGS[@]+"${REGRADE_ARGS[@]}"}"
|
||||
|
||||
[ $# -lt 2 ] && usage
|
||||
# Normalised to base 10 before any (( )) sees it: "09" passes a -ge test but is
|
||||
# invalid octal in arithmetic, which spun the dispatch loop forever.
|
||||
case "$JOBS" in
|
||||
'' | *[!0-9]*) JOBS_OK="" ;;
|
||||
*) JOBS=$((10#$JOBS)); [ "$JOBS" -ge 1 ] && JOBS_OK=1 || JOBS_OK="" ;;
|
||||
esac
|
||||
if [ -z "$JOBS_OK" ]; then
|
||||
echo "Error: --jobs must be a positive integer (got '$JOBS')." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
[ $# -lt 1 ] && usage
|
||||
TASK_DIR="$1"
|
||||
REF_RUN_DIR="$2"
|
||||
shift 2
|
||||
shift
|
||||
|
||||
# Leading non-flag positionals are reference runs; collection stops at the first
|
||||
# harbor flag so `<task> <ref> -k 4` keeps working and `4` is never read as a run.
|
||||
REF_DIRS=()
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
-*) break ;;
|
||||
*) REF_DIRS+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
|
||||
# Resolve to absolute paths — harbor cd's around internally; the replay
|
||||
# agent receives the path as an --agent-kwarg and won't know our cwd.
|
||||
@@ -101,6 +155,102 @@ TASK_DIR_ABS=$(cd "$TASK_DIR" 2>/dev/null && pwd) || {
|
||||
echo "Error: task-dir does not exist: $TASK_DIR" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Unfiltered on purpose: the child, not a name test here, decides what can be replayed.
|
||||
if [ -n "$ALL_RUNS" ]; then
|
||||
[ ${#REF_DIRS[@]} -gt 0 ] && {
|
||||
echo "Error: pass --all or explicit reference-run dirs, not both." >&2
|
||||
exit 1
|
||||
}
|
||||
REF_PARENT="$TASK_DIR_ABS/reference-runs"
|
||||
[ -d "$REF_PARENT" ] || {
|
||||
echo "Error: --all needs $REF_PARENT, which does not exist." >&2
|
||||
exit 1
|
||||
}
|
||||
for _cand in "$REF_PARENT"/*/; do
|
||||
[ -d "$_cand" ] && REF_DIRS+=("${_cand%/}")
|
||||
done
|
||||
[ ${#REF_DIRS[@]} -eq 0 ] && {
|
||||
echo "Error: $REF_PARENT holds no reference runs." >&2
|
||||
exit 1
|
||||
}
|
||||
fi
|
||||
|
||||
[ ${#REF_DIRS[@]} -eq 0 ] && usage
|
||||
|
||||
# Several runs: re-run ITSELF once each, so every child does the full preflight in an
|
||||
# output dir of its own — harbor names job dirs by the second, and sharing one corrupts.
|
||||
if [ ${#REF_DIRS[@]} -gt 1 ]; then
|
||||
OUT_BASE="${HARBOR_REGRADE_OUT:-harbor-jobs}"
|
||||
mkdir -p "$OUT_BASE"
|
||||
CHILD_FLAGS=()
|
||||
[ -n "$FAST_REQUESTED" ] && CHILD_FLAGS+=(--fast)
|
||||
[ -n "$REPLACE" ] && CHILD_FLAGS+=(--replace)
|
||||
echo "Re-grading ${#REF_DIRS[@]} reference runs, $JOBS at a time."
|
||||
FAILED=()
|
||||
FAILED_LOGS=()
|
||||
BATCH_PIDS=()
|
||||
# Without this a killed driver leaves its children running, and on a cloud backend
|
||||
# each one is a sandbox that bills until something else reaps it.
|
||||
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 143' TERM
|
||||
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 130' INT
|
||||
IDX=0
|
||||
TOTAL=${#REF_DIRS[@]}
|
||||
while [ "$IDX" -lt "$TOTAL" ]; do
|
||||
BATCH_PIDS=()
|
||||
BATCH_REFS=()
|
||||
BATCH_LOGS=()
|
||||
for ((_j = 0; _j < JOBS && IDX < TOTAL; _j++)); do
|
||||
REF="${REF_DIRS[$IDX]}"
|
||||
RUN_ID=$(basename "$REF")
|
||||
# --job-name, not a nested output dir: children keep harbor's own
|
||||
# harbor-jobs/<job>/<trial> shape. Indexed so duplicate args cannot collide.
|
||||
JOB_NAME="regrade-$((IDX + 1))-$RUN_ID"
|
||||
LOG="$OUT_BASE/$JOB_NAME.log"
|
||||
echo " starting $RUN_ID (log: $LOG)"
|
||||
"$SCRIPT_DIR/harbor-regrade" "$TASK_DIR_ABS" "$REF" \
|
||||
${CHILD_FLAGS[@]+"${CHILD_FLAGS[@]}"} "$@" \
|
||||
--job-name "$JOB_NAME" > "$LOG" 2>&1 &
|
||||
BATCH_PIDS+=($!)
|
||||
BATCH_REFS+=("$RUN_ID")
|
||||
BATCH_LOGS+=("$LOG")
|
||||
IDX=$((IDX + 1))
|
||||
done
|
||||
for ((_i = 0; _i < ${#BATCH_PIDS[@]}; _i++)); do
|
||||
if wait "${BATCH_PIDS[$_i]}"; then
|
||||
echo " ok ${BATCH_REFS[$_i]}"
|
||||
else
|
||||
echo " FAILED ${BATCH_REFS[$_i]}"
|
||||
FAILED+=("${BATCH_REFS[$_i]}")
|
||||
FAILED_LOGS+=("${BATCH_LOGS[$_i]}")
|
||||
fi
|
||||
done
|
||||
done
|
||||
if [ ${#FAILED[@]} -gt 0 ]; then
|
||||
echo "${#FAILED[@]} of $TOTAL failed. Their output, in full — re-running is safe:" >&2
|
||||
for _f in "${FAILED_LOGS[@]}"; do echo " $_f" >&2; done
|
||||
exit 1
|
||||
fi
|
||||
echo "All $TOTAL re-graded."
|
||||
# Children file their own grades into per-run logs we do not echo, so count the
|
||||
# results rather than claiming them. Only the atomic destination is countable:
|
||||
# a superseded run is gone, so there is no before/after to compare against.
|
||||
case "${HARBOR_GRADER_MODE:-}" in
|
||||
rubric-*)
|
||||
FILED=0
|
||||
for _r in "${REF_DIRS[@]}"; do
|
||||
[ -d "$TASK_DIR_ABS/rubric-regrades/$(basename "$_r")" ] && FILED=$((FILED + 1))
|
||||
done
|
||||
echo "Filed $FILED of $TOTAL into $TASK_DIR_ABS/rubric-regrades/."
|
||||
if [ "$FILED" -lt "$TOTAL" ]; then
|
||||
echo "The rest are still in their job dirs; their logs above say why." >&2
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
exit 0
|
||||
fi
|
||||
|
||||
REF_RUN_DIR="${REF_DIRS[0]}"
|
||||
REF_RUN_DIR_ABS=$(cd "$REF_RUN_DIR" 2>/dev/null && pwd) || {
|
||||
echo "Error: reference-run-dir does not exist: $REF_RUN_DIR" >&2
|
||||
exit 1
|
||||
@@ -114,7 +264,10 @@ if [ -d "$TASK_DIR_ABS/environment" ]; then
|
||||
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
|
||||
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
|
||||
if [ -f "$DNSJAIL_SRC" ]; then
|
||||
cp "$DNSJAIL_SRC" "$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
|
||||
# Rename into place: children share this task dir, and a half-written
|
||||
# script is one a sibling's image build can pick up.
|
||||
_dnsjail_dst="$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
|
||||
cp "$DNSJAIL_SRC" "$_dnsjail_dst.tmp.$$" && mv -f "$_dnsjail_dst.tmp.$$" "$_dnsjail_dst"
|
||||
break
|
||||
fi
|
||||
done
|
||||
@@ -135,7 +288,8 @@ case "${HARBOR_GRADER_MODE:-}" in
|
||||
[ -f "$RUBRIC_RENDER_SRC" ] || continue
|
||||
if [ ! -f "$RUBRIC_RENDER_DEST" ] || ! cmp -s "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"; then
|
||||
mkdir -p "$TASK_DIR_ABS/tests"
|
||||
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"
|
||||
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST.tmp.$$" &&
|
||||
mv -f "$RUBRIC_RENDER_DEST.tmp.$$" "$RUBRIC_RENDER_DEST"
|
||||
echo "Note: staged the current shared render-rubric-grade.py into $TASK_DIR_ABS/tests (rubric grader mode)." >&2
|
||||
fi
|
||||
break
|
||||
@@ -206,13 +360,16 @@ GRADER_MODE_FLAG=()
|
||||
# inside the unchanged agentic harness. See raccoon-shared/test.sh GRADER_MODEL.
|
||||
[ -n "${HARBOR_GRADER_MODEL:-}" ] && GRADER_MODE_FLAG+=(--verifier-env "GRADER_MODEL=$HARBOR_GRADER_MODEL")
|
||||
|
||||
# Optional sample count: HARBOR_GRADER_SAMPLES=1 overrides the default 3 samples.
|
||||
# For measuring per-sample properties of the grader (e.g. how often it emits a
|
||||
# schema-valid grade.json), 1 sample across N distinct trajectories is a better
|
||||
# estimator than 3 samples across N/3 — it decorrelates per-task effects for the
|
||||
# same token spend. See raccoon-shared/test.sh GRADER_SAMPLES.
|
||||
[ -n "${HARBOR_GRADER_SAMPLES:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$HARBOR_GRADER_SAMPLES")
|
||||
# Optional sample count, under either name. Unset, the task's own frozen tests/test.sh
|
||||
# decides: tasks created before Sep 2026 average 3 samples, newer ones grade once.
|
||||
_SAMPLES="${_CALLER_SAMPLES:-${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}}"
|
||||
if [ -n "$_SAMPLES" ]; then
|
||||
if ! [ "$_SAMPLES" -ge 1 ] 2>/dev/null; then
|
||||
echo "harbor-regrade: ERROR — GRADER_SAMPLES must be a positive integer (got '$_SAMPLES')." >&2
|
||||
exit 1
|
||||
fi
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$_SAMPLES")
|
||||
fi
|
||||
|
||||
# Optional API timeout: HARBOR_API_TIMEOUT_MS=<ms> raises the per-request API
|
||||
# timeout of the grader's claude calls via verifier env (API_TIMEOUT_MS is the
|
||||
@@ -283,9 +440,83 @@ fi
|
||||
[ -n "${HARBOR_GRADING_STANDARD:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADING_STANDARD=$HARBOR_GRADING_STANDARD")
|
||||
|
||||
# A regrade is all verifier, so every call it makes is the grader's.
|
||||
if [ -f "$REPO_ROOT/scripts/lib/call-origin.sh" ]; then
|
||||
. "$REPO_ROOT/scripts/lib/call-origin.sh"
|
||||
_GRADER_METADATA="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
|
||||
# `if`, not `[ ... ] && ...`: an AND-list is this block's last statement, so under
|
||||
# `set -e` an empty value would abort the regrade rather than just skip the header.
|
||||
if [ -n "$_GRADER_METADATA" ]; then
|
||||
GRADER_MODE_FLAG+=(
|
||||
--verifier-env
|
||||
"ANTHROPIC_CUSTOM_HEADERS=$CALL_ORIGIN_HEADER: $_GRADER_METADATA"
|
||||
)
|
||||
fi
|
||||
fi
|
||||
|
||||
# Where this grade would be filed, and whether filing it destroys anything. An atomic
|
||||
# regrade lands beside the run in rubric-regrades/<run>; a holistic one replaces the
|
||||
# run's own grade, so its destination is always occupied and never files unasked.
|
||||
rel() { case "$1" in "$PWD"/*) printf '%s' "${1#"$PWD"/}" ;; *) printf '%s' "$1" ;; esac; }
|
||||
|
||||
RUN_ID=$(basename "$REF_RUN_DIR_ABS")
|
||||
RUBRIC_MODE=""
|
||||
case "${HARBOR_GRADER_MODE:-}" in rubric-*) RUBRIC_MODE=1 ;; esac
|
||||
if [ -n "$RUBRIC_MODE" ]; then
|
||||
FILE_DEST="$TASK_DIR_ABS/rubric-regrades/$RUN_ID"
|
||||
else
|
||||
FILE_DEST="$REF_RUN_DIR_ABS"
|
||||
fi
|
||||
|
||||
AUTOFILE=""
|
||||
if [ ! -e "$FILE_DEST" ] || [ -n "$REPLACE" ]; then
|
||||
AUTOFILE=1
|
||||
fi
|
||||
if [ -z "$AUTOFILE" ]; then
|
||||
echo "Note: $(rel "$FILE_DEST")" >&2
|
||||
echo " already holds a grade, so this regrade will not be filed." >&2
|
||||
echo " Where it landed is printed when it finishes." >&2
|
||||
fi
|
||||
|
||||
# Where the trial will land. Harbor's own job name is a second-granularity timestamp,
|
||||
# so naming it here is what makes the trial findable afterwards (the fan-out passes one).
|
||||
CALLER_JOB_NAME=""
|
||||
CALLER_OUT=""
|
||||
_prev=""
|
||||
for _a in "$@"; do
|
||||
case "$_prev" in
|
||||
--job-name) CALLER_JOB_NAME="$_a" ;;
|
||||
-o | --output-dir) CALLER_OUT="$_a" ;;
|
||||
esac
|
||||
case "$_a" in
|
||||
--job-name=*) CALLER_JOB_NAME="${_a#--job-name=}" ;;
|
||||
--output-dir=*) CALLER_OUT="${_a#--output-dir=}" ;;
|
||||
esac
|
||||
_prev="$_a"
|
||||
done
|
||||
|
||||
JOB_NAME="$CALLER_JOB_NAME"
|
||||
JOB_NAME_FLAG=()
|
||||
if [ -n "$AUTOFILE" ] && [ -n "$CALLER_OUT" ]; then
|
||||
echo "Note: -o/--output-dir passed, so this regrade will not be filed into" >&2
|
||||
echo " $(rel "$FILE_DEST") — copy it yourself when it finishes." >&2
|
||||
AUTOFILE=""
|
||||
fi
|
||||
if [ -z "$JOB_NAME" ] && [ -z "$CALLER_OUT" ]; then
|
||||
# $$ as well as the epoch: harbor refuses an existing job dir outright, and two
|
||||
# regrades of one run can start in the same second.
|
||||
JOB_NAME="regrade-$RUN_ID-$(date +%s)-$$"
|
||||
JOB_NAME_FLAG=(--job-name "$JOB_NAME")
|
||||
fi
|
||||
|
||||
# RACCOON_DNS_JAIL is inert here on purpose: replay_agent runs no agent turn, so there is
|
||||
# nothing to restrict — only the verifier runs, and it needs the proxy.
|
||||
exec harbor run \
|
||||
#
|
||||
# harbor runs as a child, not under `exec`, so the filing below runs after it exits.
|
||||
# TERM/INT are forwarded so a kill here never orphans a job (on cloud, a billing sandbox).
|
||||
HARBOR_EXIT=0
|
||||
HARBOR_SIGNALLED=""
|
||||
harbor run \
|
||||
-p "$TASK_DIR_ABS" \
|
||||
--agent-import-path replay_agent:ReplayAgent \
|
||||
--ak "reference_run_dir=$REF_RUN_DIR_ABS" \
|
||||
@@ -296,4 +527,74 @@ exec harbor run \
|
||||
--yes \
|
||||
-o "$OUT_DIR" \
|
||||
${GRADER_MODE_FLAG[@]+"${GRADER_MODE_FLAG[@]}"} \
|
||||
"$@"
|
||||
${JOB_NAME_FLAG[@]+"${JOB_NAME_FLAG[@]}"} \
|
||||
"$@" &
|
||||
HARBOR_PID=$!
|
||||
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
|
||||
trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT
|
||||
wait "$HARBOR_PID" || HARBOR_EXIT=$?
|
||||
if [ -n "$HARBOR_SIGNALLED" ]; then
|
||||
# The first wait was interrupted by the trap; wait again for harbor's real status.
|
||||
wait "$HARBOR_PID" || HARBOR_EXIT=$?
|
||||
fi
|
||||
trap - TERM INT
|
||||
|
||||
# A failed grade has nothing to report on, and a caller-directed -o put the trial
|
||||
# somewhere this script cannot name.
|
||||
if [ "$HARBOR_EXIT" -ne 0 ] || [ -n "$CALLER_OUT" ]; then
|
||||
exit "$HARBOR_EXIT"
|
||||
fi
|
||||
|
||||
JOB_DIR="$OUT_DIR/$JOB_NAME"
|
||||
TRIALS=()
|
||||
for _t in "$JOB_DIR"/*__*/; do
|
||||
[ -d "$_t" ] && TRIALS+=("${_t%/}")
|
||||
done
|
||||
|
||||
# Atomic grades sit beside the run; a holistic one replaces it, which means removing
|
||||
# the directory it supersedes (see copy-reference-run.ts for why swap, not merge).
|
||||
COPY_FLAGS=(--rubric-regrade)
|
||||
[ -z "$RUBRIC_MODE" ] && COPY_FLAGS=(--supersede "$(rel "$REF_RUN_DIR_ABS")")
|
||||
FILE_CMD="npx tsx $(rel "$SCRIPT_DIR")/copy-reference-run.ts ${COPY_FLAGS[*]}"
|
||||
|
||||
if [ ${#TRIALS[@]} -eq 0 ]; then
|
||||
echo "Note: no trial directory under $JOB_DIR, so there is no grade." >&2
|
||||
exit "$HARBOR_EXIT"
|
||||
fi
|
||||
if [ ${#TRIALS[@]} -gt 1 ]; then
|
||||
echo "Note: ${#TRIALS[@]} trials under $JOB_DIR (-k grades one run repeatedly)." >&2
|
||||
echo " Pick the one to keep and file it: $FILE_CMD <trial-dir>" >&2
|
||||
exit "$HARBOR_EXIT"
|
||||
fi
|
||||
|
||||
# Not filing: say where the grade is and how the two compare, since comparing them is
|
||||
# the whole reason it was left alone. Adopting is a copy — re-running with --replace
|
||||
# would spend the 15-30 minutes again for a grade already sitting on disk.
|
||||
if [ -z "$AUTOFILE" ]; then
|
||||
NEW_REWARD=$(cat "${TRIALS[0]}/verifier/reward.txt" 2>/dev/null || echo "?")
|
||||
# A stored atomic grade is a whole trial, so its reward sits under verifier/;
|
||||
# a reference run keeps its own at the top.
|
||||
OLD_REWARD=$(cat "$FILE_DEST/verifier/reward.txt" 2>/dev/null ||
|
||||
cat "$FILE_DEST/reward.txt" 2>/dev/null || echo "?")
|
||||
echo "" >&2
|
||||
echo "Graded, not filed — that run already holds a grade." >&2
|
||||
echo " new $NEW_REWARD $(rel "${TRIALS[0]}")" >&2
|
||||
echo " filed $OLD_REWARD $(rel "$FILE_DEST")" >&2
|
||||
echo "" >&2
|
||||
echo "Adopt this grade:" >&2
|
||||
echo " npx tsx $(rel "$SCRIPT_DIR")/copy-reference-run.ts \\" >&2
|
||||
echo " ${COPY_FLAGS[*]} \\" >&2
|
||||
echo " $(rel "${TRIALS[0]}")" >&2
|
||||
echo "" >&2
|
||||
echo "Or pass --replace next time to file it without this step." >&2
|
||||
exit "$HARBOR_EXIT"
|
||||
fi
|
||||
|
||||
# Best-effort from here: the grade already cost 15-30 minutes, so a filing problem
|
||||
# prints the command to finish by hand rather than failing the regrade.
|
||||
if ! npx tsx "$SCRIPT_DIR/copy-reference-run.ts" "${COPY_FLAGS[@]}" "${TRIALS[0]}" >&2; then
|
||||
echo "Warning: the grade is in ${TRIALS[0]} but could not be filed. Retry with:" >&2
|
||||
echo " $FILE_CMD ${TRIALS[0]}" >&2
|
||||
fi
|
||||
|
||||
exit "$HARBOR_EXIT"
|
||||
|
||||
Reference in New Issue
Block a user