added potion-polyglot worker folder w/o repos
This commit is contained in:
299
worker-toolkit-potion-polyglot/scripts/harbor-regrade
Executable file
299
worker-toolkit-potion-polyglot/scripts/harbor-regrade
Executable file
@@ -0,0 +1,299 @@
|
||||
#!/bin/bash
|
||||
# Re-grade an existing reference run without re-invoking the agent.
|
||||
#
|
||||
# Spins up a normal harbor trial, but plugs in scripts/replay_agent.py
|
||||
# instead of a real agent. The replay agent overlays the captured
|
||||
# agent-output into /workspace, applies any captured deletions, drops
|
||||
# the captured trajectory at /logs/agent/trajectory.json so the grader
|
||||
# reads the same transcript it would for the original run, then exits.
|
||||
# The verifier (real test.sh, real LLM grader if present) runs as it
|
||||
# would for any other trial.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
|
||||
#
|
||||
# Examples:
|
||||
# # Single regrade
|
||||
# scripts/harbor-regrade \
|
||||
# harbor-tasks/<slug> \
|
||||
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg
|
||||
#
|
||||
# # Ten regrades of the same reference run (independent grader trials)
|
||||
# scripts/harbor-regrade \
|
||||
# harbor-tasks/<slug> \
|
||||
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg \
|
||||
# -k 10
|
||||
#
|
||||
# --fast runs the GRADER in claude's fast serving mode (faster output at a higher
|
||||
# token rate). The replay agent runs no model, so the grader is the only model in
|
||||
# this path. Serving speed and cost change; the grade itself is not steered.
|
||||
#
|
||||
# See scripts/replay_agent.py for what the agent actually does, and the
|
||||
# `verifier: capture tracked-file deletions in agent-output` PR for the
|
||||
# capture half of this flow (_HARBOR_DELETIONS.txt).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
|
||||
# Source API key + any verifier env from the repo's .env
|
||||
if [ -f "$REPO_ROOT/.env" ]; then
|
||||
set -a
|
||||
source "$REPO_ROOT/.env"
|
||||
set +a
|
||||
fi
|
||||
|
||||
# The grader authenticates with ANTHROPIC_API_KEY straight out of the .env sourced above, so
|
||||
# a .env saved on Windows would hand it a value with a carriage return still attached.
|
||||
if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then
|
||||
# shellcheck disable=SC1091
|
||||
HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true
|
||||
harness_setup_credentials >/dev/null 2>&1 || true
|
||||
fi
|
||||
|
||||
if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
|
||||
. "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
|
||||
fi
|
||||
|
||||
# When running inside a devcontainer, harbor needs HOST paths for docker
|
||||
# bind mounts (the docker daemon is on the host).
|
||||
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
|
||||
cd "$HOST_WORKSPACE"
|
||||
fi
|
||||
|
||||
usage() {
|
||||
cat >&2 <<EOF
|
||||
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
|
||||
|
||||
Required arguments:
|
||||
<task-dir> harbor-tasks/<slug> — same dir you'd pass to scripts/harbor-run.
|
||||
<reference-run-dir> harbor-tasks/<slug>/reference-runs/<run-id> — must contain
|
||||
agent-output/ (and ideally agent/trajectory.json).
|
||||
|
||||
Optional arguments:
|
||||
--fast grade in claude's fast serving mode (higher token rate,
|
||||
faster output). Anything else is passed through to harbor.
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
|
||||
# `--fast` is consumed here (harbor has no such flag); everything else stays an opaque
|
||||
# passthrough. Stripped first so it may appear anywhere, not just after the positionals.
|
||||
FAST_REQUESTED=""
|
||||
REGRADE_ARGS=()
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--fast) FAST_REQUESTED=1; shift ;;
|
||||
*) REGRADE_ARGS+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
set -- "${REGRADE_ARGS[@]+"${REGRADE_ARGS[@]}"}"
|
||||
|
||||
[ $# -lt 2 ] && usage
|
||||
TASK_DIR="$1"
|
||||
REF_RUN_DIR="$2"
|
||||
shift 2
|
||||
|
||||
# Resolve to absolute paths — harbor cd's around internally; the replay
|
||||
# agent receives the path as an --agent-kwarg and won't know our cwd.
|
||||
TASK_DIR_ABS=$(cd "$TASK_DIR" 2>/dev/null && pwd) || {
|
||||
echo "Error: task-dir does not exist: $TASK_DIR" >&2
|
||||
exit 1
|
||||
}
|
||||
REF_RUN_DIR_ABS=$(cd "$REF_RUN_DIR" 2>/dev/null && pwd) || {
|
||||
echo "Error: reference-run-dir does not exist: $REF_RUN_DIR" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Newer-generation Dockerfiles COPY environment/dns-jail/, a derived directory
|
||||
# that harbor-run pre-stages but a bare regrade context may lack — the sandbox
|
||||
# build then fails before the verifier ever starts. Recreate it the same way.
|
||||
if [ -d "$TASK_DIR_ABS/environment" ]; then
|
||||
mkdir -p "$TASK_DIR_ABS/environment/dns-jail"
|
||||
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
|
||||
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
|
||||
if [ -f "$DNSJAIL_SRC" ]; then
|
||||
cp "$DNSJAIL_SRC" "$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
|
||||
break
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# Rubric grader modes (HARBOR_GRADER_MODE=rubric-*) read tests/render-rubric-grade.py,
|
||||
# a shared asset like the dns-jail script above. A task created before the rubric
|
||||
# renderer shipped has no copy, and a stale copy aggregates with outdated weights;
|
||||
# either way the regrade must run against the current shared copy. Stage it
|
||||
# host-side — the container only ever sees the task directory. The source lives at
|
||||
# harbor-tasks/raccoon-shared/ in the internal repo and task-shared/ in a worker
|
||||
# toolkit checkout; first one present wins.
|
||||
case "${HARBOR_GRADER_MODE:-}" in
|
||||
rubric-*)
|
||||
RUBRIC_RENDER_DEST="$TASK_DIR_ABS/tests/render-rubric-grade.py"
|
||||
for RUBRIC_RENDER_SRC in "$REPO_ROOT/harbor-tasks/raccoon-shared/render-rubric-grade.py" \
|
||||
"$REPO_ROOT/task-shared/render-rubric-grade.py"; do
|
||||
[ -f "$RUBRIC_RENDER_SRC" ] || continue
|
||||
if [ ! -f "$RUBRIC_RENDER_DEST" ] || ! cmp -s "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"; then
|
||||
mkdir -p "$TASK_DIR_ABS/tests"
|
||||
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"
|
||||
echo "Note: staged the current shared render-rubric-grade.py into $TASK_DIR_ABS/tests (rubric grader mode)." >&2
|
||||
fi
|
||||
break
|
||||
done
|
||||
;;
|
||||
esac
|
||||
|
||||
# NOTE: agent-output/ is intentionally NOT required here. Advisory tasks (the
|
||||
# agent only reads + answers in chat) make no workspace edits, so a faithful
|
||||
# capture has an empty/absent agent-output/ — the deliverable lives in the
|
||||
# captured transcript (agent/trajectory.json) that the grader reads. ReplayAgent
|
||||
# overlays agent-output/ when present and otherwise grades base-workspace +
|
||||
# transcript, but FAILS LOUDLY if the transcript shows file-mutating tool calls
|
||||
# with no agent-output/ (genuine lost edits). So we let it make that call.
|
||||
if [ ! -d "$REF_RUN_DIR_ABS/agent-output" ]; then
|
||||
echo "Note: $REF_RUN_DIR_ABS has no agent-output/ — replaying as an" >&2
|
||||
echo " advisory run (base workspace + captured transcript). See" >&2
|
||||
echo " scripts/replay_agent.py for the lost-edits safety guard." >&2
|
||||
fi
|
||||
|
||||
# Make scripts/ importable so harbor can find replay_agent:ReplayAgent.
|
||||
# ${PYTHONPATH:+...} so an unset PYTHONPATH doesn't leave a trailing colon —
|
||||
# python treats the resulting empty entry as the CWD, silently putting
|
||||
# whatever directory the user ran this from on harbor's sys.path.
|
||||
export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
|
||||
|
||||
# Environment backend. Explicit HARBOR_ENV wins; otherwise default to docker in
|
||||
# a worker-toolkit checkout (detected by toolkit.json at the repo root) and
|
||||
# daytona in the internal repo. See scripts/harbor-run for the full rationale
|
||||
# (why the toolkit needs docker, why the marker is a workspace file not an image
|
||||
# env, and why daytona must NOT pass --no-delete — billed sandbox).
|
||||
if [ -n "${HARBOR_ENV:-}" ]; then
|
||||
ENV_TYPE="$HARBOR_ENV"
|
||||
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
|
||||
ENV_TYPE="docker"
|
||||
else
|
||||
ENV_TYPE="daytona"
|
||||
fi
|
||||
DELETE_FLAGS="--no-delete"
|
||||
case "$ENV_TYPE" in daytona|modal) DELETE_FLAGS="" ;; esac
|
||||
|
||||
# Modal bills a sandbox's REQUESTED cpu/memory for its whole life, and harbor's default
|
||||
# `auto` requests the task's full cpus/memory_mb; `limit` requests Modal's floor and caps
|
||||
# at the declared values. Override: HARBOR_CPUS_POLICY / HARBOR_MEMORY_POLICY.
|
||||
RESOURCE_FLAGS=""
|
||||
[ "$ENV_TYPE" = "modal" ] && RESOURCE_FLAGS="--cpus ${HARBOR_CPUS_POLICY:-limit} --memory ${HARBOR_MEMORY_POLICY:-limit}"
|
||||
|
||||
if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
|
||||
echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
|
||||
echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Output dir. harbor names the job subdir by second-granularity timestamp, so
|
||||
# many regrades launched in the same second under one -o collide
|
||||
# ("Job directory ... already exists and cannot be resumed"). Set
|
||||
# HARBOR_REGRADE_OUT to a per-run unique dir when running a parallel sweep.
|
||||
OUT_DIR="${HARBOR_REGRADE_OUT:-harbor-jobs}"
|
||||
|
||||
# Optional grader mode: HARBOR_GRADER_MODE=one-shot flips the task's test.sh into
|
||||
# the no-tools one-shot grader (vs the default agentic grader) via verifier env —
|
||||
# lets us A/B the agenticity gap without forking the task. See raccoon-shared/test.sh.
|
||||
GRADER_MODE_FLAG=()
|
||||
[ -n "${HARBOR_GRADER_MODE:-}" ] && GRADER_MODE_FLAG=(--verifier-env "GRADER_MODE=$HARBOR_GRADER_MODE")
|
||||
|
||||
# Optional grader model: HARBOR_GRADER_MODEL=claude-fable-5-1 overrides the grader's
|
||||
# model (default: the `opus` alias) via verifier env — lets us A/B the grader model
|
||||
# inside the unchanged agentic harness. See raccoon-shared/test.sh GRADER_MODEL.
|
||||
[ -n "${HARBOR_GRADER_MODEL:-}" ] && GRADER_MODE_FLAG+=(--verifier-env "GRADER_MODEL=$HARBOR_GRADER_MODEL")
|
||||
|
||||
# Optional sample count: HARBOR_GRADER_SAMPLES=1 overrides the default 3 samples.
|
||||
# For measuring per-sample properties of the grader (e.g. how often it emits a
|
||||
# schema-valid grade.json), 1 sample across N distinct trajectories is a better
|
||||
# estimator than 3 samples across N/3 — it decorrelates per-task effects for the
|
||||
# same token spend. See raccoon-shared/test.sh GRADER_SAMPLES.
|
||||
[ -n "${HARBOR_GRADER_SAMPLES:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$HARBOR_GRADER_SAMPLES")
|
||||
|
||||
# Optional API timeout: HARBOR_API_TIMEOUT_MS=<ms> raises the per-request API
|
||||
# timeout of the grader's claude calls via verifier env (API_TIMEOUT_MS is the
|
||||
# env var Claude Code itself reads, so test.sh needs no knowledge of it). For
|
||||
# proxies whose responses outlast the CLI default.
|
||||
[ -n "${HARBOR_API_TIMEOUT_MS:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "API_TIMEOUT_MS=$HARBOR_API_TIMEOUT_MS")
|
||||
|
||||
# Fast serving mode for the grader's own claude calls (--fast). Per-task tests/ assets
|
||||
# are frozen at creation, so say when this task's copy cannot act on the flag. The note
|
||||
# names that copy, not a shared path — this script ships to workers under another layout.
|
||||
if [ -n "$FAST_REQUESTED" ]; then
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADER_FAST_MODE=true")
|
||||
if [ -f "$TASK_DIR_ABS/tests/test.sh" ] &&
|
||||
! grep -q 'GRADER_FAST_MODE' "$TASK_DIR_ABS/tests/test.sh"; then
|
||||
echo "Note: --fast passed, but this task's tests/test.sh does not read" >&2
|
||||
echo " GRADER_FAST_MODE, so the grader will run at normal speed —" >&2
|
||||
echo " its copy predates the flag. Refresh the task's tests/test.sh" >&2
|
||||
echo " from the current shared grader assets to enable it." >&2
|
||||
fi
|
||||
fi
|
||||
|
||||
# Carry the SOURCE run's agent identity + model into this replay's own record.
|
||||
#
|
||||
# A replay reports `replay_agent:ReplayAgent` with model_name null, because no model
|
||||
# ran — the behaviour being graded came from the source run. Recording only
|
||||
# reference_run_dir makes that a pointer, and pointers dangle: a regrade is normally
|
||||
# copied back over the run it regraded, so the source usually no longer exists (501 of
|
||||
# 643 on-disk replays already point at a missing dir, none of them in the published
|
||||
# manifest either). Stamping the values here makes the replay self-describing, so the
|
||||
# originating harness and model survive the source's deletion.
|
||||
#
|
||||
# Regrading a REGRADE means the source is itself a replay, so copying its own identity
|
||||
# forward would overwrite the real provenance with a self-reference: inherit what it
|
||||
# inherited instead.
|
||||
#
|
||||
# Read with python3 rather than jq — jq is not guaranteed on a worker's box, and a
|
||||
# missing source result.json must degrade to "unknown", never abort the regrade.
|
||||
SOURCE_PROV_FLAGS=()
|
||||
if [ -f "$REF_RUN_DIR_ABS/result.json" ]; then
|
||||
SOURCE_PROV=$(python3 -c '
|
||||
import json, sys
|
||||
try:
|
||||
a = (json.load(open(sys.argv[1])).get("config") or {}).get("agent") or {}
|
||||
except Exception:
|
||||
sys.exit(0)
|
||||
kw = a.get("kwargs") or {}
|
||||
REPLAY = "replay_agent:ReplayAgent"
|
||||
if (a.get("import_path") or a.get("name")) == REPLAY:
|
||||
agent, model = kw.get("source_agent_import_path"), kw.get("source_model_name")
|
||||
else:
|
||||
agent, model = a.get("import_path") or a.get("name"), a.get("model_name")
|
||||
# An already-damaged chain cannot be recovered; leave it honestly unstamped rather
|
||||
# than propagating a source that names the replay agent itself.
|
||||
print("" if agent == REPLAY else agent or "")
|
||||
print(model or "")
|
||||
' "$REF_RUN_DIR_ABS/result.json" 2>/dev/null || true)
|
||||
SOURCE_AGENT=$(printf '%s\n' "$SOURCE_PROV" | sed -n 1p)
|
||||
SOURCE_MODEL=$(printf '%s\n' "$SOURCE_PROV" | sed -n 2p)
|
||||
[ -n "$SOURCE_AGENT" ] && SOURCE_PROV_FLAGS+=(--ak "source_agent_import_path=$SOURCE_AGENT")
|
||||
[ -n "$SOURCE_MODEL" ] && SOURCE_PROV_FLAGS+=(--ak "source_model_name=$SOURCE_MODEL")
|
||||
fi
|
||||
|
||||
# HARBOR_GRADING_STANDARD, when set, is passed through to the task's own
|
||||
# tests/test.sh as the GRADING_STANDARD verifier env var. What (if anything)
|
||||
# it does is decided by the scripts inside that tests/ directory; a test.sh
|
||||
# that reads no such variable ignores it.
|
||||
[ -n "${HARBOR_GRADING_STANDARD:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADING_STANDARD=$HARBOR_GRADING_STANDARD")
|
||||
|
||||
# RACCOON_DNS_JAIL is inert here on purpose: replay_agent runs no agent turn, so there is
|
||||
# nothing to restrict — only the verifier runs, and it needs the proxy.
|
||||
exec harbor run \
|
||||
-p "$TASK_DIR_ABS" \
|
||||
--agent-import-path replay_agent:ReplayAgent \
|
||||
--ak "reference_run_dir=$REF_RUN_DIR_ABS" \
|
||||
${SOURCE_PROV_FLAGS[@]+"${SOURCE_PROV_FLAGS[@]}"} \
|
||||
-e "$ENV_TYPE" \
|
||||
$DELETE_FLAGS \
|
||||
$RESOURCE_FLAGS \
|
||||
--yes \
|
||||
-o "$OUT_DIR" \
|
||||
${GRADER_MODE_FLAG[@]+"${GRADER_MODE_FLAG[@]}"} \
|
||||
"$@"
|
||||
Reference in New Issue
Block a user