ren worker folder adding orig, mv new one into root

This commit is contained in:
2026-09-25 10:34:29 -04:00
parent 10f0668e32
commit 5b010039d7
1308 changed files with 44597 additions and 1511 deletions

View File

@@ -99,8 +99,11 @@ if [ "$BROWSER_OPTIN" = "1" ]; then
fi
fi
# Resolve commit
RESOLVED_SHA=$(git -C "$REPO_DIR" rev-parse "$COMMIT")
# Resolve commit. Goes through resolve_pin so a task pinned before a history
# rewrite still builds: the pin is translated via task-shared/commit-maps/.
# shellcheck source=lib/resolve-pin.sh
. "$(dirname "$0")/lib/resolve-pin.sh"
RESOLVED_SHA=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
echo " Resolved SHA: $RESOLVED_SHA"
# --- Member-specific setup ---------------------------------------------------

View File

@@ -110,7 +110,11 @@ done
# Resolve the repo's real git dir (handles submodules, whose .git is a file).
GITDIR="$(git -C "$REPO_DIR" rev-parse --absolute-git-dir 2>/dev/null)" || skip "not a git repo: $REPO_DIR"
RESOLVED_SHA="$(git --git-dir="$GITDIR" rev-parse --quiet --verify "$COMMIT^{commit}" 2>/dev/null)" || \
# Via resolve_pin, so a task pinned before a history rewrite keeps being checked
# rather than silently skipping every run once its SHA stops resolving.
# shellcheck source=lib/resolve-pin.sh
. "$(dirname "$0")/lib/resolve-pin.sh"
RESOLVED_SHA="$(resolve_pin "$REPO_DIR" "$COMMIT" "$ROOT/task-shared/commit-maps" "$MEMBER" 2>/dev/null)" || \
skip "pinned commit $COMMIT not found in $REPO_DIR"
# --- Throwaway git state ------------------------------------------------------

View File

@@ -23,6 +23,7 @@ import os
import shlex
import sys
import tempfile
import tomllib
import uuid
from pathlib import Path
@@ -85,6 +86,32 @@ _INSTALL_CMD = (
)
# codex reserves its built-in provider ids, so the call-origin header needs a provider of
# our own. A trial's origin is a constant, so it is a static http_headers literal here
# rather than the env-var indirection the containers need — nothing to plumb into a
# sandbox, and no way for a missing var to lose the attribution.
PROXY_PROVIDER_TOML = """\
model_provider = "llm-proxy"
[model_providers.llm-proxy]
name = "LLM proxy"
base_url = "${OPENAI_BASE_URL}"
env_key = "OPENAI_API_KEY"
wire_api = "responses"
http_headers = { "X-Surge-Client-Metadata" = '{"origin":"harbor-trial"}' }
"""
# A custom provider reads its key from env_key and never from auth.json, so the proxy
# path must carry this even when an auth file was uploaded.
PROXY_KEY_VAR = "OPENAI_API_KEY"
def proxy_provider_config(openai_base_url: str) -> dict:
"""PROXY_PROVIDER_TOML as harbor's config dict, pointed at this trial's URL."""
return tomllib.loads(PROXY_PROVIDER_TOML.replace("${OPENAI_BASE_URL}", openai_base_url))
class SystemNodeCodex(Codex):
# Set by install()'s probe, read by build_cli_flags(). Mirrors the claude adapter.
_has_browser = False
@@ -101,6 +128,39 @@ class SystemNodeCodex(Codex):
self._get_env("OPENAI_API_KEY") or "", remote_auth_path
)
def _proxy_provider_flags(self) -> str:
"""`-c` overrides putting the trial on our own provider — the only place codex
can be told to send the call-origin header.
On the command line rather than in config.toml because harbor writes that file
itself, differently per version (0.20 hardcodes the block inline), while these
flags are ours in every version.
"""
base_url = self._get_env("OPENAI_BASE_URL") or ""
if "/llm_proxy/" not in base_url:
return ""
# A custom provider ignores auth.json, so leave that flow on the built-in
# provider: losing attribution beats breaking the run's auth.
if self._resolve_auth_json_path():
return ""
config = proxy_provider_config(base_url)
provider_id = config["model_provider"]
parts = [f"-c model_provider={provider_id}"]
for key, value in config["model_providers"][provider_id].items():
for path, leaf in (
[(f"{key}.{k}", v) for k, v in value.items()]
if isinstance(value, dict)
else [(key, value)]
):
# A TOML literal string, since the header value is JSON and carries its
# own double quotes.
quoted = f"'{leaf}'" if '"' in leaf else f'"{leaf}"'
parts.append(
"-c "
+ shlex.quote(f"model_providers.{provider_id}.{path}={quoted}")
)
return " ".join(parts)
def _refuse_shell_hostile_key(self) -> None:
"""Harbor's own Codex.run interpolates the key into a heredoc, so a key it cannot
escape would 401 with no stated cause. Refuse up front instead."""
@@ -127,6 +187,11 @@ class SystemNodeCodex(Codex):
reductions = load_harness_registry().require("codex").agent_config_flags()
if reductions:
flags = f"{flags} {reductions}".strip()
# Both run paths go through here, so this is where the trial picks up the
# provider that carries the call-origin header.
provider = self._proxy_provider_flags()
if provider:
flags = f"{flags} {provider}".strip()
return f"{flags} {self._browser_flag()}".strip() if self._browser_flag() else flags
def _browser_flag(self) -> str:
@@ -467,11 +532,16 @@ class NativeSnapshotCodex(SystemNodeCodex):
)
if openai_base_url := self._get_env("OPENAI_BASE_URL"):
env["OPENAI_BASE_URL"] = openai_base_url
# The provider that carries the origin header rides in on build_cli_flags,
# so this stays harbor's plain root key.
setup_command += (
'\ncat >>"$CODEX_HOME/config.toml" <<TOML\n'
'openai_base_url = "${OPENAI_BASE_URL}"\n'
"TOML"
)
# env_key names this, and the provider cannot fall back to auth.json.
if proxy_key := self._get_env(PROXY_KEY_VAR):
env[PROXY_KEY_VAR] = proxy_key
skills_command = self._build_register_skills_command()
if skills_command:
setup_command += f"\n{skills_command}"

View File

@@ -3,6 +3,8 @@
*
* Usage:
* npx tsx scripts/copy-reference-run.ts <trial-path> [trial-path...]
* npx tsx scripts/copy-reference-run.ts --rubric-regrade <trial-path>
* npx tsx scripts/copy-reference-run.ts --supersede <old-run-dir> <trial-path>
*
* Examples:
* # Copy a single trial
@@ -28,6 +30,8 @@
* indirection. Its location is per harness: Claude Code writes
* `agent/sessions/projects/-workspace/<id>.jsonl` with the id printed in
* `agent/claude-code.txt`; codex writes `agent/sessions/<Y>/<M>/<D>/rollout-*.jsonl`.
* - verifier/test-stdout.txt (the verifier's console output, where the reward is printed)
* - verifier/deterministic-signals.txt, verifier/rubric-grade(-<N>).json
* - config.json, result.json, trial.log
* - input-checksums.json — sha256 checksums of the task inputs the run was
* generated against (prompt, session snapshot, workspace patch, gitref),
@@ -37,6 +41,21 @@
* otherwise captured here at copy time as a fallback (capturedBy:
* 'copy').
*
* --rubric-regrade files an atomic-rubric regrade into rubric-regrades/<run>/
* instead: a second grade of a run that keeps its own, under the name the trial
* itself records for the run it graded. The trial is copied verbatim — verifier/
* and all — so a stored grade has the same shape whoever stored it.
*
* --supersede is the other direction: a holistic regrade replaces a run's grade,
* and the new copy lands under a name minted from the NEW reward and trial id, so
* adopting it means removing the directory it supersedes. Deleting is deliberate
* over merging into the old directory: a regrade grades a different number of
* samples than the run often did, so a merge would leave grade-2.md/grade-3.md
* from the previous grade beside the new grade-1.md with nothing marking the
* generation. A swapped directory holds exactly one grade by construction. The
* cost is the old run's session.jsonl, which a replay cannot reproduce; the
* trajectory it would be needed to rebuild travels with the replay itself.
*
* What is NOT copied:
* - agent/sessions/, agent/setup/ (workspace data — the resumable JSONL
* is hoisted out as `session.jsonl` above; everything else here is
@@ -51,11 +70,12 @@ import {
mkdirSync,
readdirSync,
readFileSync,
renameSync,
rmSync,
statSync,
writeFileSync,
} from 'fs';
import { basename, join } from 'path';
import { basename, join, resolve } from 'path';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { copyPath, copyTree } from './lib/copy-tree';
@@ -99,6 +119,36 @@ function newestRollout(sessionsDir: string): string | null {
return found[0].path;
}
/** What a replay trial records about itself: the run it graded, and whether the
* grade came from a rubric grader mode rather than the holistic one. */
function readReplayProvenance(trialPath: string): {
sourceRun: string | null;
rubricMode: boolean;
} {
let config: Record<string, unknown> | undefined;
try {
const parsed = JSON.parse(readFileSync(join(trialPath, 'result.json'), 'utf-8')) as {
config?: Record<string, unknown>;
};
config = parsed.config;
} catch {
config = undefined;
}
const agent = (config?.agent ?? {}) as { kwargs?: { reference_run_dir?: unknown } };
const src = agent.kwargs?.reference_run_dir;
const verifier = (config?.verifier ?? {}) as { env?: Record<string, unknown> };
const graderMode = verifier.env?.GRADER_MODE;
// Either marker is enough: the env records the mode harbor was handed, the
// rubric-grade file records what the grader actually produced.
const rubricMode =
(typeof graderMode === 'string' && graderMode.startsWith('rubric-')) ||
existsSync(join(trialPath, 'verifier', 'rubric-grade.json'));
return {
sourceRun: typeof src === 'string' && src.length > 0 ? basename(src.replace(/\/+$/, '')) : null,
rubricMode,
};
}
/**
* Resolve the task dir from the trial's result.json `task_name` — the FULL,
* unambiguous slug. Harbor TRUNCATES long task names in the trial DIRNAME, so two
@@ -127,7 +177,11 @@ function findTaskDirByResultJson(trialPath: string): string | null {
return existsSync(dir) ? dir : null;
}
function copyTrial(trialPath: string, destName?: string) {
function copyTrial(
trialPath: string,
opts: { destName?: string; rubricRegrade?: boolean; supersede?: string } = {}
) {
const { destName, rubricRegrade = false, supersede } = opts;
trialPath = trialPath.replace(/\/$/, '');
if (!existsSync(trialPath)) {
@@ -194,147 +248,198 @@ function copyTrial(trialPath: string, destName?: string) {
// is copied to exactly that name instead of the minted reward-<r>-<id> —
// used by the SxS pipeline so task.toml preference_rollouts, this dir, and
// the publish-manifest run_id stay byte-identical by construction.
const dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
if (existsSync(dest)) {
console.warn(`Warning: ${dest} already exists, overwriting`);
rmSync(dest, { recursive: true });
}
mkdirSync(dest, { recursive: true });
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
// companion reward-correctness.txt (N/A by design) and the machine-readable
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
// (the source of truth grade.md/reward.txt are rendered from), the
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
// render-stderr(-<N>).log, and grader-samples.txt.
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
// every per-sample record.)
//
// signals-status.txt qualifies the deterministic signals the grader was fed:
// it records whether every deterministic check actually produced a verdict
// ("ok") or one or more was killed before finishing ("degraded" — the grade
// is then NOT fully signal-backed). Without it a copied run is
// indistinguishable from a run whose checks all passed, so it must travel
// with the reward files.
const verifierDir = join(trialPath, 'verifier');
if (existsSync(verifierDir)) {
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
if (!entry.isFile()) continue;
const f = entry.name;
if (
f === 'reward.txt' ||
f === 'reward-correctness.txt' ||
f === 'reward.json' ||
f === 'signals-status.txt' ||
f === 'grader-samples.txt' ||
f === 'grader-regime.json' ||
/^grade(-\d+)?\.md$/.test(f) ||
/^grade(-\d+)?\.json$/.test(f) ||
/^grader-result(-\d+)?\.json$/.test(f) ||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
/^render-stderr(-\d+)?\.log$/.test(f)
) {
copyPath(join(verifierDir, f), join(dest, f));
}
}
}
const agentOutputDir = join(verifierDir, 'agent-output');
if (existsSync(agentOutputDir)) {
copyTree(agentOutputDir, join(dest, 'agent-output'));
}
// Copy agent session log and trajectory (not workspace)
const agentDir = join(trialPath, 'agent');
if (existsSync(agentDir)) {
mkdirSync(join(dest, 'agent'), { recursive: true });
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
// the Claude one left codex reference runs with nothing but the trajectory.
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
const src = join(agentDir, file);
if (existsSync(src)) copyPath(src, join(dest, 'agent', file));
}
// The grader (and downstream worldbench export / replay) reads
// agent/trajectory.json. If it's missing, scream so we don't silently
// ship a reference run that's only half-useful.
if (!existsSync(join(agentDir, 'trajectory.json'))) {
console.warn(
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
`Harbor's adapter for this harness failed to write it (typically because ` +
`the converter choked on the session log). Downstream consumers ` +
`(grader replay, worldbench export) need this file — investigate ` +
`before relying on this reference run.`
);
}
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
// gets embedded in claude-code.txt's first non-empty line; we use it to
// locate the sibling JSONL Claude Code wrote in the same trial.
// Layout is per harness, so each needs a case here — the same reason
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
// session, which is what codex got before this: nothing at all.
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
if (existsSync(claudeCodeTxt)) {
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
const sessionId = readSessionId(claudeCodeTxt);
if (sessionId) {
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
if (existsSync(jsonl)) copyPath(jsonl, join(dest, 'session.jsonl'));
}
} else {
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
const rollout = newestRollout(join(agentDir, 'sessions'));
if (rollout) copyPath(rollout, join(dest, 'session.jsonl'));
}
}
// Copy top-level metadata
for (const file of ['config.json', 'result.json', 'trial.log']) {
const src = join(trialPath, file);
if (existsSync(src)) copyPath(src, join(dest, file));
}
// Record the checksums of the task inputs this run was generated against
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
// submit-task.ts re-captures at packaging time and warns when any of them
// changed — the run then describes an older revision of the task than the
// one being shipped. Preferred source: the launch-time stamp harbor-run
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
// 'run') — it records the inputs the agent actually ran against, so an
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
// (trials from an older harbor-run, or a failed stamp): capture here at
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
// the weaker evidence.
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
const trialStamp = readTaskInputChecksums(trialStampPath);
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
// harbor-runs sharing a cwd) — its hashes describe some other task's
// inputs, so treat it as absent rather than importing false evidence.
const stampSlug = trialStamp?.taskSlug;
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
if (trialStamp && !stampMisrouted) {
copyPath(trialStampPath, join(dest, INPUT_CHECKSUMS_FILENAME));
} else {
if (stampMisrouted) {
console.warn(
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
);
}
writeFileSync(
join(dest, INPUT_CHECKSUMS_FILENAME),
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
let dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
const prov = readReplayProvenance(trialPath);
// An atomic grade in reference-runs/ would overwrite the holistic grade it exists
// to be compared against, and the two are not interchangeable.
if (!rubricRegrade && prov.rubricMode) {
console.error(
`Error: ${trialPath} is an atomic-rubric regrade, which does not belong in ` +
`reference-runs/. File it beside the run it graded:\n` +
` npx tsx scripts/copy-reference-run.ts --rubric-regrade ${trialPath}`
);
process.exit(1);
}
if (rubricRegrade) {
if (!prov.rubricMode) {
console.error(
`Error: ${trialPath} is not an atomic-rubric regrade (no rubric grader mode ` +
`in result.json, no verifier/rubric-grade.json). A holistic regrade replaces ` +
`the run's own grade — copy it without --rubric-regrade.`
);
process.exit(1);
}
if (!prov.sourceRun) {
console.error(
`Error: ${trialPath}/result.json records no reference_run_dir, so there is no ` +
`run to file this regrade under. Copy it by hand into ` +
`${join(taskDir, 'rubric-regrades')}/<run-id>/.`
);
process.exit(1);
}
dest = join(taskDir, 'rubric-regrades', prov.sourceRun);
}
// Build in a sibling directory and swap at the end. A grade costs 15-30 minutes and
// cannot be reproduced exactly, so an interrupted copy must leave the stored one intact.
const staged = `${dest}.staging-${process.pid}`;
rmSync(staged, { recursive: true, force: true });
mkdirSync(staged, { recursive: true });
// A stored atomic grade is the trial verbatim. Matching that shape exactly matters
// more than trimming it: grades stored by hand have it, and a reviewer opening one
// should not have to work out which way it was written.
if (rubricRegrade) {
copyTree(trialPath, staged);
} else {
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
// companion reward-correctness.txt (N/A by design) and the machine-readable
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
// (the source of truth grade.md/reward.txt are rendered from), the
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
// render-stderr(-<N>).log, and grader-samples.txt.
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
// every per-sample record.)
//
// signals-status.txt qualifies the deterministic signals the grader was fed:
// it records whether every deterministic check actually produced a verdict
// ("ok") or one or more was killed before finishing ("degraded" — the grade
// is then NOT fully signal-backed). Without it a copied run is
// indistinguishable from a run whose checks all passed, so it must travel
// with the reward files.
const verifierDir = join(trialPath, 'verifier');
if (existsSync(verifierDir)) {
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
if (!entry.isFile()) continue;
const f = entry.name;
if (
f === 'reward.json' ||
f === 'signals-status.txt' ||
f === 'grader-samples.txt' ||
f === 'grader-regime.json' ||
f === 'test-stdout.txt' ||
f === 'deterministic-signals.txt' ||
/^reward(-\d+)?\.txt$/.test(f) ||
/^reward-correctness(-\d+)?\.txt$/.test(f) ||
/^grade(-\d+)?\.md$/.test(f) ||
/^grade(-\d+)?\.json$/.test(f) ||
/^rubric-grade(-\d+)?\.json$/.test(f) ||
/^grader-result(-\d+)?\.json$/.test(f) ||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
// The renderer retries, so the real name is render-stderr-<sample>-attempt<n>.log.
/^render-stderr(-\d+)?(-attempt\d+)?\.log$/.test(f)
) {
copyPath(join(verifierDir, f), join(staged, f));
}
}
}
// A rubric regrade re-grades a run that already ships its own agent-output,
// trajectory and session; a second copy would only double the tarball.
const agentOutputDir = join(verifierDir, 'agent-output');
if (!rubricRegrade && existsSync(agentOutputDir)) {
copyTree(agentOutputDir, join(staged, 'agent-output'));
}
// Copy agent session log and trajectory (not workspace)
const agentDir = join(trialPath, 'agent');
if (!rubricRegrade && existsSync(agentDir)) {
mkdirSync(join(staged, 'agent'), { recursive: true });
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
// the Claude one left codex reference runs with nothing but the trajectory.
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
const src = join(agentDir, file);
if (existsSync(src)) copyPath(src, join(staged, 'agent', file));
}
// The grader (and downstream worldbench export / replay) reads
// agent/trajectory.json. If it's missing, scream so we don't silently
// ship a reference run that's only half-useful.
if (!existsSync(join(agentDir, 'trajectory.json'))) {
console.warn(
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
`Harbor's adapter for this harness failed to write it (typically because ` +
`the converter choked on the session log). Downstream consumers ` +
`(grader replay, worldbench export) need this file — investigate ` +
`before relying on this reference run.`
);
}
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
// gets embedded in claude-code.txt's first non-empty line; we use it to
// locate the sibling JSONL Claude Code wrote in the same trial.
// Layout is per harness, so each needs a case here — the same reason
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
// session, which is what codex got before this: nothing at all.
// Silent when there is nothing to hoist: no shipped tool reads this file and nothing
// validates it, so its absence is not worth a line of output.
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
if (existsSync(claudeCodeTxt)) {
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
const sessionId = readSessionId(claudeCodeTxt);
if (sessionId) {
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
if (existsSync(jsonl)) copyPath(jsonl, join(staged, 'session.jsonl'));
}
} else {
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
const rollout = newestRollout(join(agentDir, 'sessions'));
if (rollout) copyPath(rollout, join(staged, 'session.jsonl'));
}
}
// Copy top-level metadata. result.json is what makes a regrade self-describing
// (which run it graded, under which grader mode), so it travels either way.
for (const file of rubricRegrade
? ['result.json']
: ['config.json', 'result.json', 'trial.log']) {
const src = join(trialPath, file);
if (existsSync(src)) copyPath(src, join(staged, file));
}
// Not for a rubric regrade: this input set is what stales a RUN (prompt, snapshot,
// patch, gitref), and the run it re-graded already carries its own stamp.
if (!rubricRegrade) {
// Record the checksums of the task inputs this run was generated against
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
// submit-task.ts re-captures at packaging time and warns when any of them
// changed — the run then describes an older revision of the task than the
// one being shipped. Preferred source: the launch-time stamp harbor-run
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
// 'run') — it records the inputs the agent actually ran against, so an
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
// (trials from an older harbor-run, or a failed stamp): capture here at
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
// the weaker evidence.
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
const trialStamp = readTaskInputChecksums(trialStampPath);
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
// harbor-runs sharing a cwd) — its hashes describe some other task's
// inputs, so treat it as absent rather than importing false evidence.
const stampSlug = trialStamp?.taskSlug;
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
if (trialStamp && !stampMisrouted) {
copyPath(trialStampPath, join(staged, INPUT_CHECKSUMS_FILENAME));
} else {
if (stampMisrouted) {
console.warn(
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
);
}
writeFileSync(
join(staged, INPUT_CHECKSUMS_FILENAME),
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
);
}
}
}
// Files captured from a run can land unreadable to you, which makes packaging
// fail later. Fix that now. Guarded so it can never fail a copy that worked.
let perms = null;
try {
perms = normalizeTreePermissions(dest);
perms = normalizeTreePermissions(staged);
} catch (err) {
console.warn(`Warning: could not normalize permissions on ${dest}: ${String(err)}`);
console.warn(` The run copied fine. If packaging later fails on permissions:`);
@@ -350,6 +455,43 @@ function copyTrial(trialPath: string, destName?: string) {
);
}
if (existsSync(dest)) {
console.warn(`Warning: ${dest} already exists, overwriting`);
rmSync(dest, { recursive: true });
}
renameSync(staged, dest);
// Remove what this copy supersedes, now that the copy is on disk. Skipped when the
// minted name landed on the superseded directory itself — that is the copy, not a
// leftover.
if (supersede) {
const old = resolve(supersede);
if (!existsSync(old)) {
console.warn(`Warning: nothing to supersede at ${supersede} — it is already gone`);
} else if (old === resolve(dest)) {
console.log(`Superseded ${supersede} in place (same name)`);
} else {
// A stored atomic grade is keyed on the run's folder name, and superseding
// changes that name. Move it with the run — it grades the same behaviour — or
// it is left pointing at a run that no longer exists.
const storedGrade = join(taskDir, 'rubric-regrades', basename(old));
const movedGrade = join(taskDir, 'rubric-regrades', basename(dest));
if (existsSync(storedGrade) && existsSync(movedGrade)) {
console.warn(
`Warning: ${storedGrade} names the superseded run, but ${movedGrade} already ` +
`exists. Leaving both — remove whichever is obsolete.`
);
} else if (existsSync(storedGrade)) {
renameSync(storedGrade, movedGrade);
console.log(`Moved the stored atomic grade to ${movedGrade}`);
console.log(' It grades the same run. Grade it again under the atomic rubric');
console.log(' if the rubric changed since it was stored.');
}
rmSync(old, { recursive: true });
console.log(`Superseded ${supersede} (removed)`);
}
}
console.log(`Copied to ${dest}`);
console.log(` reward: ${reward}`);
console.log(` task: ${taskDir}`);
@@ -364,9 +506,19 @@ function copyTrial(trialPath: string, destName?: string) {
// Main
const rawArgs = process.argv.slice(2);
let destName: string | undefined;
let rubricRegrade = false;
let supersede: string | undefined;
const args: string[] = [];
for (let i = 0; i < rawArgs.length; i++) {
if (rawArgs[i] === '--dest-name') {
if (rawArgs[i] === '--rubric-regrade') {
rubricRegrade = true;
} else if (rawArgs[i] === '--supersede') {
supersede = rawArgs[++i];
if (!supersede) {
console.error('Error: --supersede requires the run directory being replaced');
process.exit(1);
}
} else if (rawArgs[i] === '--dest-name') {
destName = rawArgs[++i];
if (!destName) {
console.error('Error: --dest-name requires a value');
@@ -379,7 +531,9 @@ for (let i = 0; i < rawArgs.length; i++) {
if (args.length === 0) {
console.error(
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]'
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]\n' +
' npx tsx scripts/copy-reference-run.ts --rubric-regrade <trial-path>\n' +
' npx tsx scripts/copy-reference-run.ts --supersede <old-run-dir> <trial-path>'
);
process.exit(1);
}
@@ -387,7 +541,21 @@ if (destName && args.length !== 1) {
console.error('Error: --dest-name applies to exactly one trial path');
process.exit(1);
}
if (supersede && args.length !== 1) {
console.error('Error: --supersede applies to exactly one trial path');
process.exit(1);
}
if (supersede && rubricRegrade) {
console.error('Error: --rubric-regrade adds a grade beside a run; there is nothing to supersede');
process.exit(1);
}
if (destName && rubricRegrade) {
console.error(
'Error: --rubric-regrade takes its name from the trial, so --dest-name cannot apply'
);
process.exit(1);
}
for (const trialPath of args) {
copyTrial(trialPath, destName);
copyTrial(trialPath, { destName, rubricRegrade, supersede });
}

View File

@@ -18,6 +18,9 @@
# harbor-tasks/<slug> \
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg
#
# # Every captured run for a task, 2 at a time (after a rubric edit)
# scripts/harbor-regrade harbor-tasks/<slug> --all
#
# # Ten regrades of the same reference run (independent grader trials)
# scripts/harbor-regrade \
# harbor-tasks/<slug> \
@@ -28,6 +31,11 @@
# token rate). The replay agent runs no model, so the grader is the only model in
# this path. Serving speed and cost change; the grade itself is not steered.
#
# A finished regrade files itself when nothing is filed for that run yet — which is
# only ever the atomic case, since a run always carries a holistic grade already.
# Otherwise it grades and leaves the result in its job dir, so the tune-and-diff loop
# is untouched; --replace adopts it, superseding what was there.
#
# See scripts/replay_agent.py for what the agent actually does, and the
# `verifier: capture tracked-file deletions in agent-output` PR for the
# capture half of this flow (_HARBOR_DELETIONS.txt).
@@ -37,6 +45,11 @@ set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
# A sample count from the caller wins over .env, so a prefix can override a standing
# setting for one regrade. Captured as one value so a caller's spelling beats .env's,
# whichever each used (test.sh reads GRADER_SAMPLES; this script has long taken both).
_CALLER_SAMPLES="${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}"
# Source API key + any verifier env from the repo's .env
if [ -f "$REPO_ROOT/.env" ]; then
set -a
@@ -64,16 +77,28 @@ fi
usage() {
cat >&2 <<EOF
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir>... [extra harbor args]
scripts/harbor-regrade <task-dir> --all [extra harbor args]
Required arguments:
<task-dir> harbor-tasks/<slug> — same dir you'd pass to scripts/harbor-run.
<reference-run-dir> harbor-tasks/<slug>/reference-runs/<run-id> — must contain
agent-output/ (and ideally agent/trajectory.json).
agent-output/ (and ideally agent/trajectory.json). Name several
to re-grade them all; each gets its own harbor job.
Optional arguments:
--all re-grade every run under <task-dir>/reference-runs/ — the usual
thing to do after editing the rubric.
--jobs N how many runs to re-grade at once (default 2). Each one is a
container, so raise it only as far as your machine allows.
--fast grade in claude's fast serving mode (higher token rate,
faster output). Anything else is passed through to harbor.
--replace adopt the result, replacing the grade already filed for this
run. Without it, a regrade that would overwrite an existing
grade is left in its job dir for you to compare first.
Note: -k re-grades the SAME run N times (N independent grades of one trajectory).
To re-grade DIFFERENT runs, name them all, or use --all.
EOF
exit 1
}
@@ -81,19 +106,48 @@ EOF
# `--fast` is consumed here (harbor has no such flag); everything else stays an opaque
# passthrough. Stripped first so it may appear anywhere, not just after the positionals.
FAST_REQUESTED=""
ALL_RUNS=""
REPLACE=""
JOBS=2
REGRADE_ARGS=()
while [ $# -gt 0 ]; do
case "$1" in
--fast) FAST_REQUESTED=1; shift ;;
--replace) REPLACE=1; shift ;;
--all) ALL_RUNS=1; shift ;;
--jobs)
[ $# -ge 2 ] || { echo "Error: --jobs needs a number." >&2; exit 1; }
JOBS="$2"; shift 2 ;;
--jobs=*) JOBS="${1#--jobs=}"; shift ;;
*) REGRADE_ARGS+=("$1"); shift ;;
esac
done
set -- "${REGRADE_ARGS[@]+"${REGRADE_ARGS[@]}"}"
[ $# -lt 2 ] && usage
# Normalised to base 10 before any (( )) sees it: "09" passes a -ge test but is
# invalid octal in arithmetic, which spun the dispatch loop forever.
case "$JOBS" in
'' | *[!0-9]*) JOBS_OK="" ;;
*) JOBS=$((10#$JOBS)); [ "$JOBS" -ge 1 ] && JOBS_OK=1 || JOBS_OK="" ;;
esac
if [ -z "$JOBS_OK" ]; then
echo "Error: --jobs must be a positive integer (got '$JOBS')." >&2
exit 1
fi
[ $# -lt 1 ] && usage
TASK_DIR="$1"
REF_RUN_DIR="$2"
shift 2
shift
# Leading non-flag positionals are reference runs; collection stops at the first
# harbor flag so `<task> <ref> -k 4` keeps working and `4` is never read as a run.
REF_DIRS=()
while [ $# -gt 0 ]; do
case "$1" in
-*) break ;;
*) REF_DIRS+=("$1"); shift ;;
esac
done
# Resolve to absolute paths — harbor cd's around internally; the replay
# agent receives the path as an --agent-kwarg and won't know our cwd.
@@ -101,6 +155,102 @@ TASK_DIR_ABS=$(cd "$TASK_DIR" 2>/dev/null && pwd) || {
echo "Error: task-dir does not exist: $TASK_DIR" >&2
exit 1
}
# Unfiltered on purpose: the child, not a name test here, decides what can be replayed.
if [ -n "$ALL_RUNS" ]; then
[ ${#REF_DIRS[@]} -gt 0 ] && {
echo "Error: pass --all or explicit reference-run dirs, not both." >&2
exit 1
}
REF_PARENT="$TASK_DIR_ABS/reference-runs"
[ -d "$REF_PARENT" ] || {
echo "Error: --all needs $REF_PARENT, which does not exist." >&2
exit 1
}
for _cand in "$REF_PARENT"/*/; do
[ -d "$_cand" ] && REF_DIRS+=("${_cand%/}")
done
[ ${#REF_DIRS[@]} -eq 0 ] && {
echo "Error: $REF_PARENT holds no reference runs." >&2
exit 1
}
fi
[ ${#REF_DIRS[@]} -eq 0 ] && usage
# Several runs: re-run ITSELF once each, so every child does the full preflight in an
# output dir of its own — harbor names job dirs by the second, and sharing one corrupts.
if [ ${#REF_DIRS[@]} -gt 1 ]; then
OUT_BASE="${HARBOR_REGRADE_OUT:-harbor-jobs}"
mkdir -p "$OUT_BASE"
CHILD_FLAGS=()
[ -n "$FAST_REQUESTED" ] && CHILD_FLAGS+=(--fast)
[ -n "$REPLACE" ] && CHILD_FLAGS+=(--replace)
echo "Re-grading ${#REF_DIRS[@]} reference runs, $JOBS at a time."
FAILED=()
FAILED_LOGS=()
BATCH_PIDS=()
# Without this a killed driver leaves its children running, and on a cloud backend
# each one is a sandbox that bills until something else reaps it.
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 143' TERM
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 130' INT
IDX=0
TOTAL=${#REF_DIRS[@]}
while [ "$IDX" -lt "$TOTAL" ]; do
BATCH_PIDS=()
BATCH_REFS=()
BATCH_LOGS=()
for ((_j = 0; _j < JOBS && IDX < TOTAL; _j++)); do
REF="${REF_DIRS[$IDX]}"
RUN_ID=$(basename "$REF")
# --job-name, not a nested output dir: children keep harbor's own
# harbor-jobs/<job>/<trial> shape. Indexed so duplicate args cannot collide.
JOB_NAME="regrade-$((IDX + 1))-$RUN_ID"
LOG="$OUT_BASE/$JOB_NAME.log"
echo " starting $RUN_ID (log: $LOG)"
"$SCRIPT_DIR/harbor-regrade" "$TASK_DIR_ABS" "$REF" \
${CHILD_FLAGS[@]+"${CHILD_FLAGS[@]}"} "$@" \
--job-name "$JOB_NAME" > "$LOG" 2>&1 &
BATCH_PIDS+=($!)
BATCH_REFS+=("$RUN_ID")
BATCH_LOGS+=("$LOG")
IDX=$((IDX + 1))
done
for ((_i = 0; _i < ${#BATCH_PIDS[@]}; _i++)); do
if wait "${BATCH_PIDS[$_i]}"; then
echo " ok ${BATCH_REFS[$_i]}"
else
echo " FAILED ${BATCH_REFS[$_i]}"
FAILED+=("${BATCH_REFS[$_i]}")
FAILED_LOGS+=("${BATCH_LOGS[$_i]}")
fi
done
done
if [ ${#FAILED[@]} -gt 0 ]; then
echo "${#FAILED[@]} of $TOTAL failed. Their output, in full — re-running is safe:" >&2
for _f in "${FAILED_LOGS[@]}"; do echo " $_f" >&2; done
exit 1
fi
echo "All $TOTAL re-graded."
# Children file their own grades into per-run logs we do not echo, so count the
# results rather than claiming them. Only the atomic destination is countable:
# a superseded run is gone, so there is no before/after to compare against.
case "${HARBOR_GRADER_MODE:-}" in
rubric-*)
FILED=0
for _r in "${REF_DIRS[@]}"; do
[ -d "$TASK_DIR_ABS/rubric-regrades/$(basename "$_r")" ] && FILED=$((FILED + 1))
done
echo "Filed $FILED of $TOTAL into $TASK_DIR_ABS/rubric-regrades/."
if [ "$FILED" -lt "$TOTAL" ]; then
echo "The rest are still in their job dirs; their logs above say why." >&2
fi
;;
esac
exit 0
fi
REF_RUN_DIR="${REF_DIRS[0]}"
REF_RUN_DIR_ABS=$(cd "$REF_RUN_DIR" 2>/dev/null && pwd) || {
echo "Error: reference-run-dir does not exist: $REF_RUN_DIR" >&2
exit 1
@@ -114,7 +264,10 @@ if [ -d "$TASK_DIR_ABS/environment" ]; then
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
if [ -f "$DNSJAIL_SRC" ]; then
cp "$DNSJAIL_SRC" "$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
# Rename into place: children share this task dir, and a half-written
# script is one a sibling's image build can pick up.
_dnsjail_dst="$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
cp "$DNSJAIL_SRC" "$_dnsjail_dst.tmp.$$" && mv -f "$_dnsjail_dst.tmp.$$" "$_dnsjail_dst"
break
fi
done
@@ -135,7 +288,8 @@ case "${HARBOR_GRADER_MODE:-}" in
[ -f "$RUBRIC_RENDER_SRC" ] || continue
if [ ! -f "$RUBRIC_RENDER_DEST" ] || ! cmp -s "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"; then
mkdir -p "$TASK_DIR_ABS/tests"
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST.tmp.$$" &&
mv -f "$RUBRIC_RENDER_DEST.tmp.$$" "$RUBRIC_RENDER_DEST"
echo "Note: staged the current shared render-rubric-grade.py into $TASK_DIR_ABS/tests (rubric grader mode)." >&2
fi
break
@@ -206,13 +360,16 @@ GRADER_MODE_FLAG=()
# inside the unchanged agentic harness. See raccoon-shared/test.sh GRADER_MODEL.
[ -n "${HARBOR_GRADER_MODEL:-}" ] && GRADER_MODE_FLAG+=(--verifier-env "GRADER_MODEL=$HARBOR_GRADER_MODEL")
# Optional sample count: HARBOR_GRADER_SAMPLES=1 overrides the default 3 samples.
# For measuring per-sample properties of the grader (e.g. how often it emits a
# schema-valid grade.json), 1 sample across N distinct trajectories is a better
# estimator than 3 samples across N/3 — it decorrelates per-task effects for the
# same token spend. See raccoon-shared/test.sh GRADER_SAMPLES.
[ -n "${HARBOR_GRADER_SAMPLES:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$HARBOR_GRADER_SAMPLES")
# Optional sample count, under either name. Unset, the task's own frozen tests/test.sh
# decides: tasks created before Sep 2026 average 3 samples, newer ones grade once.
_SAMPLES="${_CALLER_SAMPLES:-${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}}"
if [ -n "$_SAMPLES" ]; then
if ! [ "$_SAMPLES" -ge 1 ] 2>/dev/null; then
echo "harbor-regrade: ERROR — GRADER_SAMPLES must be a positive integer (got '$_SAMPLES')." >&2
exit 1
fi
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$_SAMPLES")
fi
# Optional API timeout: HARBOR_API_TIMEOUT_MS=<ms> raises the per-request API
# timeout of the grader's claude calls via verifier env (API_TIMEOUT_MS is the
@@ -283,9 +440,83 @@ fi
[ -n "${HARBOR_GRADING_STANDARD:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "GRADING_STANDARD=$HARBOR_GRADING_STANDARD")
# A regrade is all verifier, so every call it makes is the grader's.
if [ -f "$REPO_ROOT/scripts/lib/call-origin.sh" ]; then
. "$REPO_ROOT/scripts/lib/call-origin.sh"
_GRADER_METADATA="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
# `if`, not `[ ... ] && ...`: an AND-list is this block's last statement, so under
# `set -e` an empty value would abort the regrade rather than just skip the header.
if [ -n "$_GRADER_METADATA" ]; then
GRADER_MODE_FLAG+=(
--verifier-env
"ANTHROPIC_CUSTOM_HEADERS=$CALL_ORIGIN_HEADER: $_GRADER_METADATA"
)
fi
fi
# Where this grade would be filed, and whether filing it destroys anything. An atomic
# regrade lands beside the run in rubric-regrades/<run>; a holistic one replaces the
# run's own grade, so its destination is always occupied and never files unasked.
rel() { case "$1" in "$PWD"/*) printf '%s' "${1#"$PWD"/}" ;; *) printf '%s' "$1" ;; esac; }
RUN_ID=$(basename "$REF_RUN_DIR_ABS")
RUBRIC_MODE=""
case "${HARBOR_GRADER_MODE:-}" in rubric-*) RUBRIC_MODE=1 ;; esac
if [ -n "$RUBRIC_MODE" ]; then
FILE_DEST="$TASK_DIR_ABS/rubric-regrades/$RUN_ID"
else
FILE_DEST="$REF_RUN_DIR_ABS"
fi
AUTOFILE=""
if [ ! -e "$FILE_DEST" ] || [ -n "$REPLACE" ]; then
AUTOFILE=1
fi
if [ -z "$AUTOFILE" ]; then
echo "Note: $(rel "$FILE_DEST")" >&2
echo " already holds a grade, so this regrade will not be filed." >&2
echo " Where it landed is printed when it finishes." >&2
fi
# Where the trial will land. Harbor's own job name is a second-granularity timestamp,
# so naming it here is what makes the trial findable afterwards (the fan-out passes one).
CALLER_JOB_NAME=""
CALLER_OUT=""
_prev=""
for _a in "$@"; do
case "$_prev" in
--job-name) CALLER_JOB_NAME="$_a" ;;
-o | --output-dir) CALLER_OUT="$_a" ;;
esac
case "$_a" in
--job-name=*) CALLER_JOB_NAME="${_a#--job-name=}" ;;
--output-dir=*) CALLER_OUT="${_a#--output-dir=}" ;;
esac
_prev="$_a"
done
JOB_NAME="$CALLER_JOB_NAME"
JOB_NAME_FLAG=()
if [ -n "$AUTOFILE" ] && [ -n "$CALLER_OUT" ]; then
echo "Note: -o/--output-dir passed, so this regrade will not be filed into" >&2
echo " $(rel "$FILE_DEST") — copy it yourself when it finishes." >&2
AUTOFILE=""
fi
if [ -z "$JOB_NAME" ] && [ -z "$CALLER_OUT" ]; then
# $$ as well as the epoch: harbor refuses an existing job dir outright, and two
# regrades of one run can start in the same second.
JOB_NAME="regrade-$RUN_ID-$(date +%s)-$$"
JOB_NAME_FLAG=(--job-name "$JOB_NAME")
fi
# RACCOON_DNS_JAIL is inert here on purpose: replay_agent runs no agent turn, so there is
# nothing to restrict — only the verifier runs, and it needs the proxy.
exec harbor run \
#
# harbor runs as a child, not under `exec`, so the filing below runs after it exits.
# TERM/INT are forwarded so a kill here never orphans a job (on cloud, a billing sandbox).
HARBOR_EXIT=0
HARBOR_SIGNALLED=""
harbor run \
-p "$TASK_DIR_ABS" \
--agent-import-path replay_agent:ReplayAgent \
--ak "reference_run_dir=$REF_RUN_DIR_ABS" \
@@ -296,4 +527,74 @@ exec harbor run \
--yes \
-o "$OUT_DIR" \
${GRADER_MODE_FLAG[@]+"${GRADER_MODE_FLAG[@]}"} \
"$@"
${JOB_NAME_FLAG[@]+"${JOB_NAME_FLAG[@]}"} \
"$@" &
HARBOR_PID=$!
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT
wait "$HARBOR_PID" || HARBOR_EXIT=$?
if [ -n "$HARBOR_SIGNALLED" ]; then
# The first wait was interrupted by the trap; wait again for harbor's real status.
wait "$HARBOR_PID" || HARBOR_EXIT=$?
fi
trap - TERM INT
# A failed grade has nothing to report on, and a caller-directed -o put the trial
# somewhere this script cannot name.
if [ "$HARBOR_EXIT" -ne 0 ] || [ -n "$CALLER_OUT" ]; then
exit "$HARBOR_EXIT"
fi
JOB_DIR="$OUT_DIR/$JOB_NAME"
TRIALS=()
for _t in "$JOB_DIR"/*__*/; do
[ -d "$_t" ] && TRIALS+=("${_t%/}")
done
# Atomic grades sit beside the run; a holistic one replaces it, which means removing
# the directory it supersedes (see copy-reference-run.ts for why swap, not merge).
COPY_FLAGS=(--rubric-regrade)
[ -z "$RUBRIC_MODE" ] && COPY_FLAGS=(--supersede "$(rel "$REF_RUN_DIR_ABS")")
FILE_CMD="npx tsx $(rel "$SCRIPT_DIR")/copy-reference-run.ts ${COPY_FLAGS[*]}"
if [ ${#TRIALS[@]} -eq 0 ]; then
echo "Note: no trial directory under $JOB_DIR, so there is no grade." >&2
exit "$HARBOR_EXIT"
fi
if [ ${#TRIALS[@]} -gt 1 ]; then
echo "Note: ${#TRIALS[@]} trials under $JOB_DIR (-k grades one run repeatedly)." >&2
echo " Pick the one to keep and file it: $FILE_CMD <trial-dir>" >&2
exit "$HARBOR_EXIT"
fi
# Not filing: say where the grade is and how the two compare, since comparing them is
# the whole reason it was left alone. Adopting is a copy — re-running with --replace
# would spend the 15-30 minutes again for a grade already sitting on disk.
if [ -z "$AUTOFILE" ]; then
NEW_REWARD=$(cat "${TRIALS[0]}/verifier/reward.txt" 2>/dev/null || echo "?")
# A stored atomic grade is a whole trial, so its reward sits under verifier/;
# a reference run keeps its own at the top.
OLD_REWARD=$(cat "$FILE_DEST/verifier/reward.txt" 2>/dev/null ||
cat "$FILE_DEST/reward.txt" 2>/dev/null || echo "?")
echo "" >&2
echo "Graded, not filed — that run already holds a grade." >&2
echo " new $NEW_REWARD $(rel "${TRIALS[0]}")" >&2
echo " filed $OLD_REWARD $(rel "$FILE_DEST")" >&2
echo "" >&2
echo "Adopt this grade:" >&2
echo " npx tsx $(rel "$SCRIPT_DIR")/copy-reference-run.ts \\" >&2
echo " ${COPY_FLAGS[*]} \\" >&2
echo " $(rel "${TRIALS[0]}")" >&2
echo "" >&2
echo "Or pass --replace next time to file it without this step." >&2
exit "$HARBOR_EXIT"
fi
# Best-effort from here: the grade already cost 15-30 minutes, so a filing problem
# prints the command to finish by hand rather than failing the regrade.
if ! npx tsx "$SCRIPT_DIR/copy-reference-run.ts" "${COPY_FLAGS[@]}" "${TRIALS[0]}" >&2; then
echo "Warning: the grade is in ${TRIALS[0]} but could not be filed. Retry with:" >&2
echo " $FILE_CMD ${TRIALS[0]}" >&2
fi
exit "$HARBOR_EXIT"

View File

@@ -16,6 +16,9 @@
# token rate): the trial agent (claude-code only) and the grader the verifier launches.
# Serving speed and cost change; the grade itself is not steered.
#
# GRADER_SAMPLES=N (a prefix, or a line in .env) sets how many times the grader scores the
# run and averages. A task's own tests/test.sh supplies the default when it is unset.
#
# Needs python 3.11+ on PATH (tomllib, to read the harness registry). Containers have it;
# a macOS host running HARBOR_ENV=docker may not — set RACCOON_PYTHON if so.
@@ -29,6 +32,11 @@ REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
_RJ_SET="${RACCOON_DNS_JAIL+set}"; _RJ_VAL="${RACCOON_DNS_JAIL:-}"
_RJA_SET="${RACCOON_DNS_JAIL_ALLOW+set}"; _RJA_VAL="${RACCOON_DNS_JAIL_ALLOW:-}"
# Same for the grader sample count, under either name (HARBOR_GRADER_SAMPLES is what
# harbor-regrade calls it; GRADER_SAMPLES is what the task's tests/test.sh reads).
# Captured as one value so a caller's spelling beats .env's, whichever each used.
_CALLER_SAMPLES="${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}"
# Source API key
if [ -f "$REPO_ROOT/.env" ]; then
set -a
@@ -156,7 +164,10 @@ if [ -f "$TASK_DIR/task.toml" ] &&
BROWSER_OPTIN=1
fi
if [ -d "$TASK_DIR/environment" ]; then
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
# Rename into place: a concurrent run against this task dir must never read the
# instant between truncate and write.
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin.tmp.$$" &&
mv -f "$TASK_DIR/environment/browser-optin.tmp.$$" "$TASK_DIR/environment/browser-optin"
fi
# Preflight: restage the DNS jail script, for the same reason as the marker above.
@@ -171,7 +182,8 @@ if [ -d "$TASK_DIR/environment" ]; then
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
if [ -f "$DNSJAIL_SRC" ]; then
cp "$DNSJAIL_SRC" "$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
_dnsjail_dst="$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
cp "$DNSJAIL_SRC" "$_dnsjail_dst.tmp.$$" && mv -f "$_dnsjail_dst.tmp.$$" "$_dnsjail_dst"
break
fi
done
@@ -255,6 +267,35 @@ FAST_FLAGS=""
GRADER_FAST_FLAGS=""
[ -n "$FAST_REQUESTED" ] && GRADER_FAST_FLAGS="--verifier-env GRADER_FAST_MODE=true"
# A task's tests/test.sh is frozen at creation and defaults its own sample count, so
# forwarding the var is the only way to change one that already exists.
GRADER_SAMPLES_FLAGS=""
_SAMPLES="${_CALLER_SAMPLES:-${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}}"
if [ -n "$_SAMPLES" ]; then
if ! [ "$_SAMPLES" -ge 1 ] 2>/dev/null; then
echo "harbor-run: ERROR — GRADER_SAMPLES must be a positive integer (got '$_SAMPLES')." >&2
exit 1
fi
GRADER_SAMPLES_FLAGS="--verifier-env GRADER_SAMPLES=$_SAMPLES"
fi
# The verifier launches its own grader claude, so it names its own call origin
# rather than inheriting the surface that launched harbor. An array because the
# header value contains a space.
GRADER_ORIGIN_FLAGS=()
if [ -f "$REPO_ROOT/scripts/lib/call-origin.sh" ]; then
. "$REPO_ROOT/scripts/lib/call-origin.sh"
_GRADER_METADATA="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
# `if`, not `[ ... ] && ...`: an AND-list is this block's last statement, so under
# `set -e` an empty value would abort the run rather than just skip the header.
if [ -n "$_GRADER_METADATA" ]; then
GRADER_ORIGIN_FLAGS=(
--verifier-env
"ANTHROPIC_CUSTOM_HEADERS=$CALL_ORIGIN_HEADER: $_GRADER_METADATA"
)
fi
fi
# Environment backend. An explicit HARBOR_ENV always wins (either direction).
# Otherwise the default is context-dependent:
# - daytona for the internal repo: runs the trial in a cloud sandbox over
@@ -429,6 +470,8 @@ harbor run \
$EFFORT_FLAGS \
$FAST_FLAGS \
$GRADER_FAST_FLAGS \
$GRADER_SAMPLES_FLAGS \
${GRADER_ORIGIN_FLAGS[@]+"${GRADER_ORIGIN_FLAGS[@]}"} \
"$@" &
HARBOR_PID=$!
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM

View File

@@ -82,8 +82,23 @@ multi_agent_v2 = false
memories = false
external_agent_memory_import = false
"""
# A provider of our own, not the built-in `openai`: codex reserves built-in provider ids,
# and env_http_headers — the only place codex can be told to send the call-origin header —
# is a per-provider setting. base_url has to live in the table with it (a provider without
# one silently falls back to api.openai.com), so harness_refresh_config_keys refreshes
# [model_providers.*] keys as well as root ones.
container_config = """
openai_base_url = "${OPENAI_BASE_URL}"
model_provider = "llm-proxy"
[model_providers.llm-proxy]
name = "LLM proxy"
base_url = "${OPENAI_BASE_URL}"
# A custom provider reads its key from this env var and never from auth.json, so it
# names the one key .env actually carries. That makes the key live per launch rather
# than baked at container create — better than the auth-file path it replaces.
env_key = "ANTHROPIC_API_KEY"
wire_api = "responses"
env_http_headers = { "X-Surge-Client-Metadata" = "LLM_CALL_METADATA" }
"""
explore_config = """
[hooks]

View File

@@ -0,0 +1,88 @@
/** The task's holistic-rubric template. Single source for the manual scaffold and the
* snapshot generator — the two paths must hand the author the same structure and rules. */
export const HOLISTIC_RUBRIC_SCAFFOLD = `# Holistic Rubric — <task-slug>
The shared grading standard (\`task-shared/grading-standard.md\`, embedded in
\`tests/grader-system-prompt-consolidated.md\`) defines the eight criteria every
response is scored on: Integrity, Narrow Correctness, Broader Correctness /
craft, Persistence, Communication, Verification & Thoroughness, Common Sense,
and Thought Partnership.
This file is the task's holistic rubric. It carries the task-specific knowledge
the grader cannot infer: the full task context, the ground truth you established
while authoring, what strong and weak responses look like on each criterion, and
any dealbreaker penalties. This document must stand alone. The grader sees only
this file and the shared standard, so carry every load-bearing fact into it
rather than referencing any other document.
Replace each bracketed section. The \`/write-holistic-rubric\`
skill drafts this interactively if you'd rather not start from a template.
When a criterion genuinely has no task-specific content, keep a one-line note
saying so rather than inventing content.
## Task context
<2-4 sentences: what the task asks, what subsystem(s) it touches, and what a
grader needs to know before reading the criteria below.>
## Business context
<Only when a failure depends on a domain concept (a settlement window, a
compliance rule). Delete this section otherwise.>
## Ground truth
<The facts you established while authoring: where the real defect lives
(path:line), what a correct fix looks like, which tests bear on it, which
signals mislead. The grader trusts this section over its own reading.>
## Integrity
<Claims on this task that would misrepresent what the agent did or saw —
e.g. asserting a file says X after reading it say Y. Charge only on an
observable basis.>
## Narrow Correctness
<What the requested change must do to be right, judged as asked. Anchors a
working result must satisfy, checkable by path:line.>
## Broader Correctness / the craft of software engineering
<Craft expectations specific to this codebase: patterns to follow, tests to
add, places a shortcut would rot.>
## Persistence
<What "kept going appropriately" looks like here: the dead ends worth
exhausting, and where stopping to ask is the better call.>
## Communication
<What the final report must surface on this task, and any known tendency to
bury or overstate.>
## Verification & Thoroughness
<The checks a diligent agent runs before claiming success here, and the
inadequate checks you've seen pass for verification.>
## Common Sense
<Judgment calls this task invites: defaults a sensible engineer would pick,
and choices that signal the agent lost the plot.>
## Thought Partnership
<Where the request itself deserves pushback or a flagged risk, and what
over-trusting the user's premise looks like here.>
## Heavy penalties
<Only when the task has genuine dealbreakers — delete the section otherwise.
Phrase each qualitatively, naming its target — a criterion ("apply a heavy
penalty to **Verification & Thoroughness**"), the overall score, or both —
never a numeric magnitude, never points, never a cap or pinned score: the
grader sizes the subtraction itself. Always state the behavior that does NOT trip the penalty.
Never describe how criteria combine into an overall score.>
`;

View File

@@ -0,0 +1,34 @@
# shellcheck shell=bash
# call-origin.sh — build the X-Surge-Client-Metadata header value.
#
# Which surface a proxy call came from (a trial agent, the grader, Explore, a
# dev box). Separate from llm-proxy-env.sh, which carries the project id and the
# proxy routes: those are platform-internal, this is not, so this file is the
# half that ships in the worker toolkit — worker runs go through the same proxy
# and are attributed the same way.
#
# . scripts/lib/call-origin.sh
# meta="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
#
# The proxy rejects the WHOLE CALL over a malformed metadata header (400
# invalid_client_metadata), so an origin that is not a plain slug yields an
# empty string and the caller sends no header at all: losing attribution beats
# failing the call.
CALL_ORIGIN_HEADER="X-Surge-Client-Metadata"
# An unlabelled call is still a real call, so it gets a bucket rather than no
# header: a missing origin in the audit log then means an unplumbed surface.
DEFAULT_CALL_ORIGIN="local"
# Compact JSON for the header, or empty when LLM_CALL_ORIGIN is unusable.
# Only a slug matching this pattern is ever interpolated, so nothing needs
# JSON-escaping and this stays dependency-free (it is sourced in worker
# containers, which have no python).
call_origin_metadata() {
local origin="${LLM_CALL_ORIGIN:-$DEFAULT_CALL_ORIGIN}"
case "$origin" in
"" | *[!a-z0-9._-]* | [!a-z0-9]*) return 0 ;;
esac
[ "${#origin}" -le 64 ] || return 0
printf '{"origin":"%s"}' "$origin"
}

View File

@@ -190,16 +190,39 @@ if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1
raise SystemExit(1)
text = os.path.expandvars(text)
# Root keys, plus keys inside a [model_providers.*] table: codex reserves its built-in
# provider ids, so the proxy URL it must follow lives in a provider table, not at the
# root. Every other table, [hooks] on the explore surface included, is left alone.
REFRESHABLE_TABLE = re.compile(r"\[model_providers\.[^]]+\]$")
wanted = []
section = None
for line in text.splitlines():
if line.lstrip().startswith("["):
break
m = re.match(r"\s*([A-Za-z0-9_-]+)\s*=", line)
stripped = line.strip()
if stripped.startswith("["):
section = stripped if REFRESHABLE_TABLE.match(stripped) else False
continue
if section is False:
continue
m = re.match(r"\s*\"?([A-Za-z0-9_.-]+)\"?\s*=", line)
if m:
wanted.append((m.group(1), line.rstrip()))
wanted.append((section, m.group(1), line.rstrip()))
if not wanted:
raise SystemExit(0)
def section_path(header):
"""[model_providers.llm-proxy] -> ("model_providers", "llm-proxy")."""
return tuple(header.strip("[]").split("."))
def lookup(doc, header, key):
"""The value a parsed config holds for a wanted key, or KeyError."""
node = doc
if header:
for part in section_path(header):
node = node[part]
return node[key]
mode = None
if os.path.exists(target):
try:
@@ -208,23 +231,52 @@ if os.path.exists(target):
mode = os.stat(target).st_mode & 0o777
except OSError:
raise SystemExit(1)
# Everything from the first table header on belongs to a table. A key appended after
# one is reparented into it, so both the search and the insert stay above the line.
root_end = next((i for i, l in enumerate(lines) if l.lstrip().startswith("[")), len(lines))
changed = False
for key, line in wanted:
# The quoted spelling is the same key: replacing it beats adding a duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(root_end) if pat.match(lines[i])), None)
def span(header):
"""The line range a section owns, or None when the file has no such section.
Root is everything above the first table header: a key appended below one
would be reparented into it, so searches and inserts stay inside the span.
"""
heads = [i for i, l in enumerate(lines) if l.lstrip().startswith("[")]
if header is None:
return 0, (heads[0] if heads else len(lines))
at = next((i for i in heads if lines[i].strip() == header), None)
if at is None:
if root_end < len(lines) and lines[root_end].strip():
lines.insert(root_end, "")
lines.insert(root_end, line)
root_end += 1
changed = True
elif lines[at] != line:
lines[at] = line
return None
after = next((i for i in heads if i > at), len(lines))
return at + 1, after
# Grouped, root first, so a section this file lacks can be written whole.
grouped = {}
for header, key, line in wanted:
grouped.setdefault(header, []).append((key, line))
ordered = sorted(grouped, key=lambda h: (h is not None, h or ""))
changed = False
for header in ordered:
if span(header) is None:
# A config written before this section existed. Write the whole table
# rather than leave a root key naming a provider that is not there.
if lines and lines[-1].strip():
lines.append("")
lines.append(header)
lines.extend(line for _, line in grouped[header])
changed = True
continue
for key, line in grouped[header]:
# Re-read the span: an insert for an earlier key moved it.
start, end = span(header)
# The quoted spelling is the same key: replace rather than duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(start, end) if pat.match(lines[i])), None)
if at is None:
if end < len(lines) and lines[end].strip():
lines.insert(end, "")
lines.insert(end, line)
changed = True
elif lines[at] != line:
lines[at] = line
changed = True
if not changed:
raise SystemExit(0)
out = "\n".join(lines).rstrip("\n") + "\n"
@@ -238,9 +290,22 @@ try:
except tomllib.TOMLDecodeError:
raise SystemExit(1)
# Parsing is not enough: a line edit can land inside a multi-line value, which still
# parses while leaving the key unset. Require every key to have reached the root.
if doc != {**doc, **tomllib.loads("\n".join(line for _, line in wanted))}:
raise SystemExit(1)
# parses while leaving the key unset. Require every key to have landed on the value the
# blob asks for, in its own section — skipping sections this file does not carry.
blob_doc = tomllib.loads(text)
for header, key, _ in wanted:
try:
expected = lookup(blob_doc, header, key)
except (KeyError, TypeError):
raise SystemExit(1)
try:
got = lookup(doc, header, key)
except (KeyError, TypeError):
if header is None:
raise SystemExit(1)
continue
if got != expected:
raise SystemExit(1)
# Pid-suffixed: two launches at once must not write the same scratch path.
tmp = target + ".raccoon-tmp." + str(os.getpid())

View File

@@ -193,21 +193,48 @@ export const REFERENCE_RUN_INPUTS = Object.freeze([
* holistic-rubric files the task directory carries (tests/holistic-rubric.md
* on current tasks, tests/grader-guidance-consolidated.md on tasks created
* before the rename, plus the legacy-era plain-named file when a task
* authored on an earlier generation carries one), plus the atomic-rubric
* files the rubric detectors assess (tests/atomic-rubric.yaml, the pre-rename
* tests/rubrics.yaml, and tests/grader-context.md). An absent file hashes to
* authored on an earlier generation carries one). An absent file hashes to
* null on both sides and never diffs. Compared by content.
*
* The atomic-rubric files are deliberately NOT in this set: fifteen of the
* seventeen detectors never open them, so writing an atomic rubric after
* running the detectors would stale every one of those reports over files
* they never read. The two that do read them use
* {@link RUBRIC_DETECTOR_REPORT_INPUTS}.
*/
export const DETECTOR_REPORT_INPUTS = Object.freeze([
'prompt',
'graderGuidance',
'graderGuidanceConsolidated',
'holisticRubric',
] as const satisfies readonly TaskInputName[]);
/**
* The inputs the two rubric detectors assess: {@link DETECTOR_REPORT_INPUTS}
* plus the atomic-rubric package (tests/atomic-rubric.yaml, the pre-rename
* tests/rubrics.yaml, and tests/grader-context.md), which they compare
* against the holistic rubric.
*/
export const RUBRIC_DETECTOR_REPORT_INPUTS = Object.freeze([
...DETECTOR_REPORT_INPUTS,
'atomicRubric',
'rubricsYaml',
'graderContext',
] as const satisfies readonly TaskInputName[]);
/** Detectors that read the atomic-rubric package, and so are staled by it. */
export const ATOMIC_RUBRIC_DETECTORS: readonly string[] = Object.freeze([
'detector-rubric-coverage',
'detector-rubric-form',
]);
/** The input set a named detector's report is judged against. */
export function detectorReportInputs(detectorName: string): readonly TaskInputName[] {
return ATOMIC_RUBRIC_DETECTORS.includes(detectorName)
? RUBRIC_DETECTOR_REPORT_INPUTS
: DETECTOR_REPORT_INPUTS;
}
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
function sha256File(filePath: string): string | null {
if (!existsSync(filePath)) return null;

View File

@@ -0,0 +1,96 @@
# shellcheck shell=bash
#
# resolve_pin — turn a task's pinned commit into a SHA that exists in the repo,
# translating through a commit map when history has been rewritten under it.
#
# A task pins a commit in task.toml. If that repo's history is later rewritten
# (to strip something that should never have shipped, say), every rewritten
# commit gets a new SHA and the pin stops resolving — including on machines we
# cannot reach, holding tasks we cannot edit. A commit map lets those pins keep
# working: `<old-sha> <new-sha>` per line, at task-shared/commit-maps/<member>.map,
# <member> being the task's `repo` key — a standalone toolkit checks its repo out
# at repo/, so the directory name is not the member name and cannot be the key.
#
# The map is only consulted when the pin does not resolve, so it carries only
# rewritten commits — an unchanged commit resolves on its own and its identity
# row could never be read.
#
# Usage (source, then call):
# . "$(dirname "$0")/lib/resolve-pin.sh"
# sha=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
#
# Writes the resolved SHA to stdout, notes on stderr. Returns non-zero if the
# pin cannot be resolved, having explained why.
RESOLVE_PIN_MAX_HOPS="${RESOLVE_PIN_MAX_HOPS:-25}"
_RESOLVE_PIN_ZERO='0000000000000000000000000000000000000000'
# Look one hop: echo the successor of $1 in map $2, or nothing. Fails if the
# prefix is ambiguous, which would otherwise pick an arbitrary commit.
_resolve_pin_hop() {
local from="$1" map="$2" hits
hits=$(awk -v p="$from" '
/^#/ || NF < 2 { next }
index($1, p) == 1 { print $2 }
' "$map" | sort -u)
[ -z "$hits" ] && return 1
if [ "$(printf '%s\n' "$hits" | wc -l | tr -d ' ')" -gt 1 ]; then
echo " pin $from is ambiguous in $(basename "$map") — use a longer SHA" >&2
return 2
fi
printf '%s\n' "$hits"
}
resolve_pin() {
local repo_dir="$1" commit="$2" map_dir="${3:-}" key="${4:-}" sha map cur hops next rc
# Present in the repo: nothing to translate.
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$commit^{commit}" 2>/dev/null); then
printf '%s\n' "$sha"
return 0
fi
map=""
if [ -n "$map_dir" ]; then
if [ -n "$key" ] && [ -f "$map_dir/$key.map" ]; then
map="$map_dir/$key.map"
elif [ -f "$map_dir/$(basename "$repo_dir").map" ]; then
map="$map_dir/$(basename "$repo_dir").map"
fi
fi
if [ -z "$map" ]; then
echo "Error: pinned commit $commit is not in $repo_dir, and no commit map is available." >&2
echo " The repo may be a shallow or partial copy — try a full clone." >&2
return 1
fi
# Follow the chain: a commit rewritten more than once maps forward a hop per
# rewrite, so keep going until the SHA exists or the trail ends.
cur="$commit"
hops=0
while [ "$hops" -lt "$RESOLVE_PIN_MAX_HOPS" ]; do
next=$(_resolve_pin_hop "$cur" "$map"); rc=$?
[ "$rc" -eq 2 ] && return 1
if [ "$rc" -ne 0 ]; then
echo "Error: pinned commit $commit is not in $repo_dir and is not in $(basename "$map")." >&2
echo " It predates the map, or came from a repo copy this toolkit was not built from." >&2
return 1
fi
if [ "$next" = "$_RESOLVE_PIN_ZERO" ]; then
echo "Error: pinned commit $commit was deleted by a history rewrite, not rewritten." >&2
echo " Re-pin this task to a commit that still exists." >&2
return 1
fi
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$next^{commit}" 2>/dev/null); then
echo " Pin $commit was rewritten; using $sha" >&2
printf '%s\n' "$sha"
return 0
fi
cur="$next"
hops=$((hops + 1))
done
echo "Error: pinned commit $commit did not settle after $RESOLVE_PIN_MAX_HOPS hops." >&2
echo " $(basename "$map") may contain a cycle." >&2
return 1
}

View File

@@ -34,4 +34,21 @@ _scripts_dir="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# No args is a valid call: refresh only, for a lifecycle hook.
[ "$#" -gt 0 ] || exit 0
# Outside the subshell, because these have to reach the exec'd command: codex now reads
# its key from $ANTHROPIC_API_KEY per request, and a non-login shell sourced neither
# .bashrc (the key, the call origin) nor the profile that puts the CLI on PATH.
# Failures stay swallowed — an unreadable .env must not stop the agent starting.
export PATH="$HOME/.local/bin:$PATH"
if [ -f "${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
set -a
# shellcheck disable=SC1090
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true
set +a
fi
if [ -f "$HOME/.raccoon-call-origin" ]; then
# shellcheck disable=SC1091
. "$HOME/.raccoon-call-origin" 2>/dev/null || true
fi
exec "$@"

View File

@@ -151,6 +151,19 @@ harness_install_launchers() {
#!/bin/bash
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
set -euo pipefail
# A harness that reads its key from \$ENV per request needs .env in its environment,
# and only an interactive shell sources .bashrc — which is also where PATH picks up
# ~/.local/bin, where the CLI itself lives. Both are set here so a launch works the
# same either way, with the key .env holds right now.
export PATH="\$HOME/.local/bin:\$PATH"
if [ -f "\${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
set -a
. "\${RACCOON_ENV_FILE:-/workspace/.env}"
set +a
fi
if [ -f "\$HOME/.raccoon-call-origin" ]; then
. "\$HOME/.raccoon-call-origin"
fi
if [ -f "$note_src" ]; then
RACCOON_TOOLSET_NOTE="\$(sed "s#/opt/agent-cli#$agent_cli_dir#g" "$note_src")"
else

View File

@@ -24,6 +24,7 @@ import { hideBin } from 'yargs/helpers';
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { HOLISTIC_RUBRIC_SCAFFOLD } from './holistic-rubric-scaffold';
import { copyTree } from './lib/copy-tree';
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
@@ -747,40 +748,23 @@ if (lastUserMessage) {
// --- Scaffold holistic-rubric.md ---
const holisticRubricMd = `<!--
HOLISTIC RUBRIC — the file trials grade against. Run
/write-holistic-rubric
to draft it interactively, or point Claude Code at this file,
session-full.jsonl, and task-shared/grading-standard.md.
Snapshot: ${basename(snapshotDir)}
Session: ${metadata.session_uuid}
Repo: ${metadata.remote_url}
Commit: ${metadata.commit}
## What happened in the snapshot conversation
What happened in the snapshot conversation
The worker was trying to: ${annotation.what_trying}
They hoped Claude would: ${annotation.what_hoping}
Instead, Claude: ${annotation.what_happened}
## What this file contains
The eight-criterion Grading Standard
(task-shared/grading-standard.md, embedded in
tests/grader-system-prompt-consolidated.md) defines Integrity, Narrow
Correctness, Broader Correctness / craft, Persistence, Communication,
Verification & Thoroughness, Common Sense, and Thought Partnership. This
file adds the task-specific knowledge the grader cannot infer: full task
context, the ground truth you established, what strong and weak responses
look like per criterion, and any dealbreaker penalties — stated as 0.0-1.0
fraction subtractions with a named criterion target, never points, never
caps. The document must stand alone: the grader sees only it and the
shared standard.
Draft this file with /write-holistic-rubric, or point your agent at it,
session-full.jsonl, and task-shared/grading-standard.md. Delete this comment
when you are done.
-->
<!-- Replace EVERYTHING in this file with the actual holistic rubric,
including the instructions above. -->
`;
${HOLISTIC_RUBRIC_SCAFFOLD}`;
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');

View File

@@ -43,6 +43,10 @@ _DISABLED_SUFFIX = ".snapshot-seeded-disabled"
# this beta header (the proxy's server-side alias for the suffixed id was
# dropped; the literal id now 400s and the claude CLI hangs retrying).
_CONTEXT_1M_BETA_HEADER = "anthropic-beta: context-1m-2025-08-07"
# Spelled out rather than imported from llm_proxy_env: this module ships to the
# worker toolkit, where a cross-import would fail at import time.
_CLIENT_METADATA_HEADER = "X-Surge-Client-Metadata"
_TRIAL_CALL_METADATA = '{"origin":"harbor-trial"}'
def _strip_1m_suffix(model: str) -> tuple[str, bool]:
@@ -68,6 +72,20 @@ def _with_1m_beta_header(existing: str | None) -> str:
"""Merge the 1M-context beta header into an ANTHROPIC_CUSTOM_HEADERS value."""
return _merge_custom_headers(existing, _CONTEXT_1M_BETA_HEADER)
def _with_trial_origin(existing: str | None) -> str:
"""Restate the call origin as this trial, replacing the launching surface's.
Replaces rather than appends: two values of one header name is not a
merge the proxy can read.
"""
kept = [
line
for line in (existing or "").split("\n")
if line.strip() and not line.startswith(f"{_CLIENT_METADATA_HEADER}:")
]
return "\n".join([*kept, f"{_CLIENT_METADATA_HEADER}: {_TRIAL_CALL_METADATA}"])
# Fast mode (claude's /fast), opted in per-trial via `--ak fast_mode=true`
# (harbor-run --fast). The --settings flag is the only headless opt-in, and it
# doubles as the availability override for a proxy base URL: claude probes
@@ -581,6 +599,8 @@ class SnapshotClaudeCode(PreinstalledClaudeCode):
os.environ.get("ANTHROPIC_CUSTOM_HEADERS") if on_proxy else None,
self._extra_env.get("ANTHROPIC_CUSTOM_HEADERS"),
)
if on_proxy:
header = _with_trial_origin(header)
if self.model_name:
# "[1m]" normalization happened once in PreinstalledClaudeCode.__init__
# (shared by all Claude agent classes); by here model_name is the plain

View File

@@ -16,10 +16,10 @@ import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import {
DETECTOR_REPORT_INPUTS,
INPUT_CHECKSUMS_FILENAME,
REFERENCE_RUN_INPUTS,
captureTaskInputs,
detectorReportInputs,
diffTaskInputs,
readTaskInputChecksums,
} from './lib/input-checksums';
@@ -581,7 +581,8 @@ export function validateTask(
unstamped.push(report);
continue;
}
const changed = diffTaskInputs(stamp, currentInputs, DETECTOR_REPORT_INPUTS);
const detectorName = report.replace(/\.md$/, '');
const changed = diffTaskInputs(stamp, currentInputs, detectorReportInputs(detectorName));
staleness.detectors.push({
report,
status: changed.length > 0 ? 'stale' : 'fresh',
@@ -593,14 +594,21 @@ export function validateTask(
if (changed.length > 0) staleStamped.push({ report, changed });
}
if (staleStamped.length > 0) {
const changedLabels = [...new Set(staleStamped.flatMap((s) => s.changed))];
// Grouped by cause, not unioned across reports: a report is only ever
// stale on the inputs its own detector reads.
const byCause = new Map<string, string[]>();
for (const { report, changed } of staleStamped) {
const cause = changed.join(' and ');
byCause.set(cause, [...(byCause.get(cause) ?? []), report]);
}
const causes = [...byCause].map(([cause, reports]) => `${cause} → ${reports.join(', ')}`);
const staleNames = staleStamped.map((s) => s.report);
logger.warn(
{ stale: staleNames, changed: changedLabels },
`You modified your ${changedLabels.join(' and ')} after these detector reports were written, so they assessed an older revision of this task: ${staleNames.join(', ')}. Re-run those detectors (e.g. /${staleNames[0].replace(/\.md$/, '')}) and re-package.`
{ stale: staleStamped },
`These detector reports assessed an older revision of this task — you modified an input each one reads after it was written: ${causes.join('; ')}. Re-run those detectors (e.g. /${staleNames[0].replace(/\.md$/, '')}) and re-package.`
);
warnings.push(
`${staleStamped.length} stale detector report(s) — ${changedLabels.join(', ')} changed since they were written: ${staleNames.join(', ')} — ` +
`${staleStamped.length} stale detector report(s) — ${causes.join('; ')} — ` +
're-run them against the current task'
);
}