ren worker folder adding orig, mv new one into root
This commit is contained in:
@@ -99,8 +99,11 @@ if [ "$BROWSER_OPTIN" = "1" ]; then
|
||||
fi
|
||||
fi
|
||||
|
||||
# Resolve commit
|
||||
RESOLVED_SHA=$(git -C "$REPO_DIR" rev-parse "$COMMIT")
|
||||
# Resolve commit. Goes through resolve_pin so a task pinned before a history
|
||||
# rewrite still builds: the pin is translated via task-shared/commit-maps/.
|
||||
# shellcheck source=lib/resolve-pin.sh
|
||||
. "$(dirname "$0")/lib/resolve-pin.sh"
|
||||
RESOLVED_SHA=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
|
||||
echo " Resolved SHA: $RESOLVED_SHA"
|
||||
|
||||
# --- Member-specific setup ---------------------------------------------------
|
||||
|
||||
@@ -110,7 +110,11 @@ done
|
||||
|
||||
# Resolve the repo's real git dir (handles submodules, whose .git is a file).
|
||||
GITDIR="$(git -C "$REPO_DIR" rev-parse --absolute-git-dir 2>/dev/null)" || skip "not a git repo: $REPO_DIR"
|
||||
RESOLVED_SHA="$(git --git-dir="$GITDIR" rev-parse --quiet --verify "$COMMIT^{commit}" 2>/dev/null)" || \
|
||||
# Via resolve_pin, so a task pinned before a history rewrite keeps being checked
|
||||
# rather than silently skipping every run once its SHA stops resolving.
|
||||
# shellcheck source=lib/resolve-pin.sh
|
||||
. "$(dirname "$0")/lib/resolve-pin.sh"
|
||||
RESOLVED_SHA="$(resolve_pin "$REPO_DIR" "$COMMIT" "$ROOT/task-shared/commit-maps" "$MEMBER" 2>/dev/null)" || \
|
||||
skip "pinned commit $COMMIT not found in $REPO_DIR"
|
||||
|
||||
# --- Throwaway git state ------------------------------------------------------
|
||||
|
||||
@@ -23,6 +23,7 @@ import os
|
||||
import shlex
|
||||
import sys
|
||||
import tempfile
|
||||
import tomllib
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
@@ -85,6 +86,32 @@ _INSTALL_CMD = (
|
||||
)
|
||||
|
||||
|
||||
# codex reserves its built-in provider ids, so the call-origin header needs a provider of
|
||||
# our own. A trial's origin is a constant, so it is a static http_headers literal here
|
||||
# rather than the env-var indirection the containers need — nothing to plumb into a
|
||||
# sandbox, and no way for a missing var to lose the attribution.
|
||||
PROXY_PROVIDER_TOML = """\
|
||||
model_provider = "llm-proxy"
|
||||
|
||||
[model_providers.llm-proxy]
|
||||
name = "LLM proxy"
|
||||
base_url = "${OPENAI_BASE_URL}"
|
||||
env_key = "OPENAI_API_KEY"
|
||||
wire_api = "responses"
|
||||
http_headers = { "X-Surge-Client-Metadata" = '{"origin":"harbor-trial"}' }
|
||||
"""
|
||||
# A custom provider reads its key from env_key and never from auth.json, so the proxy
|
||||
# path must carry this even when an auth file was uploaded.
|
||||
PROXY_KEY_VAR = "OPENAI_API_KEY"
|
||||
|
||||
|
||||
def proxy_provider_config(openai_base_url: str) -> dict:
|
||||
"""PROXY_PROVIDER_TOML as harbor's config dict, pointed at this trial's URL."""
|
||||
return tomllib.loads(PROXY_PROVIDER_TOML.replace("${OPENAI_BASE_URL}", openai_base_url))
|
||||
|
||||
|
||||
|
||||
|
||||
class SystemNodeCodex(Codex):
|
||||
# Set by install()'s probe, read by build_cli_flags(). Mirrors the claude adapter.
|
||||
_has_browser = False
|
||||
@@ -101,6 +128,39 @@ class SystemNodeCodex(Codex):
|
||||
self._get_env("OPENAI_API_KEY") or "", remote_auth_path
|
||||
)
|
||||
|
||||
def _proxy_provider_flags(self) -> str:
|
||||
"""`-c` overrides putting the trial on our own provider — the only place codex
|
||||
can be told to send the call-origin header.
|
||||
|
||||
On the command line rather than in config.toml because harbor writes that file
|
||||
itself, differently per version (0.20 hardcodes the block inline), while these
|
||||
flags are ours in every version.
|
||||
"""
|
||||
base_url = self._get_env("OPENAI_BASE_URL") or ""
|
||||
if "/llm_proxy/" not in base_url:
|
||||
return ""
|
||||
# A custom provider ignores auth.json, so leave that flow on the built-in
|
||||
# provider: losing attribution beats breaking the run's auth.
|
||||
if self._resolve_auth_json_path():
|
||||
return ""
|
||||
config = proxy_provider_config(base_url)
|
||||
provider_id = config["model_provider"]
|
||||
parts = [f"-c model_provider={provider_id}"]
|
||||
for key, value in config["model_providers"][provider_id].items():
|
||||
for path, leaf in (
|
||||
[(f"{key}.{k}", v) for k, v in value.items()]
|
||||
if isinstance(value, dict)
|
||||
else [(key, value)]
|
||||
):
|
||||
# A TOML literal string, since the header value is JSON and carries its
|
||||
# own double quotes.
|
||||
quoted = f"'{leaf}'" if '"' in leaf else f'"{leaf}"'
|
||||
parts.append(
|
||||
"-c "
|
||||
+ shlex.quote(f"model_providers.{provider_id}.{path}={quoted}")
|
||||
)
|
||||
return " ".join(parts)
|
||||
|
||||
def _refuse_shell_hostile_key(self) -> None:
|
||||
"""Harbor's own Codex.run interpolates the key into a heredoc, so a key it cannot
|
||||
escape would 401 with no stated cause. Refuse up front instead."""
|
||||
@@ -127,6 +187,11 @@ class SystemNodeCodex(Codex):
|
||||
reductions = load_harness_registry().require("codex").agent_config_flags()
|
||||
if reductions:
|
||||
flags = f"{flags} {reductions}".strip()
|
||||
# Both run paths go through here, so this is where the trial picks up the
|
||||
# provider that carries the call-origin header.
|
||||
provider = self._proxy_provider_flags()
|
||||
if provider:
|
||||
flags = f"{flags} {provider}".strip()
|
||||
return f"{flags} {self._browser_flag()}".strip() if self._browser_flag() else flags
|
||||
|
||||
def _browser_flag(self) -> str:
|
||||
@@ -467,11 +532,16 @@ class NativeSnapshotCodex(SystemNodeCodex):
|
||||
)
|
||||
if openai_base_url := self._get_env("OPENAI_BASE_URL"):
|
||||
env["OPENAI_BASE_URL"] = openai_base_url
|
||||
# The provider that carries the origin header rides in on build_cli_flags,
|
||||
# so this stays harbor's plain root key.
|
||||
setup_command += (
|
||||
'\ncat >>"$CODEX_HOME/config.toml" <<TOML\n'
|
||||
'openai_base_url = "${OPENAI_BASE_URL}"\n'
|
||||
"TOML"
|
||||
)
|
||||
# env_key names this, and the provider cannot fall back to auth.json.
|
||||
if proxy_key := self._get_env(PROXY_KEY_VAR):
|
||||
env[PROXY_KEY_VAR] = proxy_key
|
||||
skills_command = self._build_register_skills_command()
|
||||
if skills_command:
|
||||
setup_command += f"\n{skills_command}"
|
||||
|
||||
@@ -3,6 +3,8 @@
|
||||
*
|
||||
* Usage:
|
||||
* npx tsx scripts/copy-reference-run.ts <trial-path> [trial-path...]
|
||||
* npx tsx scripts/copy-reference-run.ts --rubric-regrade <trial-path>
|
||||
* npx tsx scripts/copy-reference-run.ts --supersede <old-run-dir> <trial-path>
|
||||
*
|
||||
* Examples:
|
||||
* # Copy a single trial
|
||||
@@ -28,6 +30,8 @@
|
||||
* indirection. Its location is per harness: Claude Code writes
|
||||
* `agent/sessions/projects/-workspace/<id>.jsonl` with the id printed in
|
||||
* `agent/claude-code.txt`; codex writes `agent/sessions/<Y>/<M>/<D>/rollout-*.jsonl`.
|
||||
* - verifier/test-stdout.txt (the verifier's console output, where the reward is printed)
|
||||
* - verifier/deterministic-signals.txt, verifier/rubric-grade(-<N>).json
|
||||
* - config.json, result.json, trial.log
|
||||
* - input-checksums.json — sha256 checksums of the task inputs the run was
|
||||
* generated against (prompt, session snapshot, workspace patch, gitref),
|
||||
@@ -37,6 +41,21 @@
|
||||
* otherwise captured here at copy time as a fallback (capturedBy:
|
||||
* 'copy').
|
||||
*
|
||||
* --rubric-regrade files an atomic-rubric regrade into rubric-regrades/<run>/
|
||||
* instead: a second grade of a run that keeps its own, under the name the trial
|
||||
* itself records for the run it graded. The trial is copied verbatim — verifier/
|
||||
* and all — so a stored grade has the same shape whoever stored it.
|
||||
*
|
||||
* --supersede is the other direction: a holistic regrade replaces a run's grade,
|
||||
* and the new copy lands under a name minted from the NEW reward and trial id, so
|
||||
* adopting it means removing the directory it supersedes. Deleting is deliberate
|
||||
* over merging into the old directory: a regrade grades a different number of
|
||||
* samples than the run often did, so a merge would leave grade-2.md/grade-3.md
|
||||
* from the previous grade beside the new grade-1.md with nothing marking the
|
||||
* generation. A swapped directory holds exactly one grade by construction. The
|
||||
* cost is the old run's session.jsonl, which a replay cannot reproduce; the
|
||||
* trajectory it would be needed to rebuild travels with the replay itself.
|
||||
*
|
||||
* What is NOT copied:
|
||||
* - agent/sessions/, agent/setup/ (workspace data — the resumable JSONL
|
||||
* is hoisted out as `session.jsonl` above; everything else here is
|
||||
@@ -51,11 +70,12 @@ import {
|
||||
mkdirSync,
|
||||
readdirSync,
|
||||
readFileSync,
|
||||
renameSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
writeFileSync,
|
||||
} from 'fs';
|
||||
import { basename, join } from 'path';
|
||||
import { basename, join, resolve } from 'path';
|
||||
|
||||
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
|
||||
import { copyPath, copyTree } from './lib/copy-tree';
|
||||
@@ -99,6 +119,36 @@ function newestRollout(sessionsDir: string): string | null {
|
||||
return found[0].path;
|
||||
}
|
||||
|
||||
/** What a replay trial records about itself: the run it graded, and whether the
|
||||
* grade came from a rubric grader mode rather than the holistic one. */
|
||||
function readReplayProvenance(trialPath: string): {
|
||||
sourceRun: string | null;
|
||||
rubricMode: boolean;
|
||||
} {
|
||||
let config: Record<string, unknown> | undefined;
|
||||
try {
|
||||
const parsed = JSON.parse(readFileSync(join(trialPath, 'result.json'), 'utf-8')) as {
|
||||
config?: Record<string, unknown>;
|
||||
};
|
||||
config = parsed.config;
|
||||
} catch {
|
||||
config = undefined;
|
||||
}
|
||||
const agent = (config?.agent ?? {}) as { kwargs?: { reference_run_dir?: unknown } };
|
||||
const src = agent.kwargs?.reference_run_dir;
|
||||
const verifier = (config?.verifier ?? {}) as { env?: Record<string, unknown> };
|
||||
const graderMode = verifier.env?.GRADER_MODE;
|
||||
// Either marker is enough: the env records the mode harbor was handed, the
|
||||
// rubric-grade file records what the grader actually produced.
|
||||
const rubricMode =
|
||||
(typeof graderMode === 'string' && graderMode.startsWith('rubric-')) ||
|
||||
existsSync(join(trialPath, 'verifier', 'rubric-grade.json'));
|
||||
return {
|
||||
sourceRun: typeof src === 'string' && src.length > 0 ? basename(src.replace(/\/+$/, '')) : null,
|
||||
rubricMode,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the task dir from the trial's result.json `task_name` — the FULL,
|
||||
* unambiguous slug. Harbor TRUNCATES long task names in the trial DIRNAME, so two
|
||||
@@ -127,7 +177,11 @@ function findTaskDirByResultJson(trialPath: string): string | null {
|
||||
return existsSync(dir) ? dir : null;
|
||||
}
|
||||
|
||||
function copyTrial(trialPath: string, destName?: string) {
|
||||
function copyTrial(
|
||||
trialPath: string,
|
||||
opts: { destName?: string; rubricRegrade?: boolean; supersede?: string } = {}
|
||||
) {
|
||||
const { destName, rubricRegrade = false, supersede } = opts;
|
||||
trialPath = trialPath.replace(/\/$/, '');
|
||||
|
||||
if (!existsSync(trialPath)) {
|
||||
@@ -194,147 +248,198 @@ function copyTrial(trialPath: string, destName?: string) {
|
||||
// is copied to exactly that name instead of the minted reward-<r>-<id> —
|
||||
// used by the SxS pipeline so task.toml preference_rollouts, this dir, and
|
||||
// the publish-manifest run_id stay byte-identical by construction.
|
||||
const dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
|
||||
|
||||
if (existsSync(dest)) {
|
||||
console.warn(`Warning: ${dest} already exists, overwriting`);
|
||||
rmSync(dest, { recursive: true });
|
||||
}
|
||||
|
||||
mkdirSync(dest, { recursive: true });
|
||||
|
||||
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
|
||||
// companion reward-correctness.txt (N/A by design) and the machine-readable
|
||||
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
|
||||
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
|
||||
// (the source of truth grade.md/reward.txt are rendered from), the
|
||||
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
|
||||
// render-stderr(-<N>).log, and grader-samples.txt.
|
||||
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
|
||||
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
|
||||
// every per-sample record.)
|
||||
//
|
||||
// signals-status.txt qualifies the deterministic signals the grader was fed:
|
||||
// it records whether every deterministic check actually produced a verdict
|
||||
// ("ok") or one or more was killed before finishing ("degraded" — the grade
|
||||
// is then NOT fully signal-backed). Without it a copied run is
|
||||
// indistinguishable from a run whose checks all passed, so it must travel
|
||||
// with the reward files.
|
||||
const verifierDir = join(trialPath, 'verifier');
|
||||
if (existsSync(verifierDir)) {
|
||||
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
|
||||
if (!entry.isFile()) continue;
|
||||
const f = entry.name;
|
||||
if (
|
||||
f === 'reward.txt' ||
|
||||
f === 'reward-correctness.txt' ||
|
||||
f === 'reward.json' ||
|
||||
f === 'signals-status.txt' ||
|
||||
f === 'grader-samples.txt' ||
|
||||
f === 'grader-regime.json' ||
|
||||
/^grade(-\d+)?\.md$/.test(f) ||
|
||||
/^grade(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-result(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
|
||||
/^render-stderr(-\d+)?\.log$/.test(f)
|
||||
) {
|
||||
copyPath(join(verifierDir, f), join(dest, f));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const agentOutputDir = join(verifierDir, 'agent-output');
|
||||
if (existsSync(agentOutputDir)) {
|
||||
copyTree(agentOutputDir, join(dest, 'agent-output'));
|
||||
}
|
||||
|
||||
// Copy agent session log and trajectory (not workspace)
|
||||
const agentDir = join(trialPath, 'agent');
|
||||
if (existsSync(agentDir)) {
|
||||
mkdirSync(join(dest, 'agent'), { recursive: true });
|
||||
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
|
||||
// the Claude one left codex reference runs with nothing but the trajectory.
|
||||
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
|
||||
const src = join(agentDir, file);
|
||||
if (existsSync(src)) copyPath(src, join(dest, 'agent', file));
|
||||
}
|
||||
// The grader (and downstream worldbench export / replay) reads
|
||||
// agent/trajectory.json. If it's missing, scream so we don't silently
|
||||
// ship a reference run that's only half-useful.
|
||||
if (!existsSync(join(agentDir, 'trajectory.json'))) {
|
||||
console.warn(
|
||||
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
|
||||
`Harbor's adapter for this harness failed to write it (typically because ` +
|
||||
`the converter choked on the session log). Downstream consumers ` +
|
||||
`(grader replay, worldbench export) need this file — investigate ` +
|
||||
`before relying on this reference run.`
|
||||
);
|
||||
}
|
||||
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
|
||||
// gets embedded in claude-code.txt's first non-empty line; we use it to
|
||||
// locate the sibling JSONL Claude Code wrote in the same trial.
|
||||
// Layout is per harness, so each needs a case here — the same reason
|
||||
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
|
||||
// session, which is what codex got before this: nothing at all.
|
||||
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
|
||||
if (existsSync(claudeCodeTxt)) {
|
||||
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
|
||||
const sessionId = readSessionId(claudeCodeTxt);
|
||||
if (sessionId) {
|
||||
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
|
||||
if (existsSync(jsonl)) copyPath(jsonl, join(dest, 'session.jsonl'));
|
||||
}
|
||||
} else {
|
||||
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
|
||||
const rollout = newestRollout(join(agentDir, 'sessions'));
|
||||
if (rollout) copyPath(rollout, join(dest, 'session.jsonl'));
|
||||
}
|
||||
}
|
||||
|
||||
// Copy top-level metadata
|
||||
for (const file of ['config.json', 'result.json', 'trial.log']) {
|
||||
const src = join(trialPath, file);
|
||||
if (existsSync(src)) copyPath(src, join(dest, file));
|
||||
}
|
||||
|
||||
// Record the checksums of the task inputs this run was generated against
|
||||
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
|
||||
// submit-task.ts re-captures at packaging time and warns when any of them
|
||||
// changed — the run then describes an older revision of the task than the
|
||||
// one being shipped. Preferred source: the launch-time stamp harbor-run
|
||||
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
|
||||
// 'run') — it records the inputs the agent actually ran against, so an
|
||||
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
|
||||
// (trials from an older harbor-run, or a failed stamp): capture here at
|
||||
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
|
||||
// the weaker evidence.
|
||||
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
|
||||
const trialStamp = readTaskInputChecksums(trialStampPath);
|
||||
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
|
||||
// harbor-runs sharing a cwd) — its hashes describe some other task's
|
||||
// inputs, so treat it as absent rather than importing false evidence.
|
||||
const stampSlug = trialStamp?.taskSlug;
|
||||
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
|
||||
if (trialStamp && !stampMisrouted) {
|
||||
copyPath(trialStampPath, join(dest, INPUT_CHECKSUMS_FILENAME));
|
||||
} else {
|
||||
if (stampMisrouted) {
|
||||
console.warn(
|
||||
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
|
||||
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
|
||||
);
|
||||
}
|
||||
writeFileSync(
|
||||
join(dest, INPUT_CHECKSUMS_FILENAME),
|
||||
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
|
||||
let dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
|
||||
const prov = readReplayProvenance(trialPath);
|
||||
// An atomic grade in reference-runs/ would overwrite the holistic grade it exists
|
||||
// to be compared against, and the two are not interchangeable.
|
||||
if (!rubricRegrade && prov.rubricMode) {
|
||||
console.error(
|
||||
`Error: ${trialPath} is an atomic-rubric regrade, which does not belong in ` +
|
||||
`reference-runs/. File it beside the run it graded:\n` +
|
||||
` npx tsx scripts/copy-reference-run.ts --rubric-regrade ${trialPath}`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
if (rubricRegrade) {
|
||||
if (!prov.rubricMode) {
|
||||
console.error(
|
||||
`Error: ${trialPath} is not an atomic-rubric regrade (no rubric grader mode ` +
|
||||
`in result.json, no verifier/rubric-grade.json). A holistic regrade replaces ` +
|
||||
`the run's own grade — copy it without --rubric-regrade.`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
if (!prov.sourceRun) {
|
||||
console.error(
|
||||
`Error: ${trialPath}/result.json records no reference_run_dir, so there is no ` +
|
||||
`run to file this regrade under. Copy it by hand into ` +
|
||||
`${join(taskDir, 'rubric-regrades')}/<run-id>/.`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
dest = join(taskDir, 'rubric-regrades', prov.sourceRun);
|
||||
}
|
||||
|
||||
// Build in a sibling directory and swap at the end. A grade costs 15-30 minutes and
|
||||
// cannot be reproduced exactly, so an interrupted copy must leave the stored one intact.
|
||||
const staged = `${dest}.staging-${process.pid}`;
|
||||
rmSync(staged, { recursive: true, force: true });
|
||||
mkdirSync(staged, { recursive: true });
|
||||
|
||||
// A stored atomic grade is the trial verbatim. Matching that shape exactly matters
|
||||
// more than trimming it: grades stored by hand have it, and a reviewer opening one
|
||||
// should not have to work out which way it was written.
|
||||
if (rubricRegrade) {
|
||||
copyTree(trialPath, staged);
|
||||
} else {
|
||||
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
|
||||
// companion reward-correctness.txt (N/A by design) and the machine-readable
|
||||
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
|
||||
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
|
||||
// (the source of truth grade.md/reward.txt are rendered from), the
|
||||
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
|
||||
// render-stderr(-<N>).log, and grader-samples.txt.
|
||||
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
|
||||
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
|
||||
// every per-sample record.)
|
||||
//
|
||||
// signals-status.txt qualifies the deterministic signals the grader was fed:
|
||||
// it records whether every deterministic check actually produced a verdict
|
||||
// ("ok") or one or more was killed before finishing ("degraded" — the grade
|
||||
// is then NOT fully signal-backed). Without it a copied run is
|
||||
// indistinguishable from a run whose checks all passed, so it must travel
|
||||
// with the reward files.
|
||||
const verifierDir = join(trialPath, 'verifier');
|
||||
if (existsSync(verifierDir)) {
|
||||
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
|
||||
if (!entry.isFile()) continue;
|
||||
const f = entry.name;
|
||||
if (
|
||||
f === 'reward.json' ||
|
||||
f === 'signals-status.txt' ||
|
||||
f === 'grader-samples.txt' ||
|
||||
f === 'grader-regime.json' ||
|
||||
f === 'test-stdout.txt' ||
|
||||
f === 'deterministic-signals.txt' ||
|
||||
/^reward(-\d+)?\.txt$/.test(f) ||
|
||||
/^reward-correctness(-\d+)?\.txt$/.test(f) ||
|
||||
/^grade(-\d+)?\.md$/.test(f) ||
|
||||
/^grade(-\d+)?\.json$/.test(f) ||
|
||||
/^rubric-grade(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-result(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
|
||||
// The renderer retries, so the real name is render-stderr-<sample>-attempt<n>.log.
|
||||
/^render-stderr(-\d+)?(-attempt\d+)?\.log$/.test(f)
|
||||
) {
|
||||
copyPath(join(verifierDir, f), join(staged, f));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A rubric regrade re-grades a run that already ships its own agent-output,
|
||||
// trajectory and session; a second copy would only double the tarball.
|
||||
const agentOutputDir = join(verifierDir, 'agent-output');
|
||||
if (!rubricRegrade && existsSync(agentOutputDir)) {
|
||||
copyTree(agentOutputDir, join(staged, 'agent-output'));
|
||||
}
|
||||
|
||||
// Copy agent session log and trajectory (not workspace)
|
||||
const agentDir = join(trialPath, 'agent');
|
||||
if (!rubricRegrade && existsSync(agentDir)) {
|
||||
mkdirSync(join(staged, 'agent'), { recursive: true });
|
||||
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
|
||||
// the Claude one left codex reference runs with nothing but the trajectory.
|
||||
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
|
||||
const src = join(agentDir, file);
|
||||
if (existsSync(src)) copyPath(src, join(staged, 'agent', file));
|
||||
}
|
||||
// The grader (and downstream worldbench export / replay) reads
|
||||
// agent/trajectory.json. If it's missing, scream so we don't silently
|
||||
// ship a reference run that's only half-useful.
|
||||
if (!existsSync(join(agentDir, 'trajectory.json'))) {
|
||||
console.warn(
|
||||
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
|
||||
`Harbor's adapter for this harness failed to write it (typically because ` +
|
||||
`the converter choked on the session log). Downstream consumers ` +
|
||||
`(grader replay, worldbench export) need this file — investigate ` +
|
||||
`before relying on this reference run.`
|
||||
);
|
||||
}
|
||||
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
|
||||
// gets embedded in claude-code.txt's first non-empty line; we use it to
|
||||
// locate the sibling JSONL Claude Code wrote in the same trial.
|
||||
// Layout is per harness, so each needs a case here — the same reason
|
||||
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
|
||||
// session, which is what codex got before this: nothing at all.
|
||||
// Silent when there is nothing to hoist: no shipped tool reads this file and nothing
|
||||
// validates it, so its absence is not worth a line of output.
|
||||
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
|
||||
if (existsSync(claudeCodeTxt)) {
|
||||
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
|
||||
const sessionId = readSessionId(claudeCodeTxt);
|
||||
if (sessionId) {
|
||||
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
|
||||
if (existsSync(jsonl)) copyPath(jsonl, join(staged, 'session.jsonl'));
|
||||
}
|
||||
} else {
|
||||
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
|
||||
const rollout = newestRollout(join(agentDir, 'sessions'));
|
||||
if (rollout) copyPath(rollout, join(staged, 'session.jsonl'));
|
||||
}
|
||||
}
|
||||
|
||||
// Copy top-level metadata. result.json is what makes a regrade self-describing
|
||||
// (which run it graded, under which grader mode), so it travels either way.
|
||||
for (const file of rubricRegrade
|
||||
? ['result.json']
|
||||
: ['config.json', 'result.json', 'trial.log']) {
|
||||
const src = join(trialPath, file);
|
||||
if (existsSync(src)) copyPath(src, join(staged, file));
|
||||
}
|
||||
|
||||
// Not for a rubric regrade: this input set is what stales a RUN (prompt, snapshot,
|
||||
// patch, gitref), and the run it re-graded already carries its own stamp.
|
||||
if (!rubricRegrade) {
|
||||
// Record the checksums of the task inputs this run was generated against
|
||||
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
|
||||
// submit-task.ts re-captures at packaging time and warns when any of them
|
||||
// changed — the run then describes an older revision of the task than the
|
||||
// one being shipped. Preferred source: the launch-time stamp harbor-run
|
||||
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
|
||||
// 'run') — it records the inputs the agent actually ran against, so an
|
||||
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
|
||||
// (trials from an older harbor-run, or a failed stamp): capture here at
|
||||
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
|
||||
// the weaker evidence.
|
||||
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
|
||||
const trialStamp = readTaskInputChecksums(trialStampPath);
|
||||
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
|
||||
// harbor-runs sharing a cwd) — its hashes describe some other task's
|
||||
// inputs, so treat it as absent rather than importing false evidence.
|
||||
const stampSlug = trialStamp?.taskSlug;
|
||||
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
|
||||
if (trialStamp && !stampMisrouted) {
|
||||
copyPath(trialStampPath, join(staged, INPUT_CHECKSUMS_FILENAME));
|
||||
} else {
|
||||
if (stampMisrouted) {
|
||||
console.warn(
|
||||
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
|
||||
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
|
||||
);
|
||||
}
|
||||
writeFileSync(
|
||||
join(staged, INPUT_CHECKSUMS_FILENAME),
|
||||
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Files captured from a run can land unreadable to you, which makes packaging
|
||||
// fail later. Fix that now. Guarded so it can never fail a copy that worked.
|
||||
let perms = null;
|
||||
try {
|
||||
perms = normalizeTreePermissions(dest);
|
||||
perms = normalizeTreePermissions(staged);
|
||||
} catch (err) {
|
||||
console.warn(`Warning: could not normalize permissions on ${dest}: ${String(err)}`);
|
||||
console.warn(` The run copied fine. If packaging later fails on permissions:`);
|
||||
@@ -350,6 +455,43 @@ function copyTrial(trialPath: string, destName?: string) {
|
||||
);
|
||||
}
|
||||
|
||||
if (existsSync(dest)) {
|
||||
console.warn(`Warning: ${dest} already exists, overwriting`);
|
||||
rmSync(dest, { recursive: true });
|
||||
}
|
||||
renameSync(staged, dest);
|
||||
|
||||
// Remove what this copy supersedes, now that the copy is on disk. Skipped when the
|
||||
// minted name landed on the superseded directory itself — that is the copy, not a
|
||||
// leftover.
|
||||
if (supersede) {
|
||||
const old = resolve(supersede);
|
||||
if (!existsSync(old)) {
|
||||
console.warn(`Warning: nothing to supersede at ${supersede} — it is already gone`);
|
||||
} else if (old === resolve(dest)) {
|
||||
console.log(`Superseded ${supersede} in place (same name)`);
|
||||
} else {
|
||||
// A stored atomic grade is keyed on the run's folder name, and superseding
|
||||
// changes that name. Move it with the run — it grades the same behaviour — or
|
||||
// it is left pointing at a run that no longer exists.
|
||||
const storedGrade = join(taskDir, 'rubric-regrades', basename(old));
|
||||
const movedGrade = join(taskDir, 'rubric-regrades', basename(dest));
|
||||
if (existsSync(storedGrade) && existsSync(movedGrade)) {
|
||||
console.warn(
|
||||
`Warning: ${storedGrade} names the superseded run, but ${movedGrade} already ` +
|
||||
`exists. Leaving both — remove whichever is obsolete.`
|
||||
);
|
||||
} else if (existsSync(storedGrade)) {
|
||||
renameSync(storedGrade, movedGrade);
|
||||
console.log(`Moved the stored atomic grade to ${movedGrade}`);
|
||||
console.log(' It grades the same run. Grade it again under the atomic rubric');
|
||||
console.log(' if the rubric changed since it was stored.');
|
||||
}
|
||||
rmSync(old, { recursive: true });
|
||||
console.log(`Superseded ${supersede} (removed)`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`Copied to ${dest}`);
|
||||
console.log(` reward: ${reward}`);
|
||||
console.log(` task: ${taskDir}`);
|
||||
@@ -364,9 +506,19 @@ function copyTrial(trialPath: string, destName?: string) {
|
||||
// Main
|
||||
const rawArgs = process.argv.slice(2);
|
||||
let destName: string | undefined;
|
||||
let rubricRegrade = false;
|
||||
let supersede: string | undefined;
|
||||
const args: string[] = [];
|
||||
for (let i = 0; i < rawArgs.length; i++) {
|
||||
if (rawArgs[i] === '--dest-name') {
|
||||
if (rawArgs[i] === '--rubric-regrade') {
|
||||
rubricRegrade = true;
|
||||
} else if (rawArgs[i] === '--supersede') {
|
||||
supersede = rawArgs[++i];
|
||||
if (!supersede) {
|
||||
console.error('Error: --supersede requires the run directory being replaced');
|
||||
process.exit(1);
|
||||
}
|
||||
} else if (rawArgs[i] === '--dest-name') {
|
||||
destName = rawArgs[++i];
|
||||
if (!destName) {
|
||||
console.error('Error: --dest-name requires a value');
|
||||
@@ -379,7 +531,9 @@ for (let i = 0; i < rawArgs.length; i++) {
|
||||
|
||||
if (args.length === 0) {
|
||||
console.error(
|
||||
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]'
|
||||
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]\n' +
|
||||
' npx tsx scripts/copy-reference-run.ts --rubric-regrade <trial-path>\n' +
|
||||
' npx tsx scripts/copy-reference-run.ts --supersede <old-run-dir> <trial-path>'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
@@ -387,7 +541,21 @@ if (destName && args.length !== 1) {
|
||||
console.error('Error: --dest-name applies to exactly one trial path');
|
||||
process.exit(1);
|
||||
}
|
||||
if (supersede && args.length !== 1) {
|
||||
console.error('Error: --supersede applies to exactly one trial path');
|
||||
process.exit(1);
|
||||
}
|
||||
if (supersede && rubricRegrade) {
|
||||
console.error('Error: --rubric-regrade adds a grade beside a run; there is nothing to supersede');
|
||||
process.exit(1);
|
||||
}
|
||||
if (destName && rubricRegrade) {
|
||||
console.error(
|
||||
'Error: --rubric-regrade takes its name from the trial, so --dest-name cannot apply'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
for (const trialPath of args) {
|
||||
copyTrial(trialPath, destName);
|
||||
copyTrial(trialPath, { destName, rubricRegrade, supersede });
|
||||
}
|
||||
|
||||
@@ -18,6 +18,9 @@
|
||||
# harbor-tasks/<slug> \
|
||||
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg
|
||||
#
|
||||
# # Every captured run for a task, 2 at a time (after a rubric edit)
|
||||
# scripts/harbor-regrade harbor-tasks/<slug> --all
|
||||
#
|
||||
# # Ten regrades of the same reference run (independent grader trials)
|
||||
# scripts/harbor-regrade \
|
||||
# harbor-tasks/<slug> \
|
||||
@@ -28,6 +31,11 @@
|
||||
# token rate). The replay agent runs no model, so the grader is the only model in
|
||||
# this path. Serving speed and cost change; the grade itself is not steered.
|
||||
#
|
||||
# A finished regrade files itself when nothing is filed for that run yet — which is
|
||||
# only ever the atomic case, since a run always carries a holistic grade already.
|
||||
# Otherwise it grades and leaves the result in its job dir, so the tune-and-diff loop
|
||||
# is untouched; --replace adopts it, superseding what was there.
|
||||
#
|
||||
# See scripts/replay_agent.py for what the agent actually does, and the
|
||||
# `verifier: capture tracked-file deletions in agent-output` PR for the
|
||||
# capture half of this flow (_HARBOR_DELETIONS.txt).
|
||||
@@ -37,6 +45,11 @@ set -euo pipefail
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
|
||||
# A sample count from the caller wins over .env, so a prefix can override a standing
|
||||
# setting for one regrade. Captured as one value so a caller's spelling beats .env's,
|
||||
# whichever each used (test.sh reads GRADER_SAMPLES; this script has long taken both).
|
||||
_CALLER_SAMPLES="${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}"
|
||||
|
||||
# Source API key + any verifier env from the repo's .env
|
||||
if [ -f "$REPO_ROOT/.env" ]; then
|
||||
set -a
|
||||
@@ -64,16 +77,28 @@ fi
|
||||
|
||||
usage() {
|
||||
cat >&2 <<EOF
|
||||
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
|
||||
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir>... [extra harbor args]
|
||||
scripts/harbor-regrade <task-dir> --all [extra harbor args]
|
||||
|
||||
Required arguments:
|
||||
<task-dir> harbor-tasks/<slug> — same dir you'd pass to scripts/harbor-run.
|
||||
<reference-run-dir> harbor-tasks/<slug>/reference-runs/<run-id> — must contain
|
||||
agent-output/ (and ideally agent/trajectory.json).
|
||||
agent-output/ (and ideally agent/trajectory.json). Name several
|
||||
to re-grade them all; each gets its own harbor job.
|
||||
|
||||
Optional arguments:
|
||||
--all re-grade every run under <task-dir>/reference-runs/ — the usual
|
||||
thing to do after editing the rubric.
|
||||
--jobs N how many runs to re-grade at once (default 2). Each one is a
|
||||
container, so raise it only as far as your machine allows.
|
||||
--fast grade in claude's fast serving mode (higher token rate,
|
||||
faster output). Anything else is passed through to harbor.
|
||||
--replace adopt the result, replacing the grade already filed for this
|
||||
run. Without it, a regrade that would overwrite an existing
|
||||
grade is left in its job dir for you to compare first.
|
||||
|
||||
Note: -k re-grades the SAME run N times (N independent grades of one trajectory).
|
||||
To re-grade DIFFERENT runs, name them all, or use --all.
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
@@ -81,19 +106,48 @@ EOF
|
||||
# `--fast` is consumed here (harbor has no such flag); everything else stays an opaque
|
||||
# passthrough. Stripped first so it may appear anywhere, not just after the positionals.
|
||||
FAST_REQUESTED=""
|
||||
ALL_RUNS=""
|
||||
REPLACE=""
|
||||
JOBS=2
|
||||
REGRADE_ARGS=()
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--fast) FAST_REQUESTED=1; shift ;;
|
||||
--replace) REPLACE=1; shift ;;
|
||||
--all) ALL_RUNS=1; shift ;;
|
||||
--jobs)
|
||||
[ $# -ge 2 ] || { echo "Error: --jobs needs a number." >&2; exit 1; }
|
||||
JOBS="$2"; shift 2 ;;
|
||||
--jobs=*) JOBS="${1#--jobs=}"; shift ;;
|
||||
*) REGRADE_ARGS+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
set -- "${REGRADE_ARGS[@]+"${REGRADE_ARGS[@]}"}"
|
||||
|
||||
[ $# -lt 2 ] && usage
|
||||
# Normalised to base 10 before any (( )) sees it: "09" passes a -ge test but is
|
||||
# invalid octal in arithmetic, which spun the dispatch loop forever.
|
||||
case "$JOBS" in
|
||||
'' | *[!0-9]*) JOBS_OK="" ;;
|
||||
*) JOBS=$((10#$JOBS)); [ "$JOBS" -ge 1 ] && JOBS_OK=1 || JOBS_OK="" ;;
|
||||
esac
|
||||
if [ -z "$JOBS_OK" ]; then
|
||||
echo "Error: --jobs must be a positive integer (got '$JOBS')." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
[ $# -lt 1 ] && usage
|
||||
TASK_DIR="$1"
|
||||
REF_RUN_DIR="$2"
|
||||
shift 2
|
||||
shift
|
||||
|
||||
# Leading non-flag positionals are reference runs; collection stops at the first
|
||||
# harbor flag so `<task> <ref> -k 4` keeps working and `4` is never read as a run.
|
||||
REF_DIRS=()
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
-*) break ;;
|
||||
*) REF_DIRS+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
|
||||
# Resolve to absolute paths — harbor cd's around internally; the replay
|
||||
# agent receives the path as an --agent-kwarg and won't know our cwd.
|
||||
@@ -101,6 +155,102 @@ TASK_DIR_ABS=$(cd "$TASK_DIR" 2>/dev/null && pwd) || {
|
||||
echo "Error: task-dir does not exist: $TASK_DIR" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Unfiltered on purpose: the child, not a name test here, decides what can be replayed.
|
||||
if [ -n "$ALL_RUNS" ]; then
|
||||
[ ${#REF_DIRS[@]} -gt 0 ] && {
|
||||
echo "Error: pass --all or explicit reference-run dirs, not both." >&2
|
||||
exit 1
|
||||
}
|
||||
REF_PARENT="$TASK_DIR_ABS/reference-runs"
|
||||
[ -d "$REF_PARENT" ] || {
|
||||
echo "Error: --all needs $REF_PARENT, which does not exist." >&2
|
||||
exit 1
|
||||
}
|
||||
for _cand in "$REF_PARENT"/*/; do
|
||||
[ -d "$_cand" ] && REF_DIRS+=("${_cand%/}")
|
||||
done
|
||||
[ ${#REF_DIRS[@]} -eq 0 ] && {
|
||||
echo "Error: $REF_PARENT holds no reference runs." >&2
|
||||
exit 1
|
||||
}
|
||||
fi
|
||||
|
||||
[ ${#REF_DIRS[@]} -eq 0 ] && usage
|
||||
|
||||
# Several runs: re-run ITSELF once each, so every child does the full preflight in an
|
||||
# output dir of its own — harbor names job dirs by the second, and sharing one corrupts.
|
||||
if [ ${#REF_DIRS[@]} -gt 1 ]; then
|
||||
OUT_BASE="${HARBOR_REGRADE_OUT:-harbor-jobs}"
|
||||
mkdir -p "$OUT_BASE"
|
||||
CHILD_FLAGS=()
|
||||
[ -n "$FAST_REQUESTED" ] && CHILD_FLAGS+=(--fast)
|
||||
[ -n "$REPLACE" ] && CHILD_FLAGS+=(--replace)
|
||||
echo "Re-grading ${#REF_DIRS[@]} reference runs, $JOBS at a time."
|
||||
FAILED=()
|
||||
FAILED_LOGS=()
|
||||
BATCH_PIDS=()
|
||||
# Without this a killed driver leaves its children running, and on a cloud backend
|
||||
# each one is a sandbox that bills until something else reaps it.
|
||||
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 143' TERM
|
||||
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 130' INT
|
||||
IDX=0
|
||||
TOTAL=${#REF_DIRS[@]}
|
||||
while [ "$IDX" -lt "$TOTAL" ]; do
|
||||
BATCH_PIDS=()
|
||||
BATCH_REFS=()
|
||||
BATCH_LOGS=()
|
||||
for ((_j = 0; _j < JOBS && IDX < TOTAL; _j++)); do
|
||||
REF="${REF_DIRS[$IDX]}"
|
||||
RUN_ID=$(basename "$REF")
|
||||
# --job-name, not a nested output dir: children keep harbor's own
|
||||
# harbor-jobs/<job>/<trial> shape. Indexed so duplicate args cannot collide.
|
||||
JOB_NAME="regrade-$((IDX + 1))-$RUN_ID"
|
||||
LOG="$OUT_BASE/$JOB_NAME.log"
|
||||
echo " starting $RUN_ID (log: $LOG)"
|
||||
"$SCRIPT_DIR/harbor-regrade" "$TASK_DIR_ABS" "$REF" \
|
||||
${CHILD_FLAGS[@]+"${CHILD_FLAGS[@]}"} "$@" \
|
||||
--job-name "$JOB_NAME" > "$LOG" 2>&1 &
|
||||
BATCH_PIDS+=($!)
|
||||
BATCH_REFS+=("$RUN_ID")
|
||||
BATCH_LOGS+=("$LOG")
|
||||
IDX=$((IDX + 1))
|
||||
done
|
||||
for ((_i = 0; _i < ${#BATCH_PIDS[@]}; _i++)); do
|
||||
if wait "${BATCH_PIDS[$_i]}"; then
|
||||
echo " ok ${BATCH_REFS[$_i]}"
|
||||
else
|
||||
echo " FAILED ${BATCH_REFS[$_i]}"
|
||||
FAILED+=("${BATCH_REFS[$_i]}")
|
||||
FAILED_LOGS+=("${BATCH_LOGS[$_i]}")
|
||||
fi
|
||||
done
|
||||
done
|
||||
if [ ${#FAILED[@]} -gt 0 ]; then
|
||||
echo "${#FAILED[@]} of $TOTAL failed. Their output, in full — re-running is safe:" >&2
|
||||
for _f in "${FAILED_LOGS[@]}"; do echo " $_f" >&2; done
|
||||
exit 1
|
||||
fi
|
||||
echo "All $TOTAL re-graded."
|
||||
# Children file their own grades into per-run logs we do not echo, so count the
|
||||
# results rather than claiming them. Only the atomic destination is countable:
|
||||
# a superseded run is gone, so there is no before/after to compare against.
|
||||
case "${HARBOR_GRADER_MODE:-}" in
|
||||
rubric-*)
|
||||
FILED=0
|
||||
for _r in "${REF_DIRS[@]}"; do
|
||||
[ -d "$TASK_DIR_ABS/rubric-regrades/$(basename "$_r")" ] && FILED=$((FILED + 1))
|
||||
done
|
||||
echo "Filed $FILED of $TOTAL into $TASK_DIR_ABS/rubric-regrades/."
|
||||
if [ "$FILED" -lt "$TOTAL" ]; then
|
||||
echo "The rest are still in their job dirs; their logs above say why." >&2
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
exit 0
|
||||
fi
|
||||
|
||||
REF_RUN_DIR="${REF_DIRS[0]}"
|
||||
REF_RUN_DIR_ABS=$(cd "$REF_RUN_DIR" 2>/dev/null && pwd) || {
|
||||
echo "Error: reference-run-dir does not exist: $REF_RUN_DIR" >&2
|
||||
exit 1
|
||||
@@ -114,7 +264,10 @@ if [ -d "$TASK_DIR_ABS/environment" ]; then
|
||||
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
|
||||
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
|
||||
if [ -f "$DNSJAIL_SRC" ]; then
|
||||
cp "$DNSJAIL_SRC" "$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
|
||||
# Rename into place: children share this task dir, and a half-written
|
||||
# script is one a sibling's image build can pick up.
|
||||
_dnsjail_dst="$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
|
||||
cp "$DNSJAIL_SRC" "$_dnsjail_dst.tmp.$$" && mv -f "$_dnsjail_dst.tmp.$$" "$_dnsjail_dst"
|
||||
break
|
||||
fi
|
||||
done
|
||||
@@ -135,7 +288,8 @@ case "${HARBOR_GRADER_MODE:-}" in
|
||||
[ -f "$RUBRIC_RENDER_SRC" ] || continue
|
||||
if [ ! -f "$RUBRIC_RENDER_DEST" ] || ! cmp -s "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"; then
|
||||
mkdir -p "$TASK_DIR_ABS/tests"
|
||||
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"
|
||||
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST.tmp.$$" &&
|
||||
mv -f "$RUBRIC_RENDER_DEST.tmp.$$" "$RUBRIC_RENDER_DEST"
|
||||
echo "Note: staged the current shared render-rubric-grade.py into $TASK_DIR_ABS/tests (rubric grader mode)." >&2
|
||||
fi
|
||||
break
|
||||
@@ -206,13 +360,16 @@ GRADER_MODE_FLAG=()
|
||||
# inside the unchanged agentic harness. See raccoon-shared/test.sh GRADER_MODEL.
|
||||
[ -n "${HARBOR_GRADER_MODEL:-}" ] && GRADER_MODE_FLAG+=(--verifier-env "GRADER_MODEL=$HARBOR_GRADER_MODEL")
|
||||
|
||||
# Optional sample count: HARBOR_GRADER_SAMPLES=1 overrides the default 3 samples.
|
||||
# For measuring per-sample properties of the grader (e.g. how often it emits a
|
||||
# schema-valid grade.json), 1 sample across N distinct trajectories is a better
|
||||
# estimator than 3 samples across N/3 — it decorrelates per-task effects for the
|
||||
# same token spend. See raccoon-shared/test.sh GRADER_SAMPLES.
|
||||
[ -n "${HARBOR_GRADER_SAMPLES:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$HARBOR_GRADER_SAMPLES")
|
||||
# Optional sample count, under either name. Unset, the task's own frozen tests/test.sh
|
||||
# decides: tasks created before Sep 2026 average 3 samples, newer ones grade once.
|
||||
_SAMPLES="${_CALLER_SAMPLES:-${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}}"
|
||||
if [ -n "$_SAMPLES" ]; then
|
||||
if ! [ "$_SAMPLES" -ge 1 ] 2>/dev/null; then
|
||||
echo "harbor-regrade: ERROR — GRADER_SAMPLES must be a positive integer (got '$_SAMPLES')." >&2
|
||||
exit 1
|
||||
fi
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$_SAMPLES")
|
||||
fi
|
||||
|
||||
# Optional API timeout: HARBOR_API_TIMEOUT_MS=<ms> raises the per-request API
|
||||
# timeout of the grader's claude calls via verifier env (API_TIMEOUT_MS is the
|
||||
@@ -283,9 +440,83 @@ fi
|
||||
[ -n "${HARBOR_GRADING_STANDARD:-}" ] &&
|
||||
GRADER_MODE_FLAG+=(--verifier-env "GRADING_STANDARD=$HARBOR_GRADING_STANDARD")
|
||||
|
||||
# A regrade is all verifier, so every call it makes is the grader's.
|
||||
if [ -f "$REPO_ROOT/scripts/lib/call-origin.sh" ]; then
|
||||
. "$REPO_ROOT/scripts/lib/call-origin.sh"
|
||||
_GRADER_METADATA="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
|
||||
# `if`, not `[ ... ] && ...`: an AND-list is this block's last statement, so under
|
||||
# `set -e` an empty value would abort the regrade rather than just skip the header.
|
||||
if [ -n "$_GRADER_METADATA" ]; then
|
||||
GRADER_MODE_FLAG+=(
|
||||
--verifier-env
|
||||
"ANTHROPIC_CUSTOM_HEADERS=$CALL_ORIGIN_HEADER: $_GRADER_METADATA"
|
||||
)
|
||||
fi
|
||||
fi
|
||||
|
||||
# Where this grade would be filed, and whether filing it destroys anything. An atomic
|
||||
# regrade lands beside the run in rubric-regrades/<run>; a holistic one replaces the
|
||||
# run's own grade, so its destination is always occupied and never files unasked.
|
||||
rel() { case "$1" in "$PWD"/*) printf '%s' "${1#"$PWD"/}" ;; *) printf '%s' "$1" ;; esac; }
|
||||
|
||||
RUN_ID=$(basename "$REF_RUN_DIR_ABS")
|
||||
RUBRIC_MODE=""
|
||||
case "${HARBOR_GRADER_MODE:-}" in rubric-*) RUBRIC_MODE=1 ;; esac
|
||||
if [ -n "$RUBRIC_MODE" ]; then
|
||||
FILE_DEST="$TASK_DIR_ABS/rubric-regrades/$RUN_ID"
|
||||
else
|
||||
FILE_DEST="$REF_RUN_DIR_ABS"
|
||||
fi
|
||||
|
||||
AUTOFILE=""
|
||||
if [ ! -e "$FILE_DEST" ] || [ -n "$REPLACE" ]; then
|
||||
AUTOFILE=1
|
||||
fi
|
||||
if [ -z "$AUTOFILE" ]; then
|
||||
echo "Note: $(rel "$FILE_DEST")" >&2
|
||||
echo " already holds a grade, so this regrade will not be filed." >&2
|
||||
echo " Where it landed is printed when it finishes." >&2
|
||||
fi
|
||||
|
||||
# Where the trial will land. Harbor's own job name is a second-granularity timestamp,
|
||||
# so naming it here is what makes the trial findable afterwards (the fan-out passes one).
|
||||
CALLER_JOB_NAME=""
|
||||
CALLER_OUT=""
|
||||
_prev=""
|
||||
for _a in "$@"; do
|
||||
case "$_prev" in
|
||||
--job-name) CALLER_JOB_NAME="$_a" ;;
|
||||
-o | --output-dir) CALLER_OUT="$_a" ;;
|
||||
esac
|
||||
case "$_a" in
|
||||
--job-name=*) CALLER_JOB_NAME="${_a#--job-name=}" ;;
|
||||
--output-dir=*) CALLER_OUT="${_a#--output-dir=}" ;;
|
||||
esac
|
||||
_prev="$_a"
|
||||
done
|
||||
|
||||
JOB_NAME="$CALLER_JOB_NAME"
|
||||
JOB_NAME_FLAG=()
|
||||
if [ -n "$AUTOFILE" ] && [ -n "$CALLER_OUT" ]; then
|
||||
echo "Note: -o/--output-dir passed, so this regrade will not be filed into" >&2
|
||||
echo " $(rel "$FILE_DEST") — copy it yourself when it finishes." >&2
|
||||
AUTOFILE=""
|
||||
fi
|
||||
if [ -z "$JOB_NAME" ] && [ -z "$CALLER_OUT" ]; then
|
||||
# $$ as well as the epoch: harbor refuses an existing job dir outright, and two
|
||||
# regrades of one run can start in the same second.
|
||||
JOB_NAME="regrade-$RUN_ID-$(date +%s)-$$"
|
||||
JOB_NAME_FLAG=(--job-name "$JOB_NAME")
|
||||
fi
|
||||
|
||||
# RACCOON_DNS_JAIL is inert here on purpose: replay_agent runs no agent turn, so there is
|
||||
# nothing to restrict — only the verifier runs, and it needs the proxy.
|
||||
exec harbor run \
|
||||
#
|
||||
# harbor runs as a child, not under `exec`, so the filing below runs after it exits.
|
||||
# TERM/INT are forwarded so a kill here never orphans a job (on cloud, a billing sandbox).
|
||||
HARBOR_EXIT=0
|
||||
HARBOR_SIGNALLED=""
|
||||
harbor run \
|
||||
-p "$TASK_DIR_ABS" \
|
||||
--agent-import-path replay_agent:ReplayAgent \
|
||||
--ak "reference_run_dir=$REF_RUN_DIR_ABS" \
|
||||
@@ -296,4 +527,74 @@ exec harbor run \
|
||||
--yes \
|
||||
-o "$OUT_DIR" \
|
||||
${GRADER_MODE_FLAG[@]+"${GRADER_MODE_FLAG[@]}"} \
|
||||
"$@"
|
||||
${JOB_NAME_FLAG[@]+"${JOB_NAME_FLAG[@]}"} \
|
||||
"$@" &
|
||||
HARBOR_PID=$!
|
||||
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
|
||||
trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT
|
||||
wait "$HARBOR_PID" || HARBOR_EXIT=$?
|
||||
if [ -n "$HARBOR_SIGNALLED" ]; then
|
||||
# The first wait was interrupted by the trap; wait again for harbor's real status.
|
||||
wait "$HARBOR_PID" || HARBOR_EXIT=$?
|
||||
fi
|
||||
trap - TERM INT
|
||||
|
||||
# A failed grade has nothing to report on, and a caller-directed -o put the trial
|
||||
# somewhere this script cannot name.
|
||||
if [ "$HARBOR_EXIT" -ne 0 ] || [ -n "$CALLER_OUT" ]; then
|
||||
exit "$HARBOR_EXIT"
|
||||
fi
|
||||
|
||||
JOB_DIR="$OUT_DIR/$JOB_NAME"
|
||||
TRIALS=()
|
||||
for _t in "$JOB_DIR"/*__*/; do
|
||||
[ -d "$_t" ] && TRIALS+=("${_t%/}")
|
||||
done
|
||||
|
||||
# Atomic grades sit beside the run; a holistic one replaces it, which means removing
|
||||
# the directory it supersedes (see copy-reference-run.ts for why swap, not merge).
|
||||
COPY_FLAGS=(--rubric-regrade)
|
||||
[ -z "$RUBRIC_MODE" ] && COPY_FLAGS=(--supersede "$(rel "$REF_RUN_DIR_ABS")")
|
||||
FILE_CMD="npx tsx $(rel "$SCRIPT_DIR")/copy-reference-run.ts ${COPY_FLAGS[*]}"
|
||||
|
||||
if [ ${#TRIALS[@]} -eq 0 ]; then
|
||||
echo "Note: no trial directory under $JOB_DIR, so there is no grade." >&2
|
||||
exit "$HARBOR_EXIT"
|
||||
fi
|
||||
if [ ${#TRIALS[@]} -gt 1 ]; then
|
||||
echo "Note: ${#TRIALS[@]} trials under $JOB_DIR (-k grades one run repeatedly)." >&2
|
||||
echo " Pick the one to keep and file it: $FILE_CMD <trial-dir>" >&2
|
||||
exit "$HARBOR_EXIT"
|
||||
fi
|
||||
|
||||
# Not filing: say where the grade is and how the two compare, since comparing them is
|
||||
# the whole reason it was left alone. Adopting is a copy — re-running with --replace
|
||||
# would spend the 15-30 minutes again for a grade already sitting on disk.
|
||||
if [ -z "$AUTOFILE" ]; then
|
||||
NEW_REWARD=$(cat "${TRIALS[0]}/verifier/reward.txt" 2>/dev/null || echo "?")
|
||||
# A stored atomic grade is a whole trial, so its reward sits under verifier/;
|
||||
# a reference run keeps its own at the top.
|
||||
OLD_REWARD=$(cat "$FILE_DEST/verifier/reward.txt" 2>/dev/null ||
|
||||
cat "$FILE_DEST/reward.txt" 2>/dev/null || echo "?")
|
||||
echo "" >&2
|
||||
echo "Graded, not filed — that run already holds a grade." >&2
|
||||
echo " new $NEW_REWARD $(rel "${TRIALS[0]}")" >&2
|
||||
echo " filed $OLD_REWARD $(rel "$FILE_DEST")" >&2
|
||||
echo "" >&2
|
||||
echo "Adopt this grade:" >&2
|
||||
echo " npx tsx $(rel "$SCRIPT_DIR")/copy-reference-run.ts \\" >&2
|
||||
echo " ${COPY_FLAGS[*]} \\" >&2
|
||||
echo " $(rel "${TRIALS[0]}")" >&2
|
||||
echo "" >&2
|
||||
echo "Or pass --replace next time to file it without this step." >&2
|
||||
exit "$HARBOR_EXIT"
|
||||
fi
|
||||
|
||||
# Best-effort from here: the grade already cost 15-30 minutes, so a filing problem
|
||||
# prints the command to finish by hand rather than failing the regrade.
|
||||
if ! npx tsx "$SCRIPT_DIR/copy-reference-run.ts" "${COPY_FLAGS[@]}" "${TRIALS[0]}" >&2; then
|
||||
echo "Warning: the grade is in ${TRIALS[0]} but could not be filed. Retry with:" >&2
|
||||
echo " $FILE_CMD ${TRIALS[0]}" >&2
|
||||
fi
|
||||
|
||||
exit "$HARBOR_EXIT"
|
||||
|
||||
@@ -16,6 +16,9 @@
|
||||
# token rate): the trial agent (claude-code only) and the grader the verifier launches.
|
||||
# Serving speed and cost change; the grade itself is not steered.
|
||||
#
|
||||
# GRADER_SAMPLES=N (a prefix, or a line in .env) sets how many times the grader scores the
|
||||
# run and averages. A task's own tests/test.sh supplies the default when it is unset.
|
||||
#
|
||||
# Needs python 3.11+ on PATH (tomllib, to read the harness registry). Containers have it;
|
||||
# a macOS host running HARBOR_ENV=docker may not — set RACCOON_PYTHON if so.
|
||||
|
||||
@@ -29,6 +32,11 @@ REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
_RJ_SET="${RACCOON_DNS_JAIL+set}"; _RJ_VAL="${RACCOON_DNS_JAIL:-}"
|
||||
_RJA_SET="${RACCOON_DNS_JAIL_ALLOW+set}"; _RJA_VAL="${RACCOON_DNS_JAIL_ALLOW:-}"
|
||||
|
||||
# Same for the grader sample count, under either name (HARBOR_GRADER_SAMPLES is what
|
||||
# harbor-regrade calls it; GRADER_SAMPLES is what the task's tests/test.sh reads).
|
||||
# Captured as one value so a caller's spelling beats .env's, whichever each used.
|
||||
_CALLER_SAMPLES="${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}"
|
||||
|
||||
# Source API key
|
||||
if [ -f "$REPO_ROOT/.env" ]; then
|
||||
set -a
|
||||
@@ -156,7 +164,10 @@ if [ -f "$TASK_DIR/task.toml" ] &&
|
||||
BROWSER_OPTIN=1
|
||||
fi
|
||||
if [ -d "$TASK_DIR/environment" ]; then
|
||||
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
|
||||
# Rename into place: a concurrent run against this task dir must never read the
|
||||
# instant between truncate and write.
|
||||
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin.tmp.$$" &&
|
||||
mv -f "$TASK_DIR/environment/browser-optin.tmp.$$" "$TASK_DIR/environment/browser-optin"
|
||||
fi
|
||||
|
||||
# Preflight: restage the DNS jail script, for the same reason as the marker above.
|
||||
@@ -171,7 +182,8 @@ if [ -d "$TASK_DIR/environment" ]; then
|
||||
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
|
||||
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
|
||||
if [ -f "$DNSJAIL_SRC" ]; then
|
||||
cp "$DNSJAIL_SRC" "$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
|
||||
_dnsjail_dst="$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
|
||||
cp "$DNSJAIL_SRC" "$_dnsjail_dst.tmp.$$" && mv -f "$_dnsjail_dst.tmp.$$" "$_dnsjail_dst"
|
||||
break
|
||||
fi
|
||||
done
|
||||
@@ -255,6 +267,35 @@ FAST_FLAGS=""
|
||||
GRADER_FAST_FLAGS=""
|
||||
[ -n "$FAST_REQUESTED" ] && GRADER_FAST_FLAGS="--verifier-env GRADER_FAST_MODE=true"
|
||||
|
||||
# A task's tests/test.sh is frozen at creation and defaults its own sample count, so
|
||||
# forwarding the var is the only way to change one that already exists.
|
||||
GRADER_SAMPLES_FLAGS=""
|
||||
_SAMPLES="${_CALLER_SAMPLES:-${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}}"
|
||||
if [ -n "$_SAMPLES" ]; then
|
||||
if ! [ "$_SAMPLES" -ge 1 ] 2>/dev/null; then
|
||||
echo "harbor-run: ERROR — GRADER_SAMPLES must be a positive integer (got '$_SAMPLES')." >&2
|
||||
exit 1
|
||||
fi
|
||||
GRADER_SAMPLES_FLAGS="--verifier-env GRADER_SAMPLES=$_SAMPLES"
|
||||
fi
|
||||
|
||||
# The verifier launches its own grader claude, so it names its own call origin
|
||||
# rather than inheriting the surface that launched harbor. An array because the
|
||||
# header value contains a space.
|
||||
GRADER_ORIGIN_FLAGS=()
|
||||
if [ -f "$REPO_ROOT/scripts/lib/call-origin.sh" ]; then
|
||||
. "$REPO_ROOT/scripts/lib/call-origin.sh"
|
||||
_GRADER_METADATA="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
|
||||
# `if`, not `[ ... ] && ...`: an AND-list is this block's last statement, so under
|
||||
# `set -e` an empty value would abort the run rather than just skip the header.
|
||||
if [ -n "$_GRADER_METADATA" ]; then
|
||||
GRADER_ORIGIN_FLAGS=(
|
||||
--verifier-env
|
||||
"ANTHROPIC_CUSTOM_HEADERS=$CALL_ORIGIN_HEADER: $_GRADER_METADATA"
|
||||
)
|
||||
fi
|
||||
fi
|
||||
|
||||
# Environment backend. An explicit HARBOR_ENV always wins (either direction).
|
||||
# Otherwise the default is context-dependent:
|
||||
# - daytona for the internal repo: runs the trial in a cloud sandbox over
|
||||
@@ -429,6 +470,8 @@ harbor run \
|
||||
$EFFORT_FLAGS \
|
||||
$FAST_FLAGS \
|
||||
$GRADER_FAST_FLAGS \
|
||||
$GRADER_SAMPLES_FLAGS \
|
||||
${GRADER_ORIGIN_FLAGS[@]+"${GRADER_ORIGIN_FLAGS[@]}"} \
|
||||
"$@" &
|
||||
HARBOR_PID=$!
|
||||
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
|
||||
|
||||
@@ -82,8 +82,23 @@ multi_agent_v2 = false
|
||||
memories = false
|
||||
external_agent_memory_import = false
|
||||
"""
|
||||
# A provider of our own, not the built-in `openai`: codex reserves built-in provider ids,
|
||||
# and env_http_headers — the only place codex can be told to send the call-origin header —
|
||||
# is a per-provider setting. base_url has to live in the table with it (a provider without
|
||||
# one silently falls back to api.openai.com), so harness_refresh_config_keys refreshes
|
||||
# [model_providers.*] keys as well as root ones.
|
||||
container_config = """
|
||||
openai_base_url = "${OPENAI_BASE_URL}"
|
||||
model_provider = "llm-proxy"
|
||||
|
||||
[model_providers.llm-proxy]
|
||||
name = "LLM proxy"
|
||||
base_url = "${OPENAI_BASE_URL}"
|
||||
# A custom provider reads its key from this env var and never from auth.json, so it
|
||||
# names the one key .env actually carries. That makes the key live per launch rather
|
||||
# than baked at container create — better than the auth-file path it replaces.
|
||||
env_key = "ANTHROPIC_API_KEY"
|
||||
wire_api = "responses"
|
||||
env_http_headers = { "X-Surge-Client-Metadata" = "LLM_CALL_METADATA" }
|
||||
"""
|
||||
explore_config = """
|
||||
[hooks]
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
/** The task's holistic-rubric template. Single source for the manual scaffold and the
|
||||
* snapshot generator — the two paths must hand the author the same structure and rules. */
|
||||
export const HOLISTIC_RUBRIC_SCAFFOLD = `# Holistic Rubric — <task-slug>
|
||||
|
||||
The shared grading standard (\`task-shared/grading-standard.md\`, embedded in
|
||||
\`tests/grader-system-prompt-consolidated.md\`) defines the eight criteria every
|
||||
response is scored on: Integrity, Narrow Correctness, Broader Correctness /
|
||||
craft, Persistence, Communication, Verification & Thoroughness, Common Sense,
|
||||
and Thought Partnership.
|
||||
|
||||
This file is the task's holistic rubric. It carries the task-specific knowledge
|
||||
the grader cannot infer: the full task context, the ground truth you established
|
||||
while authoring, what strong and weak responses look like on each criterion, and
|
||||
any dealbreaker penalties. This document must stand alone. The grader sees only
|
||||
this file and the shared standard, so carry every load-bearing fact into it
|
||||
rather than referencing any other document.
|
||||
|
||||
Replace each bracketed section. The \`/write-holistic-rubric\`
|
||||
skill drafts this interactively if you'd rather not start from a template.
|
||||
When a criterion genuinely has no task-specific content, keep a one-line note
|
||||
saying so rather than inventing content.
|
||||
|
||||
## Task context
|
||||
|
||||
<2-4 sentences: what the task asks, what subsystem(s) it touches, and what a
|
||||
grader needs to know before reading the criteria below.>
|
||||
|
||||
## Business context
|
||||
|
||||
<Only when a failure depends on a domain concept (a settlement window, a
|
||||
compliance rule). Delete this section otherwise.>
|
||||
|
||||
## Ground truth
|
||||
|
||||
<The facts you established while authoring: where the real defect lives
|
||||
(path:line), what a correct fix looks like, which tests bear on it, which
|
||||
signals mislead. The grader trusts this section over its own reading.>
|
||||
|
||||
## Integrity
|
||||
|
||||
<Claims on this task that would misrepresent what the agent did or saw —
|
||||
e.g. asserting a file says X after reading it say Y. Charge only on an
|
||||
observable basis.>
|
||||
|
||||
## Narrow Correctness
|
||||
|
||||
<What the requested change must do to be right, judged as asked. Anchors a
|
||||
working result must satisfy, checkable by path:line.>
|
||||
|
||||
## Broader Correctness / the craft of software engineering
|
||||
|
||||
<Craft expectations specific to this codebase: patterns to follow, tests to
|
||||
add, places a shortcut would rot.>
|
||||
|
||||
## Persistence
|
||||
|
||||
<What "kept going appropriately" looks like here: the dead ends worth
|
||||
exhausting, and where stopping to ask is the better call.>
|
||||
|
||||
## Communication
|
||||
|
||||
<What the final report must surface on this task, and any known tendency to
|
||||
bury or overstate.>
|
||||
|
||||
## Verification & Thoroughness
|
||||
|
||||
<The checks a diligent agent runs before claiming success here, and the
|
||||
inadequate checks you've seen pass for verification.>
|
||||
|
||||
## Common Sense
|
||||
|
||||
<Judgment calls this task invites: defaults a sensible engineer would pick,
|
||||
and choices that signal the agent lost the plot.>
|
||||
|
||||
## Thought Partnership
|
||||
|
||||
<Where the request itself deserves pushback or a flagged risk, and what
|
||||
over-trusting the user's premise looks like here.>
|
||||
|
||||
## Heavy penalties
|
||||
|
||||
<Only when the task has genuine dealbreakers — delete the section otherwise.
|
||||
Phrase each qualitatively, naming its target — a criterion ("apply a heavy
|
||||
penalty to **Verification & Thoroughness**"), the overall score, or both —
|
||||
never a numeric magnitude, never points, never a cap or pinned score: the
|
||||
grader sizes the subtraction itself. Always state the behavior that does NOT trip the penalty.
|
||||
Never describe how criteria combine into an overall score.>
|
||||
`;
|
||||
34
worker-toolkit-potion-polyglot/scripts/lib/call-origin.sh
Normal file
34
worker-toolkit-potion-polyglot/scripts/lib/call-origin.sh
Normal file
@@ -0,0 +1,34 @@
|
||||
# shellcheck shell=bash
|
||||
# call-origin.sh — build the X-Surge-Client-Metadata header value.
|
||||
#
|
||||
# Which surface a proxy call came from (a trial agent, the grader, Explore, a
|
||||
# dev box). Separate from llm-proxy-env.sh, which carries the project id and the
|
||||
# proxy routes: those are platform-internal, this is not, so this file is the
|
||||
# half that ships in the worker toolkit — worker runs go through the same proxy
|
||||
# and are attributed the same way.
|
||||
#
|
||||
# . scripts/lib/call-origin.sh
|
||||
# meta="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
|
||||
#
|
||||
# The proxy rejects the WHOLE CALL over a malformed metadata header (400
|
||||
# invalid_client_metadata), so an origin that is not a plain slug yields an
|
||||
# empty string and the caller sends no header at all: losing attribution beats
|
||||
# failing the call.
|
||||
|
||||
CALL_ORIGIN_HEADER="X-Surge-Client-Metadata"
|
||||
# An unlabelled call is still a real call, so it gets a bucket rather than no
|
||||
# header: a missing origin in the audit log then means an unplumbed surface.
|
||||
DEFAULT_CALL_ORIGIN="local"
|
||||
|
||||
# Compact JSON for the header, or empty when LLM_CALL_ORIGIN is unusable.
|
||||
# Only a slug matching this pattern is ever interpolated, so nothing needs
|
||||
# JSON-escaping and this stays dependency-free (it is sourced in worker
|
||||
# containers, which have no python).
|
||||
call_origin_metadata() {
|
||||
local origin="${LLM_CALL_ORIGIN:-$DEFAULT_CALL_ORIGIN}"
|
||||
case "$origin" in
|
||||
"" | *[!a-z0-9._-]* | [!a-z0-9]*) return 0 ;;
|
||||
esac
|
||||
[ "${#origin}" -le 64 ] || return 0
|
||||
printf '{"origin":"%s"}' "$origin"
|
||||
}
|
||||
@@ -190,16 +190,39 @@ if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1
|
||||
raise SystemExit(1)
|
||||
text = os.path.expandvars(text)
|
||||
|
||||
# Root keys, plus keys inside a [model_providers.*] table: codex reserves its built-in
|
||||
# provider ids, so the proxy URL it must follow lives in a provider table, not at the
|
||||
# root. Every other table, [hooks] on the explore surface included, is left alone.
|
||||
REFRESHABLE_TABLE = re.compile(r"\[model_providers\.[^]]+\]$")
|
||||
wanted = []
|
||||
section = None
|
||||
for line in text.splitlines():
|
||||
if line.lstrip().startswith("["):
|
||||
break
|
||||
m = re.match(r"\s*([A-Za-z0-9_-]+)\s*=", line)
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("["):
|
||||
section = stripped if REFRESHABLE_TABLE.match(stripped) else False
|
||||
continue
|
||||
if section is False:
|
||||
continue
|
||||
m = re.match(r"\s*\"?([A-Za-z0-9_.-]+)\"?\s*=", line)
|
||||
if m:
|
||||
wanted.append((m.group(1), line.rstrip()))
|
||||
wanted.append((section, m.group(1), line.rstrip()))
|
||||
if not wanted:
|
||||
raise SystemExit(0)
|
||||
|
||||
|
||||
def section_path(header):
|
||||
"""[model_providers.llm-proxy] -> ("model_providers", "llm-proxy")."""
|
||||
return tuple(header.strip("[]").split("."))
|
||||
|
||||
|
||||
def lookup(doc, header, key):
|
||||
"""The value a parsed config holds for a wanted key, or KeyError."""
|
||||
node = doc
|
||||
if header:
|
||||
for part in section_path(header):
|
||||
node = node[part]
|
||||
return node[key]
|
||||
|
||||
mode = None
|
||||
if os.path.exists(target):
|
||||
try:
|
||||
@@ -208,23 +231,52 @@ if os.path.exists(target):
|
||||
mode = os.stat(target).st_mode & 0o777
|
||||
except OSError:
|
||||
raise SystemExit(1)
|
||||
# Everything from the first table header on belongs to a table. A key appended after
|
||||
# one is reparented into it, so both the search and the insert stay above the line.
|
||||
root_end = next((i for i, l in enumerate(lines) if l.lstrip().startswith("[")), len(lines))
|
||||
changed = False
|
||||
for key, line in wanted:
|
||||
# The quoted spelling is the same key: replacing it beats adding a duplicate.
|
||||
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
|
||||
at = next((i for i in range(root_end) if pat.match(lines[i])), None)
|
||||
def span(header):
|
||||
"""The line range a section owns, or None when the file has no such section.
|
||||
|
||||
Root is everything above the first table header: a key appended below one
|
||||
would be reparented into it, so searches and inserts stay inside the span.
|
||||
"""
|
||||
heads = [i for i, l in enumerate(lines) if l.lstrip().startswith("[")]
|
||||
if header is None:
|
||||
return 0, (heads[0] if heads else len(lines))
|
||||
at = next((i for i in heads if lines[i].strip() == header), None)
|
||||
if at is None:
|
||||
if root_end < len(lines) and lines[root_end].strip():
|
||||
lines.insert(root_end, "")
|
||||
lines.insert(root_end, line)
|
||||
root_end += 1
|
||||
changed = True
|
||||
elif lines[at] != line:
|
||||
lines[at] = line
|
||||
return None
|
||||
after = next((i for i in heads if i > at), len(lines))
|
||||
return at + 1, after
|
||||
|
||||
# Grouped, root first, so a section this file lacks can be written whole.
|
||||
grouped = {}
|
||||
for header, key, line in wanted:
|
||||
grouped.setdefault(header, []).append((key, line))
|
||||
ordered = sorted(grouped, key=lambda h: (h is not None, h or ""))
|
||||
|
||||
changed = False
|
||||
for header in ordered:
|
||||
if span(header) is None:
|
||||
# A config written before this section existed. Write the whole table
|
||||
# rather than leave a root key naming a provider that is not there.
|
||||
if lines and lines[-1].strip():
|
||||
lines.append("")
|
||||
lines.append(header)
|
||||
lines.extend(line for _, line in grouped[header])
|
||||
changed = True
|
||||
continue
|
||||
for key, line in grouped[header]:
|
||||
# Re-read the span: an insert for an earlier key moved it.
|
||||
start, end = span(header)
|
||||
# The quoted spelling is the same key: replace rather than duplicate.
|
||||
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
|
||||
at = next((i for i in range(start, end) if pat.match(lines[i])), None)
|
||||
if at is None:
|
||||
if end < len(lines) and lines[end].strip():
|
||||
lines.insert(end, "")
|
||||
lines.insert(end, line)
|
||||
changed = True
|
||||
elif lines[at] != line:
|
||||
lines[at] = line
|
||||
changed = True
|
||||
if not changed:
|
||||
raise SystemExit(0)
|
||||
out = "\n".join(lines).rstrip("\n") + "\n"
|
||||
@@ -238,9 +290,22 @@ try:
|
||||
except tomllib.TOMLDecodeError:
|
||||
raise SystemExit(1)
|
||||
# Parsing is not enough: a line edit can land inside a multi-line value, which still
|
||||
# parses while leaving the key unset. Require every key to have reached the root.
|
||||
if doc != {**doc, **tomllib.loads("\n".join(line for _, line in wanted))}:
|
||||
raise SystemExit(1)
|
||||
# parses while leaving the key unset. Require every key to have landed on the value the
|
||||
# blob asks for, in its own section — skipping sections this file does not carry.
|
||||
blob_doc = tomllib.loads(text)
|
||||
for header, key, _ in wanted:
|
||||
try:
|
||||
expected = lookup(blob_doc, header, key)
|
||||
except (KeyError, TypeError):
|
||||
raise SystemExit(1)
|
||||
try:
|
||||
got = lookup(doc, header, key)
|
||||
except (KeyError, TypeError):
|
||||
if header is None:
|
||||
raise SystemExit(1)
|
||||
continue
|
||||
if got != expected:
|
||||
raise SystemExit(1)
|
||||
|
||||
# Pid-suffixed: two launches at once must not write the same scratch path.
|
||||
tmp = target + ".raccoon-tmp." + str(os.getpid())
|
||||
|
||||
@@ -193,21 +193,48 @@ export const REFERENCE_RUN_INPUTS = Object.freeze([
|
||||
* holistic-rubric files the task directory carries (tests/holistic-rubric.md
|
||||
* on current tasks, tests/grader-guidance-consolidated.md on tasks created
|
||||
* before the rename, plus the legacy-era plain-named file when a task
|
||||
* authored on an earlier generation carries one), plus the atomic-rubric
|
||||
* files the rubric detectors assess (tests/atomic-rubric.yaml, the pre-rename
|
||||
* tests/rubrics.yaml, and tests/grader-context.md). An absent file hashes to
|
||||
* authored on an earlier generation carries one). An absent file hashes to
|
||||
* null on both sides and never diffs. Compared by content.
|
||||
*
|
||||
* The atomic-rubric files are deliberately NOT in this set: fifteen of the
|
||||
* seventeen detectors never open them, so writing an atomic rubric after
|
||||
* running the detectors would stale every one of those reports over files
|
||||
* they never read. The two that do read them use
|
||||
* {@link RUBRIC_DETECTOR_REPORT_INPUTS}.
|
||||
*/
|
||||
export const DETECTOR_REPORT_INPUTS = Object.freeze([
|
||||
'prompt',
|
||||
'graderGuidance',
|
||||
'graderGuidanceConsolidated',
|
||||
'holisticRubric',
|
||||
] as const satisfies readonly TaskInputName[]);
|
||||
|
||||
/**
|
||||
* The inputs the two rubric detectors assess: {@link DETECTOR_REPORT_INPUTS}
|
||||
* plus the atomic-rubric package (tests/atomic-rubric.yaml, the pre-rename
|
||||
* tests/rubrics.yaml, and tests/grader-context.md), which they compare
|
||||
* against the holistic rubric.
|
||||
*/
|
||||
export const RUBRIC_DETECTOR_REPORT_INPUTS = Object.freeze([
|
||||
...DETECTOR_REPORT_INPUTS,
|
||||
'atomicRubric',
|
||||
'rubricsYaml',
|
||||
'graderContext',
|
||||
] as const satisfies readonly TaskInputName[]);
|
||||
|
||||
/** Detectors that read the atomic-rubric package, and so are staled by it. */
|
||||
export const ATOMIC_RUBRIC_DETECTORS: readonly string[] = Object.freeze([
|
||||
'detector-rubric-coverage',
|
||||
'detector-rubric-form',
|
||||
]);
|
||||
|
||||
/** The input set a named detector's report is judged against. */
|
||||
export function detectorReportInputs(detectorName: string): readonly TaskInputName[] {
|
||||
return ATOMIC_RUBRIC_DETECTORS.includes(detectorName)
|
||||
? RUBRIC_DETECTOR_REPORT_INPUTS
|
||||
: DETECTOR_REPORT_INPUTS;
|
||||
}
|
||||
|
||||
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
|
||||
function sha256File(filePath: string): string | null {
|
||||
if (!existsSync(filePath)) return null;
|
||||
|
||||
96
worker-toolkit-potion-polyglot/scripts/lib/resolve-pin.sh
Normal file
96
worker-toolkit-potion-polyglot/scripts/lib/resolve-pin.sh
Normal file
@@ -0,0 +1,96 @@
|
||||
# shellcheck shell=bash
|
||||
#
|
||||
# resolve_pin — turn a task's pinned commit into a SHA that exists in the repo,
|
||||
# translating through a commit map when history has been rewritten under it.
|
||||
#
|
||||
# A task pins a commit in task.toml. If that repo's history is later rewritten
|
||||
# (to strip something that should never have shipped, say), every rewritten
|
||||
# commit gets a new SHA and the pin stops resolving — including on machines we
|
||||
# cannot reach, holding tasks we cannot edit. A commit map lets those pins keep
|
||||
# working: `<old-sha> <new-sha>` per line, at task-shared/commit-maps/<member>.map,
|
||||
# <member> being the task's `repo` key — a standalone toolkit checks its repo out
|
||||
# at repo/, so the directory name is not the member name and cannot be the key.
|
||||
#
|
||||
# The map is only consulted when the pin does not resolve, so it carries only
|
||||
# rewritten commits — an unchanged commit resolves on its own and its identity
|
||||
# row could never be read.
|
||||
#
|
||||
# Usage (source, then call):
|
||||
# . "$(dirname "$0")/lib/resolve-pin.sh"
|
||||
# sha=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
|
||||
#
|
||||
# Writes the resolved SHA to stdout, notes on stderr. Returns non-zero if the
|
||||
# pin cannot be resolved, having explained why.
|
||||
|
||||
RESOLVE_PIN_MAX_HOPS="${RESOLVE_PIN_MAX_HOPS:-25}"
|
||||
_RESOLVE_PIN_ZERO='0000000000000000000000000000000000000000'
|
||||
|
||||
# Look one hop: echo the successor of $1 in map $2, or nothing. Fails if the
|
||||
# prefix is ambiguous, which would otherwise pick an arbitrary commit.
|
||||
_resolve_pin_hop() {
|
||||
local from="$1" map="$2" hits
|
||||
hits=$(awk -v p="$from" '
|
||||
/^#/ || NF < 2 { next }
|
||||
index($1, p) == 1 { print $2 }
|
||||
' "$map" | sort -u)
|
||||
[ -z "$hits" ] && return 1
|
||||
if [ "$(printf '%s\n' "$hits" | wc -l | tr -d ' ')" -gt 1 ]; then
|
||||
echo " pin $from is ambiguous in $(basename "$map") — use a longer SHA" >&2
|
||||
return 2
|
||||
fi
|
||||
printf '%s\n' "$hits"
|
||||
}
|
||||
|
||||
resolve_pin() {
|
||||
local repo_dir="$1" commit="$2" map_dir="${3:-}" key="${4:-}" sha map cur hops next rc
|
||||
|
||||
# Present in the repo: nothing to translate.
|
||||
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$commit^{commit}" 2>/dev/null); then
|
||||
printf '%s\n' "$sha"
|
||||
return 0
|
||||
fi
|
||||
|
||||
map=""
|
||||
if [ -n "$map_dir" ]; then
|
||||
if [ -n "$key" ] && [ -f "$map_dir/$key.map" ]; then
|
||||
map="$map_dir/$key.map"
|
||||
elif [ -f "$map_dir/$(basename "$repo_dir").map" ]; then
|
||||
map="$map_dir/$(basename "$repo_dir").map"
|
||||
fi
|
||||
fi
|
||||
if [ -z "$map" ]; then
|
||||
echo "Error: pinned commit $commit is not in $repo_dir, and no commit map is available." >&2
|
||||
echo " The repo may be a shallow or partial copy — try a full clone." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Follow the chain: a commit rewritten more than once maps forward a hop per
|
||||
# rewrite, so keep going until the SHA exists or the trail ends.
|
||||
cur="$commit"
|
||||
hops=0
|
||||
while [ "$hops" -lt "$RESOLVE_PIN_MAX_HOPS" ]; do
|
||||
next=$(_resolve_pin_hop "$cur" "$map"); rc=$?
|
||||
[ "$rc" -eq 2 ] && return 1
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo "Error: pinned commit $commit is not in $repo_dir and is not in $(basename "$map")." >&2
|
||||
echo " It predates the map, or came from a repo copy this toolkit was not built from." >&2
|
||||
return 1
|
||||
fi
|
||||
if [ "$next" = "$_RESOLVE_PIN_ZERO" ]; then
|
||||
echo "Error: pinned commit $commit was deleted by a history rewrite, not rewritten." >&2
|
||||
echo " Re-pin this task to a commit that still exists." >&2
|
||||
return 1
|
||||
fi
|
||||
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$next^{commit}" 2>/dev/null); then
|
||||
echo " Pin $commit was rewritten; using $sha" >&2
|
||||
printf '%s\n' "$sha"
|
||||
return 0
|
||||
fi
|
||||
cur="$next"
|
||||
hops=$((hops + 1))
|
||||
done
|
||||
|
||||
echo "Error: pinned commit $commit did not settle after $RESOLVE_PIN_MAX_HOPS hops." >&2
|
||||
echo " $(basename "$map") may contain a cycle." >&2
|
||||
return 1
|
||||
}
|
||||
@@ -34,4 +34,21 @@ _scripts_dir="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
|
||||
|
||||
# No args is a valid call: refresh only, for a lifecycle hook.
|
||||
[ "$#" -gt 0 ] || exit 0
|
||||
|
||||
# Outside the subshell, because these have to reach the exec'd command: codex now reads
|
||||
# its key from $ANTHROPIC_API_KEY per request, and a non-login shell sourced neither
|
||||
# .bashrc (the key, the call origin) nor the profile that puts the CLI on PATH.
|
||||
# Failures stay swallowed — an unreadable .env must not stop the agent starting.
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
if [ -f "${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
|
||||
set -a
|
||||
# shellcheck disable=SC1090
|
||||
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true
|
||||
set +a
|
||||
fi
|
||||
if [ -f "$HOME/.raccoon-call-origin" ]; then
|
||||
# shellcheck disable=SC1091
|
||||
. "$HOME/.raccoon-call-origin" 2>/dev/null || true
|
||||
fi
|
||||
|
||||
exec "$@"
|
||||
|
||||
@@ -151,6 +151,19 @@ harness_install_launchers() {
|
||||
#!/bin/bash
|
||||
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
|
||||
set -euo pipefail
|
||||
# A harness that reads its key from \$ENV per request needs .env in its environment,
|
||||
# and only an interactive shell sources .bashrc — which is also where PATH picks up
|
||||
# ~/.local/bin, where the CLI itself lives. Both are set here so a launch works the
|
||||
# same either way, with the key .env holds right now.
|
||||
export PATH="\$HOME/.local/bin:\$PATH"
|
||||
if [ -f "\${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
|
||||
set -a
|
||||
. "\${RACCOON_ENV_FILE:-/workspace/.env}"
|
||||
set +a
|
||||
fi
|
||||
if [ -f "\$HOME/.raccoon-call-origin" ]; then
|
||||
. "\$HOME/.raccoon-call-origin"
|
||||
fi
|
||||
if [ -f "$note_src" ]; then
|
||||
RACCOON_TOOLSET_NOTE="\$(sed "s#/opt/agent-cli#$agent_cli_dir#g" "$note_src")"
|
||||
else
|
||||
|
||||
@@ -24,6 +24,7 @@ import { hideBin } from 'yargs/helpers';
|
||||
|
||||
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
|
||||
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
|
||||
import { HOLISTIC_RUBRIC_SCAFFOLD } from './holistic-rubric-scaffold';
|
||||
import { copyTree } from './lib/copy-tree';
|
||||
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
|
||||
|
||||
@@ -747,40 +748,23 @@ if (lastUserMessage) {
|
||||
// --- Scaffold holistic-rubric.md ---
|
||||
|
||||
const holisticRubricMd = `<!--
|
||||
HOLISTIC RUBRIC — the file trials grade against. Run
|
||||
/write-holistic-rubric
|
||||
to draft it interactively, or point Claude Code at this file,
|
||||
session-full.jsonl, and task-shared/grading-standard.md.
|
||||
|
||||
Snapshot: ${basename(snapshotDir)}
|
||||
Session: ${metadata.session_uuid}
|
||||
Repo: ${metadata.remote_url}
|
||||
Commit: ${metadata.commit}
|
||||
|
||||
## What happened in the snapshot conversation
|
||||
What happened in the snapshot conversation
|
||||
|
||||
The worker was trying to: ${annotation.what_trying}
|
||||
They hoped Claude would: ${annotation.what_hoping}
|
||||
Instead, Claude: ${annotation.what_happened}
|
||||
|
||||
## What this file contains
|
||||
|
||||
The eight-criterion Grading Standard
|
||||
(task-shared/grading-standard.md, embedded in
|
||||
tests/grader-system-prompt-consolidated.md) defines Integrity, Narrow
|
||||
Correctness, Broader Correctness / craft, Persistence, Communication,
|
||||
Verification & Thoroughness, Common Sense, and Thought Partnership. This
|
||||
file adds the task-specific knowledge the grader cannot infer: full task
|
||||
context, the ground truth you established, what strong and weak responses
|
||||
look like per criterion, and any dealbreaker penalties — stated as 0.0-1.0
|
||||
fraction subtractions with a named criterion target, never points, never
|
||||
caps. The document must stand alone: the grader sees only it and the
|
||||
shared standard.
|
||||
Draft this file with /write-holistic-rubric, or point your agent at it,
|
||||
session-full.jsonl, and task-shared/grading-standard.md. Delete this comment
|
||||
when you are done.
|
||||
-->
|
||||
|
||||
<!-- Replace EVERYTHING in this file with the actual holistic rubric,
|
||||
including the instructions above. -->
|
||||
`;
|
||||
${HOLISTIC_RUBRIC_SCAFFOLD}`;
|
||||
|
||||
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
|
||||
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');
|
||||
|
||||
@@ -43,6 +43,10 @@ _DISABLED_SUFFIX = ".snapshot-seeded-disabled"
|
||||
# this beta header (the proxy's server-side alias for the suffixed id was
|
||||
# dropped; the literal id now 400s and the claude CLI hangs retrying).
|
||||
_CONTEXT_1M_BETA_HEADER = "anthropic-beta: context-1m-2025-08-07"
|
||||
# Spelled out rather than imported from llm_proxy_env: this module ships to the
|
||||
# worker toolkit, where a cross-import would fail at import time.
|
||||
_CLIENT_METADATA_HEADER = "X-Surge-Client-Metadata"
|
||||
_TRIAL_CALL_METADATA = '{"origin":"harbor-trial"}'
|
||||
|
||||
|
||||
def _strip_1m_suffix(model: str) -> tuple[str, bool]:
|
||||
@@ -68,6 +72,20 @@ def _with_1m_beta_header(existing: str | None) -> str:
|
||||
"""Merge the 1M-context beta header into an ANTHROPIC_CUSTOM_HEADERS value."""
|
||||
return _merge_custom_headers(existing, _CONTEXT_1M_BETA_HEADER)
|
||||
|
||||
|
||||
def _with_trial_origin(existing: str | None) -> str:
|
||||
"""Restate the call origin as this trial, replacing the launching surface's.
|
||||
|
||||
Replaces rather than appends: two values of one header name is not a
|
||||
merge the proxy can read.
|
||||
"""
|
||||
kept = [
|
||||
line
|
||||
for line in (existing or "").split("\n")
|
||||
if line.strip() and not line.startswith(f"{_CLIENT_METADATA_HEADER}:")
|
||||
]
|
||||
return "\n".join([*kept, f"{_CLIENT_METADATA_HEADER}: {_TRIAL_CALL_METADATA}"])
|
||||
|
||||
# Fast mode (claude's /fast), opted in per-trial via `--ak fast_mode=true`
|
||||
# (harbor-run --fast). The --settings flag is the only headless opt-in, and it
|
||||
# doubles as the availability override for a proxy base URL: claude probes
|
||||
@@ -581,6 +599,8 @@ class SnapshotClaudeCode(PreinstalledClaudeCode):
|
||||
os.environ.get("ANTHROPIC_CUSTOM_HEADERS") if on_proxy else None,
|
||||
self._extra_env.get("ANTHROPIC_CUSTOM_HEADERS"),
|
||||
)
|
||||
if on_proxy:
|
||||
header = _with_trial_origin(header)
|
||||
if self.model_name:
|
||||
# "[1m]" normalization happened once in PreinstalledClaudeCode.__init__
|
||||
# (shared by all Claude agent classes); by here model_name is the plain
|
||||
|
||||
@@ -16,10 +16,10 @@ import yargs from 'yargs';
|
||||
import { hideBin } from 'yargs/helpers';
|
||||
|
||||
import {
|
||||
DETECTOR_REPORT_INPUTS,
|
||||
INPUT_CHECKSUMS_FILENAME,
|
||||
REFERENCE_RUN_INPUTS,
|
||||
captureTaskInputs,
|
||||
detectorReportInputs,
|
||||
diffTaskInputs,
|
||||
readTaskInputChecksums,
|
||||
} from './lib/input-checksums';
|
||||
@@ -581,7 +581,8 @@ export function validateTask(
|
||||
unstamped.push(report);
|
||||
continue;
|
||||
}
|
||||
const changed = diffTaskInputs(stamp, currentInputs, DETECTOR_REPORT_INPUTS);
|
||||
const detectorName = report.replace(/\.md$/, '');
|
||||
const changed = diffTaskInputs(stamp, currentInputs, detectorReportInputs(detectorName));
|
||||
staleness.detectors.push({
|
||||
report,
|
||||
status: changed.length > 0 ? 'stale' : 'fresh',
|
||||
@@ -593,14 +594,21 @@ export function validateTask(
|
||||
if (changed.length > 0) staleStamped.push({ report, changed });
|
||||
}
|
||||
if (staleStamped.length > 0) {
|
||||
const changedLabels = [...new Set(staleStamped.flatMap((s) => s.changed))];
|
||||
// Grouped by cause, not unioned across reports: a report is only ever
|
||||
// stale on the inputs its own detector reads.
|
||||
const byCause = new Map<string, string[]>();
|
||||
for (const { report, changed } of staleStamped) {
|
||||
const cause = changed.join(' and ');
|
||||
byCause.set(cause, [...(byCause.get(cause) ?? []), report]);
|
||||
}
|
||||
const causes = [...byCause].map(([cause, reports]) => `${cause} → ${reports.join(', ')}`);
|
||||
const staleNames = staleStamped.map((s) => s.report);
|
||||
logger.warn(
|
||||
{ stale: staleNames, changed: changedLabels },
|
||||
`You modified your ${changedLabels.join(' and ')} after these detector reports were written, so they assessed an older revision of this task: ${staleNames.join(', ')}. Re-run those detectors (e.g. /${staleNames[0].replace(/\.md$/, '')}) and re-package.`
|
||||
{ stale: staleStamped },
|
||||
`These detector reports assessed an older revision of this task — you modified an input each one reads after it was written: ${causes.join('; ')}. Re-run those detectors (e.g. /${staleNames[0].replace(/\.md$/, '')}) and re-package.`
|
||||
);
|
||||
warnings.push(
|
||||
`${staleStamped.length} stale detector report(s) — ${changedLabels.join(', ')} changed since they were written: ${staleNames.join(', ')} — ` +
|
||||
`${staleStamped.length} stale detector report(s) — ${causes.join('; ')} — ` +
|
||||
're-run them against the current task'
|
||||
);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user