ren worker folder adding orig, mv new one into root
This commit is contained in:
@@ -3,6 +3,8 @@
|
||||
*
|
||||
* Usage:
|
||||
* npx tsx scripts/copy-reference-run.ts <trial-path> [trial-path...]
|
||||
* npx tsx scripts/copy-reference-run.ts --rubric-regrade <trial-path>
|
||||
* npx tsx scripts/copy-reference-run.ts --supersede <old-run-dir> <trial-path>
|
||||
*
|
||||
* Examples:
|
||||
* # Copy a single trial
|
||||
@@ -28,6 +30,8 @@
|
||||
* indirection. Its location is per harness: Claude Code writes
|
||||
* `agent/sessions/projects/-workspace/<id>.jsonl` with the id printed in
|
||||
* `agent/claude-code.txt`; codex writes `agent/sessions/<Y>/<M>/<D>/rollout-*.jsonl`.
|
||||
* - verifier/test-stdout.txt (the verifier's console output, where the reward is printed)
|
||||
* - verifier/deterministic-signals.txt, verifier/rubric-grade(-<N>).json
|
||||
* - config.json, result.json, trial.log
|
||||
* - input-checksums.json — sha256 checksums of the task inputs the run was
|
||||
* generated against (prompt, session snapshot, workspace patch, gitref),
|
||||
@@ -37,6 +41,21 @@
|
||||
* otherwise captured here at copy time as a fallback (capturedBy:
|
||||
* 'copy').
|
||||
*
|
||||
* --rubric-regrade files an atomic-rubric regrade into rubric-regrades/<run>/
|
||||
* instead: a second grade of a run that keeps its own, under the name the trial
|
||||
* itself records for the run it graded. The trial is copied verbatim — verifier/
|
||||
* and all — so a stored grade has the same shape whoever stored it.
|
||||
*
|
||||
* --supersede is the other direction: a holistic regrade replaces a run's grade,
|
||||
* and the new copy lands under a name minted from the NEW reward and trial id, so
|
||||
* adopting it means removing the directory it supersedes. Deleting is deliberate
|
||||
* over merging into the old directory: a regrade grades a different number of
|
||||
* samples than the run often did, so a merge would leave grade-2.md/grade-3.md
|
||||
* from the previous grade beside the new grade-1.md with nothing marking the
|
||||
* generation. A swapped directory holds exactly one grade by construction. The
|
||||
* cost is the old run's session.jsonl, which a replay cannot reproduce; the
|
||||
* trajectory it would be needed to rebuild travels with the replay itself.
|
||||
*
|
||||
* What is NOT copied:
|
||||
* - agent/sessions/, agent/setup/ (workspace data — the resumable JSONL
|
||||
* is hoisted out as `session.jsonl` above; everything else here is
|
||||
@@ -51,11 +70,12 @@ import {
|
||||
mkdirSync,
|
||||
readdirSync,
|
||||
readFileSync,
|
||||
renameSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
writeFileSync,
|
||||
} from 'fs';
|
||||
import { basename, join } from 'path';
|
||||
import { basename, join, resolve } from 'path';
|
||||
|
||||
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
|
||||
import { copyPath, copyTree } from './lib/copy-tree';
|
||||
@@ -99,6 +119,36 @@ function newestRollout(sessionsDir: string): string | null {
|
||||
return found[0].path;
|
||||
}
|
||||
|
||||
/** What a replay trial records about itself: the run it graded, and whether the
|
||||
* grade came from a rubric grader mode rather than the holistic one. */
|
||||
function readReplayProvenance(trialPath: string): {
|
||||
sourceRun: string | null;
|
||||
rubricMode: boolean;
|
||||
} {
|
||||
let config: Record<string, unknown> | undefined;
|
||||
try {
|
||||
const parsed = JSON.parse(readFileSync(join(trialPath, 'result.json'), 'utf-8')) as {
|
||||
config?: Record<string, unknown>;
|
||||
};
|
||||
config = parsed.config;
|
||||
} catch {
|
||||
config = undefined;
|
||||
}
|
||||
const agent = (config?.agent ?? {}) as { kwargs?: { reference_run_dir?: unknown } };
|
||||
const src = agent.kwargs?.reference_run_dir;
|
||||
const verifier = (config?.verifier ?? {}) as { env?: Record<string, unknown> };
|
||||
const graderMode = verifier.env?.GRADER_MODE;
|
||||
// Either marker is enough: the env records the mode harbor was handed, the
|
||||
// rubric-grade file records what the grader actually produced.
|
||||
const rubricMode =
|
||||
(typeof graderMode === 'string' && graderMode.startsWith('rubric-')) ||
|
||||
existsSync(join(trialPath, 'verifier', 'rubric-grade.json'));
|
||||
return {
|
||||
sourceRun: typeof src === 'string' && src.length > 0 ? basename(src.replace(/\/+$/, '')) : null,
|
||||
rubricMode,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the task dir from the trial's result.json `task_name` — the FULL,
|
||||
* unambiguous slug. Harbor TRUNCATES long task names in the trial DIRNAME, so two
|
||||
@@ -127,7 +177,11 @@ function findTaskDirByResultJson(trialPath: string): string | null {
|
||||
return existsSync(dir) ? dir : null;
|
||||
}
|
||||
|
||||
function copyTrial(trialPath: string, destName?: string) {
|
||||
function copyTrial(
|
||||
trialPath: string,
|
||||
opts: { destName?: string; rubricRegrade?: boolean; supersede?: string } = {}
|
||||
) {
|
||||
const { destName, rubricRegrade = false, supersede } = opts;
|
||||
trialPath = trialPath.replace(/\/$/, '');
|
||||
|
||||
if (!existsSync(trialPath)) {
|
||||
@@ -194,147 +248,198 @@ function copyTrial(trialPath: string, destName?: string) {
|
||||
// is copied to exactly that name instead of the minted reward-<r>-<id> —
|
||||
// used by the SxS pipeline so task.toml preference_rollouts, this dir, and
|
||||
// the publish-manifest run_id stay byte-identical by construction.
|
||||
const dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
|
||||
|
||||
if (existsSync(dest)) {
|
||||
console.warn(`Warning: ${dest} already exists, overwriting`);
|
||||
rmSync(dest, { recursive: true });
|
||||
}
|
||||
|
||||
mkdirSync(dest, { recursive: true });
|
||||
|
||||
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
|
||||
// companion reward-correctness.txt (N/A by design) and the machine-readable
|
||||
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
|
||||
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
|
||||
// (the source of truth grade.md/reward.txt are rendered from), the
|
||||
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
|
||||
// render-stderr(-<N>).log, and grader-samples.txt.
|
||||
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
|
||||
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
|
||||
// every per-sample record.)
|
||||
//
|
||||
// signals-status.txt qualifies the deterministic signals the grader was fed:
|
||||
// it records whether every deterministic check actually produced a verdict
|
||||
// ("ok") or one or more was killed before finishing ("degraded" — the grade
|
||||
// is then NOT fully signal-backed). Without it a copied run is
|
||||
// indistinguishable from a run whose checks all passed, so it must travel
|
||||
// with the reward files.
|
||||
const verifierDir = join(trialPath, 'verifier');
|
||||
if (existsSync(verifierDir)) {
|
||||
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
|
||||
if (!entry.isFile()) continue;
|
||||
const f = entry.name;
|
||||
if (
|
||||
f === 'reward.txt' ||
|
||||
f === 'reward-correctness.txt' ||
|
||||
f === 'reward.json' ||
|
||||
f === 'signals-status.txt' ||
|
||||
f === 'grader-samples.txt' ||
|
||||
f === 'grader-regime.json' ||
|
||||
/^grade(-\d+)?\.md$/.test(f) ||
|
||||
/^grade(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-result(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
|
||||
/^render-stderr(-\d+)?\.log$/.test(f)
|
||||
) {
|
||||
copyPath(join(verifierDir, f), join(dest, f));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const agentOutputDir = join(verifierDir, 'agent-output');
|
||||
if (existsSync(agentOutputDir)) {
|
||||
copyTree(agentOutputDir, join(dest, 'agent-output'));
|
||||
}
|
||||
|
||||
// Copy agent session log and trajectory (not workspace)
|
||||
const agentDir = join(trialPath, 'agent');
|
||||
if (existsSync(agentDir)) {
|
||||
mkdirSync(join(dest, 'agent'), { recursive: true });
|
||||
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
|
||||
// the Claude one left codex reference runs with nothing but the trajectory.
|
||||
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
|
||||
const src = join(agentDir, file);
|
||||
if (existsSync(src)) copyPath(src, join(dest, 'agent', file));
|
||||
}
|
||||
// The grader (and downstream worldbench export / replay) reads
|
||||
// agent/trajectory.json. If it's missing, scream so we don't silently
|
||||
// ship a reference run that's only half-useful.
|
||||
if (!existsSync(join(agentDir, 'trajectory.json'))) {
|
||||
console.warn(
|
||||
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
|
||||
`Harbor's adapter for this harness failed to write it (typically because ` +
|
||||
`the converter choked on the session log). Downstream consumers ` +
|
||||
`(grader replay, worldbench export) need this file — investigate ` +
|
||||
`before relying on this reference run.`
|
||||
);
|
||||
}
|
||||
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
|
||||
// gets embedded in claude-code.txt's first non-empty line; we use it to
|
||||
// locate the sibling JSONL Claude Code wrote in the same trial.
|
||||
// Layout is per harness, so each needs a case here — the same reason
|
||||
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
|
||||
// session, which is what codex got before this: nothing at all.
|
||||
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
|
||||
if (existsSync(claudeCodeTxt)) {
|
||||
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
|
||||
const sessionId = readSessionId(claudeCodeTxt);
|
||||
if (sessionId) {
|
||||
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
|
||||
if (existsSync(jsonl)) copyPath(jsonl, join(dest, 'session.jsonl'));
|
||||
}
|
||||
} else {
|
||||
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
|
||||
const rollout = newestRollout(join(agentDir, 'sessions'));
|
||||
if (rollout) copyPath(rollout, join(dest, 'session.jsonl'));
|
||||
}
|
||||
}
|
||||
|
||||
// Copy top-level metadata
|
||||
for (const file of ['config.json', 'result.json', 'trial.log']) {
|
||||
const src = join(trialPath, file);
|
||||
if (existsSync(src)) copyPath(src, join(dest, file));
|
||||
}
|
||||
|
||||
// Record the checksums of the task inputs this run was generated against
|
||||
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
|
||||
// submit-task.ts re-captures at packaging time and warns when any of them
|
||||
// changed — the run then describes an older revision of the task than the
|
||||
// one being shipped. Preferred source: the launch-time stamp harbor-run
|
||||
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
|
||||
// 'run') — it records the inputs the agent actually ran against, so an
|
||||
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
|
||||
// (trials from an older harbor-run, or a failed stamp): capture here at
|
||||
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
|
||||
// the weaker evidence.
|
||||
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
|
||||
const trialStamp = readTaskInputChecksums(trialStampPath);
|
||||
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
|
||||
// harbor-runs sharing a cwd) — its hashes describe some other task's
|
||||
// inputs, so treat it as absent rather than importing false evidence.
|
||||
const stampSlug = trialStamp?.taskSlug;
|
||||
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
|
||||
if (trialStamp && !stampMisrouted) {
|
||||
copyPath(trialStampPath, join(dest, INPUT_CHECKSUMS_FILENAME));
|
||||
} else {
|
||||
if (stampMisrouted) {
|
||||
console.warn(
|
||||
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
|
||||
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
|
||||
);
|
||||
}
|
||||
writeFileSync(
|
||||
join(dest, INPUT_CHECKSUMS_FILENAME),
|
||||
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
|
||||
let dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
|
||||
const prov = readReplayProvenance(trialPath);
|
||||
// An atomic grade in reference-runs/ would overwrite the holistic grade it exists
|
||||
// to be compared against, and the two are not interchangeable.
|
||||
if (!rubricRegrade && prov.rubricMode) {
|
||||
console.error(
|
||||
`Error: ${trialPath} is an atomic-rubric regrade, which does not belong in ` +
|
||||
`reference-runs/. File it beside the run it graded:\n` +
|
||||
` npx tsx scripts/copy-reference-run.ts --rubric-regrade ${trialPath}`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
if (rubricRegrade) {
|
||||
if (!prov.rubricMode) {
|
||||
console.error(
|
||||
`Error: ${trialPath} is not an atomic-rubric regrade (no rubric grader mode ` +
|
||||
`in result.json, no verifier/rubric-grade.json). A holistic regrade replaces ` +
|
||||
`the run's own grade — copy it without --rubric-regrade.`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
if (!prov.sourceRun) {
|
||||
console.error(
|
||||
`Error: ${trialPath}/result.json records no reference_run_dir, so there is no ` +
|
||||
`run to file this regrade under. Copy it by hand into ` +
|
||||
`${join(taskDir, 'rubric-regrades')}/<run-id>/.`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
dest = join(taskDir, 'rubric-regrades', prov.sourceRun);
|
||||
}
|
||||
|
||||
// Build in a sibling directory and swap at the end. A grade costs 15-30 minutes and
|
||||
// cannot be reproduced exactly, so an interrupted copy must leave the stored one intact.
|
||||
const staged = `${dest}.staging-${process.pid}`;
|
||||
rmSync(staged, { recursive: true, force: true });
|
||||
mkdirSync(staged, { recursive: true });
|
||||
|
||||
// A stored atomic grade is the trial verbatim. Matching that shape exactly matters
|
||||
// more than trimming it: grades stored by hand have it, and a reviewer opening one
|
||||
// should not have to work out which way it was written.
|
||||
if (rubricRegrade) {
|
||||
copyTree(trialPath, staged);
|
||||
} else {
|
||||
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
|
||||
// companion reward-correctness.txt (N/A by design) and the machine-readable
|
||||
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
|
||||
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
|
||||
// (the source of truth grade.md/reward.txt are rendered from), the
|
||||
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
|
||||
// render-stderr(-<N>).log, and grader-samples.txt.
|
||||
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
|
||||
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
|
||||
// every per-sample record.)
|
||||
//
|
||||
// signals-status.txt qualifies the deterministic signals the grader was fed:
|
||||
// it records whether every deterministic check actually produced a verdict
|
||||
// ("ok") or one or more was killed before finishing ("degraded" — the grade
|
||||
// is then NOT fully signal-backed). Without it a copied run is
|
||||
// indistinguishable from a run whose checks all passed, so it must travel
|
||||
// with the reward files.
|
||||
const verifierDir = join(trialPath, 'verifier');
|
||||
if (existsSync(verifierDir)) {
|
||||
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
|
||||
if (!entry.isFile()) continue;
|
||||
const f = entry.name;
|
||||
if (
|
||||
f === 'reward.json' ||
|
||||
f === 'signals-status.txt' ||
|
||||
f === 'grader-samples.txt' ||
|
||||
f === 'grader-regime.json' ||
|
||||
f === 'test-stdout.txt' ||
|
||||
f === 'deterministic-signals.txt' ||
|
||||
/^reward(-\d+)?\.txt$/.test(f) ||
|
||||
/^reward-correctness(-\d+)?\.txt$/.test(f) ||
|
||||
/^grade(-\d+)?\.md$/.test(f) ||
|
||||
/^grade(-\d+)?\.json$/.test(f) ||
|
||||
/^rubric-grade(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-result(-\d+)?\.json$/.test(f) ||
|
||||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
|
||||
// The renderer retries, so the real name is render-stderr-<sample>-attempt<n>.log.
|
||||
/^render-stderr(-\d+)?(-attempt\d+)?\.log$/.test(f)
|
||||
) {
|
||||
copyPath(join(verifierDir, f), join(staged, f));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A rubric regrade re-grades a run that already ships its own agent-output,
|
||||
// trajectory and session; a second copy would only double the tarball.
|
||||
const agentOutputDir = join(verifierDir, 'agent-output');
|
||||
if (!rubricRegrade && existsSync(agentOutputDir)) {
|
||||
copyTree(agentOutputDir, join(staged, 'agent-output'));
|
||||
}
|
||||
|
||||
// Copy agent session log and trajectory (not workspace)
|
||||
const agentDir = join(trialPath, 'agent');
|
||||
if (!rubricRegrade && existsSync(agentDir)) {
|
||||
mkdirSync(join(staged, 'agent'), { recursive: true });
|
||||
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
|
||||
// the Claude one left codex reference runs with nothing but the trajectory.
|
||||
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
|
||||
const src = join(agentDir, file);
|
||||
if (existsSync(src)) copyPath(src, join(staged, 'agent', file));
|
||||
}
|
||||
// The grader (and downstream worldbench export / replay) reads
|
||||
// agent/trajectory.json. If it's missing, scream so we don't silently
|
||||
// ship a reference run that's only half-useful.
|
||||
if (!existsSync(join(agentDir, 'trajectory.json'))) {
|
||||
console.warn(
|
||||
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
|
||||
`Harbor's adapter for this harness failed to write it (typically because ` +
|
||||
`the converter choked on the session log). Downstream consumers ` +
|
||||
`(grader replay, worldbench export) need this file — investigate ` +
|
||||
`before relying on this reference run.`
|
||||
);
|
||||
}
|
||||
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
|
||||
// gets embedded in claude-code.txt's first non-empty line; we use it to
|
||||
// locate the sibling JSONL Claude Code wrote in the same trial.
|
||||
// Layout is per harness, so each needs a case here — the same reason
|
||||
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
|
||||
// session, which is what codex got before this: nothing at all.
|
||||
// Silent when there is nothing to hoist: no shipped tool reads this file and nothing
|
||||
// validates it, so its absence is not worth a line of output.
|
||||
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
|
||||
if (existsSync(claudeCodeTxt)) {
|
||||
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
|
||||
const sessionId = readSessionId(claudeCodeTxt);
|
||||
if (sessionId) {
|
||||
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
|
||||
if (existsSync(jsonl)) copyPath(jsonl, join(staged, 'session.jsonl'));
|
||||
}
|
||||
} else {
|
||||
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
|
||||
const rollout = newestRollout(join(agentDir, 'sessions'));
|
||||
if (rollout) copyPath(rollout, join(staged, 'session.jsonl'));
|
||||
}
|
||||
}
|
||||
|
||||
// Copy top-level metadata. result.json is what makes a regrade self-describing
|
||||
// (which run it graded, under which grader mode), so it travels either way.
|
||||
for (const file of rubricRegrade
|
||||
? ['result.json']
|
||||
: ['config.json', 'result.json', 'trial.log']) {
|
||||
const src = join(trialPath, file);
|
||||
if (existsSync(src)) copyPath(src, join(staged, file));
|
||||
}
|
||||
|
||||
// Not for a rubric regrade: this input set is what stales a RUN (prompt, snapshot,
|
||||
// patch, gitref), and the run it re-graded already carries its own stamp.
|
||||
if (!rubricRegrade) {
|
||||
// Record the checksums of the task inputs this run was generated against
|
||||
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
|
||||
// submit-task.ts re-captures at packaging time and warns when any of them
|
||||
// changed — the run then describes an older revision of the task than the
|
||||
// one being shipped. Preferred source: the launch-time stamp harbor-run
|
||||
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
|
||||
// 'run') — it records the inputs the agent actually ran against, so an
|
||||
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
|
||||
// (trials from an older harbor-run, or a failed stamp): capture here at
|
||||
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
|
||||
// the weaker evidence.
|
||||
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
|
||||
const trialStamp = readTaskInputChecksums(trialStampPath);
|
||||
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
|
||||
// harbor-runs sharing a cwd) — its hashes describe some other task's
|
||||
// inputs, so treat it as absent rather than importing false evidence.
|
||||
const stampSlug = trialStamp?.taskSlug;
|
||||
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
|
||||
if (trialStamp && !stampMisrouted) {
|
||||
copyPath(trialStampPath, join(staged, INPUT_CHECKSUMS_FILENAME));
|
||||
} else {
|
||||
if (stampMisrouted) {
|
||||
console.warn(
|
||||
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
|
||||
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
|
||||
);
|
||||
}
|
||||
writeFileSync(
|
||||
join(staged, INPUT_CHECKSUMS_FILENAME),
|
||||
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Files captured from a run can land unreadable to you, which makes packaging
|
||||
// fail later. Fix that now. Guarded so it can never fail a copy that worked.
|
||||
let perms = null;
|
||||
try {
|
||||
perms = normalizeTreePermissions(dest);
|
||||
perms = normalizeTreePermissions(staged);
|
||||
} catch (err) {
|
||||
console.warn(`Warning: could not normalize permissions on ${dest}: ${String(err)}`);
|
||||
console.warn(` The run copied fine. If packaging later fails on permissions:`);
|
||||
@@ -350,6 +455,43 @@ function copyTrial(trialPath: string, destName?: string) {
|
||||
);
|
||||
}
|
||||
|
||||
if (existsSync(dest)) {
|
||||
console.warn(`Warning: ${dest} already exists, overwriting`);
|
||||
rmSync(dest, { recursive: true });
|
||||
}
|
||||
renameSync(staged, dest);
|
||||
|
||||
// Remove what this copy supersedes, now that the copy is on disk. Skipped when the
|
||||
// minted name landed on the superseded directory itself — that is the copy, not a
|
||||
// leftover.
|
||||
if (supersede) {
|
||||
const old = resolve(supersede);
|
||||
if (!existsSync(old)) {
|
||||
console.warn(`Warning: nothing to supersede at ${supersede} — it is already gone`);
|
||||
} else if (old === resolve(dest)) {
|
||||
console.log(`Superseded ${supersede} in place (same name)`);
|
||||
} else {
|
||||
// A stored atomic grade is keyed on the run's folder name, and superseding
|
||||
// changes that name. Move it with the run — it grades the same behaviour — or
|
||||
// it is left pointing at a run that no longer exists.
|
||||
const storedGrade = join(taskDir, 'rubric-regrades', basename(old));
|
||||
const movedGrade = join(taskDir, 'rubric-regrades', basename(dest));
|
||||
if (existsSync(storedGrade) && existsSync(movedGrade)) {
|
||||
console.warn(
|
||||
`Warning: ${storedGrade} names the superseded run, but ${movedGrade} already ` +
|
||||
`exists. Leaving both — remove whichever is obsolete.`
|
||||
);
|
||||
} else if (existsSync(storedGrade)) {
|
||||
renameSync(storedGrade, movedGrade);
|
||||
console.log(`Moved the stored atomic grade to ${movedGrade}`);
|
||||
console.log(' It grades the same run. Grade it again under the atomic rubric');
|
||||
console.log(' if the rubric changed since it was stored.');
|
||||
}
|
||||
rmSync(old, { recursive: true });
|
||||
console.log(`Superseded ${supersede} (removed)`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`Copied to ${dest}`);
|
||||
console.log(` reward: ${reward}`);
|
||||
console.log(` task: ${taskDir}`);
|
||||
@@ -364,9 +506,19 @@ function copyTrial(trialPath: string, destName?: string) {
|
||||
// Main
|
||||
const rawArgs = process.argv.slice(2);
|
||||
let destName: string | undefined;
|
||||
let rubricRegrade = false;
|
||||
let supersede: string | undefined;
|
||||
const args: string[] = [];
|
||||
for (let i = 0; i < rawArgs.length; i++) {
|
||||
if (rawArgs[i] === '--dest-name') {
|
||||
if (rawArgs[i] === '--rubric-regrade') {
|
||||
rubricRegrade = true;
|
||||
} else if (rawArgs[i] === '--supersede') {
|
||||
supersede = rawArgs[++i];
|
||||
if (!supersede) {
|
||||
console.error('Error: --supersede requires the run directory being replaced');
|
||||
process.exit(1);
|
||||
}
|
||||
} else if (rawArgs[i] === '--dest-name') {
|
||||
destName = rawArgs[++i];
|
||||
if (!destName) {
|
||||
console.error('Error: --dest-name requires a value');
|
||||
@@ -379,7 +531,9 @@ for (let i = 0; i < rawArgs.length; i++) {
|
||||
|
||||
if (args.length === 0) {
|
||||
console.error(
|
||||
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]'
|
||||
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]\n' +
|
||||
' npx tsx scripts/copy-reference-run.ts --rubric-regrade <trial-path>\n' +
|
||||
' npx tsx scripts/copy-reference-run.ts --supersede <old-run-dir> <trial-path>'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
@@ -387,7 +541,21 @@ if (destName && args.length !== 1) {
|
||||
console.error('Error: --dest-name applies to exactly one trial path');
|
||||
process.exit(1);
|
||||
}
|
||||
if (supersede && args.length !== 1) {
|
||||
console.error('Error: --supersede applies to exactly one trial path');
|
||||
process.exit(1);
|
||||
}
|
||||
if (supersede && rubricRegrade) {
|
||||
console.error('Error: --rubric-regrade adds a grade beside a run; there is nothing to supersede');
|
||||
process.exit(1);
|
||||
}
|
||||
if (destName && rubricRegrade) {
|
||||
console.error(
|
||||
'Error: --rubric-regrade takes its name from the trial, so --dest-name cannot apply'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
for (const trialPath of args) {
|
||||
copyTrial(trialPath, destName);
|
||||
copyTrial(trialPath, { destName, rubricRegrade, supersede });
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user