/** * submit-task.ts — Validate and package a task directory for submission. * * Usage: * npx tsx scripts/submit-task.ts * npx tsx scripts/submit-task.ts my-cool-task --json */ import { execFileSync, execSync } from 'child_process'; import { existsSync, readFileSync, readdirSync, statSync, writeFileSync } from 'fs'; import { fileURLToPath } from 'node:url'; import { join, resolve } from 'path'; import pino from 'pino'; import pinoPretty from 'pino-pretty'; import yargs from 'yargs'; import { hideBin } from 'yargs/helpers'; import { DETECTOR_REPORT_INPUTS, INPUT_CHECKSUMS_FILENAME, REFERENCE_RUN_INPUTS, captureTaskInputs, diffTaskInputs, readTaskInputChecksums, } from './lib/input-checksums'; import { bannerize, checkTaskInfraIntegrity, formatIntegrityReport, } from './lib/task-infra-integrity.js'; import { didRepair, manualRepairHint, normalizeTreePermissions } from './lib/tree-permissions'; /** Written to harbor-tasks//staleness.json by {@link validateTask}. */ export const STALENESS_REPORT_FILENAME = 'staleness.json'; /** * Written to harbor-tasks//toolkit-files.json by {@link validateTask}: the * per-file verdict on the toolkit-managed files. Shipped in the package because the * check needs the toolkit's `task-shared/` and so cannot be recomputed downstream. */ export const INFRA_REPORT_FILENAME = 'toolkit-files.json'; /** * Staleness of one reference run against the task inputs being packaged. * `unknown` = the run predates input-checksum tracking (no recorded * checksums), so freshness can't be verified either way. */ export interface ReferenceRunStaleness { runId: string; status: 'fresh' | 'stale' | 'unknown'; /** Human-readable labels of the inputs that changed since capture. */ changed: string[]; /** When the run's checksums were captured (recorded runs only). */ capturedAt?: string; /** * Lifecycle point of the capture (recorded runs only): 'run' = stamped at * trial launch by harbor-run (strongest — immune to edits made between the * run and the copy); 'copy' = captured at copy-reference-run time (an input * edited between harbor-run and the copy is recorded post-edit). Absent on * records that predate the field. */ capturedBy?: string; } /** * Staleness of one detector report. `mtime` = legacy unstamped report; the * mtime heuristic can prove a report is OLDER than the docs but never that it * matched their content, so an mtime record is `stale` or `unknown` — only a * checksum stamp can say `fresh`. */ export interface DetectorReportStaleness { report: string; status: 'fresh' | 'stale' | 'unknown'; method: 'checksums' | 'mtime'; changed: string[]; capturedAt?: string; /** Lifecycle point of the stamp's capture (checksum records only). */ capturedBy?: string; } /** * The machine-readable staleness record shipped inside the tarball (and * refreshed whenever validateTask runs), so a review — on any pipeline — can * display exactly which runs/reports predate which input edits even if the * worker packaged past the warnings. */ export interface TaskStalenessReport { version: number; generatedAt: string; referenceRuns: ReferenceRunStaleness[]; detectors: DetectorReportStaleness[]; } export interface ValidationResult { hasErrors: boolean; /** Hard-failure messages — each one means the package violates a submission guarantee. */ errors: string[]; /** Advisory messages — the package is submittable but could be improved. */ warnings: string[]; refCount: number; /** Behavioral scores (reward.txt) across the reference runs. */ scores: number[]; /** Correctness scores (reward-correctness.txt), excluding `N/A` runs. */ correctnessScores: number[]; workspaceFiles: number; /** Per-run / per-report staleness (also written to staleness.json). */ staleness: TaskStalenessReport; /** Loud notice about toolkit-managed files, re-printed last by the CLI. */ infraNotice?: string; } /** * First interpreter that can `import tomllib`, or null. `python3` is not always new * enough (3.11+), and a resolver run under an old one fails in a way that looks like a * malformed task.toml. Twin of `_raccoon_python` in scripts/lib/harness-credentials.sh * and the search in snapshot-to-task.ts — kept separate because those live in trees that * cannot import each other. */ function pythonWithTomllib(): string | null { for (const candidate of [ process.env.RACCOON_PYTHON, 'python3', 'python3.13', 'python3.12', 'python3.11', ]) { if (!candidate) continue; try { execFileSync(candidate, ['-c', 'import tomllib'], { stdio: 'ignore' }); return candidate; } catch { // try the next one } } return null; } /** * Validate a task package against the submission guarantees. This is the SINGLE * source of truth for "what a well-formed tarball looks like": workers run it * via `submit-task` before packaging, and the review pipeline runs the exact * same function on its side after unpacking, so a package that wouldn't pass * here is cleanly flagged rather than crashing a detector or being half-reviewed. * * Returns structured `errors` / `warnings` (in addition to logging them) so a * caller can record WHY a package was rejected, not just that it was. */ /** * One harness per task: every reference run must have been produced by the harness * the task declares. * * Why it's a hard error rather than a warning. A task's reference runs ARE its * calibration — detectors and reviewers reason over them, and the benchmark groups by * (harness, model). If the runs came from a harness other than the declared one, every * one of those readings is attributed to the wrong agent, and nothing downstream can * detect it: the scores look perfectly ordinary. The fix is also cheap (correct the * declaration, or re-run), which is what makes refusing reasonable. * * Resolution goes through `resolve_harness.py` rather than string-comparing identities, * for two reasons. Raw identities under-report agreement — `SnapshotClaudeCode` and * `PreinstalledClaudeCode` are both `claude-code`, and a task whose runs straddle the * multi-turn-detection change legitimately has both. And the toolkit has no TOML parser * for TS, so the declared harness must be read by something that actually parses * task.toml; a regex would match a commented line or the wrong table. * * Replays (`harbor-regrade`) are not a harness: they report `replay_agent:ReplayAgent` * with no model, because no model ran. Their harness is the SOURCE run's, carried on * `kwargs.source_agent_import_path`. Older replays predate that field and are simply * unattributable — skipped, not failed, because refusing them would reject every task * with a regraded reference run. */ function checkHarnessConsistency( taskDir: string, runDirs: string[], cwd: string, logger: pino.Logger ): { errors: string[]; warnings: string[] } { const errors: string[] = []; const warnings: string[] = []; const resolver = join(cwd, 'scripts', 'resolve_harness.py'); if (!existsSync(resolver) || runDirs.length === 0) return { errors, warnings }; const python = pythonWithTomllib(); /** * Ask the resolver something. * * A refusal and a failure to run are different answers and must not be conflated: the * resolver refusing means the task really is malformed, while a missing interpreter or * a broken import means we learned nothing and have no business blocking a submission * over it. `resolve_harness.py` prefixes every deliberate refusal with * `resolve-harness: ERROR:`, which is the only reliable discriminator — a crash also * exits 1. */ type ResolverAnswer = | { ok: true; stdout: string } | { ok: false; refused: boolean; detail: string }; const askResolver = (args: string[]): ResolverAnswer => { if (!python) { return { ok: false, refused: false, detail: 'no python3.11+ with tomllib on PATH' }; } try { return { ok: true, stdout: execFileSync(python, [resolver, ...args], { cwd, encoding: 'utf-8', stdio: ['ignore', 'pipe', 'pipe'], }), }; } catch (err) { const e = err as { stderr?: Buffer | string; message?: string }; const stderr = String(e.stderr ?? ''); return { ok: false, refused: stderr.includes('resolve-harness: ERROR:'), detail: (stderr || e.message || 'unknown failure').trim().split('\n').slice(-3).join(' '), }; } }; const declaredAnswer = askResolver(['--declared-harness', '--task-dir', taskDir]); if (!declaredAnswer.ok && declaredAnswer.refused) { logger.error( { file: 'task.toml', detail: declaredAnswer.detail }, 'Could not read [agent] harness from task.toml. The likely cause is a SECOND [agent] table — tasks already carry [agent] timeout_sec, so a harness line must go inside that table, not start a new one. TOML rejects a duplicate table, which makes the whole file unreadable.' ); errors.push('task.toml could not be parsed for [agent] harness'); return { errors, warnings }; } if (!declaredAnswer.ok) { logger.warn( { detail: declaredAnswer.detail }, 'Skipping the harness-consistency check: could not run scripts/resolve_harness.py. Your task is not affected — this check only compares the harness your reference runs came from against task.toml.' ); warnings.push('harness-consistency check skipped (resolver could not run)'); return { errors, warnings }; } const declared = declaredAnswer.stdout.trim(); // Each run's effective identity: its own agent, or for a replay the source it // replayed. Undefined when unattributable. const identityByRun = new Map(); for (const runId of runDirs) { const resultPath = join(taskDir, 'reference-runs', runId, 'result.json'); if (!existsSync(resultPath)) continue; let agent: { import_path?: unknown; name?: unknown; kwargs?: unknown } | undefined; try { agent = JSON.parse(readFileSync(resultPath, 'utf-8'))?.config?.agent; } catch { continue; } const own = [agent?.import_path, agent?.name].find( (v): v is string => typeof v === 'string' && v.length > 0 ); if (!own) continue; const source = (agent?.kwargs as { source_agent_import_path?: unknown } | undefined) ?.source_agent_import_path; const effective = own === 'replay_agent:ReplayAgent' ? typeof source === 'string' && source.length > 0 ? source : undefined : own; if (effective) identityByRun.set(runId, effective); } if (identityByRun.size === 0) return { errors, warnings }; const distinct = [...new Set(identityByRun.values())]; const resolvedAnswer = askResolver(distinct.flatMap((i) => ['--resolve-identity', i])); // Nothing to compare against if this can't run; the declared harness was already read. if (!resolvedAnswer.ok) return { errors, warnings }; const harnessByIdentity = new Map( resolvedAnswer.stdout .split('\n') .filter(Boolean) .map((line) => line.split('\t') as [string, string]) ); const harnessByRun = new Map(); for (const [runId, identity] of identityByRun) { const harness = harnessByIdentity.get(identity)?.trim(); // An identity the registry doesn't know isn't a mismatch — it's an agent nobody // has registered. Say so once; don't fail the submission over it. if (!harness) { warnings.push(`reference run ${runId} ran an unregistered agent (${identity})`); continue; } harnessByRun.set(runId, harness); } const present = [...new Set(harnessByRun.values())].sort(); if (present.length === 0) return { errors, warnings }; if (present.length > 1) { const breakdown = [...harnessByRun.entries()].map(([runId, h]) => `${runId} → ${h}`).join(', '); logger.error( { harnesses: present }, `Reference runs came from MORE THAN ONE harness (${present.join(', ')}). A task is calibrated against one harness — mixed runs make the reference set unreadable, because detectors and reviewers can't tell which agent's behaviour they're looking at. Re-run the trials with a single harness: ${breakdown}` ); errors.push(`reference runs span multiple harnesses (${present.join(', ')})`); return { errors, warnings }; } const actual = present[0]; if (!declared) { logger.warn( { harness: actual }, `task.toml declares no [agent] harness, but the reference runs were produced by "${actual}". Add it inside the existing [agent] table so the trial and every downstream consumer know what this task is calibrated against:\n [agent]\n harness = "${actual}"` ); warnings.push(`task.toml declares no [agent] harness (reference runs used ${actual})`); } else if (declared !== actual) { logger.error( { declared, actual }, `task.toml declares harness "${declared}" but the reference runs were produced by "${actual}". One of the two is wrong, and downstream can't tell which: the runs would be attributed to "${declared}" and scored as if that agent produced them. Either correct the declaration or re-run the trials with "${declared}".` ); errors.push(`declared harness "${declared}" does not match reference runs ("${actual}")`); } return { errors, warnings }; } export function validateTask( slug: string, logger: pino.Logger, opts: { cwd?: string } = {} ): ValidationResult { const cwd = opts.cwd ?? process.cwd(); const taskDir = join(cwd, 'harbor-tasks', slug); const errors: string[] = []; const warnings: string[] = []; const staleness: TaskStalenessReport = { version: 1, generatedAt: new Date().toISOString(), referenceRuns: [], detectors: [], }; if (!existsSync(taskDir)) { logger.fatal({ taskDir }, 'Task directory not found'); errors.push(`task directory not found: harbor-tasks/${slug}`); return { hasErrors: true, errors, warnings, refCount: 0, scores: [], correctnessScores: [], workspaceFiles: 0, staleness, }; } // The task inputs as they stand right now — the baseline every recorded // capture (reference runs, detector-report stamps) is compared against. const currentInputs = captureTaskInputs(taskDir); logger.info({ slug }, 'Validating task'); interface Check { path: string; label: string; required: boolean; } const checks: Check[] = [ { path: 'instruction.md', label: 'Task prompt', required: true }, { path: 'task.toml', label: 'Task config', required: true }, { path: 'tests/test.sh', label: 'Grader orchestration', required: true }, { path: 'tests/grader-system-prompt.md', label: 'Grader system prompt', required: true }, { path: 'environment/Dockerfile', label: 'Dockerfile', required: true }, ]; for (const check of checks) { const fullPath = join(taskDir, check.path); if (existsSync(fullPath)) { logger.debug({ file: check.path }, `${check.label}: found`); } else if (check.required) { logger.error({ file: check.path }, `${check.label}: MISSING`); errors.push(`missing required file: ${check.path} (${check.label})`); } else { logger.warn({ file: check.path }, `${check.label}: not found (optional)`); warnings.push(`optional file not found: ${check.path}`); } } // Grader guidance: at least one of the legacy and consolidated files must be // present — a transition-era task carries both, a consolidated-only task just // the latter. Whichever is present is placeholder-checked below. const guidancePaths = ['tests/grader-guidance.md', 'tests/grader-guidance-consolidated.md']; const presentGuidance = guidancePaths.filter((p) => existsSync(join(taskDir, p))); if (presentGuidance.length > 0) { for (const p of presentGuidance) logger.debug({ file: p }, 'Grader guidance: found'); } else { logger.error({ file: guidancePaths.join(' or ') }, 'Grader guidance: MISSING'); errors.push( `missing required file: ${guidancePaths[0]} or ${guidancePaths[1]} (Grader guidance)` ); } // Placeholder detection const placeholderFiles = ['instruction.md', ...presentGuidance]; for (const file of placeholderFiles) { const fullPath = join(taskDir, file); if (existsSync(fullPath)) { const content = readFileSync(fullPath, 'utf-8'); if (content.includes('Replace this with')) { logger.warn({ file }, 'Still contains scaffold placeholder text'); errors.push(`scaffold placeholder text not replaced in ${file}`); } } } // Deterministic checks: available for this member but not staged? // // tests/test-commands.sh is optional-by-absence — tests/test.sh just skips the // deterministic signals when it isn't there, and correctness gets judged from the // code alone. That makes a forgotten copy look exactly like a member that // legitimately has no runnable checks, which is the whole problem: silence means // two different things. Resolving the member against task-shared/ tells them // apart, so warn ONLY when checks exist and weren't staged. if (!existsSync(join(taskDir, 'tests', 'test-commands.sh'))) { const taskToml = join(taskDir, 'task.toml'); const member = existsSync(taskToml) ? /^repo\s*=\s*"?([^"\n]+)"?/m.exec(readFileSync(taskToml, 'utf-8'))?.[1]?.trim() : undefined; const sharedDir = join(cwd, 'task-shared'); const candidates = [ ...(member ? [join(sharedDir, `test-commands.${member.toLowerCase()}.sh`)] : []), join(sharedDir, 'test-commands.sh'), ]; const available = candidates.find((p) => existsSync(p)); if (available) { logger.warn( { member, available: available.replace(`${cwd}/`, '') }, `This repo has deterministic checks (tests/typecheck/lint) but tests/test-commands.sh isn't in the task, so the grader had no test signal behind the correctness score. Stage it and re-run your trials:` ); logger.warn(` bash scripts/build-workspace.sh ${slug}`); warnings.push( `tests/test-commands.sh missing though checks exist for ${member ?? 'this repo'} — ` + 'correctness was not signal-backed; run `bash scripts/build-workspace.sh ` and re-run trials' ); } } // Detector self-check reports. The toolkit ships one skill per worker-runnable // detector, each writing harbor-tasks//detectors/.md. The expected // set is enumerated from the WORKER-SHIPPED skills directory rather than // hardcoded, so it tracks detector additions and retirements — and the // resolution order is load-bearing: validateTask also runs repo-side // (fetch-submission and the review pipeline both pass the repo root as cwd), // where .claude/skills additionally holds pipeline-only detectors that // workers cannot run and must never be warned about. // raccoon-worker-toolkit/static/.claude/skills is therefore preferred; // in a packed toolkit only .claude/skills exists, and it IS the shipped set. // Missing reports stay a warning, not an error: the package is reviewable // without them, but a reviewer has to regenerate the set by hand, which stalls // the queue. A zero-byte report counts as missing — a stub carries no verdict. const detectorSkillsDir = [ join(cwd, 'raccoon-worker-toolkit', 'static', '.claude', 'skills'), join(cwd, '.claude', 'skills'), ].find((p) => existsSync(p)); const expectedDetectors = detectorSkillsDir ? readdirSync(detectorSkillsDir).filter( (d) => d.startsWith('detector-') && statSync(join(detectorSkillsDir, d)).isDirectory() ) : []; const detectorsDir = join(taskDir, 'detectors'); if (expectedDetectors.length > 0) { const missing = expectedDetectors.filter((name) => { const report = join(detectorsDir, `${name}.md`); return !existsSync(report) || statSync(report).size === 0; }); if (missing.length > 0) { logger.warn( { missing, expected: expectedDetectors.length }, `${missing.length} of ${expectedDetectors.length} detector reports are missing from detectors/. Reviewers expect the full self-check set — regenerating it during review is the most common avoidable review-queue delay. Run each missing detector (e.g. /${missing[0]}) and re-package.` ); warnings.push( `missing ${missing.length} of ${expectedDetectors.length} detector reports: ${missing.join(', ')} — ` + 'run the detector skills and re-package' ); } } // Stale detector reports. A report written before the current instruction.md or // tests/grader-guidance.md assessed an earlier revision of the task, and nothing // on the review page reveals that. Two generations of evidence: // // 1. Input checksums (durable). Detector skills stamp each report with a // capture of the task inputs it assessed (detectors/.inputs.json, // written by scripts/record-detector-inputs.ts). Content hashes can't be // faked or masked by whole-tree touches (re-clone, checkout), which is // exactly where mtimes lie. A stamped report is judged by its stamp — // including a fresh verdict when its mtime happens to look old. // 2. mtime (legacy fallback). Unstamped reports predate stamping; for them // the original comparison against the newest source doc still applies — // mtimes are meaningful on both sides of the pipeline (the worker // authored these files in place, and tar preserves timestamps through // pack/unpack). Equal timestamps don't warn (same-operation touches). if (existsSync(detectorsDir)) { const reports = readdirSync(detectorsDir).filter((f) => f.endsWith('.md')); const unstamped: string[] = []; const staleStamped: Array<{ report: string; changed: string[] }> = []; for (const report of reports) { const stamp = readTaskInputChecksums( join(detectorsDir, report.replace(/\.md$/, '.inputs.json')) ); if (!stamp) { unstamped.push(report); continue; } const changed = diffTaskInputs(stamp, currentInputs, DETECTOR_REPORT_INPUTS); staleness.detectors.push({ report, status: changed.length > 0 ? 'stale' : 'fresh', method: 'checksums', changed, capturedAt: stamp.capturedAt, ...(stamp.capturedBy ? { capturedBy: stamp.capturedBy } : {}), }); if (changed.length > 0) staleStamped.push({ report, changed }); } if (staleStamped.length > 0) { const changedLabels = [...new Set(staleStamped.flatMap((s) => s.changed))]; const staleNames = staleStamped.map((s) => s.report); logger.warn( { stale: staleNames, changed: changedLabels }, `You modified your ${changedLabels.join(' and ')} after these detector reports were written, so they assessed an older revision of this task: ${staleNames.join(', ')}. Re-run those detectors (e.g. /${staleNames[0].replace(/\.md$/, '')}) and re-package.` ); warnings.push( `${staleStamped.length} stale detector report(s) — ${changedLabels.join(', ')} changed since they were written: ${staleNames.join(', ')} — ` + 're-run them against the current task' ); } const newestSource = ['instruction.md', join('tests', 'grader-guidance.md')] .map((rel) => ({ rel, full: join(taskDir, rel) })) .filter(({ full }) => existsSync(full)) .reduce( (acc, doc) => { const mtimeMs = statSync(doc.full).mtimeMs; return mtimeMs > acc.mtimeMs ? { rel: doc.rel, mtimeMs } : acc; }, { rel: '', mtimeMs: -Infinity } ); const stale = unstamped.filter( (f) => statSync(join(detectorsDir, f)).mtimeMs < newestSource.mtimeMs ); // Unstamped reports carry no durable evidence either way, so the mtime // heuristic never asserts freshness: it can escalate to `stale` (the // report provably predates a source doc) but otherwise the record says // `unknown` — same standard as a reference run with no recorded // checksums. `changed` stays empty on mtime records: the newest source // doc is what the report predates, not a diff of what changed (only a // checksum comparison can name that), so it goes in the log line below // rather than masquerading as per-component precision in staleness.json. for (const report of unstamped) { staleness.detectors.push({ report, status: stale.includes(report) ? 'stale' : 'unknown', method: 'mtime', changed: [], }); } if (unstamped.length > 0) { // Adoption visibility: stamping is agentic (detector skills follow // _detector-worker-shell.md), so non-adoption would otherwise be // silent — an unstamped report just quietly falls back to the weaker // mtime heuristic. Say so, so gaps are observable (and so we know when // the mtime branch can be deleted). Advisory log only — deliberately // not pushed onto `warnings`, so every pre-stamping report doesn't add // a packaging warning to every in-flight submission. logger.warn( { reports: unstamped }, `${unstamped.length} detector report(s) have no input stamp (detectors/.inputs.json), so staleness is judged by file mtime only. Reports written by a current toolkit are stamped automatically when the detector skill runs record-detector-inputs.ts.` ); } if (stale.length > 0) { logger.warn( { stale, newerSource: newestSource.rel }, `${stale.length} detector report(s) predate ${newestSource.rel} — they assessed an older revision of this task than the one being packaged. Re-run those detectors so the reports match what ships.` ); warnings.push( `${stale.length} detector report(s) older than ${newestSource.rel}: ${stale.join(', ')} — ` + 're-run them against the current task' ); } } // Toolkit-managed files. environment/Dockerfile, tests/test.sh and // tests/grader-system-prompt.md ship from task-shared/ and are not the author's to // change: they decide how the trial runs and how the grade is produced, so an edit // makes the packaged reference runs mean something different from every other task's. // Skipped silently when the toolkit ships no baselines (see task-infra-integrity.ts). let infraNotice: string | undefined; const integrity = checkTaskInfraIntegrity(taskDir, cwd); if (integrity.checked) { const message = formatIntegrityReport(integrity); if (message) { infraNotice = bannerize(message, integrity); // Warnings, never errors. An author who edited one of these did it to get // unstuck, not knowing we'd rather hear about the problem — refusing to // package their finished work over that is the wrong trade. The verdict is // written into the package instead (below), so a reviewer sees it even // though nothing stopped the submission. logger.warn( { edited: integrity.modified.map((f) => f.taskPath), outdated: integrity.outdated.map((f) => f.taskPath), unverifiable: integrity.unverifiable.map((f) => f.taskPath), }, 'Toolkit-managed files need a look — full notice at the end of this run' ); for (const f of integrity.modified) { warnings.push( `toolkit-managed file looks edited: ${f.taskPath} — restore the shipped copy` ); } for (const f of integrity.outdated) { warnings.push( `toolkit-managed file is from an older release: ${f.taskPath} — scores not directly comparable` ); } for (const f of integrity.unverifiable) { warnings.push( `toolkit-managed file matches no shipped copy: ${f.taskPath} — either an older release or an edit` ); } } // Ship the verdict with the task. Nothing blocks any more, and the check can't // run on the far side (it needs the toolkit's task-shared/), so persisting it // here is the only way the signal survives packaging. writeFileSync( join(taskDir, INFRA_REPORT_FILENAME), `${JSON.stringify( { version: 1, generatedAt: new Date().toISOString(), files: integrity.files }, null, 2 )}\n` ); } // Session JSONL: check for leaked snapshot commands const sessionJsonl = join(taskDir, 'environment', 'session.jsonl'); if (existsSync(sessionJsonl)) { const sessionLines = readFileSync(sessionJsonl, 'utf-8').trimEnd().split('\n'); const hasSnapshotCommand = sessionLines.some((line) => { try { const entry = JSON.parse(line) as { type?: string; message?: { content?: unknown } }; return ( entry.type === 'user' && typeof entry.message?.content === 'string' && entry.message.content.includes('create-snapshot:snapshot') ); } catch { return false; } }); if (hasSnapshotCommand) { logger.error( { file: 'environment/session.jsonl' }, `The session history contains /create-snapshot commands. We don't want this, because snapshot creation is an artifact of our observation, and it wouldn't exist in the real world. 1. Long-term fix: next time you're creating multiple snapshots from a conversation, remember to /rewind after taking a snapshot so future convo steps don't have previous snapshots in the history. 2. Immediate fix: manually edit environment/session.jsonl to remove the snapshot command lines. Then you'll need to revalidate the task to make sure that the agent still repros the mistake.` ); errors.push( 'environment/session.jsonl contains a /create-snapshot command (observation artifact)' ); } } // Reference runs. Zero is a hard error, not a warning: detector-run-behaviors and // detector-meaningful-failure reason OVER the reference runs, so with none there's no // signal to extract and the task can't be fully reviewed. (1-3 is allowed but // thin — a warning.) // Any subdirectory counts as a run — naming varies across toolkit versions // (reward-- today, other shapes in older tarballs), so nothing below // keys off the dir-name format. const refDir = join(taskDir, 'reference-runs'); const runDirs = existsSync(refDir) ? readdirSync(refDir).filter((d) => statSync(join(refDir, d)).isDirectory()) : []; const refCount = runDirs.length; const scores: number[] = []; const correctnessScores: number[] = []; const RECOMMENDED_REF_COUNT = 4; if (refCount === 0) { logger.error('No reference runs found. Run the task first:'); logger.error(` scripts/harbor-run harbor-tasks/${slug} -k 4`); errors.push( 'no reference runs — detector-run-behaviors and detector-meaningful-failure cannot be assessed; ' + 'run `scripts/harbor-run harbor-tasks/ -k 4` and include reference-runs/' ); } else { if (refCount < RECOMMENDED_REF_COUNT) { logger.warn( { count: refCount, recommended: RECOMMENDED_REF_COUNT }, `Only ${refCount} reference run(s) found; ${RECOMMENDED_REF_COUNT}+ are recommended for finalized submissions to capture behavioral signal variance. Continuing — but consider running more trials before finalizing:` ); logger.warn(` scripts/harbor-run harbor-tasks/${slug} -k 4`); warnings.push( `only ${refCount} reference run(s); ${RECOMMENDED_REF_COUNT}+ recommended for behavioral-signal variance` ); } logger.info({ count: refCount }, 'Reference runs found'); // The grader writes two independent scores per run: behavioral (reward.txt) // and correctness (reward-correctness.txt, which may be the literal `N/A` // when the agent produced nothing substantive to check). Report both — an // all-N/A correctness column on a task that DOES ask for a deliverable is a // signal worth surfacing before submission. let correctnessNA = 0; for (const run of runDirs) { const rewardFile = join(refDir, run, 'reward.txt'); if (existsSync(rewardFile)) { const score = parseFloat(readFileSync(rewardFile, 'utf-8').trim()); if (!isNaN(score)) scores.push(score); } const correctnessFile = join(refDir, run, 'reward-correctness.txt'); if (existsSync(correctnessFile)) { const raw = readFileSync(correctnessFile, 'utf-8').trim(); if (/^n\/?a$/i.test(raw)) { correctnessNA++; } else { const score = parseFloat(raw); if (!isNaN(score)) correctnessScores.push(score); } } } if (scores.length > 0) { scores.sort((a, b) => a - b); const mean = scores.reduce((a, b) => a + b, 0) / scores.length; logger.info( { scores: scores.map((s) => s.toFixed(2)), mean: mean.toFixed(2) }, 'Behavioral score distribution' ); } if (correctnessScores.length > 0) { correctnessScores.sort((a, b) => a - b); const mean = correctnessScores.reduce((a, b) => a + b, 0) / correctnessScores.length; logger.info( { scores: correctnessScores.map((s) => s.toFixed(2)), mean: mean.toFixed(2), na: correctnessNA, }, 'Correctness score distribution' ); } else if (correctnessNA > 0) { // Expected for assessment / diagnosis / pushback tasks, where the agent // isn't meant to produce a checkable deliverable. Worth a look otherwise. logger.warn( { runs: correctnessNA }, `Correctness is N/A on all ${correctnessNA} reference run(s). That's expected if this task asks for an assessment, a diagnosis, or a pushback rather than a deliverable. If it does ask the agent to write code or make substantive technical claims, check verifier/test-stdout.txt for a SIGNALS_DEGRADED line — the deterministic checks may not be running.` ); warnings.push( `correctness N/A on all ${correctnessNA} reference run(s) (expected for assessment-shaped tasks; check tests/test-commands.sh otherwise)` ); } } // One harness per task — see checkHarnessConsistency above. { const harnessCheck = checkHarnessConsistency(taskDir, runDirs, cwd, logger); errors.push(...harnessCheck.errors); warnings.push(...harnessCheck.warnings); } // Stale reference runs. Each run records the checksums of the task inputs // it was generated against (input-checksums.json, written by // copy-reference-run.ts): the prompt, the session snapshot, the workspace // patch, and the gitref. If any of those changed since, the run shows an // agent doing a DIFFERENT task than the one being packaged — the most common // way this happens is a worker editing their prompt after their trials and // forgetting to redo them. Grader-guidance edits are deliberately not in // this set (they stale the grade, not the run — that's a regrade, not a // rerun). Runs with no recorded checksums predate this tracking; their // staleness is unknowable, which gets said once rather than presumed either // way. const staleRunGroups = new Map(); const unverifiableRuns: string[] = []; for (const run of runDirs) { const recorded = readTaskInputChecksums(join(refDir, run, INPUT_CHECKSUMS_FILENAME)); if (!recorded) { unverifiableRuns.push(run); staleness.referenceRuns.push({ runId: run, status: 'unknown', changed: [] }); continue; } const changed = diffTaskInputs(recorded, currentInputs, REFERENCE_RUN_INPUTS); staleness.referenceRuns.push({ runId: run, status: changed.length > 0 ? 'stale' : 'fresh', changed, capturedAt: recorded.capturedAt, ...(recorded.capturedBy ? { capturedBy: recorded.capturedBy } : {}), }); if (changed.length > 0) { const key = changed.join('|'); const group = staleRunGroups.get(key) ?? { changed, runIds: [] }; group.runIds.push(run); staleRunGroups.set(key, group); } } for (const { changed, runIds } of staleRunGroups.values()) { logger.warn( { runs: runIds, changed }, `You modified your ${changed.join(' and ')} after these reference runs were captured, so they are stale: ${runIds.join(', ')}. Re-run the task and re-copy the runs:` ); logger.warn(` scripts/harbor-run harbor-tasks/${slug} -k 4`); warnings.push( `${runIds.length} stale reference run(s) — ${changed.join(', ')} changed since they were captured: ${runIds.join(', ')} — ` + 're-run the task and re-copy the runs' ); } if (unverifiableRuns.length > 0) { logger.warn( { runs: unverifiableRuns }, // The last sentence is transitional: it exists because every run captured // before checksum tracking shipped triggers this warning, so everyone's // first submission after the toolkit update shows it. Remove that sentence // in the next toolkit release once the skew clears (around 2026-08-05) — // tracked in the source repo's issue #748. (Issue number, not URL: this // file ships in the worker toolkit, which stays free of internal-infra // references.) `${unverifiableRuns.length} reference run(s) predate input-checksum tracking, so staleness can't be verified for them. If you've edited your prompt, session snapshot, workspace patch, or gitref since these runs were captured, re-run the task and re-copy the runs. If this is your first submission since updating the toolkit, this warning is expected — runs captured before the update have no recorded checksums; re-copying the reference runs will record them.` ); warnings.push( `${unverifiableRuns.length} reference run(s) have no recorded input checksums (captured with an older toolkit) — ` + `staleness unverifiable: ${unverifiableRuns.join(', ')}` ); } // Workspace const workspaceDir = join(taskDir, 'environment', 'workspace'); let workspaceFiles = 0; if (existsSync(workspaceDir)) { const wsFiles = execSync(`find "${workspaceDir}" -type f | wc -l`, { encoding: 'utf-8', }).trim(); workspaceFiles = parseInt(wsFiles, 10); logger.info({ files: workspaceFiles }, 'Workspace included'); } else { logger.warn('No workspace directory — task will need workspace built before running'); warnings.push('no environment/workspace directory (workspace must be built before running)'); } // Persist the staleness record inside the task dir so it ships in the // tarball. Warnings are easy to package past; this file is the durable // copy a reviewer sees — and since the review pipeline runs validateTask // again after unpacking, the record is refreshed there too, so a review can // rely on it no matter which pipeline (or toolkit version) produced the // tarball. A write failure downgrades to a warning: the record is advisory // and must never block validation itself. try { writeFileSync( join(taskDir, STALENESS_REPORT_FILENAME), JSON.stringify(staleness, null, 2) + '\n' ); } catch (err) { logger.warn({ err }, `Could not write ${STALENESS_REPORT_FILENAME}`); } return { hasErrors: errors.length > 0, infraNotice, errors, warnings, refCount, scores, correctnessScores, workspaceFiles, staleness, }; } // --- CLI entrypoint --- if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1])) { const argv = yargs(hideBin(process.argv)) .usage('$0 ', 'Validate and package a task for submission', (y) => y.positional('slug', { type: 'string', demandOption: true, describe: 'Task slug' }) ) .option('json', { type: 'boolean', describe: 'Output structured JSON logs', default: false, }) .help() .parseSync(); const slug = argv.slug as string; const log = pino( { name: 'submit-task', level: 'info' }, argv.json ? process.stdout : pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' }) ); const result = validateTask(slug, log); if (result.hasErrors) { log.fatal('Validation failed. Fix the issues above before submitting.'); process.exit(1); } if (result.warnings.length > 0) { log.warn( { count: result.warnings.length }, `Packaging with ${result.warnings.length} warning(s) — reviewers run these same checks and will see them too. Fixing them before submitting saves a review round-trip.` ); } // Create tarball. Exclude environment/corpus/ — it's a large, regenerable build artifact // (staged by build-workspace when a task references the reference-data corpus); shipping it // would bloat the upload by gigabytes. It's re-materialized at build time from the toolkit's // own copy, so the deliverable only needs the Dockerfile's reference to it. const tarball = `${slug}.tar.gz`; const tarCmd = `tar czf "${tarball}" --exclude="${slug}/environment/corpus" -C harbor-tasks "${slug}"`; const runTar = (): void => { execSync(tarCmd, { stdio: 'pipe' }); }; const taskDir = join('harbor-tasks', slug); // tar records each file's mode as-is, and this container runs as root — so a // write-only file (Claude Code writes subagent session records --w-------) is // archived, not refused, and every later extraction of it is unreadable. // Normalize before packing; the catch below still repairs what only tar can see. try { const pre = normalizeTreePermissions(taskDir); if (didRepair(pre)) { log.info( { ownerFixed: pre.ownerFixed.length, modeFixed: pre.modeFixed.length }, 'Normalized workspace permissions before packaging' ); } if (pre.failures.length > 0) { log.warn( { count: pre.failures.length, paths: pre.failures.slice(0, 5).map((f) => f.path) }, `Could not normalize some paths. If the tarball has unreadable files, run:\n ${manualRepairHint(taskDir)}` ); } } catch (repairErr) { log.warn( { err: repairErr }, `Permission normalization failed — packaging anyway. If the tarball has unreadable files, run:\n ${manualRepairHint(taskDir)}` ); } try { runTar(); } catch (err) { // Usually a permissions problem: something under harbor-tasks/ isn't // readable by you, so tar can't walk it. Say which paths, repair, retry once. const stderr = String((err as { stderr?: Buffer }).stderr ?? ''); const denied = [...stderr.matchAll(/^tar: (.+?): Cannot (?:stat|open)/gm)].map((m) => m[1]); if (denied.length === 0) throw err; log.warn( { paths: denied.slice(0, 5), total: denied.length }, 'Packaging hit a permissions error — attempting to repair ownership and modes' ); for (const rel of denied.slice(0, 5)) { const abs = join('harbor-tasks', rel); try { const st = statSync(abs); log.warn( { path: rel, uid: st.uid, gid: st.gid, mode: (st.mode & 0o7777).toString(8) }, 'offending path' ); } catch { // Can't even stat it from here — the repair below is the only recourse. } } // Guarded so a failed repair can't mask the real packaging error. let perms; try { perms = normalizeTreePermissions(taskDir); } catch (repairErr) { log.error({ err: repairErr }, 'Permission repair failed unexpectedly'); log.fatal( `Could not repair automatically. Run this on the host and try again:\n` + ` ${manualRepairHint(taskDir)}` ); throw err; } if (!didRepair(perms)) { log.fatal( `Nothing could be repaired automatically. Run this on the host and try again:\n` + ` ${manualRepairHint(taskDir)}` ); throw err; } log.info( { ownerFixed: perms.ownerFixed.length, modeFixed: perms.modeFixed.length }, 'Repaired workspace permissions — retrying' ); try { runTar(); } catch (retryErr) { log.fatal( `Still failing after repair. Run this on the host and try again:\n` + ` ${manualRepairHint(taskDir)}` ); throw retryErr; } } const tarSize = statSync(tarball).size; const tarSizeMB = (tarSize / 1024 / 1024).toFixed(1); log.info( { tarball, sizeMB: tarSizeMB, referenceRuns: result.refCount }, 'Task packaged — upload the tarball to the project' ); // Last word, deliberately: nothing about the managed files stops the package, so a // notice printed before the packaging summary is one the author has already // scrolled past by the time they read "Task packaged". // // Emitted through the logger rather than written to a stream directly. pino-pretty's // destination is async, so a synchronous process.stdout/stderr write races ahead of // pino's queued lines and lands near the TOP of a piped log — which is exactly the // "nobody reads it" problem this placement exists to solve. if (result.infraNotice) { log.warn(`\n\n${result.infraNotice}`); } }