Files
project-work/worker-toolkit-stocks-in-the-future/scripts/lib/input-checksums.ts

252 lines
11 KiB
TypeScript

/**
* input-checksums.ts — capture and compare sha256 checksums of the task inputs
* that reference runs and detector reports depend on.
*
* A reference run is only meaningful for the task inputs it actually ran
* against: the prompt (instruction.md), the snapshot session
* (environment/session.jsonl), the workspace patch
* (environment/workspace.patch), and the gitref the workspace is built from
* (task.toml `[metadata].commit`). Detector reports likewise assess a specific
* revision of instruction.md + tests/grader-guidance.md. When any of those
* change after the artifact was produced, the artifact is stale — it describes
* an older revision of the task than the one being packaged.
*
* This module is the single source of truth for WHAT gets checksummed and how
* captures are compared. Capture sites (copy-reference-run.ts,
* record-detector-inputs.ts) write a {@link TaskInputChecksums} record next to
* the artifact; submit-task.ts re-captures at packaging time and diffs.
* Content hashes rather than mtimes: a re-clone / whole-tree touch can fake or
* mask an mtime, but can't change a sha256.
*
* Lives in the worker toolkit's shipped file set (static/scripts/lib/), so in
* a packed toolkit it sits at scripts/lib/ next to both consumers. Repo-side
* callers go through the scripts/lib/input-checksums.ts re-export shim, which
* occupies the same relative path there (mirroring the check-devcontainer
* pattern).
*
* Related but deliberately separate: `computeDeliveryHash` (repo-side
* delivery script — grep the internal repo for it; not shipped with the
* toolkit) hashes an overlapping input set for delivery idempotency. It is
* NOT built on this module because its hash format is load-bearing (a
* changed hash re-delivers every task); if you change WHAT counts as a task
* input here, check whether the delivery hash needs the same change.
*/
import { createHash } from 'node:crypto';
import { existsSync, readFileSync } from 'node:fs';
import { join } from 'node:path';
/** Bump when the record shape changes incompatibly. */
export const INPUT_CHECKSUMS_VERSION = 1;
/** Filename of the record inside a reference-run directory. */
export const INPUT_CHECKSUMS_FILENAME = 'input-checksums.json';
/**
* Where in the artifact lifecycle a capture happened. The moment matters for
* how much a "fresh" verdict can be trusted:
*
* - 'run' — at trial launch (scripts/harbor-run stamps the trial dir).
* The strongest evidence: the record is what the agent ran
* against, whatever got edited afterwards.
* - 'copy' — at copy-reference-run time, the fallback when a trial carries
* no run-time stamp. An input edited between harbor-run and the
* copy is recorded at its post-edit state, so a stale run can
* read fresh.
* - 'stamp' — record-detector-inputs.ts, right after a detector skill
* writes its report.
* - 'mirror' — fetch-detectors --write-dir, when canonical detector
* reports are re-materialized to disk from their remote
* store (repo-side flow only; never written in a worker
* checkout).
*
* Absent on records written before this field existed.
*/
export type TaskInputCaptureMethod = 'run' | 'copy' | 'stamp' | 'mirror';
/**
* The checksums of a task's inputs as they stood at capture time. Every hash
* field is a sha256 hex digest, or `null` when the file didn't exist (a null
* that later becomes a hash — or vice versa — is a change like any other).
* `gitref` is the raw `[metadata].commit` string, recorded verbatim rather
* than hashed so a mismatch message can show it.
*/
export interface TaskInputChecksums {
version: number;
/** ISO-8601 timestamp of the capture. */
capturedAt: string;
/** Lifecycle point of the capture ({@link TaskInputCaptureMethod}). */
capturedBy?: TaskInputCaptureMethod;
/**
* The task slug the capture was taken from (harbor-tasks/<slug>). Written
* by launch-time captures so stamping can be scoped to the right task's
* trial dirs when several harbor-runs share a cwd — and so a mis-routed
* stamp is detectable after the fact.
*/
taskSlug?: string;
/**
* Set when the finalize flow re-stamped the prompt/graderGuidance hashes
* after the harbor-path scrub deliberately rewrote those docs
* (scripts/restamp-task-inputs.ts) — the run/report still reflects the
* task; only the doc bytes were normalized.
*/
restampedAt?: string;
inputs: {
prompt: string | null;
graderGuidance: string | null;
sessionJsonl: string | null;
workspacePatch: string | null;
gitref: string | null;
/**
* tests/grader-guidance-consolidated.md — the consolidated-standard
* guidance. Absent (undefined) on records captured before the field
* existed; comparisons skip a field the record predates, so old captures
* stay fresh until they are re-stamped. Last in field order because the
* task-checksum digest serializes fields in this order and appends new
* fields at the end (see scripts/lib/grader-run-checksums.ts).
*/
graderGuidanceConsolidated?: string | null;
};
}
export type TaskInputName = keyof TaskInputChecksums['inputs'];
/** Human-readable component names, used verbatim in staleness warnings. */
export const INPUT_LABELS: Record<TaskInputName, string> = {
prompt: 'prompt (instruction.md)',
graderGuidance: 'grader guidance (tests/grader-guidance.md)',
graderGuidanceConsolidated:
'consolidated grader guidance (tests/grader-guidance-consolidated.md)',
sessionJsonl: 'session snapshot (environment/session.jsonl)',
workspacePatch: 'workspace patch (environment/workspace.patch)',
gitref: 'gitref (task.toml commit)',
};
/**
* The inputs that shape what the AGENT saw and did. Changing any of them means
* a captured reference run no longer reflects the task being packaged, and
* only re-running the agent can fix that. The grader-guidance files (legacy
* and consolidated) are deliberately NOT in this set: editing the rubric
* stales the run's GRADE, not the run itself, and `scripts/harbor-regrade`
* re-derives grades without re-running the agent.
*/
export const REFERENCE_RUN_INPUTS: readonly TaskInputName[] = [
'prompt',
'sessionJsonl',
'workspacePatch',
'gitref',
];
/**
* The inputs a detector report assesses — instruction.md plus whichever
* grader-guidance files the task carries (the legacy pair is what the old
* mtime comparison in submit-task.ts watched; the consolidated file joined
* when grading moved to the consolidated standard). Compared by content.
*/
export const DETECTOR_REPORT_INPUTS: readonly TaskInputName[] = [
'prompt',
'graderGuidance',
'graderGuidanceConsolidated',
];
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
function sha256File(filePath: string): string | null {
if (!existsSync(filePath)) return null;
return createHash('sha256').update(readFileSync(filePath)).digest('hex');
}
/**
* The gitref (`[metadata].commit`) from a task.toml, or null when the file is
* missing, unreadable, or has no commit line — a read failure downgrades to
* "absent" rather than crashing a capture or a validation sweep.
*
* Deliberately a regex, not a TOML parser: this module ships in the worker
* toolkit, where a new runtime dep would break packaging for every worker
* whose container predates the dep (npm install runs only on container
* create, and containers survive toolkit upgrades). build-workspace.sh reads
* the same key with the same grep-a-`commit`-line approach. The one `commit`
* key in a task.toml is `[metadata].commit`, so anchoring to the first
* `commit = "…"` line is exact in practice.
*/
function readGitref(taskDir: string): string | null {
const tomlPath = join(taskDir, 'task.toml');
if (!existsSync(tomlPath)) return null;
try {
const match = /^\s*commit\s*=\s*(?:"([^"\n]+)"|'([^'\n]+)')\s*(?:#.*)?$/m.exec(
readFileSync(tomlPath, 'utf-8')
);
const commit = match?.[1] ?? match?.[2];
return commit && commit.length > 0 ? commit : null;
} catch {
return null;
}
}
/** Checksum the task inputs as they stand right now under `taskDir`. */
export function captureTaskInputs(
taskDir: string,
capturedBy?: TaskInputCaptureMethod
): TaskInputChecksums {
return {
version: INPUT_CHECKSUMS_VERSION,
capturedAt: new Date().toISOString(),
...(capturedBy ? { capturedBy } : {}),
inputs: {
prompt: sha256File(join(taskDir, 'instruction.md')),
graderGuidance: sha256File(join(taskDir, 'tests', 'grader-guidance.md')),
sessionJsonl: sha256File(join(taskDir, 'environment', 'session.jsonl')),
workspacePatch: sha256File(join(taskDir, 'environment', 'workspace.patch')),
gitref: readGitref(taskDir),
graderGuidanceConsolidated: sha256File(
join(taskDir, 'tests', 'grader-guidance-consolidated.md')
),
},
};
}
/**
* Read a previously captured record. Returns null when the file is missing or
* doesn't look like a capture (pre-tracking artifact, hand-edited JSON, a
* future incompatible version) — callers treat null as "staleness unknowable",
* never as an error. No zod here: this module ships in the worker toolkit,
* whose dependency set stays minimal, so the guard is manual.
*/
export function readTaskInputChecksums(filePath: string): TaskInputChecksums | null {
if (!existsSync(filePath)) return null;
let parsed: unknown;
try {
parsed = JSON.parse(readFileSync(filePath, 'utf-8'));
} catch {
return null;
}
if (typeof parsed !== 'object' || parsed === null) return null;
const record = parsed as TaskInputChecksums;
if (record.version !== INPUT_CHECKSUMS_VERSION) return null;
if (typeof record.inputs !== 'object' || record.inputs === null) return null;
for (const name of Object.keys(INPUT_LABELS) as TaskInputName[]) {
const value = record.inputs[name];
// undefined = the record predates this input field; still a valid capture.
if (value !== undefined && value !== null && typeof value !== 'string') return null;
}
return record;
}
/**
* Which of `names` changed between a recorded capture and the current state?
* Returns the human-readable labels ({@link INPUT_LABELS}) of every component
* whose value differs — including absent→present and present→absent flips. A
* field the recorded capture predates (the key is not in the record at all)
* is skipped: freshness on that axis is unknowable, and flagging every old
* record the moment a new axis ships would drown the real signal.
*/
export function diffTaskInputs(
recorded: TaskInputChecksums,
current: TaskInputChecksums,
names: readonly TaskInputName[]
): string[] {
return names
.filter((name) => name in recorded.inputs)
.filter((name) => recorded.inputs[name] !== current.inputs[name])
.map((name) => INPUT_LABELS[name]);
}