lots of change - all to start my 3rd redo

This commit is contained in:
2026-09-26 14:31:52 -04:00
parent 7f4d388e19
commit bceb52e8ee
1046 changed files with 4476 additions and 0 deletions

View File

@@ -1,8 +0,0 @@
{
"name": "create-snapshot",
"description": "Capture conversation context and repo state as a snapshot",
"version": "0.1.0",
"author": {
"name": "raccoon"
}
}

View File

@@ -1,788 +0,0 @@
#!/usr/bin/env node
import { execSync } from 'node:child_process';
import crypto from 'node:crypto';
import fs from 'node:fs';
import path from 'node:path';
import { linearSnapshotLines, readSession } from './harness-session.mjs';
// --- Argument parsing ---
function parseArgs(argv) {
const args = {};
for (let i = 2; i < argv.length; i++) {
if (argv[i].startsWith('--')) {
const key = argv[i].slice(2);
const val = argv[i + 1];
if (!val || val.startsWith('--')) {
args[key] = true;
} else {
args[key] = val;
i++;
}
}
}
return args;
}
// Last-resort data dir. Claude Code's plugin runtime always provides one, so this is
// what makes the start marker and session record work under any other harness.
function defaultDataDir() {
const home = process.env.HOME || '/root';
return path.join(home, '.raccoon', 'snapshot-data');
}
const args = parseArgs(process.argv);
const slug = args.slug;
const annotationPath = args.annotation;
const outputDir = args['output-dir'];
if (!args['mark-start'] && (!slug || !annotationPath || !outputDir)) {
console.error(
'Usage: capture-snapshot.mjs --slug <slug> --annotation <path> --output-dir <dir> [--plugin-data <path>]\n' +
' capture-snapshot.mjs --mark-start [--harness <id>]'
);
process.exit(1);
}
// --- Locate session info ---
// --harness wins over the launcher's RACCOON_HARNESS so a caller that knows which
// conversation it is capturing can say so; the default keeps Claude Code's plugin
// working unchanged. Claude Code has an exact cut point (the snapshot slash command)
// and a message tree to prune, so it keeps the bespoke path below; other harnesses go
// through the shared reader.
const HARNESS = args.harness || process.env.RACCOON_HARNESS || 'claude-code';
const IS_CLAUDE = HARNESS === 'claude-code';
// The generated restore.sh writes one of exactly two session layouts, and everything
// below branches on IS_CLAUDE — so a third harness would silently be handed codex's
// $CODEX_HOME/sessions paths. Refuse instead; adding a harness means adding a layout.
if (!IS_CLAUDE && HARNESS !== 'codex') {
console.error(
`capture-snapshot: no session-restore layout for harness "${HARNESS}". ` +
'Add one to capture-snapshot.mjs (and harness-session.mjs) before capturing with it.'
);
process.exit(1);
}
// Try multiple strategies to find the current session transcript:
// 1. Plugin data dir (from SessionStart hook)
// 2. Scan ~/.claude/projects/ for the most recently modified JSONL
let session_id = null;
let transcript_path = null;
const dataDir =
args['plugin-data'] ||
process.env.RACCOON_SNAPSHOT_DATA ||
process.env.CLAUDE_PLUGIN_DATA ||
(process.env.CLAUDE_PLUGIN_ROOT && path.join(process.env.CLAUDE_PLUGIN_ROOT, '.data')) ||
defaultDataDir();
// The SessionStart hook records the live session for every harness, so prefer it over
// guessing. `readSession` falls back to the newest file on disk when it is absent.
let recordedSession = null;
if (dataDir) {
const sessionInfoPath = path.join(dataDir, 'current-session.json');
if (fs.existsSync(sessionInfoPath)) {
try {
recordedSession = JSON.parse(fs.readFileSync(sessionInfoPath, 'utf8'));
} catch {
recordedSession = null;
}
}
}
const startMarkerPath = dataDir ? path.join(dataDir, 'snapshot-start.json') : null;
// `--mark-start` runs BEFORE the annotation Q&A and records how long the conversation
// was at that moment. It is the linear-harness stand-in for Claude Code's slash-command
// line: without it, capture would stage the snapshot's own Q&A as conversation.
if (args['mark-start']) {
const session = readSession(HARNESS);
if (!session) {
console.error(`No ${HARNESS} session found to mark.`);
process.exit(1);
}
if (!startMarkerPath) {
console.error('No data dir available to record the snapshot start marker.');
process.exit(1);
}
fs.mkdirSync(path.dirname(startMarkerPath), { recursive: true });
fs.writeFileSync(
startMarkerPath,
JSON.stringify(
{ harness: HARNESS, transcript_path: session.rawPath, line_count: session.lines.length },
null,
2
) + '\n'
);
console.log(`Snapshot start marked at ${session.lines.length} records.`);
process.exit(0);
}
let harnessSession = null;
if (!IS_CLAUDE) {
harnessSession = readSession(HARNESS, recordedSession?.transcript_path);
if (!harnessSession) {
console.error(
`Could not find a ${HARNESS} session to capture. Capture has to run from inside the ${HARNESS} conversation you want to snapshot.`
);
process.exit(1);
}
transcript_path = harnessSession.rawPath;
// Fall back to a fresh id only if the harness records none — restore.sh names the
// installed session by it, so it has to match what `resume` will look up.
session_id = recordedSession?.session_id || harnessSession.sessionId || crypto.randomUUID();
} else if (recordedSession) {
session_id = recordedSession.session_id;
transcript_path = recordedSession.transcript_path;
}
// Fallback: find the most recently modified JSONL in ~/.claude/projects/
if (!transcript_path) {
const homeDir = process.env.HOME || '/root';
const projectsDir = path.join(homeDir, '.claude', 'projects');
if (fs.existsSync(projectsDir)) {
let newest = null;
let newestMtime = 0;
for (const projEntry of fs.readdirSync(projectsDir)) {
const projDir = path.join(projectsDir, projEntry);
if (!fs.statSync(projDir).isDirectory()) continue;
for (const file of fs.readdirSync(projDir)) {
if (!file.endsWith('.jsonl')) continue;
const filePath = path.join(projDir, file);
const mtime = fs.statSync(filePath).mtimeMs;
if (mtime > newestMtime) {
newestMtime = mtime;
newest = filePath;
session_id = file.replace(/\.jsonl$/, '');
}
}
}
transcript_path = newest;
}
}
if (!transcript_path || !fs.existsSync(transcript_path)) {
console.error(
"Could not find a Claude Code session transcript. This script should be run from within a Claude Code conversation via the /create-snapshot:snapshot command. Please file a bug if you're seeing this unexpectedly."
);
process.exit(1);
}
if (!transcript_path || !fs.existsSync(transcript_path)) {
console.error(`Transcript file not found at ${transcript_path}. Please file a bug.`);
process.exit(1);
}
// --- Require a git repo at capture time ---
//
// A snapshot is "commit SHA + diff vs HEAD", reconstituted later via
// `git archive <SHA> | tar -x` + `git apply workspace.patch`. Without a git
// repo here we have no SHA to pin, no patch to record, and no way for
// downstream `build-workspace.sh` to reproduce the workspace — the resulting
// snapshot would be structurally meaningless. This check runs BEFORE the
// snapshot directory is created so a misconfigured invocation leaves no
// half-written state behind.
// Find the git repo by asking git itself — walks up from cwd looking for
// `.git`, handling submodules and worktrees correctly. Returns null when
// cwd is outside any repo, so the worker gets a clear "cd into your repo"
// error instead of silently descending into something they didn't name.
function findGitRepo() {
try {
const top = execSync('git rev-parse --show-toplevel', {
stdio: ['ignore', 'pipe', 'ignore'],
})
.toString()
.trim();
return top || null;
} catch {
return null;
}
}
const gitRepo = findGitRepo();
if (!gitRepo) {
console.error(
"Error: Not running inside a git repo. /create-snapshot needs a git repo so it can pin a commit SHA and record a diff of in-flight changes; without one the snapshot can't be reproduced as a task. cd into the repo you're exploring (the toolkit's repo/ submodule) and re-run /create-snapshot:snapshot."
);
process.exit(1);
}
// --- Create snapshot directory ---
const ts = new Date().toISOString().replace(/[-:]/g, '').replace('T', '-').slice(0, 15); // 20260403-225449
const snapshotDir = path.join(outputDir, `${ts}-${slug}`);
if (fs.existsSync(snapshotDir)) {
console.error(`Snapshot directory already exists: ${snapshotDir}\nPlease file a bug.`);
process.exit(1);
}
fs.mkdirSync(snapshotDir, { recursive: true });
// --- Copy conversation transcript (trimmed + branch-pruned) ---
// The JSONL is a tree of messages linked by parentUuid. When the user rewinds
// a conversation, old branches remain in the file. We need to:
// 1. Cut at the LAST /create-snapshot:snapshot command (later invocations
// supersede earlier ones in the same session)
// 2. Find the tip of the active branch (last message before the cut)
// 3. Walk parentUuid back to the root, collecting only messages on that path
// 4. Exclude the Q&A subgraphs of any PRIOR /create-snapshot:snapshot
// invocations in this session (their cut points are on the same
// conversation branch, so the walk would otherwise pull in the
// assistant's annotation questions and the user's answers — a
// contamination path that snapshot.patch doesn't show). Boundaries
// for prior invocations are recorded in a side file (see end of
// this script) so this run can identify them.
// 5. Drop bookkeeping entries whose content can leak rewound-branch state.
const rawLines = fs.readFileSync(transcript_path, 'utf8').trimEnd().split('\n');
// A user message is a /create-snapshot:snapshot invocation when its content
// STARTS with one of Claude Code's slash-command tags AND mentions the
// command name. The "starts with" guard distinguishes a real invocation
// from prose that quotes the command (a worker reporting a bug, the
// command-listing skill output, etc.) — prose doesn't begin with those
// tags. The whitespace-tolerant pattern survives minor format drift in
// Claude Code's slash-command rendering.
const SNAPSHOT_CMD_PATTERN =
/<command-(?:name|message)>\s*\/?\s*create-snapshot:snapshot\s*<\/command-(?:name|message)>/;
function isSnapshotCommandContent(content) {
if (typeof content !== 'string') return false;
const trimmed = content.trimStart();
if (!trimmed.startsWith('<command-name>') && !trimmed.startsWith('<command-message>')) {
return false;
}
return SNAPSHOT_CMD_PATTERN.test(content);
}
// Find every snapshot-command line index, in order. The LAST one is the
// current invocation (cut point); earlier ones bound prior Q&A subgraphs.
const snapshotCmdIndexes = [];
for (let i = 0; i < rawLines.length; i++) {
try {
const entry = JSON.parse(rawLines[i]);
if (entry.type === 'user' && isSnapshotCommandContent(entry.message?.content)) {
snapshotCmdIndexes.push(i);
}
} catch {
// Skip malformed lines
}
}
const cutIndex =
snapshotCmdIndexes.length > 0
? snapshotCmdIndexes[snapshotCmdIndexes.length - 1]
: rawLines.length;
const priorCmdIndexes = snapshotCmdIndexes.slice(0, -1);
// Load prior-snapshot boundary records so we know where each earlier
// invocation's Q&A subgraph ended. The boundary file is written at the
// end of every capture run (see below) and is keyed by session uuid.
function loadPriorBoundaries() {
if (!dataDir || !session_id) return [];
const boundariesPath = path.join(dataDir, 'snapshot-boundaries.jsonl');
if (!fs.existsSync(boundariesPath)) return [];
const lines = fs.readFileSync(boundariesPath, 'utf8').trimEnd().split('\n');
const out = [];
for (const line of lines) {
if (!line) continue;
try {
const rec = JSON.parse(line);
if (rec.sessionUuid === session_id && rec.snapshotCommandUuid) out.push(rec);
} catch {
/* skip malformed */
}
}
return out;
}
const priorBoundaries = loadPriorBoundaries();
// Compute the line ranges to exclude for each prior snapshot. The Q&A
// subgraph starts at the prior snapshot's command line and runs through
// the line whose entry uuid matches the boundary record (the last entry
// in the JSONL when that prior capture-snapshot completed).
//
// Fall back to the next snapshot command (or the current cut) when no
// matching boundary record exists — better to drop too much than to leak
// the Q&A; the visible cost is excluding any "real work" that happened
// between snapshots without a recorded boundary, which only occurs if
// the boundary log was wiped or the prior capture crashed.
function findUuidLineIndex(targetUuid, startLine, endLineExclusive) {
for (let i = startLine; i < endLineExclusive; i++) {
try {
const entry = JSON.parse(rawLines[i]);
if (entry.uuid === targetUuid) return i;
} catch {
/* skip */
}
}
return -1;
}
const priorQAExcludedLines = new Set();
for (let i = 0; i < priorCmdIndexes.length; i++) {
const startLine = priorCmdIndexes[i];
const nextCutLine = i + 1 < priorCmdIndexes.length ? priorCmdIndexes[i + 1] : cutIndex;
let snapshotCmdUuid = null;
try {
snapshotCmdUuid = JSON.parse(rawLines[startLine]).uuid || null;
} catch {
/* unparseable command line — skip */
}
let endLine = -1;
if (snapshotCmdUuid) {
const boundary = priorBoundaries.find((b) => b.snapshotCommandUuid === snapshotCmdUuid);
if (boundary && boundary.lastEntryUuid) {
endLine = findUuidLineIndex(boundary.lastEntryUuid, startLine, nextCutLine);
}
}
// No matching boundary: bound the exclusion at the next snapshot/current
// cut so the Q&A doesn't leak even if state was lost.
if (endLine < 0) endLine = nextCutLine - 1;
for (let j = startLine; j <= endLine; j++) priorQAExcludedLines.add(j);
}
// Step 2: parse all entries before the cut, build uuid index
const preCutEntries = [];
const byUuid = {};
for (let i = 0; i < cutIndex; i++) {
try {
const entry = JSON.parse(rawLines[i]);
preCutEntries.push({ line: rawLines[i], entry, index: i });
if (entry.uuid) {
byUuid[entry.uuid] = entry;
}
} catch {
// Keep unparseable lines (they'll be included as non-message entries)
preCutEntries.push({ line: rawLines[i], entry: null, index: i });
}
}
// Step 3: find the tip of the active branch. The snapshot command's parentUuid
// points to the message the user was looking at when they ran the snapshot —
// this is authoritative even after rewinds.
let tipUuid = null;
if (cutIndex < rawLines.length) {
try {
const snapshotCmd = JSON.parse(rawLines[cutIndex]);
tipUuid = snapshotCmd.parentUuid || null;
} catch {
// not valid JSON — leave tipUuid null
}
}
// If the live tip is itself inside a prior snapshot's Q&A subgraph (e.g.
// the user ran /create-snapshot:snapshot a second time WITHOUT typing
// anything between the two — there's no "real work" gap), walk back past
// the excluded range to find the closest non-excluded ancestor. Otherwise
// activeBranchUuids would be empty and we'd produce an empty snapshot.
function nearestNonExcludedAncestor(startUuid) {
let cur = startUuid;
while (cur) {
const e = byUuid[cur];
if (!e) return cur; // unknown uuid — best effort, keep
// Find the line index of this entry to check exclusion.
// (Line index isn't stored on the entry; recompute via preCutEntries.)
const found = preCutEntries.find((p) => p.entry?.uuid === cur);
if (!found || !priorQAExcludedLines.has(found.index)) return cur;
cur = e.parentUuid || null;
}
return null;
}
if (tipUuid) tipUuid = nearestNonExcludedAncestor(tipUuid);
// Fallback: if no snapshot command found, use the last entry with a uuid
if (!tipUuid) {
for (let i = preCutEntries.length - 1; i >= 0; i--) {
if (preCutEntries[i].entry?.uuid && !priorQAExcludedLines.has(preCutEntries[i].index)) {
tipUuid = preCutEntries[i].entry.uuid;
break;
}
}
}
// Collect all uuids on the active branch
const activeBranchUuids = new Set();
let current = tipUuid;
while (current) {
activeBranchUuids.add(current);
current = byUuid[current]?.parentUuid || null;
}
// Step 4: filter — keep entries on the active branch.
//
// Claude Code writes several bookkeeping entry types alongside the message
// tree that don't carry a branch uuid. Their content references whatever
// branch was active when they were written, so if the user has rewound,
// these will leak rewound-branch state (file backups, prior prompt text,
// stale titles, queued prompts, PR links, etc.) into the snapshot — a leak
// snapshot.patch doesn't show. Drop the ones we can't attribute to the
// active branch.
//
// `file-history-snapshot` is special-cased: it carries a `messageId`
// pointing at the message whose pre-edit state it tracks, so we can
// keep only those whose messageId is on the active branch. That
// preserves /rewind functionality after a snapshot is restored (rewind
// needs the file-backup metadata) while still dropping records from
// rewound branches.
const BLANKET_DROP_TYPES = new Set([
'agent-name',
'ai-title',
'custom-title',
'last-prompt',
'permission-mode',
'pr-link',
'queue-operation',
]);
function prunedClaudeLines() {
const kept = [];
for (const { line, entry, index } of preCutEntries) {
if (priorQAExcludedLines.has(index)) continue;
if (!entry) {
// Unparseable line — keep as-is so we don't lose data we can't classify.
kept.push(line);
continue;
}
if (entry.uuid) {
if (activeBranchUuids.has(entry.uuid)) kept.push(line);
continue;
}
// No uuid: bookkeeping entry.
if (entry.type === 'file-history-snapshot') {
// Keep only if the message it tracks is on the active branch.
if (entry.messageId && activeBranchUuids.has(entry.messageId)) {
kept.push(line);
}
continue;
}
if (!BLANKET_DROP_TYPES.has(entry.type)) kept.push(line);
}
return kept;
}
// Rewind branches and prior-Q&A exclusion are Claude-transcript concerns; a linear
// harness transcript just truncates at its boundary.
let startLine;
if (!IS_CLAUDE && startMarkerPath && fs.existsSync(startMarkerPath)) {
try {
const marker = JSON.parse(fs.readFileSync(startMarkerPath, 'utf8'));
if (marker.transcript_path === harnessSession.rawPath) startLine = marker.line_count;
} catch {
startLine = undefined;
}
}
if (!IS_CLAUDE && startLine === undefined) {
console.error(
'WARNING: no snapshot start marker for this session — the snapshot Q&A may be captured as conversation. Run capture-snapshot.mjs --mark-start before the annotation questions.'
);
}
const outputLines = IS_CLAUDE
? prunedClaudeLines()
: linearSnapshotLines(harnessSession, startLine);
if (outputLines.length === 0) {
console.error(
`WARNING: found no conversation to seed in ${transcript_path}, so this snapshot has no prior turns. The task will run cold from its prompt alone — fine if that is what you want, but if you meant to capture a conversation, check that the exchange you wanted came BEFORE this snapshot.`
);
}
// Zero bytes, not a lone newline, when there is nothing to seed: downstream decides
// single- vs multi-turn on the file's SIZE, so a 1-byte file would try to resume nothing.
fs.writeFileSync(
path.join(snapshotDir, 'session.jsonl'),
outputLines.length > 0 ? outputLines.join('\n') + '\n' : ''
);
// --- Write boundary record so the NEXT capture-snapshot in this session
// can identify and exclude this snapshot's Q&A subgraph ---
//
// The record pairs the current invocation's command-line uuid with the
// uuid of the last entry in the JSONL at this moment (which is whichever
// assistant turn invoked us as a tool). A subsequent capture run reads
// this file, finds these two uuids in its raw lines, and excludes the
// range — a small leak still exists for entries appended AFTER capture
// returns (the assistant's "Snapshot saved to: ..." reply), but the
// substantive annotation Q&A is fully bounded.
if (dataDir && session_id && cutIndex < rawLines.length) {
let snapshotCommandUuid = null;
try {
snapshotCommandUuid = JSON.parse(rawLines[cutIndex]).uuid || null;
} catch {
/* leave null — we'll skip writing */
}
// Re-read transcript so we pick up any lines Claude Code has appended
// since we read it above (the assistant's tool-use entry, etc.).
let lastEntryUuid = null;
try {
const liveLines = fs.readFileSync(transcript_path, 'utf8').trimEnd().split('\n');
for (let i = liveLines.length - 1; i >= 0; i--) {
try {
const e = JSON.parse(liveLines[i]);
if (e.uuid) {
lastEntryUuid = e.uuid;
break;
}
} catch {
/* skip */
}
}
} catch {
/* transcript unreadable now — skip writing */
}
if (snapshotCommandUuid && lastEntryUuid) {
const boundariesPath = path.join(dataDir, 'snapshot-boundaries.jsonl');
try {
fs.mkdirSync(dataDir, { recursive: true });
fs.appendFileSync(
boundariesPath,
JSON.stringify({
sessionUuid: session_id,
snapshotCommandUuid,
lastEntryUuid,
timestamp: new Date().toISOString(),
}) + '\n'
);
} catch {
// Best-effort: a missing boundary just means the next run falls back
// to the conservative "exclude through next snapshot" heuristic.
}
}
}
// Copy subagents and tool-results if they exist
const sessionSiblingDir = transcript_path.replace(/\.jsonl$/, '');
if (fs.existsSync(sessionSiblingDir) && fs.statSync(sessionSiblingDir).isDirectory()) {
fs.cpSync(sessionSiblingDir, path.join(snapshotDir, 'session'), { recursive: true });
// Claude Code creates subagent files with write-only permissions (--w-------).
// Fix them so downstream tools (cpSync in snapshot-to-task, Harbor's dirhash) can read them.
execSync(`chmod -R +r "${path.join(snapshotDir, 'session')}"`, { stdio: 'pipe' });
}
// --- Capture git state as a patch ---
// Returns raw stdout bytes — callers that want a single-line value must
// .trim() themselves. Don't trim here: some callers (git diff) produce
// patches where a trailing " \n" blank-context line is load-bearing, and
// stripping it corrupts the patch.
function git(cmd, opts) {
try {
return execSync(`git ${cmd}`, {
encoding: 'utf8',
maxBuffer: 50 * 1024 * 1024,
cwd: gitRepo,
stdio: ['pipe', 'pipe', 'pipe'],
...opts,
});
} catch {
return null;
}
}
const commit = git('rev-parse HEAD')?.trim() ?? null;
const branch = git('rev-parse --abbrev-ref HEAD')?.trim() ?? null;
const remoteUrl = git('remote get-url origin')?.trim() ?? null;
// Generate a unified patch representing the workspace state AT THE END OF
// THE PRIOR TURN — i.e., everything done up to but not including the turn
// being snapshotted. This is the state the trial agent should inherit so
// it gets a fresh attempt at the prompt that triggered the snapshot.
//
// The UserPromptSubmit hook checkpoints the working tree to
// `refs/raccoon/turn-checkpoint` at every turn boundary (skipping snapshot
// invocations themselves), so the latest checkpoint is exactly the state
// at the start of the snapshotted turn. We diff HEAD against that
// checkpoint to produce the patch.
//
// Falls back to the pre-checkpoint behavior (full working-tree diff) when
// no checkpoint exists — e.g., the worker took a snapshot before any
// non-snapshot user message was sent, or the hook never fired (legacy
// session, plugin re-installed mid-session, etc.).
if (gitRepo) {
try {
const tmpIndex = path.join(snapshotDir, '.tmp-git-index');
const indexEnv = { ...process.env, GIT_INDEX_FILE: tmpIndex };
// Prefer the FROZEN ref — this is set by checkpoint-workspace at the
// moment the user invokes /create-snapshot:*, before any Q&A turns
// have a chance to advance the live checkpoint past the state we
// want to capture. Fall back to the live checkpoint (then to
// working-tree diff) for backward-compat or if the freeze step failed.
let baseline = null;
try {
baseline = git('rev-parse refs/raccoon/turn-checkpoint-frozen')?.trim() ?? null;
} catch {
baseline = null;
}
if (!baseline) {
try {
baseline = git('rev-parse refs/raccoon/turn-checkpoint')?.trim() ?? null;
} catch {
baseline = null;
}
}
// --binary --full-index, on both branches: a plain `git diff` records a
// binary difference as an opaque `Binary files a/x and /dev/null differ`
// stub, and `git apply` refuses it ("without full index line"), so
// build-workspace.sh can't rebuild the task at all. Nobody has to edit a
// binary to hit this — a tracked .DS_Store the toolkit zip strips from the
// shipped checkout reads as a binary deletion in every session.
//
// maxBuffer: inlined binaries make patches far bigger than text diffs, and
// exceeding the default cap would throw away the whole patch silently.
const diffOpts = { env: indexEnv, maxBuffer: 512 * 1024 * 1024 };
let patch;
if (baseline) {
// Diff HEAD against the prior-turn checkpoint. Untracked files in
// the checkpoint have been committed to the checkpoint tree, so
// they're included automatically.
patch = git(`diff --binary --full-index HEAD ${baseline}`, diffOpts);
} else {
// No checkpoint — fall back to live working-tree diff (pre-fix
// behavior). Captures everything different from HEAD, including
// any agent edits during the current turn.
git('read-tree HEAD', { env: indexEnv });
git('add -A', { env: indexEnv });
patch = git('diff --cached --binary --full-index HEAD', diffOpts);
}
try {
fs.unlinkSync(tmpIndex);
} catch {
/* ignore */
}
if (patch) {
fs.writeFileSync(
path.join(snapshotDir, 'snapshot.patch'),
patch.endsWith('\n') ? patch : patch + '\n'
);
}
} catch {
// Read-only repo or other git error — skip patch generation
}
}
// --- Copy annotation ---
const annotation = JSON.parse(fs.readFileSync(annotationPath, 'utf8'));
fs.writeFileSync(
path.join(snapshotDir, 'annotation.json'),
JSON.stringify(annotation, null, 2) + '\n'
);
// Clean up temp file
try {
fs.unlinkSync(annotationPath);
} catch {
// Ignore cleanup failures
}
// --- Write metadata ---
const metadata = {
slug: slug,
session_uuid: session_id,
// The harness the session was actually read as, so it can't disagree with what
// was captured.
harness: HARNESS,
original_cwd: process.cwd(),
commit: commit,
branch: branch,
remote_url: remoteUrl,
timestamp: new Date().toISOString(),
plugin_version: '0.2.0',
};
fs.writeFileSync(path.join(snapshotDir, 'metadata.json'), JSON.stringify(metadata, null, 2) + '\n');
// --- Generate restore.sh ---
const restoreScript = `#!/usr/bin/env bash
set -euo pipefail
# Restore a snapshot for resuming a Claude Code conversation.
#
# Usage: ./restore.sh [target-dir]
# target-dir: directory to clone/checkout the repo into (default: ./repo)
SCRIPT_DIR="$(cd "$(dirname "\${BASH_SOURCE[0]}")" && pwd)"
TARGET_DIR="\${1:-./repo}"
# Read metadata
COMMIT=$(jq -r '.commit' "$SCRIPT_DIR/metadata.json")
REMOTE=$(jq -r '.remote_url' "$SCRIPT_DIR/metadata.json")
SESSION_UUID=$(jq -r '.session_uuid' "$SCRIPT_DIR/metadata.json")
echo "Cloning $REMOTE at $COMMIT..."
git clone "$REMOTE" "$TARGET_DIR"
cd "$TARGET_DIR"
git checkout "$COMMIT"
# Apply snapshot patch if present
if [ -f "$SCRIPT_DIR/snapshot.patch" ]; then
echo "Applying snapshot.patch..."
git apply "$SCRIPT_DIR/snapshot.patch"
fi
# Install conversation so the authoring harness can resume it
${
IS_CLAUDE
? `ENCODED_CWD=$(echo "$PWD" | sed 's|/|-|g; s|^-||')
DEST_DIR="$HOME/.claude/projects/-$ENCODED_CWD"
mkdir -p "$DEST_DIR"
cp "$SCRIPT_DIR/session.jsonl" "$DEST_DIR/$SESSION_UUID.jsonl"
if [ -d "$SCRIPT_DIR/session" ]; then
cp -r "$SCRIPT_DIR/session" "$DEST_DIR/$SESSION_UUID"
fi
echo ""
echo "Snapshot restored. To resume the conversation:"
echo " cd $TARGET_DIR"
echo " claude --resume $SESSION_UUID"`
: `DEST_DIR="\${CODEX_HOME:-$HOME/.codex}/sessions/$(date -u +%Y/%m/%d)"
mkdir -p "$DEST_DIR"
cp "$SCRIPT_DIR/session.jsonl" \\
"$DEST_DIR/rollout-$(date -u +%Y-%m-%dT%H-%M-%S).000Z-$SESSION_UUID.jsonl"
echo ""
echo "Snapshot restored. To resume the conversation:"
echo " cd $TARGET_DIR"
echo " codex resume $SESSION_UUID"`
}
`;
fs.writeFileSync(path.join(snapshotDir, 'restore.sh'), restoreScript);
fs.chmodSync(path.join(snapshotDir, 'restore.sh'), 0o755);
try {
execSync('bash -ic "_ev snapshot_created 2>/dev/null" 2>/dev/null', {
stdio: 'ignore',
timeout: 5000,
});
} catch {
// best-effort
}
// --- Done ---
const fullSnapshotDir = path.resolve(snapshotDir);
console.log(`Snapshot saved to: ${fullSnapshotDir}`);
console.log(` session.jsonl — conversation transcript`);
if (fs.existsSync(sessionSiblingDir) && fs.statSync(sessionSiblingDir).isDirectory()) {
console.log(` session/ — subagents + tool results`);
}
if (fs.existsSync(path.join(snapshotDir, 'snapshot.patch'))) {
console.log(` snapshot.patch — working tree changes`);
}
console.log(` annotation.json — worker annotations`);
console.log(` metadata.json — session metadata`);
console.log(` restore.sh — restore script for resuming`);

View File

@@ -1,485 +0,0 @@
#!/usr/bin/env node
// Rewind-aware workspace checkpointing for the reduced-toolset Explore agent.
// One script, two hook events (branches on hook_event_name):
//
// UserPromptSubmit -> CAPTURE
// Snapshot the pre-turn working tree into refs/raccoon/turn-checkpoint (the
// chain capture-snapshot uses for snapshot.patch) AND record, in
// .git/raccoon-state.json, anchor_map[tip] = checkpoint-commit and
// last_anchor = tip. `tip` is the conversation node the new prompt attaches
// to (the END of the previous turn) — exactly the node a future /rewind to
// THIS turn will branch from. Anchoring to the prior tip (not the
// just-submitted, maybe-unflushed message) makes capture race-free.
//
// PreToolUse (first tool call of a turn) -> RECONCILE
// Claude Code's /rewind restores the conversation but NOT bash-made edits,
// and fires no hook. By the first tool call the post-rewind branch message is
// reliably persisted and the agent has not yet read/edited code. We parse the
// transcript into a parentUuid DAG, pick the ACTIVE branch (leaf with the
// newest tip), and walk it for the newest checkpoint anchor that is a genuine
// rewind fork (the anchor still has an orphaned child branch — the discarded
// turns). If found, restore the working tree to that checkpoint before the tool
// runs. Idempotent per (anchor, branch): restores once per rewind.
//
// Every failure path is a safe no-op: the hook never aborts the session and
// never restores to an unverified tree.
import { execSync } from 'node:child_process';
import fs from 'node:fs';
import path from 'node:path';
const CHECKPOINT_REF = 'refs/raccoon/turn-checkpoint';
const FROZEN_REF = 'refs/raccoon/turn-checkpoint-frozen';
// Synthetic parent for every parentless transcript node, so that a rewind to the
// VERY FIRST turn (where the new prompt also has parentUuid=null) is detected by
// the same divergence machinery as any other turn.
const ROOT = '__ROOT__';
const RACCOON_AUTHOR = {
GIT_AUTHOR_NAME: 'raccoon',
GIT_AUTHOR_EMAIL: 'raccoon@local',
GIT_COMMITTER_NAME: 'raccoon',
GIT_COMMITTER_EMAIL: 'raccoon@local',
};
function makeGit(gitDir, extraEnv) {
const env = { ...process.env, ...extraEnv };
return (cmd) =>
execSync(`git ${cmd}`, {
cwd: gitDir,
env,
stdio: ['pipe', 'pipe', 'pipe'],
encoding: 'utf8',
}).trim();
}
function findGitDir(cwd) {
let gitDir = cwd;
for (let i = 0; i < 10; i++) {
if (fs.existsSync(path.join(gitDir, '.git'))) return gitDir;
const parent = path.dirname(gitDir);
if (parent === gitDir) return null;
gitDir = parent;
}
return null;
}
// ---- transcript + state ----
function readEntries(transcriptPath) {
try {
if (!transcriptPath || !fs.existsSync(transcriptPath)) return [];
const out = [];
for (const raw of fs.readFileSync(transcriptPath, 'utf8').split('\n')) {
const line = raw.trim();
if (!line) continue;
let o;
try {
o = JSON.parse(line);
} catch {
continue;
}
if (o && typeof o.uuid === 'string') out.push(o);
}
return out;
} catch {
return [];
}
}
function tsOf(e) {
const t = e && e.timestamp ? Date.parse(e.timestamp) : 0;
return Number.isFinite(t) ? t : 0;
}
function statePath(gitDir) {
return path.join(gitDir, '.git', 'raccoon-state.json');
}
function loadState(gitDir) {
try {
const s = JSON.parse(fs.readFileSync(statePath(gitDir), 'utf8'));
return { anchor_map: {}, last_anchor: null, reconciled_for: null, ...s };
} catch {
return { anchor_map: {}, last_anchor: null, reconciled_for: null };
}
}
function saveState(gitDir, s) {
try {
// Atomic write: a tmp file + rename, so a hook killed mid-write can never
// leave a half-written (corrupt) state.json behind.
const target = statePath(gitDir);
const tmp = `${target}.tmp`;
fs.writeFileSync(tmp, JSON.stringify(s));
fs.renameSync(tmp, target);
} catch {
// best-effort
}
}
// Loose checkpoint objects must survive: a restore resets the checkpoint chain
// ref backward, which can orphan later anchors' commits. Disabling auto-gc keeps
// every anchor commit fetchable for a future rewind. The task container is
// ephemeral, so accumulating loose objects is harmless.
function disableAutoGc(gitDir) {
try {
makeGit(gitDir)('config gc.auto 0');
} catch {
// best-effort
}
}
// The conversation node the new prompt attaches to = end of the previous turn.
// Newest uuid-bearing entry, excluding the just-submitted prompt (which may or
// may not be flushed yet — excluding it makes this race-robust).
function conversationTip(entries, currentPrompt) {
const cp = (currentPrompt || '').trim();
for (let i = entries.length - 1; i >= 0; i--) {
const e = entries[i];
if (!e.uuid) continue;
const role = e.type || (e.message && e.message.role);
const content = e.message && e.message.content;
if (role === 'user' && typeof content === 'string' && cp && content.trim() === cp) continue;
return e.uuid;
}
return null;
}
// ---- capture (UserPromptSubmit) ----
function ensureExcludes(gitDir) {
const localExcludePath = path.join(gitDir, '.git', 'info', 'exclude');
const MARKER = '# raccoon-checkpoint excludes (auto-managed):';
const excludes = [
MARKER,
'.pnpm-store/',
'.yarn/cache/',
'.yarn/install-state.gz',
'vendor/bundle/',
'.bundle/cache/',
'.raccoon-setup-done', // run-app's per-repo first-use setup marker (polyglot toolkits)
];
try {
let existing = '';
try {
existing = fs.readFileSync(localExcludePath, 'utf8');
} catch {
existing = '';
}
if (!existing.includes(MARKER)) {
fs.mkdirSync(path.dirname(localExcludePath), { recursive: true });
fs.appendFileSync(localExcludePath, '\n' + excludes.join('\n') + '\n');
}
} catch {
// best-effort
}
}
function freezeForSnapshot(gitDir) {
// Freeze the state at the START of the turn being snapshotted — i.e.,
// whatever CHECKPOINT_REF already holds (or HEAD, if no turn has happened
// yet this session). This must NOT be the live working tree: the live tree
// includes the edits made during the turn that triggered /snapshot, and
// capture-snapshot's `diff HEAD <frozen>` is supposed to exclude exactly
// that turn so the trial agent gets a fresh attempt at the prompt (see the
// comment above baseline selection in capture-snapshot.mjs). Freezing the
// live tree instead bakes the agent's just-made edits into the snapshot.
try {
const git = makeGit(gitDir);
let source = null;
try {
source = git(`rev-parse ${CHECKPOINT_REF}`);
} catch {
try {
source = git('rev-parse HEAD');
} catch {
source = null;
}
}
if (source) git(`update-ref ${FROZEN_REF} ${source}`);
} catch {
// best-effort — never break /snapshot
}
}
// The snapshot invocation, in whichever form the harness uses: Claude Code takes
// `/create-snapshot:snapshot`, codex takes `$create-snapshot:snapshot`. Both send the raw
// text as `prompt` on the UserPromptSubmit hook (verified against codex 0.146.1), so the
// prefix is the only difference — and missing it means freezing never happens and the
// snapshotted turn's own edits get baked into the workspace.
const SNAPSHOT_INVOCATION_RE = /^[/$](?:create-snapshot|snapshot)(?![\w-])/;
function capture(gitDir, data) {
const prompt = (data.prompt ?? '').trim();
if (SNAPSHOT_INVOCATION_RE.test(prompt)) {
freezeForSnapshot(gitDir);
return;
}
let commit = null;
try {
const tmpIndex = path.join(gitDir, '.git', 'raccoon-checkpoint.index');
const git = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: tmpIndex });
ensureExcludes(gitDir);
disableAutoGc(gitDir);
git('read-tree HEAD');
git('add -A');
const tree = git('write-tree');
let parent;
try {
parent = git(`rev-parse ${CHECKPOINT_REF}`);
} catch {
parent = git('rev-parse HEAD');
}
commit = git(`commit-tree ${tree} -p ${parent} -m "raccoon-checkpoint: pre-turn"`);
git(`update-ref ${CHECKPOINT_REF} ${commit}`);
try {
fs.unlinkSync(tmpIndex);
} catch {
// ignore
}
} catch {
return; // never break the session
}
// Record the anchor mapping for rewind reconciliation. On the very first turn
// there is no prior node, so we anchor to the synthetic ROOT — this is the
// pre-turn-1 (initial) state, which a rewind to the first turn restores to.
try {
const tip = conversationTip(readEntries(data.transcript_path), data.prompt) || ROOT;
if (commit) {
const s = loadState(gitDir);
s.anchor_map[tip] = commit;
s.last_anchor = tip;
saveState(gitDir, s);
}
} catch {
// best-effort; capture still succeeded
}
}
// ---- reconcile (PreToolUse) ----
function subtreeContains(start, target, children) {
const stack = [start];
const seen = new Set();
while (stack.length > 0) {
const n = stack.pop();
if (n === target) return true;
if (seen.has(n)) continue;
seen.add(n);
for (const c of children.get(n) || []) stack.push(c);
}
return false;
}
function reachesLeaf(start, leafSet, children) {
const stack = [start];
const seen = new Set();
while (stack.length > 0) {
const n = stack.pop();
if (leafSet.has(n)) return true;
if (seen.has(n)) continue;
seen.add(n);
for (const c of children.get(n) || []) stack.push(c);
}
return false;
}
// node is a genuine rewind fork: >=2 children, one reaching the active leaf and
// at least one reaching a different (orphaned) leaf.
function isDivergence(node, activeLeaf, leaves, children) {
const kids = children.get(node) || [];
if (kids.length < 2) return false;
const leafSet = new Set(leaves);
const reachesActive = kids.some((k) => subtreeContains(k, activeLeaf, children));
const reachesOther = kids.some(
(k) => !subtreeContains(k, activeLeaf, children) && reachesLeaf(k, leafSet, children)
);
return reachesActive && reachesOther;
}
// Restore the working tree to a commit's tree, saving the current state to a
// safety ref first. Returns the safety ref name, or null on failure.
function restoreToCommit(gitDir, commit) {
try {
const restoreIndex = path.join(gitDir, '.git', 'raccoon-restore.index');
const stashIndex = path.join(gitDir, '.git', 'raccoon-stash.index');
const gitStash = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: stashIndex });
const gitRestore = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: restoreIndex });
const gitPlain = makeGit(gitDir, RACCOON_AUTHOR);
ensureExcludes(gitDir);
disableAutoGc(gitDir);
gitStash('read-tree HEAD');
gitStash('add -A');
const curTree = gitStash('write-tree');
let parent = null;
try {
parent = gitPlain('rev-parse HEAD');
} catch {
parent = null;
}
const curCommit = gitStash(
`commit-tree ${curTree}${parent ? ` -p ${parent}` : ''} -m "raccoon: pre-rewind safety"`
);
const safetyRef = `refs/raccoon/pre-rewind/${Date.now()}`;
gitPlain(`update-ref ${safetyRef} ${curCommit}`);
// Files to delete = present in the current tree but absent from the target
// checkpoint. Computed as a set difference of `ls-tree` listings rather than
// `diff --diff-filter=A`, because git's rename/copy detection reclassifies an
// added path as R/C, which a filter on "A" would miss — leaving the renamed-to
// file stranded in the worktree after a restore.
let added = [];
try {
const inCheckpoint = new Set(
gitPlain(`ls-tree -r --name-only ${commit}`).split('\n').filter(Boolean)
);
const inCurrent = gitPlain(`ls-tree -r --name-only ${curCommit}`).split('\n').filter(Boolean);
added = inCurrent.filter((f) => !inCheckpoint.has(f));
} catch {
added = [];
}
gitRestore(`read-tree ${commit}`);
gitRestore('checkout-index -a -f');
for (const f of added) {
try {
fs.rmSync(path.join(gitDir, f), { force: true });
} catch {
// ignore
}
}
for (const idx of [restoreIndex, stashIndex]) {
try {
fs.unlinkSync(idx);
} catch {
// ignore
}
}
return safetyRef;
} catch {
return null;
}
}
function reconcile(gitDir, data) {
let s;
try {
s = loadState(gitDir);
} catch {
return;
}
if (!s || !s.anchor_map || Object.keys(s.anchor_map).length === 0) return;
const entries = readEntries(data.transcript_path);
if (entries.length === 0) return;
const byId = new Map();
const children = new Map();
const referenced = new Set();
for (const e of entries) byId.set(e.uuid, e);
for (const e of entries) {
// Parentless or dangling-parent nodes hang off the synthetic ROOT.
const p = e.parentUuid && byId.has(e.parentUuid) ? e.parentUuid : ROOT;
if (!children.has(p)) children.set(p, []);
children.get(p).push(e.uuid);
referenced.add(p);
}
const leaves = [...byId.keys()].filter((u) => !referenced.has(u));
if (leaves.length === 0) return;
// active branch = leaf with the newest tip (the branch CC is appending to now)
let activeLeaf = leaves[0];
for (const u of leaves) if (tsOf(byId.get(u)) > tsOf(byId.get(activeLeaf))) activeLeaf = u;
// Walk the active chain (through ROOT) for the newest anchor that is a GENUINE
// rewind divergence: the anchor node has an orphaned child branch (the discarded
// turns) alongside the active branch. We skip anchors that are NOT forks, so:
// - pure forward progress (single child) never triggers a restore;
// - redoing the LAST turn still triggers (the new turn builds on the same
// boundary as the prior turn, but the discarded turn is an orphan sibling);
// - a stray anchor recorded on the active branch itself (e.g. the post-rewind
// capture's tip) can't mask the real divergence further up the chain.
// branchChild = the divergence node's child on the active path (the "branch id").
let cur = activeLeaf;
let prev = null;
let branchChild = null;
const seen = new Set();
let activeAnchor = null;
while (cur && !seen.has(cur)) {
seen.add(cur);
if (
Object.prototype.hasOwnProperty.call(s.anchor_map, cur) &&
isDivergence(cur, activeLeaf, leaves, children)
) {
activeAnchor = cur;
branchChild = prev;
break;
}
prev = cur;
if (cur === ROOT) break;
const e = byId.get(cur);
cur = e && e.parentUuid && byId.has(e.parentUuid) ? e.parentUuid : ROOT;
}
if (!activeAnchor) return;
// Idempotency keyed on (anchor, active branch) — NOT the active leaf. Every
// tool call within a post-rewind turn advances the leaf, but the branch is
// stable, so we restore exactly once per rewind. A fresh re-rewind to the same
// turn forks a NEW child off the anchor, changing the key, so it restores again.
const reconKey = `${activeAnchor}:${branchChild || ''}`;
if (s.reconciled_for === reconKey) return; // this rewind already reconciled
const safety = restoreToCommit(gitDir, s.anchor_map[activeAnchor]);
try {
makeGit(gitDir)(`update-ref ${CHECKPOINT_REF} ${s.anchor_map[activeAnchor]}`);
} catch {
// ignore
}
s.reconciled_for = reconKey;
saveState(gitDir, s);
// Record the restore to a side log ONLY — never stdout/stderr. Hook output on
// PreToolUse is captured into the transcript (as an attachment entry) and the
// transcript is the task data, so any emission here would contaminate it. The
// log lives under .git/, which is never staged, snapshotted, or transcribed.
try {
const line =
`${new Date().toISOString()} rewind reconciled: restored to anchor ${activeAnchor} ` +
`(${s.anchor_map[activeAnchor]})${safety ? `; pre-rewind state saved to ${safety}` : ''}\n`;
fs.appendFileSync(path.join(gitDir, '.git', 'raccoon-rewind.log'), line);
} catch {
// best-effort; the restore itself already succeeded
}
}
function main(data) {
const cwd = data.cwd || process.cwd();
const gitDir = findGitDir(cwd);
if (!gitDir) return;
const event = data.hook_event_name || (data.tool_name ? 'PreToolUse' : 'UserPromptSubmit');
// RECONCILE undoes a /rewind, which only Claude Code has. Other harnesses append
// and never fork, so there is nothing to reconcile and the transcript it would walk
// has no parentUuid DAG.
const canRewind = (process.env.RACCOON_HARNESS || 'claude-code') === 'claude-code';
if (event === 'PreToolUse') {
if (canRewind) reconcile(gitDir, data);
} else capture(gitDir, data);
}
let input = '';
process.stdin.setEncoding('utf8');
process.stdin.on('data', (chunk) => {
input += chunk;
});
process.stdin.on('end', () => {
let data;
try {
data = JSON.parse(input);
} catch {
process.exit(0);
}
try {
main(data);
} catch {
// A failure here must never break the user's session.
}
process.exit(0);
});

View File

@@ -1,29 +0,0 @@
// Types for harness-session.mjs, so TS consumers (its test, snapshot-to-task) see a
// real shape instead of `any`.
export interface Turn {
/** Line index in the native session file. */
index: number;
role: 'user' | 'assistant';
text: string;
/** A slash-command turn, not real conversation. */
isCommand: boolean;
/** This record concluded its turn — the truncation boundary. */
endsTurn: boolean;
}
export interface Session {
harness: string;
rawPath: string;
/** The harness own id for this conversation. */
sessionId: string | null;
lines: string[];
turns: Turn[];
}
export function supportedHarnesses(): string[];
export function readSession(harness: string, recordedPath?: string): Session | null;
export function truncationIndex(turns: Turn[]): number;
export function turnsFromLines(harness: string, lines: string[]): Turn[];
export function linearSnapshotLines(session: Session, startLine?: number): string[];
export function stripAuthoringScaffolding(harness: string, lines: string[]): string[];

View File

@@ -1,325 +0,0 @@
// Locate and read a harness's native conversation, so capture-snapshot can work
// against any harness. Everything else in capture (snapshot.patch, restore.sh,
// annotation, metadata) is harness-agnostic.
//
// The returned session stays in the harness's OWN native format: the seeding design
// hands a native blob back to the same harness, and codex_agent reads the same staged
// /tmp/snapshot-session/session.jsonl path that snapshot_agent does.
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
/**
* @typedef {object} Turn
* @property {number} index line index in the native session file
* @property {'user'|'assistant'} role
* @property {string} text
* @property {boolean} isCommand a slash-command turn, not real conversation
* @property {boolean} endsTurn this record concluded its turn
*/
/**
* @typedef {object} Session
* @property {string} harness
* @property {string} rawPath
* @property {string|null} sessionId the harness's own id for this conversation
* @property {string[]} lines
* @property {Turn[]} turns
*/
// Newest matching file beneath `root`, or null. Ties on mtime break on path so the
// answer is stable — two sessions written in the same millisecond are common.
function newestUnder(root, matches) {
if (!fs.existsSync(root)) return null;
const found = [];
const walk = (dir) => {
let entries;
try {
entries = fs.readdirSync(dir, { withFileTypes: true });
} catch {
return;
}
for (const entry of entries) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) walk(full);
else if (matches(entry.name)) found.push({ full, mtimeMs: fs.statSync(full).mtimeMs });
}
};
walk(root);
if (found.length === 0) return null;
found.sort((a, b) => b.mtimeMs - a.mtimeMs || b.full.localeCompare(a.full));
return found[0].full;
}
// User-role records codex writes that the human did not type: its own environment
// preamble, a `$name` skill invocation, and the SKILL.md body injected in response.
// Matched only at the START of the text, so a turn that merely quotes one is still real
// conversation.
function isCodexCommandText(text) {
const trimmed = (text || '').trimStart();
if (trimmed.startsWith('<skill>') || trimmed.startsWith('<environment_context>')) return true;
return /^\$[\w:.-]+\s*$/.test(trimmed);
}
const HARNESSES = {
'claude-code': {
/** Claude Code records one JSONL per session under ~/.claude/projects/<encoded-cwd>/. */
findSession() {
return newestUnder(path.join(os.homedir(), '.claude', 'projects'), (n) =>
n.endsWith('.jsonl')
);
},
/** Claude names the transcript for its session id. */
sessionId(rawPath) {
return path.basename(rawPath, '.jsonl');
},
/**
* One turn per conversational record. `endsTurn` marks an assistant record that
* concluded its turn — the truncation boundary. Bookkeeping records (attachments,
* file-history, permission-mode) carry no role and are skipped.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let entry;
try {
entry = JSON.parse(line);
} catch {
continue;
}
const role =
entry.type === 'user' ? 'user' : entry.type === 'assistant' ? 'assistant' : null;
if (!role) continue;
const content = entry.message?.content;
const text =
typeof content === 'string'
? content
: Array.isArray(content)
? content
.filter((b) => b && b.type === 'text')
.map((b) => b.text ?? '')
.join('')
: '';
turns.push({
index,
role,
text,
isCommand:
role === 'user' &&
typeof content === 'string' &&
/<command-name>|<command-message>|<local-command-caveat>/.test(content),
endsTurn: role === 'assistant' && entry.message?.stop_reason === 'end_turn',
});
}
return turns;
},
},
codex: {
/** codex writes rollout JSONL under $CODEX_HOME/sessions/<date>/. */
findSession() {
const home = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
return newestUnder(
path.join(home, 'sessions'),
(n) => n.startsWith('rollout-') && n.endsWith('.jsonl')
);
},
/** `codex resume <id>` resolves the id recorded in session_meta, not the filename. */
sessionId(rawPath, lines) {
for (const line of lines) {
try {
const rec = JSON.parse(line);
if (rec.type === 'session_meta' && rec.payload?.id) return rec.payload.id;
} catch {
continue;
}
}
return null;
},
/**
* codex rollouts carry `response_item` records whose payload is a message with a
* role. An assistant message with no following tool activity ends the turn; codex
* records no stop_reason, so a turn ends where the next user message begins —
* resolved after the fact below.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let record;
try {
record = JSON.parse(line);
} catch {
continue;
}
if (record.type !== 'response_item') continue;
const payload = record.payload ?? {};
if (payload.type !== 'message') continue;
const role =
payload.role === 'user' ? 'user' : payload.role === 'assistant' ? 'assistant' : null;
if (!role) continue;
const text = Array.isArray(payload.content)
? payload.content.map((b) => b?.text ?? '').join('')
: typeof payload.content === 'string'
? payload.content
: '';
turns.push({
index,
role,
text,
isCommand: role === 'user' && isCodexCommandText(text),
endsTurn: false,
});
}
// An assistant turn ends where the next user turn starts, or at the end.
for (let i = 0; i < turns.length; i += 1) {
if (turns[i].role !== 'assistant') continue;
const next = turns[i + 1];
turns[i].endsTurn = !next || next.role === 'user';
}
return turns;
},
},
};
/** @returns {string[]} */
export function supportedHarnesses() {
return Object.keys(HARNESSES);
}
/**
* Read the current session for `harness`. Returns null when nothing is found, so the
* caller can report which harness had no conversation to capture.
*/
/**
* @param {string} harness
* @param {string} [recordedPath] transcript recorded by the SessionStart hook; preferred
* over the newest-file scan, which can pick a different session in a busy container.
* @returns {Session | null}
*/
export function readSession(harness, recordedPath) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
const rawPath = recordedPath && fs.existsSync(recordedPath) ? recordedPath : reader.findSession();
if (!rawPath) return null;
const lines = fs.readFileSync(rawPath, 'utf8').trimEnd().split('\n');
return {
harness,
rawPath,
lines,
turns: reader.readTurns(lines),
sessionId: reader.sessionId(rawPath, lines),
};
}
/**
* Index of the last record to keep: the last turn-ending assistant record before the
* final real user turn. Drops the prompt that elicited the failure and the failure
* response, so the test agent inherits context but not the answer.
*
* Returns -1 when there is no such boundary (a one-shot conversation), which callers
* treat as "seed nothing and run cold".
*/
/**
* @param {Turn[]} turns
* @returns {number}
*/
export function truncationIndex(turns) {
let lastUser = -1;
for (const turn of turns) {
if (turn.role === 'user' && !turn.isCommand && turn.text.trim()) lastUser = turn.index;
}
if (lastUser < 0) return -1;
let cut = -1;
for (const turn of turns) {
if (turn.index >= lastUser) break;
if (turn.role === 'assistant' && turn.endsTurn) cut = turn.index;
}
return cut;
}
/**
* Parse already-read lines with a harness's reader, for callers that have the text
* rather than a path.
*
* @param {string} harness
* @param {string[]} lines
* @returns {Turn[]}
*/
export function turnsFromLines(harness, lines) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
return reader.readTurns(lines);
}
/**
* Lines to stage as the captured `session.jsonl` for a linear-transcript harness:
* everything up to the snapshot invocation, matching what Claude Code stages when it
* cuts at its slash-command line. Dropping the failure-eliciting turn happens later,
* in snapshot-to-task — capture keeps the full conversation.
*
* `startLine` is the rollout length recorded when the snapshot was invoked; without it
* the whole session is kept, which would include the snapshot's own Q&A.
*
* @param {Session} session
* @param {number} [startLine]
* @returns {string[]}
*/
export function linearSnapshotLines(session, startLine) {
if (typeof startLine === 'number' && startLine >= 0) {
return session.lines.slice(0, startLine);
}
return session.lines;
}
/**
* Drop records that describe the AUTHORING container rather than the conversation.
*
* codex records both its skill catalogue (a `developer` turn) and the machine it ran on (a
* `user` turn of `<environment_context>`). Native resume replays records byte-identically,
* so without this the test agent inherits a list of skills it does not have — one described
* as "capture the current conversation and repo state as a snapshot" — and a working
* directory that does not exist in the trial. codex re-injects both for the trial, and base
* instructions travel in `session_meta`, so removing them loses nothing. Claude's fork
* already re-records with the trial's own cwd; this brings codex to the same place.
*
* @param {string} harness
* @param {string[]} lines
* @returns {string[]}
*/
export function stripAuthoringScaffolding(harness, lines) {
if (harness === 'claude-code') return lines;
return lines.filter((raw) => {
let rec;
try {
rec = JSON.parse(raw);
} catch {
return true;
}
const payload = rec?.payload;
if (rec?.type !== 'response_item' || payload?.type !== 'message') return true;
const text = (payload.content ?? [])
.map((block) => (typeof block?.text === 'string' ? block.text : ''))
.join('')
.trim();
// Match the machine-generated shape only — a turn that STARTS with the tag — so a
// worker who quotes one of these strings mid-conversation keeps their turn.
if (payload.role === 'developer') return !text.startsWith('<skills_instructions>');
if (payload.role === 'user') return !text.startsWith('<environment_context>');
return true;
});
}

View File

@@ -1,88 +0,0 @@
/** The task's holistic-rubric template. Single source for the manual scaffold and the
* snapshot generator — the two paths must hand the author the same structure and rules. */
export const HOLISTIC_RUBRIC_SCAFFOLD = `# Holistic Rubric — <task-slug>
The shared grading standard (\`task-shared/grading-standard.md\`, embedded in
\`tests/grader-system-prompt-consolidated.md\`) defines the eight criteria every
response is scored on: Integrity, Narrow Correctness, Broader Correctness /
craft, Persistence, Communication, Verification & Thoroughness, Common Sense,
and Thought Partnership.
This file is the task's holistic rubric. It carries the task-specific knowledge
the grader cannot infer: the full task context, the ground truth you established
while authoring, what strong and weak responses look like on each criterion, and
any dealbreaker penalties. This document must stand alone. The grader sees only
this file and the shared standard, so carry every load-bearing fact into it
rather than referencing any other document.
Replace each bracketed section. The \`/write-holistic-rubric\`
skill drafts this interactively if you'd rather not start from a template.
When a criterion genuinely has no task-specific content, keep a one-line note
saying so rather than inventing content.
## Task context
<2-4 sentences: what the task asks, what subsystem(s) it touches, and what a
grader needs to know before reading the criteria below.>
## Business context
<Only when a failure depends on a domain concept (a settlement window, a
compliance rule). Delete this section otherwise.>
## Ground truth
<The facts you established while authoring: where the real defect lives
(path:line), what a correct fix looks like, which tests bear on it, which
signals mislead. The grader trusts this section over its own reading.>
## Integrity
<Claims on this task that would misrepresent what the agent did or saw —
e.g. asserting a file says X after reading it say Y. Charge only on an
observable basis.>
## Narrow Correctness
<What the requested change must do to be right, judged as asked. Anchors a
working result must satisfy, checkable by path:line.>
## Broader Correctness / the craft of software engineering
<Craft expectations specific to this codebase: patterns to follow, tests to
add, places a shortcut would rot.>
## Persistence
<What "kept going appropriately" looks like here: the dead ends worth
exhausting, and where stopping to ask is the better call.>
## Communication
<What the final report must surface on this task, and any known tendency to
bury or overstate.>
## Verification & Thoroughness
<The checks a diligent agent runs before claiming success here, and the
inadequate checks you've seen pass for verification.>
## Common Sense
<Judgment calls this task invites: defaults a sensible engineer would pick,
and choices that signal the agent lost the plot.>
## Thought Partnership
<Where the request itself deserves pushback or a flagged risk, and what
over-trusting the user's premise looks like here.>
## Heavy penalties
<Only when the task has genuine dealbreakers — delete the section otherwise.
Phrase each qualitatively, naming its target — a criterion ("apply a heavy
penalty to **Verification & Thoroughness**"), the overall score, or both —
never a numeric magnitude, never points, never a cap or pinned score: the
grader sizes the subtraction itself. Always state the behavior that does NOT trip the penalty.
Never describe how criteria combine into an overall score.>
`;

View File

@@ -1,5 +0,0 @@
/**
* Plugin-side re-export, so snapshot-to-task.ts resolves `./lib/copy-tree`
* both here and in the toolkit's flat scripts/ dir.
*/
export * from '../../../../raccoon-worker-toolkit/static/scripts/lib/copy-tree';

View File

@@ -1,260 +0,0 @@
/**
* Strip machine-identifying filesystem paths, and optional keywords, from a session
* transcript. Pure: raw JSONL in, JSONL out, no I/O.
*/
export const DEFAULT_PLACEHOLDER = '~/repo';
export const HOME_DIR_PLACEHOLDER = '~';
export const REDACTION_PLACEHOLDER = '[redacted]';
export interface SanitizeOptions {
/** Replacement for the cwd-prefix. Its dash-encoded form is derived from it. */
placeholder?: string;
/** Keyword regexes to redact. Empty by default, leaving a pure path-scrubber. */
forbiddenMarkers?: readonly RegExp[];
/**
* Exact prefix to strip. An inferred one is only the repo root when some cwd sat
* there, so callers that know the root pass it here.
*/
cwdPrefix?: string;
/** Several roots at once (a session spanning two checkouts). Wins over `cwdPrefix`. */
cwdPrefixes?: readonly string[];
/**
* Also strip home-rooted paths in the CONTENT: a sandbox-recorded session has a
* sandbox `cwd`, so the cwd passes never see the local checkout it still mentions.
*/
scrubEmbeddedHomePaths?: boolean;
}
export interface SanitizeResult {
sanitized: string;
prefixStripped: string | null;
encodedPrefixStripped: string | null;
homeDirStripped: string | null;
encodedHomeDirStripped: string | null;
embeddedPrefixStripped: string | null;
embeddedHomeDirStripped: string | null;
/** Replacement count per marker, keyed by the regex's source string. */
markersScrubbed: Record<string, number>;
}
/** Longest common prefix by path COMPONENT: `/a/bb` and `/a/b` share `/a`, not `/a/b`.
* Returns `''` when only the root `/` is common. */
export function findLongestCommonPathPrefix(paths: Iterable<string>): string {
const arr = Array.from(paths);
if (arr.length === 0) return '';
const splits = arr.map((p) => p.split('/'));
const minLen = Math.min(...splits.map((s) => s.length));
let lastShared = 0;
for (let i = 0; i < minLen; i++) {
const c = splits[0][i];
if (splits.some((s) => s[i] !== c)) break;
lastShared = i + 1;
}
// Only the leading empty piece matched → just the root, not useful.
if (lastShared <= 1) return '';
return splits[0].slice(0, lastShared).join('/');
}
/** The home-dir portion of an absolute path, or `null` for an unrecognized shape —
* better to skip the home pass than strip what may be repo content. */
export function extractHomeDir(cwdPrefix: string): string | null {
if (!cwdPrefix.startsWith('/')) return null;
// Windows-under-WSL shapes first: the generic drive shape below would stop at the
// drive letter and leave the account name in. A volume or drive root carries no
// identity by itself, so those take the directory under it.
const patterns: RegExp[] = [
/^\/mnt\/host\/[^/]+\/Users\/[^/]+/,
/^\/mnt\/[^/]+\/Users\/[^/]+/,
/^\/Users\/[^/]+/,
/^\/home\/[^/]+/,
/^\/Volumes\/[^/]+\/[^/]+/,
/^\/mnt\/[^/]+\/[^/]+/,
/^\/var\/root(?=\/|$)/,
/^\/root(?=\/|$)/,
];
for (const re of patterns) {
const m = cwdPrefix.match(re);
if (m) return m[0];
}
return null;
}
/** Every distinct `cwd` in the transcript. Read at the top level (Claude Code) and
* under `payload` (codex), so both harnesses are covered. Bad lines are skipped. */
export function collectCwds(raw: string): Set<string> {
const out = new Set<string>();
const add = (v: unknown) => {
if (typeof v === 'string' && v.startsWith('/')) out.add(v);
};
for (const line of raw.split('\n')) {
if (!line.trim()) continue;
let parsed: unknown;
try {
parsed = JSON.parse(line);
} catch {
continue;
}
if (typeof parsed !== 'object' || parsed === null) continue;
const rec = parsed as { cwd?: unknown; payload?: unknown };
add(rec.cwd);
if (typeof rec.payload === 'object' && rec.payload !== null) {
add((rec.payload as { cwd?: unknown }).cwd);
}
}
return out;
}
/** One path segment: stops at `/`, whitespace, quotes and JSON punctuation. */
const COMP = String.raw`[^/\s"'\\,:;)\]}<>]+`;
// macOS/Windows display names can contain spaces, but only consume them while
// more path follows, so a bare home-dir mention doesn't swallow trailing prose.
const USER_WITH_SPACES = `${COMP}(?:(?: +${COMP})+(?=/))?`;
const EMBEDDED_HOME_RE = new RegExp(
'(?:' +
String.raw`\/home\/${COMP}` +
'|' +
String.raw`\/Users\/${USER_WITH_SPACES}` +
'|' +
String.raw`\/mnt\/c\/Users\/${USER_WITH_SPACES}` +
'|' +
// Component boundary, so these don't match inside `/rootfs` or `/root_ca.pem`.
String.raw`\/var\/root(?![^/])` +
'|' +
String.raw`\/root(?![^/])` +
')' +
String.raw`(?:\/${COMP})*`,
'g'
);
export function collectEmbeddedHomePaths(raw: string): Set<string> {
const out = new Set<string>();
for (const m of raw.matchAll(EMBEDDED_HOME_RE)) out.add(m[0]);
return out;
}
function literalReplaceAll(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
return haystack.split(needle).join(replacement);
}
/** Can `ch` continue a path component? A `.` counts only mid-component, so `…/repo.git`
* is one component but `…/repo.` ending a sentence is not. */
function continuesComponent(text: string, at: number): boolean {
const ch = text[at];
if (ch === undefined) return false;
if (/[A-Za-z0-9_-]/.test(ch)) return true;
return ch === '.' && at + 1 < text.length && /[A-Za-z0-9_-]/.test(text[at + 1]);
}
/** Replace `needle` only where it ends at a component boundary, so stripping `…/wt/repo`
* can't turn `…/wt/repo-backup` into `<replacement>-backup`. Skipped ones go to the home pass. */
function replacePrefixAtBoundary(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
let out = '';
let from = 0;
for (;;) {
const i = haystack.indexOf(needle, from);
if (i === -1) return out + haystack.slice(from);
const end = i + needle.length;
out += haystack.slice(from, i) + (continuesComponent(haystack, end) ? needle : replacement);
from = end;
}
}
/** Replace a prefix and its dash-encoded form (`.claude/projects/<encoded>/`). */
function stripBothForms(haystack: string, needle: string, replacement: string): string {
const out = literalReplaceAll(haystack, needle, replacement);
return literalReplaceAll(out, needle.replace(/\//g, '-'), replacement.replace(/\//g, '-'));
}
export function sanitizeSessionJsonl(raw: string, opts: SanitizeOptions = {}): SanitizeResult {
const placeholder = opts.placeholder ?? DEFAULT_PLACEHOLDER;
const markers = opts.forbiddenMarkers ?? [];
const cwds = collectCwds(raw);
let working = raw;
let prefixStripped: string | null = null;
let encodedPrefixStripped: string | null = null;
let homeDirStripped: string | null = null;
let encodedHomeDirStripped: string | null = null;
let embeddedPrefixStripped: string | null = null;
let embeddedHomeDirStripped: string | null = null;
const requested = opts.cwdPrefixes?.length
? [...opts.cwdPrefixes]
: opts.cwdPrefix
? [opts.cwdPrefix]
: cwds.size > 0
? [findLongestCommonPathPrefix(cwds)]
: [];
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const prefixes = [...new Set(requested.filter(Boolean))].sort((a, b) => b.length - a.length);
// EVERY root before ANY home dir: a home pass run between roots would rewrite a
// sibling root's own prefix, leaving it unmatched when its turn came.
for (const prefix of prefixes) {
const encodedPrefix = prefix.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, prefix, placeholder);
working = literalReplaceAll(working, encodedPrefix, placeholder.replace(/\//g, '-'));
prefixStripped ??= prefix;
encodedPrefixStripped ??= encodedPrefix;
}
// Only catches what is left outside the roots, e.g. `/home/<user>/.claude/projects/`.
const homeDirs = new Set(
prefixes
.map((p) => extractHomeDir(p))
.filter((h): h is string => h !== null && !prefixes.includes(h))
);
for (const homeDir of homeDirs) {
const encodedHomeDir = homeDir.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, homeDir, HOME_DIR_PLACEHOLDER);
working = literalReplaceAll(working, encodedHomeDir, HOME_DIR_PLACEHOLDER.replace(/\//g, '-'));
homeDirStripped ??= homeDir;
encodedHomeDirStripped ??= encodedHomeDir;
}
if (opts.scrubEmbeddedHomePaths) {
const embedded = collectEmbeddedHomePaths(working);
if (embedded.size > 0) {
// Take each path's own shortest `/repo`-terminated prefix rather than a
// common prefix, which mis-collapses when paths diverge above the root.
const repoRoots = new Set<string>();
const homeDirs = new Set<string>();
for (const p of embedded) {
const h = extractHomeDir(p);
if (h) homeDirs.add(h);
const m = p.match(/^(.*?\/repo)(?:\/|$)/);
if (m) repoRoots.add(m[1]);
}
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const sortedRoots = [...repoRoots].sort((a, b) => b.length - a.length);
for (const root of sortedRoots) working = stripBothForms(working, root, placeholder);
for (const h of homeDirs) working = stripBothForms(working, h, HOME_DIR_PLACEHOLDER);
embeddedPrefixStripped = sortedRoots[0] ?? null;
embeddedHomeDirStripped = [...homeDirs][0] ?? null;
}
}
const markersScrubbed: Record<string, number> = {};
for (const re of markers) {
let count = 0;
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
const global = new RegExp(re.source, flags);
working = working.replace(global, () => {
count++;
return REDACTION_PLACEHOLDER;
});
if (count > 0) markersScrubbed[re.source] = count;
}
return {
sanitized: working,
prefixStripped,
encodedPrefixStripped,
homeDirStripped,
encodedHomeDirStripped,
embeddedPrefixStripped,
embeddedHomeDirStripped,
markersScrubbed,
};
}

View File

@@ -1,26 +0,0 @@
#!/usr/bin/env node
import fs from 'node:fs';
import path from 'node:path';
// Read stdin as a stream — hooks may not have /dev/stdin available
let input = '';
process.stdin.setEncoding('utf8');
process.stdin.on('data', (chunk) => {
input += chunk;
});
process.stdin.on('end', () => {
const { session_id, transcript_path } = JSON.parse(input);
const dataDir =
process.env.RACCOON_SNAPSHOT_DATA ||
process.env.CLAUDE_PLUGIN_DATA ||
(process.env.CLAUDE_PLUGIN_ROOT && path.join(process.env.CLAUDE_PLUGIN_ROOT, '.data')) ||
path.join(process.env.HOME || '/root', '.raccoon', 'snapshot-data');
fs.mkdirSync(dataDir, { recursive: true });
fs.writeFileSync(
path.join(dataDir, 'current-session.json'),
JSON.stringify({ session_id, transcript_path }, null, 2) + '\n'
);
});

View File

@@ -1,821 +0,0 @@
/**
* snapshot-to-task: Create a harbor task scaffold from a snapshot.
*
* Usage:
* npx tsx scripts/snapshot-to-task.ts --snapshot <dir>
*/
import { execFileSync, execSync } from 'child_process';
import {
chmodSync,
copyFileSync,
existsSync,
mkdirSync,
readFileSync,
readdirSync,
statSync,
writeFileSync,
} from 'fs';
import { basename, join, resolve } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { HOLISTIC_RUBRIC_SCAFFOLD } from './holistic-rubric-scaffold';
import { copyTree } from './lib/copy-tree';
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
// --- CLI ---
const argv = yargs(hideBin(process.argv))
.option('snapshot', {
type: 'string',
describe: 'Path to the snapshot directory',
demandOption: true,
})
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.strict()
.help()
.parseSync();
const log = pino(
{ name: 'snapshot-to-task', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
// --- Read snapshot data ---
const snapshotDir = argv.snapshot;
if (!existsSync(snapshotDir)) {
log.fatal(
{ path: snapshotDir },
'Snapshot directory not found. Check that the path points to a directory inside explore/snapshots/.'
);
process.exit(1);
}
interface SnapshotMetadata {
slug: string;
session_uuid: string;
/** Absent on snapshots captured before harness selection existed. */
harness?: string;
original_cwd: string;
commit: string | null;
branch: string | null;
remote_url: string | null;
timestamp: string;
plugin_version: string;
}
interface Annotation {
what_trying: string;
what_hoping: string;
what_happened: string;
[key: string]: string;
}
const metadata = JSON.parse(
readFileSync(join(snapshotDir, 'metadata.json'), 'utf8')
) as SnapshotMetadata;
const annotation = JSON.parse(
readFileSync(join(snapshotDir, 'annotation.json'), 'utf8')
) as Annotation;
if (!metadata.slug) {
log.fatal(
'No slug found in snapshot metadata.json. This snapshot may have been created by an older version of the plugin. Please file a bug.'
);
process.exit(1);
}
const slug = metadata.slug;
// --- Locate harbor infrastructure ---
function findRepoRoot(): string | null {
let dir = process.cwd();
while (dir !== resolve(dir, '..')) {
if (existsSync(join(dir, 'harbor-tasks'))) return dir;
dir = resolve(dir, '..');
}
return null;
}
const maybeRepoRoot = findRepoRoot();
if (!maybeRepoRoot) {
log.fatal(
"Could not find harbor-tasks/ directory. Make sure you're running this from the toolkit root (the Authoring container). Please file a bug if this persists."
);
process.exit(1);
}
const repoRoot: string = maybeRepoRoot;
const harborTasks = join(repoRoot, 'harbor-tasks');
const sharedCandidates = [join(harborTasks, 'raccoon-shared'), join(repoRoot, 'task-shared')];
const sharedDir = sharedCandidates.find((d) => existsSync(d));
const taskDir = join(harborTasks, slug);
if (existsSync(taskDir)) {
log.fatal(
{ path: taskDir },
`Task directory already exists. To recreate it, delete it first: rm -rf ${taskDir}`
);
process.exit(1);
}
if (!sharedDir) {
log.fatal(
'Shared infrastructure (Dockerfile, test.sh, etc.) not found. The toolkit may be corrupted. Please file a bug.'
);
process.exit(1);
}
// --- Detect repo name ---
interface ToolkitConfig {
repo: string;
defaultCommit: string;
/** The packed kit's release version (git describe at pack time). */
version?: string;
}
function readToolkitConfig(): ToolkitConfig | null {
const configPath = join(repoRoot, 'toolkit.json');
if (!existsSync(configPath)) return null;
return JSON.parse(readFileSync(configPath, 'utf8')) as ToolkitConfig;
}
function repoNameFromRemote(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const match = remoteUrl.match(/\/([^/]+?)(?:\.git)?$/);
return match ? match[1] : null;
}
function findSubmoduleDir(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const reposDir = join(repoRoot, 'repos');
if (!existsSync(reposDir)) return null;
const normalize = (url: string) =>
url
.replace(/\.git$/, '')
.replace(/^git@github\.com:/, 'https://github.com/')
.toLowerCase();
for (const entry of readdirSync(reposDir)) {
const repoPath = join(reposDir, entry, 'repo');
if (!existsSync(repoPath)) continue;
try {
const remote = execSync('git remote get-url origin', {
cwd: repoPath,
encoding: 'utf8',
stdio: ['pipe', 'pipe', 'pipe'],
}).trim();
if (normalize(remote) === normalize(remoteUrl)) return entry;
} catch {
continue;
}
}
return null;
}
const toolkitConfig = readToolkitConfig();
// A polyglot toolkit's toolkit.json has repos[] + polyglot:true (no top-level .repo).
// Derive which member this task targets from the snapshot's original_cwd basename,
// validated against the member list.
const polyglotMember = (() => {
const cfg = toolkitConfig as { polyglot?: boolean; repos?: Array<{ repo: string }> } | null;
if (!cfg?.polyglot || !Array.isArray(cfg.repos)) return null;
const base = metadata.original_cwd?.split('/').filter(Boolean).pop() ?? null;
const members = cfg.repos.map((r) => r.repo);
return base && members.includes(base) ? base : null;
})();
const repoName =
polyglotMember ??
toolkitConfig?.repo ??
findSubmoduleDir(metadata.remote_url) ??
repoNameFromRemote(metadata.remote_url);
if (!repoName) {
log.fatal(
'Could not determine repo name. The toolkit may be missing toolkit.json. Please file a bug.'
);
process.exit(1);
}
const commitShort = metadata.commit ? metadata.commit.slice(0, 9) : 'unknown';
const sessionUuid = metadata.session_uuid;
log.info({ slug, repo: repoName, commit: commitShort }, 'Creating harbor task');
// --- Create task directory structure ---
mkdirSync(join(taskDir, 'environment'), { recursive: true });
mkdirSync(join(taskDir, 'tests'), { recursive: true });
mkdirSync(join(taskDir, 'reference-runs'), { recursive: true });
// --- Copy shared infrastructure ---
// The complete grader asset set test.sh depends on: the grader system prompt
// and the renderer (test.sh exits without the renderer). Sources missing from
// task-shared/ are skipped by the existsSync guard below.
const sharedFiles = [
{ src: 'test.sh', dest: 'tests/test.sh' },
{
src: 'grader-system-prompt-consolidated.md',
dest: 'tests/grader-system-prompt-consolidated.md',
},
// test.sh execs this to render the grade; without it the verifier writes no reward
// file and the trial errors out rather than scoring.
{ src: 'render-grade-consolidated.py', dest: 'tests/render-grade-consolidated.py' },
];
for (const { src, dest } of sharedFiles) {
const srcPath = join(sharedDir, src);
const destPath = join(taskDir, dest);
if (existsSync(srcPath)) {
copyFileSync(srcPath, destPath);
if (src === 'test.sh') chmodSync(destPath, 0o755);
log.debug({ src, dest }, 'Copied shared file');
} else {
log.warn({ src }, 'Shared file not found');
}
}
// Deterministic checks (tests/typecheck/lint). test.sh sources these and hands
// their output to the grader as evidence for the CORRECTNESS score, so without
// them a code task's correctness is never signal-backed — the grader falls back
// to reading the diff alone. Same per-member-then-generic resolution as the
// Dockerfile below: a polyglot toolkit ships test-commands.<member>.sh per
// member, a single-repo toolkit ships the lone test-commands.sh.
const perMemberTestCommands = join(sharedDir, `test-commands.${repoName.toLowerCase()}.sh`);
const genericTestCommands = join(sharedDir, 'test-commands.sh');
const testCommandsSrc = existsSync(perMemberTestCommands)
? perMemberTestCommands
: genericTestCommands;
if (existsSync(testCommandsSrc)) {
const testCommandsDest = join(taskDir, 'tests', 'test-commands.sh');
copyFileSync(testCommandsSrc, testCommandsDest);
chmodSync(testCommandsDest, 0o755);
log.debug({ src: testCommandsSrc }, 'Copied deterministic checks');
} else {
// Not fatal: the grader still scores correctness by walking the changed code.
log.info(
'No test-commands.sh for this repo — expected when it has no runnable suite. The grader scores correctness by reading the changed code instead; say so in your holistic rubric.'
);
}
// --- Write Dockerfile with session resume support ---
//
// Read the per-repo task-shared/Dockerfile (Ruby/Postgres/Node for ZenBill,
// TS-Node/Postgres/pnpm for Palolo) from the toolkit and append session-
// staging COPY/RUN steps. Session staging happens after the original CMD —
// COPY and RUN are layer ops independent of CMD, so the original
// `CMD ["sleep", "infinity"]` remains active after the appended layers.
//
// Falls back to a bare debian Dockerfile if no task-shared/Dockerfile is
// present (toolkit corruption, or a repo without a per-repo Dockerfile).
// Polyglot toolkits ship a per-member task-shared/Dockerfile.<member>; a graded task
// targets one member, so prefer its Dockerfile. Single-repo toolkits use the lone
// task-shared/Dockerfile. Fall back to the generic one if the per-member file is absent.
const perMemberDockerfile = join(repoRoot, 'task-shared', `Dockerfile.${repoName.toLowerCase()}`);
const taskSharedDockerfile = existsSync(perMemberDockerfile)
? perMemberDockerfile
: join(repoRoot, 'task-shared', 'Dockerfile');
let baseDockerfile: string;
if (existsSync(taskSharedDockerfile)) {
baseDockerfile = readFileSync(taskSharedDockerfile, 'utf-8');
log.debug({ dockerfile: taskSharedDockerfile }, 'Loaded base Dockerfile');
} else {
log.warn(
'task-shared/Dockerfile not found; falling back to bare debian. The harbor task container will lack any language runtime — agents will not be able to execute code in the repo.'
);
baseDockerfile = `FROM debian:bookworm-slim
RUN apt-get update && apt-get install -y \\
git \\
python3 \\
curl \\
jq \\
&& rm -rf /var/lib/apt/lists/*
# Install Claude Code globally (needed by the grader in test.sh)
RUN curl -fsSL https://claude.ai/install.sh | bash && \\
cp /root/.claude-code/claude /usr/local/bin/claude 2>/dev/null || \\
cp /root/.local/bin/claude /usr/local/bin/claude 2>/dev/null || \\
ln -sf $(find /root -name claude -type f 2>/dev/null | head -1) /usr/local/bin/claude
WORKDIR /workspace
COPY workspace/ .
# Block network tools — agent should only read code and write documents
RUN mkdir -p .claude && \\
echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > .claude/settings.json
RUN git init && \\
git config user.email "dev@agent" && \\
git config user.name "Dev" && \\
git add -A && \\
git commit -m "initial" --quiet
CMD ["sleep", "infinity"]
`;
}
// Wrapped in toolkit-managed sentinels so check-task-infra reads this as the
// toolkit's own append rather than an edit to the Dockerfile.
// Only Claude Code produces the sibling session/ directory (subagents, tool results).
// A COPY of an empty directory fails the build outright — buildkit does not carry empty
// directories in the context, so the layer errors with `"/session": not found`.
// Read the SNAPSHOT, not the task dir: the Dockerfile is generated before the session
// files are copied into environment/, so the task-side copy is not there yet.
const sessionSiblingDir = join(snapshotDir, 'session');
const hasSessionSibling =
existsSync(sessionSiblingDir) && readdirSync(sessionSiblingDir).length > 0;
const sessionStaging = `
# >>> toolkit-managed: snapshot-session >>>
# Stage session files for the snapshot agent adapter to install at runtime.
COPY session.jsonl /tmp/snapshot-session/session.jsonl
${hasSessionSibling ? 'COPY session/ /tmp/snapshot-session/session/\n' : ''}RUN echo '${sessionUuid}' > /tmp/snapshot-session/uuid.txt
# <<< toolkit-managed <<<
`;
const dockerfile = baseDockerfile.trimEnd() + '\n' + sessionStaging;
writeFileSync(join(taskDir, 'environment', 'Dockerfile'), dockerfile);
log.debug('Wrote Dockerfile (per-repo base + session staging)');
// --- Copy snapshot.patch as workspace.patch ---
const snapshotPatch = join(snapshotDir, 'snapshot.patch');
if (existsSync(snapshotPatch)) {
copyFileSync(snapshotPatch, join(taskDir, 'environment', 'workspace.patch'));
log.debug('Copied snapshot.patch -> workspace.patch');
}
// --- Scrub the worker's filesystem layout out of the session ---
// In Explore the recorded `cwd` is the worker's HOST checkout (explore/repo is an absolute
// symlink); rewriting the repo root to /workspace both drops the leak and matches the trial.
const WORKSPACE_MOUNT = '/workspace';
const escapeRegExp = (v: string) => v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
/** Member names when this toolkit is polyglot; empty means single-repo. */
const MEMBER_NAMES: readonly string[] = (() => {
const dir = join(repoRoot, 'repos');
if (!existsSync(dir)) return [];
try {
return readdirSync(dir, { withFileTypes: true })
.filter((e) => e.isDirectory())
.map((e) => e.name);
} catch {
return [];
}
})();
/** The repo root within a cwd — the prefix a trial mounts at /workspace. `/repos/<member>`
* anchors only on a polyglot toolkit, so a personal `~/repos/…` above it can't win. */
function repoRootOf(cwd: string): string | null {
if (MEMBER_NAMES.length > 0) {
// A real member of THIS toolkit wins; the generic shape covers a member whose
// directory the toolkit no longer has (an older snapshot, a renamed member).
for (const name of MEMBER_NAMES) {
const hit = cwd.match(new RegExp(`^(.*?/repos/${escapeRegExp(name)})(?:/|$)`));
if (hit) return hit[1];
}
const generic = cwd.match(/^(.*?\/repos\/[^/]+)(?:\/|$)/);
if (generic) return generic[1];
}
// `/repo` needs a component boundary, so it never matches inside `/repos/`.
const m = cwd.match(/^(.*?\/repo)(?:\/|$)/);
return m ? m[1] : null;
}
/** Rewrite every checkout root to /workspace, and the home dir each sits under to `~`. The
* `repo/` anchor needs no host-root list; the home pass still keys off extractHomeDir. */
function scrubWorkerPaths(raw: string): { text: string; roots: string[] } {
// Each cwd contributes its own root, longest first, so a nested root isn't clobbered
// and a session spanning two checkouts is scrubbed rather than skipped.
const roots = [...new Set([...collectCwds(raw)].map(repoRootOf))]
.filter((r): r is string => r !== null)
.sort((a, b) => b.length - a.length);
const { sanitized } = sanitizeSessionJsonl(raw, {
cwdPrefixes: roots,
placeholder: WORKSPACE_MOUNT,
});
return { text: sanitized, roots };
}
// --- Copy session files for --resume ---
//
// The full session.jsonl (including any post-end_turn entries) goes into the
// task root for reference. A truncated version — keeping everything up to
// and including the last assistant entry with stop_reason="end_turn" — goes
// into environment/ for the container. Stopping on a clean assistant turn
// avoids Claude Code's synthetic "No response requested." injection when
// the session is resumed with --fork-session and a new --print prompt.
const sessionJsonl = join(snapshotDir, 'session.jsonl');
if (existsSync(sessionJsonl)) {
// Fail-open: a session this can't scrub ships exactly as it was, because a
// leaked path is a smaller problem than a task that can't be created.
let sessionText = readFileSync(sessionJsonl, 'utf8');
try {
const { text, roots } = scrubWorkerPaths(sessionText);
if (roots.length > 0) {
sessionText = text;
log.info(
{ roots, mountedAt: WORKSPACE_MOUNT },
'Rewrote the authoring checkout path to the trial mount point'
);
} else {
log.debug('No worker-rooted cwd to rewrite; session used as-is');
}
} catch (err) {
log.warn(
{ err: err instanceof Error ? err.message : String(err) },
'Could not rewrite paths in the session; using it as-is'
);
}
// Full version for reference
writeFileSync(join(taskDir, 'session-full.jsonl'), sessionText);
log.debug('Wrote full session.jsonl to task root');
// Truncated version for the container: strip everything from the last
// user text turn onwards. This drops the failure-eliciting question
// (which `--print` will redeliver to the trial agent as the new prompt)
// AND the failure response itself (so the trial agent doesn't see its
// previous answer), while preserving conversational context up to the
// last clean assistant `end_turn`.
//
// Algorithm (refined Option B):
// 1. Find U = index of the last user-text turn that is NOT a slash
// command (use the same command-marker filter as
// extractLastUserMessage).
// 2. Walk backwards from U - 1 to find the last `assistant` entry
// with stop_reason: "end_turn".
// 3. Truncate slice(0, lastEndTurnIndex + 1).
//
// If U doesn't exist or no end_turn assistant precedes U, write an
// empty session.jsonl — the snapshot agent adapter detects this and
// skips --resume entirely, starting fresh from --print.
const sessionLines = sessionText.trimEnd().split('\n');
// A non-Claude session is not a Claude transcript, so the scan below finds no
// `stop_reason: "end_turn"` and would silently write an empty session. Its reader
// applies the same rule in that harness's own format.
const harness = metadata.harness ?? 'claude-code';
const isClaude = harness === 'claude-code';
let lastUserTextIndex = -1;
for (let i = 0; i < sessionLines.length; i++) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type !== 'user' || typeof entry.message?.content !== 'string') continue;
// Compaction summaries are synthetic user turns whose text often quotes
// earlier /create-snapshot:snapshot runs — never the command turn itself,
// so they must not trip the break below.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
// Mirror extractLastUserMessage: skip the snapshot command itself
// and any slash-command / local-command marker turns.
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserTextIndex = i;
} catch {
continue;
}
}
let lastEndTurnIndex = -1;
if (lastUserTextIndex > 0) {
for (let i = lastUserTextIndex - 1; i >= 0; i--) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
message?: { stop_reason?: unknown };
};
if (entry.type === 'assistant' && entry.message?.stop_reason === 'end_turn') {
lastEndTurnIndex = i;
break;
}
} catch {
continue;
}
}
}
if (!isClaude) {
const cut = truncationIndex(turnsFromLines(harness, sessionLines));
const kept = cut >= 0 ? sessionLines.slice(0, cut + 1) : [];
const truncated = stripAuthoringScaffolding(harness, kept);
writeFileSync(
join(taskDir, 'environment', 'session.jsonl'),
truncated.length ? truncated.join('\n') + '\n' : ''
);
log.debug(
{ harness, fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (harness reader)'
);
} else if (lastEndTurnIndex >= 0) {
const truncated = sessionLines.slice(0, lastEndTurnIndex + 1);
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), truncated.join('\n') + '\n');
log.debug(
{ fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (strips last user turn + failure response, keeps through last clean assistant end_turn)'
);
} else {
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), '');
if (lastUserTextIndex < 0) {
log.warn(
'No user text turn found in session — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
} else {
log.warn(
'No assistant entry with stop_reason="end_turn" found before the last user turn (one-shot snapshot) — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
}
}
}
const sessionDir = join(snapshotDir, 'session');
if (existsSync(sessionDir) && statSync(sessionDir).isDirectory()) {
copyTree(sessionDir, join(taskDir, 'environment', 'session'));
// Claude Code writes subagent files write-only (--w-------). Fix them so
// Harbor's dirhash can read them during environment setup.
execSync(`chmod -R +r "${join(taskDir, 'environment', 'session')}"`, { stdio: 'pipe' });
log.debug('Copied session/');
} else {
mkdirSync(join(taskDir, 'environment', 'session'), { recursive: true });
}
// The harness that captured the snapshot; the trial runs this one.
const harness =
typeof metadata.harness === 'string' && metadata.harness ? metadata.harness : 'claude-code';
/**
* The model and effort this harness defaulted to when the task was authored, recorded
* for reference only — nothing reads these back, and a trial still resolves both from
* the registry at run time. Best-effort: a task is not worth failing over a note.
*/
function authoredDefaults(harnessId: string): { model: string; effort: string } | null {
try {
const resolver = join(repoRoot, 'scripts', 'resolve_harness.py');
// Same interpreter search as `_raccoon_python` in scripts/lib/harness-credentials.sh
// and `pythonWithTomllib` in submit-task.ts: `python3` is not always 3.11+, and the
// registry needs tomllib. Best-effort, so a miss just omits the note.
let python = '';
for (const candidate of [
process.env.RACCOON_PYTHON,
'python3',
'python3.13',
'python3.12',
'python3.11',
]) {
if (!candidate) continue;
try {
execFileSync(candidate, ['-c', 'import tomllib'], { stdio: 'ignore' });
python = candidate;
break;
} catch {
continue;
}
}
if (!python) return null;
const rows = execFileSync(python, [resolver, '--defaults'], {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
});
for (const line of rows.split('\n')) {
const [id, model, effort] = line.split('\t');
if (id === harnessId && model) return { model, effort: effort ?? '' };
}
} catch {
// registry unreadable here — omit the note
}
return null;
}
const authored = authoredDefaults(harness);
// --- Write task.toml ---
// The reference-data corpus is included in every zeta task (build-workspace decides from the repo),
// so there's nothing to set here.
const taskToml = `version = "1.0"
[metadata]
program = "raccoon"
author = "rl-env-coding"
category = "sdlc/technical-writing"
repo = "${repoName}"
commit = "${commitShort}"
# The toolkit release this task was created with. Written by the toolkit —
# leave it in place: task tooling reads it to know which toolkit's assets
# this task grades with.
toolkit_version = "${toolkitConfig?.version ?? 'unknown'}"
snapshot = "${basename(snapshotDir)}"
session_uuid = "${sessionUuid}"
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
browser = false
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
[verifier]
timeout_sec = 7200.0
[agent]
harness = "${harness}"
timeout_sec = 18000.0
[environment]
build_timeout_sec = 6000.0
cpus = 2
memory_mb = 4096
storage_mb = 10240
gpus = 0
allow_internet = true
[verifier.env]
ANTHROPIC_API_KEY = "\${ANTHROPIC_API_KEY}"
ANTHROPIC_BASE_URL = "\${ANTHROPIC_BASE_URL}"
[solution.env]
`;
writeFileSync(join(taskDir, 'task.toml'), taskToml);
log.debug('Wrote task.toml');
// --- Extract instruction from session transcript ---
function extractLastUserMessage(sessionPath: string, harness: string): string | null {
if (!existsSync(sessionPath)) return null;
const lines = readFileSync(sessionPath, 'utf8').trimEnd().split('\n');
// A non-Claude session has no `type: "user"` records, so the scan below finds nothing
// and the worker silently gets a placeholder instruction. Its reader applies the same
// rule — last real user turn, ignoring command invocations — in that harness's format.
if (harness !== 'claude-code') {
const userTurns = turnsFromLines(harness, lines).filter(
(t) => t.role === 'user' && !t.isCommand && t.text.trim()
);
return userTurns.length ? userTurns[userTurns.length - 1].text : null;
}
let lastUserMessage: string | null = null;
for (const line of lines) {
try {
const entry = JSON.parse(line) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type === 'user' && typeof entry.message?.content === 'string') {
// Synthetic compaction summary — not a real user turn, and its text
// often quotes earlier /create-snapshot:snapshot runs.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserMessage = content;
}
} catch {
continue;
}
}
return lastUserMessage;
}
const lastUserMessage = extractLastUserMessage(
join(snapshotDir, 'session.jsonl'),
metadata.harness ?? 'claude-code'
);
const instructionHeader =
'# Replace this with your refined task instruction\n\n' +
"<!-- The text below was auto-extracted from your snapshot's last user message.\n" +
' Refine, condense, or rewrite to focus on the behavior you want to elicit. -->\n\n';
if (lastUserMessage) {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader + lastUserMessage.trimEnd() + '\n'
);
log.info('Wrote instruction.md (from last user message in session)');
} else {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader +
'<!-- Could not extract user message from session. Write the instruction manually. -->\n'
);
log.warn('Could not extract instruction from session — needs manual editing');
}
// --- Scaffold holistic-rubric.md ---
const holisticRubricMd = `<!--
Snapshot: ${basename(snapshotDir)}
Session: ${metadata.session_uuid}
Repo: ${metadata.remote_url}
Commit: ${metadata.commit}
What happened in the snapshot conversation
The worker was trying to: ${annotation.what_trying}
They hoped Claude would: ${annotation.what_hoping}
Instead, Claude: ${annotation.what_happened}
Draft this file with /write-holistic-rubric, or point your agent at it,
session-full.jsonl, and task-shared/grading-standard.md. Delete this comment
when you are done.
-->
${HOLISTIC_RUBRIC_SCAFFOLD}`;
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');
// --- Build workspace ---
const buildScript = join(repoRoot, 'scripts', 'build-workspace.sh');
if (existsSync(buildScript)) {
log.info({ repo: repoName, commit: commitShort }, 'Building workspace');
try {
execSync(`bash "${buildScript}" "${slug}" "${commitShort}"`, {
cwd: repoRoot,
encoding: 'utf8',
stdio: 'inherit',
// build-workspace does a bulk-file write burst (git archive|tar of the
// repo tree + a throwaway git add/commit to apply the patch, and for zeta
// toolkits a hardlink-stage of the ~126k-file reference-data corpus that
// falls back to a full copy across filesystems). On a slow bind mount
// (Docker Desktop non-VirtioFS, or WSL2 with the toolkit on a Windows/9p
// path) that legitimately runs into minutes, so a tight cap false-fails a
// working-but-slow build as "not runnable". Keep this generous — it's only
// a backstop against a true hang; the real Harbor build downstream budgets
// build_timeout_sec = 6000.
timeout: 1_200_000,
});
} catch (e: unknown) {
const msg = e instanceof Error ? e.message : String(e);
log.fatal({ error: msg }, 'Workspace build failed — task is not runnable');
log.fatal(` Retry manually: bash scripts/build-workspace.sh ${slug}`);
log.fatal(` Then: scripts/harbor-run harbor-tasks/${slug}`);
process.exit(1);
}
} else {
log.fatal('scripts/build-workspace.sh not found. Please file a bug.');
process.exit(1);
}
try {
execSync('bash -ic "_ev task_created 2>/dev/null" 2>/dev/null', {
stdio: 'ignore',
timeout: 5000,
});
} catch {
// best-effort
}
// --- Done ---
log.info({ taskDir: resolve(taskDir) }, 'Task scaffolded');
log.info('Next steps:');
log.info(' 1. Review instruction.md');
log.info(' 2. Edit tests/holistic-rubric.md — write the rubric');
log.info(' 3. Run calibration trials to validate scoring tiers');

View File

@@ -1,57 +0,0 @@
---
description: Capture a snapshot of the current conversation and repo state.
---
# Create Snapshot
You are capturing a snapshot of the current conversation and repo state so it can be replayed as an RL training task.
## Step 1: Ask annotation questions
**Important — tell the user this first, verbatim:**
> ⚠️ This snapshot captures your entire conversation history with me, not just the most
> recent turn. If you told me the answer earlier in this conversation, or steered me
> toward it, the agent will see that same context when the snapshot replays — and will
> probably solve the task without making the mistake. Your task will be contaminated.
>
> If you've leaked the answer at any point in this conversation: if your agent can rewind
> (Claude Code's `/rewind`), rewind to a point before the contamination and snapshot from
> there. If it can't — codex has no rewind — this snapshot is not salvageable: start a
> fresh session, reproduce the mistake without steering, and snapshot that instead.
Wait for the user to acknowledge before moving on.
Ask the user each of these questions **one at a time** as plain text, waiting for their response before proceeding to the next:
1. "What were you trying to do?"
2. "What were you hoping was going to happen?"
3. "What did the agent actually do instead?"
## Step 2: Propose a slug
Based on the user's answers, generate a **short kebab-case slug** (2-4 words) that captures the essence of the mistake. For example: `bad-refactor`, `wrong-test-strategy`, `missed-edge-case`.
Present your suggestion and ask the user to confirm or provide an alternative.
## Step 3: Write annotation file and run capture
Write the annotation to a temporary JSON file, then run the capture script.
Write this JSON to a temp file (use a path like `/tmp/snapshot-annotation-<timestamp>.json`):
```json
{
"what_trying": "<answer to question 1>",
"what_hoping": "<answer to question 2>",
"what_happened": "<answer to question 3>"
}
```
Then run:
```bash
"${CLAUDE_PLUGIN_ROOT}/bin/capture-snapshot.mjs" --slug <slug> --annotation <temp-file-path> --output-dir "${CLAUDE_PLUGIN_ROOT}/../../snapshots" --plugin-data "${CLAUDE_PLUGIN_DATA:-${CLAUDE_PLUGIN_ROOT}/.data}"
```
Report the script's stdout output verbatim to the user. Do not paraphrase or shorten paths.

View File

@@ -1,35 +0,0 @@
{
"hooks": {
"SessionStart": [
{
"hooks": [
{
"type": "command",
"command": "${CLAUDE_PLUGIN_ROOT}/bin/save-session-info.mjs"
}
]
}
],
"UserPromptSubmit": [
{
"hooks": [
{
"type": "command",
"command": "${CLAUDE_PLUGIN_ROOT}/bin/checkpoint-workspace.mjs"
}
]
}
],
"PreToolUse": [
{
"matcher": "Bash",
"hooks": [
{
"type": "command",
"command": "${CLAUDE_PLUGIN_ROOT}/bin/checkpoint-workspace.mjs"
}
]
}
]
}
}

View File

@@ -1,76 +0,0 @@
---
name: snapshot
description: Capture the current conversation and repo state as a snapshot, to be replayed as a task. Use when the user wants to snapshot a mistake the agent just made.
---
# Create Snapshot
You are capturing a snapshot of the current conversation and repo state so it can be
replayed as an RL training task.
## Step 0: Mark where the snapshot begins
Run this FIRST, before asking anything. It records where the conversation ended so the
questions below aren't captured as part of it:
```bash
"${RACCOON_SNAPSHOT_PLUGIN_ROOT:-/workspace/plugins/create-snapshot}/bin/capture-snapshot.mjs" \
--mark-start --harness "${RACCOON_HARNESS:?not set — start your session through the launcher (the plain agent command, e.g. \`codex\`) so the snapshot records which agent it came from}"
```
## Step 1: Ask annotation questions
**Important — tell the user this first, verbatim:**
> ⚠️ This snapshot captures your entire conversation history with me, not just the most
> recent turn. If you told me the answer earlier in this conversation, or steered me
> toward it, the agent will see that same context when the snapshot replays — and will
> probably solve the task without making the mistake. Your task will be contaminated.
>
> If you've leaked the answer at any point in this conversation: if your agent can rewind
> (Claude Code's `/rewind`), rewind to a point before the contamination and snapshot from
> there. If it can't — codex has no rewind — this snapshot is not salvageable: start a
> fresh session, reproduce the mistake without steering, and snapshot that instead.
Wait for the user to acknowledge before moving on.
Ask the user each of these questions **one at a time** as plain text, waiting for their
response before proceeding to the next:
1. "What were you trying to do?"
2. "What were you hoping was going to happen?"
3. "What did the agent actually do instead?"
## Step 2: Propose a slug
Based on the user's answers, generate a **short kebab-case slug** (2-4 words) that
captures the essence of the mistake. For example: `bad-refactor`,
`wrong-test-strategy`, `missed-edge-case`.
Present your suggestion and ask the user to confirm or provide an alternative.
## Step 3: Write annotation file and run capture
Write the annotation to a temporary JSON file, then run the capture script.
Write this JSON to a temp file (use a path like `/tmp/snapshot-annotation-<timestamp>.json`):
```json
{
"what_trying": "<answer to question 1>",
"what_hoping": "<answer to question 2>",
"what_happened": "<answer to question 3>"
}
```
Then run:
```bash
"${RACCOON_SNAPSHOT_PLUGIN_ROOT:-/workspace/plugins/create-snapshot}/bin/capture-snapshot.mjs" \
--harness "$RACCOON_HARNESS" \
--slug <slug> \
--annotation <temp-file-path> \
--output-dir /workspace/snapshots
```
Report the script's stdout output verbatim to the user. Do not paraphrase or shorten paths.