added potion-polyglot worker folder w/o repos

This commit is contained in:
2026-09-08 21:58:19 -04:00
parent f10303b8c2
commit a16a457669
211 changed files with 41614 additions and 0 deletions

View File

@@ -0,0 +1,788 @@
#!/usr/bin/env node
import { execSync } from 'node:child_process';
import crypto from 'node:crypto';
import fs from 'node:fs';
import path from 'node:path';
import { linearSnapshotLines, readSession } from './harness-session.mjs';
// --- Argument parsing ---
function parseArgs(argv) {
const args = {};
for (let i = 2; i < argv.length; i++) {
if (argv[i].startsWith('--')) {
const key = argv[i].slice(2);
const val = argv[i + 1];
if (!val || val.startsWith('--')) {
args[key] = true;
} else {
args[key] = val;
i++;
}
}
}
return args;
}
// Last-resort data dir. Claude Code's plugin runtime always provides one, so this is
// what makes the start marker and session record work under any other harness.
function defaultDataDir() {
const home = process.env.HOME || '/root';
return path.join(home, '.raccoon', 'snapshot-data');
}
const args = parseArgs(process.argv);
const slug = args.slug;
const annotationPath = args.annotation;
const outputDir = args['output-dir'];
if (!args['mark-start'] && (!slug || !annotationPath || !outputDir)) {
console.error(
'Usage: capture-snapshot.mjs --slug <slug> --annotation <path> --output-dir <dir> [--plugin-data <path>]\n' +
' capture-snapshot.mjs --mark-start [--harness <id>]'
);
process.exit(1);
}
// --- Locate session info ---
// --harness wins over the launcher's RACCOON_HARNESS so a caller that knows which
// conversation it is capturing can say so; the default keeps Claude Code's plugin
// working unchanged. Claude Code has an exact cut point (the snapshot slash command)
// and a message tree to prune, so it keeps the bespoke path below; other harnesses go
// through the shared reader.
const HARNESS = args.harness || process.env.RACCOON_HARNESS || 'claude-code';
const IS_CLAUDE = HARNESS === 'claude-code';
// The generated restore.sh writes one of exactly two session layouts, and everything
// below branches on IS_CLAUDE — so a third harness would silently be handed codex's
// $CODEX_HOME/sessions paths. Refuse instead; adding a harness means adding a layout.
if (!IS_CLAUDE && HARNESS !== 'codex') {
console.error(
`capture-snapshot: no session-restore layout for harness "${HARNESS}". ` +
'Add one to capture-snapshot.mjs (and harness-session.mjs) before capturing with it.'
);
process.exit(1);
}
// Try multiple strategies to find the current session transcript:
// 1. Plugin data dir (from SessionStart hook)
// 2. Scan ~/.claude/projects/ for the most recently modified JSONL
let session_id = null;
let transcript_path = null;
const dataDir =
args['plugin-data'] ||
process.env.RACCOON_SNAPSHOT_DATA ||
process.env.CLAUDE_PLUGIN_DATA ||
(process.env.CLAUDE_PLUGIN_ROOT && path.join(process.env.CLAUDE_PLUGIN_ROOT, '.data')) ||
defaultDataDir();
// The SessionStart hook records the live session for every harness, so prefer it over
// guessing. `readSession` falls back to the newest file on disk when it is absent.
let recordedSession = null;
if (dataDir) {
const sessionInfoPath = path.join(dataDir, 'current-session.json');
if (fs.existsSync(sessionInfoPath)) {
try {
recordedSession = JSON.parse(fs.readFileSync(sessionInfoPath, 'utf8'));
} catch {
recordedSession = null;
}
}
}
const startMarkerPath = dataDir ? path.join(dataDir, 'snapshot-start.json') : null;
// `--mark-start` runs BEFORE the annotation Q&A and records how long the conversation
// was at that moment. It is the linear-harness stand-in for Claude Code's slash-command
// line: without it, capture would stage the snapshot's own Q&A as conversation.
if (args['mark-start']) {
const session = readSession(HARNESS);
if (!session) {
console.error(`No ${HARNESS} session found to mark.`);
process.exit(1);
}
if (!startMarkerPath) {
console.error('No data dir available to record the snapshot start marker.');
process.exit(1);
}
fs.mkdirSync(path.dirname(startMarkerPath), { recursive: true });
fs.writeFileSync(
startMarkerPath,
JSON.stringify(
{ harness: HARNESS, transcript_path: session.rawPath, line_count: session.lines.length },
null,
2
) + '\n'
);
console.log(`Snapshot start marked at ${session.lines.length} records.`);
process.exit(0);
}
let harnessSession = null;
if (!IS_CLAUDE) {
harnessSession = readSession(HARNESS, recordedSession?.transcript_path);
if (!harnessSession) {
console.error(
`Could not find a ${HARNESS} session to capture. Capture has to run from inside the ${HARNESS} conversation you want to snapshot.`
);
process.exit(1);
}
transcript_path = harnessSession.rawPath;
// Fall back to a fresh id only if the harness records none — restore.sh names the
// installed session by it, so it has to match what `resume` will look up.
session_id = recordedSession?.session_id || harnessSession.sessionId || crypto.randomUUID();
} else if (recordedSession) {
session_id = recordedSession.session_id;
transcript_path = recordedSession.transcript_path;
}
// Fallback: find the most recently modified JSONL in ~/.claude/projects/
if (!transcript_path) {
const homeDir = process.env.HOME || '/root';
const projectsDir = path.join(homeDir, '.claude', 'projects');
if (fs.existsSync(projectsDir)) {
let newest = null;
let newestMtime = 0;
for (const projEntry of fs.readdirSync(projectsDir)) {
const projDir = path.join(projectsDir, projEntry);
if (!fs.statSync(projDir).isDirectory()) continue;
for (const file of fs.readdirSync(projDir)) {
if (!file.endsWith('.jsonl')) continue;
const filePath = path.join(projDir, file);
const mtime = fs.statSync(filePath).mtimeMs;
if (mtime > newestMtime) {
newestMtime = mtime;
newest = filePath;
session_id = file.replace(/\.jsonl$/, '');
}
}
}
transcript_path = newest;
}
}
if (!transcript_path || !fs.existsSync(transcript_path)) {
console.error(
"Could not find a Claude Code session transcript. This script should be run from within a Claude Code conversation via the /create-snapshot:snapshot command. Please file a bug if you're seeing this unexpectedly."
);
process.exit(1);
}
if (!transcript_path || !fs.existsSync(transcript_path)) {
console.error(`Transcript file not found at ${transcript_path}. Please file a bug.`);
process.exit(1);
}
// --- Require a git repo at capture time ---
//
// A snapshot is "commit SHA + diff vs HEAD", reconstituted later via
// `git archive <SHA> | tar -x` + `git apply workspace.patch`. Without a git
// repo here we have no SHA to pin, no patch to record, and no way for
// downstream `build-workspace.sh` to reproduce the workspace — the resulting
// snapshot would be structurally meaningless. This check runs BEFORE the
// snapshot directory is created so a misconfigured invocation leaves no
// half-written state behind.
// Find the git repo by asking git itself — walks up from cwd looking for
// `.git`, handling submodules and worktrees correctly. Returns null when
// cwd is outside any repo, so the worker gets a clear "cd into your repo"
// error instead of silently descending into something they didn't name.
function findGitRepo() {
try {
const top = execSync('git rev-parse --show-toplevel', {
stdio: ['ignore', 'pipe', 'ignore'],
})
.toString()
.trim();
return top || null;
} catch {
return null;
}
}
const gitRepo = findGitRepo();
if (!gitRepo) {
console.error(
"Error: Not running inside a git repo. /create-snapshot needs a git repo so it can pin a commit SHA and record a diff of in-flight changes; without one the snapshot can't be reproduced as a task. cd into the repo you're exploring (the toolkit's repo/ submodule) and re-run /create-snapshot:snapshot."
);
process.exit(1);
}
// --- Create snapshot directory ---
const ts = new Date().toISOString().replace(/[-:]/g, '').replace('T', '-').slice(0, 15); // 20260403-225449
const snapshotDir = path.join(outputDir, `${ts}-${slug}`);
if (fs.existsSync(snapshotDir)) {
console.error(`Snapshot directory already exists: ${snapshotDir}\nPlease file a bug.`);
process.exit(1);
}
fs.mkdirSync(snapshotDir, { recursive: true });
// --- Copy conversation transcript (trimmed + branch-pruned) ---
// The JSONL is a tree of messages linked by parentUuid. When the user rewinds
// a conversation, old branches remain in the file. We need to:
// 1. Cut at the LAST /create-snapshot:snapshot command (later invocations
// supersede earlier ones in the same session)
// 2. Find the tip of the active branch (last message before the cut)
// 3. Walk parentUuid back to the root, collecting only messages on that path
// 4. Exclude the Q&A subgraphs of any PRIOR /create-snapshot:snapshot
// invocations in this session (their cut points are on the same
// conversation branch, so the walk would otherwise pull in the
// assistant's annotation questions and the user's answers — a
// contamination path that snapshot.patch doesn't show). Boundaries
// for prior invocations are recorded in a side file (see end of
// this script) so this run can identify them.
// 5. Drop bookkeeping entries whose content can leak rewound-branch state.
const rawLines = fs.readFileSync(transcript_path, 'utf8').trimEnd().split('\n');
// A user message is a /create-snapshot:snapshot invocation when its content
// STARTS with one of Claude Code's slash-command tags AND mentions the
// command name. The "starts with" guard distinguishes a real invocation
// from prose that quotes the command (a worker reporting a bug, the
// command-listing skill output, etc.) — prose doesn't begin with those
// tags. The whitespace-tolerant pattern survives minor format drift in
// Claude Code's slash-command rendering.
const SNAPSHOT_CMD_PATTERN =
/<command-(?:name|message)>\s*\/?\s*create-snapshot:snapshot\s*<\/command-(?:name|message)>/;
function isSnapshotCommandContent(content) {
if (typeof content !== 'string') return false;
const trimmed = content.trimStart();
if (!trimmed.startsWith('<command-name>') && !trimmed.startsWith('<command-message>')) {
return false;
}
return SNAPSHOT_CMD_PATTERN.test(content);
}
// Find every snapshot-command line index, in order. The LAST one is the
// current invocation (cut point); earlier ones bound prior Q&A subgraphs.
const snapshotCmdIndexes = [];
for (let i = 0; i < rawLines.length; i++) {
try {
const entry = JSON.parse(rawLines[i]);
if (entry.type === 'user' && isSnapshotCommandContent(entry.message?.content)) {
snapshotCmdIndexes.push(i);
}
} catch {
// Skip malformed lines
}
}
const cutIndex =
snapshotCmdIndexes.length > 0
? snapshotCmdIndexes[snapshotCmdIndexes.length - 1]
: rawLines.length;
const priorCmdIndexes = snapshotCmdIndexes.slice(0, -1);
// Load prior-snapshot boundary records so we know where each earlier
// invocation's Q&A subgraph ended. The boundary file is written at the
// end of every capture run (see below) and is keyed by session uuid.
function loadPriorBoundaries() {
if (!dataDir || !session_id) return [];
const boundariesPath = path.join(dataDir, 'snapshot-boundaries.jsonl');
if (!fs.existsSync(boundariesPath)) return [];
const lines = fs.readFileSync(boundariesPath, 'utf8').trimEnd().split('\n');
const out = [];
for (const line of lines) {
if (!line) continue;
try {
const rec = JSON.parse(line);
if (rec.sessionUuid === session_id && rec.snapshotCommandUuid) out.push(rec);
} catch {
/* skip malformed */
}
}
return out;
}
const priorBoundaries = loadPriorBoundaries();
// Compute the line ranges to exclude for each prior snapshot. The Q&A
// subgraph starts at the prior snapshot's command line and runs through
// the line whose entry uuid matches the boundary record (the last entry
// in the JSONL when that prior capture-snapshot completed).
//
// Fall back to the next snapshot command (or the current cut) when no
// matching boundary record exists — better to drop too much than to leak
// the Q&A; the visible cost is excluding any "real work" that happened
// between snapshots without a recorded boundary, which only occurs if
// the boundary log was wiped or the prior capture crashed.
function findUuidLineIndex(targetUuid, startLine, endLineExclusive) {
for (let i = startLine; i < endLineExclusive; i++) {
try {
const entry = JSON.parse(rawLines[i]);
if (entry.uuid === targetUuid) return i;
} catch {
/* skip */
}
}
return -1;
}
const priorQAExcludedLines = new Set();
for (let i = 0; i < priorCmdIndexes.length; i++) {
const startLine = priorCmdIndexes[i];
const nextCutLine = i + 1 < priorCmdIndexes.length ? priorCmdIndexes[i + 1] : cutIndex;
let snapshotCmdUuid = null;
try {
snapshotCmdUuid = JSON.parse(rawLines[startLine]).uuid || null;
} catch {
/* unparseable command line — skip */
}
let endLine = -1;
if (snapshotCmdUuid) {
const boundary = priorBoundaries.find((b) => b.snapshotCommandUuid === snapshotCmdUuid);
if (boundary && boundary.lastEntryUuid) {
endLine = findUuidLineIndex(boundary.lastEntryUuid, startLine, nextCutLine);
}
}
// No matching boundary: bound the exclusion at the next snapshot/current
// cut so the Q&A doesn't leak even if state was lost.
if (endLine < 0) endLine = nextCutLine - 1;
for (let j = startLine; j <= endLine; j++) priorQAExcludedLines.add(j);
}
// Step 2: parse all entries before the cut, build uuid index
const preCutEntries = [];
const byUuid = {};
for (let i = 0; i < cutIndex; i++) {
try {
const entry = JSON.parse(rawLines[i]);
preCutEntries.push({ line: rawLines[i], entry, index: i });
if (entry.uuid) {
byUuid[entry.uuid] = entry;
}
} catch {
// Keep unparseable lines (they'll be included as non-message entries)
preCutEntries.push({ line: rawLines[i], entry: null, index: i });
}
}
// Step 3: find the tip of the active branch. The snapshot command's parentUuid
// points to the message the user was looking at when they ran the snapshot —
// this is authoritative even after rewinds.
let tipUuid = null;
if (cutIndex < rawLines.length) {
try {
const snapshotCmd = JSON.parse(rawLines[cutIndex]);
tipUuid = snapshotCmd.parentUuid || null;
} catch {
// not valid JSON — leave tipUuid null
}
}
// If the live tip is itself inside a prior snapshot's Q&A subgraph (e.g.
// the user ran /create-snapshot:snapshot a second time WITHOUT typing
// anything between the two — there's no "real work" gap), walk back past
// the excluded range to find the closest non-excluded ancestor. Otherwise
// activeBranchUuids would be empty and we'd produce an empty snapshot.
function nearestNonExcludedAncestor(startUuid) {
let cur = startUuid;
while (cur) {
const e = byUuid[cur];
if (!e) return cur; // unknown uuid — best effort, keep
// Find the line index of this entry to check exclusion.
// (Line index isn't stored on the entry; recompute via preCutEntries.)
const found = preCutEntries.find((p) => p.entry?.uuid === cur);
if (!found || !priorQAExcludedLines.has(found.index)) return cur;
cur = e.parentUuid || null;
}
return null;
}
if (tipUuid) tipUuid = nearestNonExcludedAncestor(tipUuid);
// Fallback: if no snapshot command found, use the last entry with a uuid
if (!tipUuid) {
for (let i = preCutEntries.length - 1; i >= 0; i--) {
if (preCutEntries[i].entry?.uuid && !priorQAExcludedLines.has(preCutEntries[i].index)) {
tipUuid = preCutEntries[i].entry.uuid;
break;
}
}
}
// Collect all uuids on the active branch
const activeBranchUuids = new Set();
let current = tipUuid;
while (current) {
activeBranchUuids.add(current);
current = byUuid[current]?.parentUuid || null;
}
// Step 4: filter — keep entries on the active branch.
//
// Claude Code writes several bookkeeping entry types alongside the message
// tree that don't carry a branch uuid. Their content references whatever
// branch was active when they were written, so if the user has rewound,
// these will leak rewound-branch state (file backups, prior prompt text,
// stale titles, queued prompts, PR links, etc.) into the snapshot — a leak
// snapshot.patch doesn't show. Drop the ones we can't attribute to the
// active branch.
//
// `file-history-snapshot` is special-cased: it carries a `messageId`
// pointing at the message whose pre-edit state it tracks, so we can
// keep only those whose messageId is on the active branch. That
// preserves /rewind functionality after a snapshot is restored (rewind
// needs the file-backup metadata) while still dropping records from
// rewound branches.
const BLANKET_DROP_TYPES = new Set([
'agent-name',
'ai-title',
'custom-title',
'last-prompt',
'permission-mode',
'pr-link',
'queue-operation',
]);
function prunedClaudeLines() {
const kept = [];
for (const { line, entry, index } of preCutEntries) {
if (priorQAExcludedLines.has(index)) continue;
if (!entry) {
// Unparseable line — keep as-is so we don't lose data we can't classify.
kept.push(line);
continue;
}
if (entry.uuid) {
if (activeBranchUuids.has(entry.uuid)) kept.push(line);
continue;
}
// No uuid: bookkeeping entry.
if (entry.type === 'file-history-snapshot') {
// Keep only if the message it tracks is on the active branch.
if (entry.messageId && activeBranchUuids.has(entry.messageId)) {
kept.push(line);
}
continue;
}
if (!BLANKET_DROP_TYPES.has(entry.type)) kept.push(line);
}
return kept;
}
// Rewind branches and prior-Q&A exclusion are Claude-transcript concerns; a linear
// harness transcript just truncates at its boundary.
let startLine;
if (!IS_CLAUDE && startMarkerPath && fs.existsSync(startMarkerPath)) {
try {
const marker = JSON.parse(fs.readFileSync(startMarkerPath, 'utf8'));
if (marker.transcript_path === harnessSession.rawPath) startLine = marker.line_count;
} catch {
startLine = undefined;
}
}
if (!IS_CLAUDE && startLine === undefined) {
console.error(
'WARNING: no snapshot start marker for this session — the snapshot Q&A may be captured as conversation. Run capture-snapshot.mjs --mark-start before the annotation questions.'
);
}
const outputLines = IS_CLAUDE
? prunedClaudeLines()
: linearSnapshotLines(harnessSession, startLine);
if (outputLines.length === 0) {
console.error(
`WARNING: found no conversation to seed in ${transcript_path}, so this snapshot has no prior turns. The task will run cold from its prompt alone — fine if that is what you want, but if you meant to capture a conversation, check that the exchange you wanted came BEFORE this snapshot.`
);
}
// Zero bytes, not a lone newline, when there is nothing to seed: downstream decides
// single- vs multi-turn on the file's SIZE, so a 1-byte file would try to resume nothing.
fs.writeFileSync(
path.join(snapshotDir, 'session.jsonl'),
outputLines.length > 0 ? outputLines.join('\n') + '\n' : ''
);
// --- Write boundary record so the NEXT capture-snapshot in this session
// can identify and exclude this snapshot's Q&A subgraph ---
//
// The record pairs the current invocation's command-line uuid with the
// uuid of the last entry in the JSONL at this moment (which is whichever
// assistant turn invoked us as a tool). A subsequent capture run reads
// this file, finds these two uuids in its raw lines, and excludes the
// range — a small leak still exists for entries appended AFTER capture
// returns (the assistant's "Snapshot saved to: ..." reply), but the
// substantive annotation Q&A is fully bounded.
if (dataDir && session_id && cutIndex < rawLines.length) {
let snapshotCommandUuid = null;
try {
snapshotCommandUuid = JSON.parse(rawLines[cutIndex]).uuid || null;
} catch {
/* leave null — we'll skip writing */
}
// Re-read transcript so we pick up any lines Claude Code has appended
// since we read it above (the assistant's tool-use entry, etc.).
let lastEntryUuid = null;
try {
const liveLines = fs.readFileSync(transcript_path, 'utf8').trimEnd().split('\n');
for (let i = liveLines.length - 1; i >= 0; i--) {
try {
const e = JSON.parse(liveLines[i]);
if (e.uuid) {
lastEntryUuid = e.uuid;
break;
}
} catch {
/* skip */
}
}
} catch {
/* transcript unreadable now — skip writing */
}
if (snapshotCommandUuid && lastEntryUuid) {
const boundariesPath = path.join(dataDir, 'snapshot-boundaries.jsonl');
try {
fs.mkdirSync(dataDir, { recursive: true });
fs.appendFileSync(
boundariesPath,
JSON.stringify({
sessionUuid: session_id,
snapshotCommandUuid,
lastEntryUuid,
timestamp: new Date().toISOString(),
}) + '\n'
);
} catch {
// Best-effort: a missing boundary just means the next run falls back
// to the conservative "exclude through next snapshot" heuristic.
}
}
}
// Copy subagents and tool-results if they exist
const sessionSiblingDir = transcript_path.replace(/\.jsonl$/, '');
if (fs.existsSync(sessionSiblingDir) && fs.statSync(sessionSiblingDir).isDirectory()) {
fs.cpSync(sessionSiblingDir, path.join(snapshotDir, 'session'), { recursive: true });
// Claude Code creates subagent files with write-only permissions (--w-------).
// Fix them so downstream tools (cpSync in snapshot-to-task, Harbor's dirhash) can read them.
execSync(`chmod -R +r "${path.join(snapshotDir, 'session')}"`, { stdio: 'pipe' });
}
// --- Capture git state as a patch ---
// Returns raw stdout bytes — callers that want a single-line value must
// .trim() themselves. Don't trim here: some callers (git diff) produce
// patches where a trailing " \n" blank-context line is load-bearing, and
// stripping it corrupts the patch.
function git(cmd, opts) {
try {
return execSync(`git ${cmd}`, {
encoding: 'utf8',
maxBuffer: 50 * 1024 * 1024,
cwd: gitRepo,
stdio: ['pipe', 'pipe', 'pipe'],
...opts,
});
} catch {
return null;
}
}
const commit = git('rev-parse HEAD')?.trim() ?? null;
const branch = git('rev-parse --abbrev-ref HEAD')?.trim() ?? null;
const remoteUrl = git('remote get-url origin')?.trim() ?? null;
// Generate a unified patch representing the workspace state AT THE END OF
// THE PRIOR TURN — i.e., everything done up to but not including the turn
// being snapshotted. This is the state the trial agent should inherit so
// it gets a fresh attempt at the prompt that triggered the snapshot.
//
// The UserPromptSubmit hook checkpoints the working tree to
// `refs/raccoon/turn-checkpoint` at every turn boundary (skipping snapshot
// invocations themselves), so the latest checkpoint is exactly the state
// at the start of the snapshotted turn. We diff HEAD against that
// checkpoint to produce the patch.
//
// Falls back to the pre-checkpoint behavior (full working-tree diff) when
// no checkpoint exists — e.g., the worker took a snapshot before any
// non-snapshot user message was sent, or the hook never fired (legacy
// session, plugin re-installed mid-session, etc.).
if (gitRepo) {
try {
const tmpIndex = path.join(snapshotDir, '.tmp-git-index');
const indexEnv = { ...process.env, GIT_INDEX_FILE: tmpIndex };
// Prefer the FROZEN ref — this is set by checkpoint-workspace at the
// moment the user invokes /create-snapshot:*, before any Q&A turns
// have a chance to advance the live checkpoint past the state we
// want to capture. Fall back to the live checkpoint (then to
// working-tree diff) for backward-compat or if the freeze step failed.
let baseline = null;
try {
baseline = git('rev-parse refs/raccoon/turn-checkpoint-frozen')?.trim() ?? null;
} catch {
baseline = null;
}
if (!baseline) {
try {
baseline = git('rev-parse refs/raccoon/turn-checkpoint')?.trim() ?? null;
} catch {
baseline = null;
}
}
// --binary --full-index, on both branches: a plain `git diff` records a
// binary difference as an opaque `Binary files a/x and /dev/null differ`
// stub, and `git apply` refuses it ("without full index line"), so
// build-workspace.sh can't rebuild the task at all. Nobody has to edit a
// binary to hit this — a tracked .DS_Store the toolkit zip strips from the
// shipped checkout reads as a binary deletion in every session.
//
// maxBuffer: inlined binaries make patches far bigger than text diffs, and
// exceeding the default cap would throw away the whole patch silently.
const diffOpts = { env: indexEnv, maxBuffer: 512 * 1024 * 1024 };
let patch;
if (baseline) {
// Diff HEAD against the prior-turn checkpoint. Untracked files in
// the checkpoint have been committed to the checkpoint tree, so
// they're included automatically.
patch = git(`diff --binary --full-index HEAD ${baseline}`, diffOpts);
} else {
// No checkpoint — fall back to live working-tree diff (pre-fix
// behavior). Captures everything different from HEAD, including
// any agent edits during the current turn.
git('read-tree HEAD', { env: indexEnv });
git('add -A', { env: indexEnv });
patch = git('diff --cached --binary --full-index HEAD', diffOpts);
}
try {
fs.unlinkSync(tmpIndex);
} catch {
/* ignore */
}
if (patch) {
fs.writeFileSync(
path.join(snapshotDir, 'snapshot.patch'),
patch.endsWith('\n') ? patch : patch + '\n'
);
}
} catch {
// Read-only repo or other git error — skip patch generation
}
}
// --- Copy annotation ---
const annotation = JSON.parse(fs.readFileSync(annotationPath, 'utf8'));
fs.writeFileSync(
path.join(snapshotDir, 'annotation.json'),
JSON.stringify(annotation, null, 2) + '\n'
);
// Clean up temp file
try {
fs.unlinkSync(annotationPath);
} catch {
// Ignore cleanup failures
}
// --- Write metadata ---
const metadata = {
slug: slug,
session_uuid: session_id,
// The harness the session was actually read as, so it can't disagree with what
// was captured.
harness: HARNESS,
original_cwd: process.cwd(),
commit: commit,
branch: branch,
remote_url: remoteUrl,
timestamp: new Date().toISOString(),
plugin_version: '0.2.0',
};
fs.writeFileSync(path.join(snapshotDir, 'metadata.json'), JSON.stringify(metadata, null, 2) + '\n');
// --- Generate restore.sh ---
const restoreScript = `#!/usr/bin/env bash
set -euo pipefail
# Restore a snapshot for resuming a Claude Code conversation.
#
# Usage: ./restore.sh [target-dir]
# target-dir: directory to clone/checkout the repo into (default: ./repo)
SCRIPT_DIR="$(cd "$(dirname "\${BASH_SOURCE[0]}")" && pwd)"
TARGET_DIR="\${1:-./repo}"
# Read metadata
COMMIT=$(jq -r '.commit' "$SCRIPT_DIR/metadata.json")
REMOTE=$(jq -r '.remote_url' "$SCRIPT_DIR/metadata.json")
SESSION_UUID=$(jq -r '.session_uuid' "$SCRIPT_DIR/metadata.json")
echo "Cloning $REMOTE at $COMMIT..."
git clone "$REMOTE" "$TARGET_DIR"
cd "$TARGET_DIR"
git checkout "$COMMIT"
# Apply snapshot patch if present
if [ -f "$SCRIPT_DIR/snapshot.patch" ]; then
echo "Applying snapshot.patch..."
git apply "$SCRIPT_DIR/snapshot.patch"
fi
# Install conversation so the authoring harness can resume it
${
IS_CLAUDE
? `ENCODED_CWD=$(echo "$PWD" | sed 's|/|-|g; s|^-||')
DEST_DIR="$HOME/.claude/projects/-$ENCODED_CWD"
mkdir -p "$DEST_DIR"
cp "$SCRIPT_DIR/session.jsonl" "$DEST_DIR/$SESSION_UUID.jsonl"
if [ -d "$SCRIPT_DIR/session" ]; then
cp -r "$SCRIPT_DIR/session" "$DEST_DIR/$SESSION_UUID"
fi
echo ""
echo "Snapshot restored. To resume the conversation:"
echo " cd $TARGET_DIR"
echo " claude --resume $SESSION_UUID"`
: `DEST_DIR="\${CODEX_HOME:-$HOME/.codex}/sessions/$(date -u +%Y/%m/%d)"
mkdir -p "$DEST_DIR"
cp "$SCRIPT_DIR/session.jsonl" \\
"$DEST_DIR/rollout-$(date -u +%Y-%m-%dT%H-%M-%S).000Z-$SESSION_UUID.jsonl"
echo ""
echo "Snapshot restored. To resume the conversation:"
echo " cd $TARGET_DIR"
echo " codex resume $SESSION_UUID"`
}
`;
fs.writeFileSync(path.join(snapshotDir, 'restore.sh'), restoreScript);
fs.chmodSync(path.join(snapshotDir, 'restore.sh'), 0o755);
try {
execSync('bash -ic "_ev snapshot_created 2>/dev/null" 2>/dev/null', {
stdio: 'ignore',
timeout: 5000,
});
} catch {
// best-effort
}
// --- Done ---
const fullSnapshotDir = path.resolve(snapshotDir);
console.log(`Snapshot saved to: ${fullSnapshotDir}`);
console.log(` session.jsonl — conversation transcript`);
if (fs.existsSync(sessionSiblingDir) && fs.statSync(sessionSiblingDir).isDirectory()) {
console.log(` session/ — subagents + tool results`);
}
if (fs.existsSync(path.join(snapshotDir, 'snapshot.patch'))) {
console.log(` snapshot.patch — working tree changes`);
}
console.log(` annotation.json — worker annotations`);
console.log(` metadata.json — session metadata`);
console.log(` restore.sh — restore script for resuming`);

View File

@@ -0,0 +1,485 @@
#!/usr/bin/env node
// Rewind-aware workspace checkpointing for the reduced-toolset Explore agent.
// One script, two hook events (branches on hook_event_name):
//
// UserPromptSubmit -> CAPTURE
// Snapshot the pre-turn working tree into refs/raccoon/turn-checkpoint (the
// chain capture-snapshot uses for snapshot.patch) AND record, in
// .git/raccoon-state.json, anchor_map[tip] = checkpoint-commit and
// last_anchor = tip. `tip` is the conversation node the new prompt attaches
// to (the END of the previous turn) — exactly the node a future /rewind to
// THIS turn will branch from. Anchoring to the prior tip (not the
// just-submitted, maybe-unflushed message) makes capture race-free.
//
// PreToolUse (first tool call of a turn) -> RECONCILE
// Claude Code's /rewind restores the conversation but NOT bash-made edits,
// and fires no hook. By the first tool call the post-rewind branch message is
// reliably persisted and the agent has not yet read/edited code. We parse the
// transcript into a parentUuid DAG, pick the ACTIVE branch (leaf with the
// newest tip), and walk it for the newest checkpoint anchor that is a genuine
// rewind fork (the anchor still has an orphaned child branch — the discarded
// turns). If found, restore the working tree to that checkpoint before the tool
// runs. Idempotent per (anchor, branch): restores once per rewind.
//
// Every failure path is a safe no-op: the hook never aborts the session and
// never restores to an unverified tree.
import { execSync } from 'node:child_process';
import fs from 'node:fs';
import path from 'node:path';
const CHECKPOINT_REF = 'refs/raccoon/turn-checkpoint';
const FROZEN_REF = 'refs/raccoon/turn-checkpoint-frozen';
// Synthetic parent for every parentless transcript node, so that a rewind to the
// VERY FIRST turn (where the new prompt also has parentUuid=null) is detected by
// the same divergence machinery as any other turn.
const ROOT = '__ROOT__';
const RACCOON_AUTHOR = {
GIT_AUTHOR_NAME: 'raccoon',
GIT_AUTHOR_EMAIL: 'raccoon@local',
GIT_COMMITTER_NAME: 'raccoon',
GIT_COMMITTER_EMAIL: 'raccoon@local',
};
function makeGit(gitDir, extraEnv) {
const env = { ...process.env, ...extraEnv };
return (cmd) =>
execSync(`git ${cmd}`, {
cwd: gitDir,
env,
stdio: ['pipe', 'pipe', 'pipe'],
encoding: 'utf8',
}).trim();
}
function findGitDir(cwd) {
let gitDir = cwd;
for (let i = 0; i < 10; i++) {
if (fs.existsSync(path.join(gitDir, '.git'))) return gitDir;
const parent = path.dirname(gitDir);
if (parent === gitDir) return null;
gitDir = parent;
}
return null;
}
// ---- transcript + state ----
function readEntries(transcriptPath) {
try {
if (!transcriptPath || !fs.existsSync(transcriptPath)) return [];
const out = [];
for (const raw of fs.readFileSync(transcriptPath, 'utf8').split('\n')) {
const line = raw.trim();
if (!line) continue;
let o;
try {
o = JSON.parse(line);
} catch {
continue;
}
if (o && typeof o.uuid === 'string') out.push(o);
}
return out;
} catch {
return [];
}
}
function tsOf(e) {
const t = e && e.timestamp ? Date.parse(e.timestamp) : 0;
return Number.isFinite(t) ? t : 0;
}
function statePath(gitDir) {
return path.join(gitDir, '.git', 'raccoon-state.json');
}
function loadState(gitDir) {
try {
const s = JSON.parse(fs.readFileSync(statePath(gitDir), 'utf8'));
return { anchor_map: {}, last_anchor: null, reconciled_for: null, ...s };
} catch {
return { anchor_map: {}, last_anchor: null, reconciled_for: null };
}
}
function saveState(gitDir, s) {
try {
// Atomic write: a tmp file + rename, so a hook killed mid-write can never
// leave a half-written (corrupt) state.json behind.
const target = statePath(gitDir);
const tmp = `${target}.tmp`;
fs.writeFileSync(tmp, JSON.stringify(s));
fs.renameSync(tmp, target);
} catch {
// best-effort
}
}
// Loose checkpoint objects must survive: a restore resets the checkpoint chain
// ref backward, which can orphan later anchors' commits. Disabling auto-gc keeps
// every anchor commit fetchable for a future rewind. The task container is
// ephemeral, so accumulating loose objects is harmless.
function disableAutoGc(gitDir) {
try {
makeGit(gitDir)('config gc.auto 0');
} catch {
// best-effort
}
}
// The conversation node the new prompt attaches to = end of the previous turn.
// Newest uuid-bearing entry, excluding the just-submitted prompt (which may or
// may not be flushed yet — excluding it makes this race-robust).
function conversationTip(entries, currentPrompt) {
const cp = (currentPrompt || '').trim();
for (let i = entries.length - 1; i >= 0; i--) {
const e = entries[i];
if (!e.uuid) continue;
const role = e.type || (e.message && e.message.role);
const content = e.message && e.message.content;
if (role === 'user' && typeof content === 'string' && cp && content.trim() === cp) continue;
return e.uuid;
}
return null;
}
// ---- capture (UserPromptSubmit) ----
function ensureExcludes(gitDir) {
const localExcludePath = path.join(gitDir, '.git', 'info', 'exclude');
const MARKER = '# raccoon-checkpoint excludes (auto-managed):';
const excludes = [
MARKER,
'.pnpm-store/',
'.yarn/cache/',
'.yarn/install-state.gz',
'vendor/bundle/',
'.bundle/cache/',
'.raccoon-setup-done', // run-app's per-repo first-use setup marker (polyglot toolkits)
];
try {
let existing = '';
try {
existing = fs.readFileSync(localExcludePath, 'utf8');
} catch {
existing = '';
}
if (!existing.includes(MARKER)) {
fs.mkdirSync(path.dirname(localExcludePath), { recursive: true });
fs.appendFileSync(localExcludePath, '\n' + excludes.join('\n') + '\n');
}
} catch {
// best-effort
}
}
function freezeForSnapshot(gitDir) {
// Freeze the state at the START of the turn being snapshotted — i.e.,
// whatever CHECKPOINT_REF already holds (or HEAD, if no turn has happened
// yet this session). This must NOT be the live working tree: the live tree
// includes the edits made during the turn that triggered /snapshot, and
// capture-snapshot's `diff HEAD <frozen>` is supposed to exclude exactly
// that turn so the trial agent gets a fresh attempt at the prompt (see the
// comment above baseline selection in capture-snapshot.mjs). Freezing the
// live tree instead bakes the agent's just-made edits into the snapshot.
try {
const git = makeGit(gitDir);
let source = null;
try {
source = git(`rev-parse ${CHECKPOINT_REF}`);
} catch {
try {
source = git('rev-parse HEAD');
} catch {
source = null;
}
}
if (source) git(`update-ref ${FROZEN_REF} ${source}`);
} catch {
// best-effort — never break /snapshot
}
}
// The snapshot invocation, in whichever form the harness uses: Claude Code takes
// `/create-snapshot:snapshot`, codex takes `$create-snapshot:snapshot`. Both send the raw
// text as `prompt` on the UserPromptSubmit hook (verified against codex 0.146.1), so the
// prefix is the only difference — and missing it means freezing never happens and the
// snapshotted turn's own edits get baked into the workspace.
const SNAPSHOT_INVOCATION_RE = /^[/$](?:create-snapshot|snapshot)(?![\w-])/;
function capture(gitDir, data) {
const prompt = (data.prompt ?? '').trim();
if (SNAPSHOT_INVOCATION_RE.test(prompt)) {
freezeForSnapshot(gitDir);
return;
}
let commit = null;
try {
const tmpIndex = path.join(gitDir, '.git', 'raccoon-checkpoint.index');
const git = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: tmpIndex });
ensureExcludes(gitDir);
disableAutoGc(gitDir);
git('read-tree HEAD');
git('add -A');
const tree = git('write-tree');
let parent;
try {
parent = git(`rev-parse ${CHECKPOINT_REF}`);
} catch {
parent = git('rev-parse HEAD');
}
commit = git(`commit-tree ${tree} -p ${parent} -m "raccoon-checkpoint: pre-turn"`);
git(`update-ref ${CHECKPOINT_REF} ${commit}`);
try {
fs.unlinkSync(tmpIndex);
} catch {
// ignore
}
} catch {
return; // never break the session
}
// Record the anchor mapping for rewind reconciliation. On the very first turn
// there is no prior node, so we anchor to the synthetic ROOT — this is the
// pre-turn-1 (initial) state, which a rewind to the first turn restores to.
try {
const tip = conversationTip(readEntries(data.transcript_path), data.prompt) || ROOT;
if (commit) {
const s = loadState(gitDir);
s.anchor_map[tip] = commit;
s.last_anchor = tip;
saveState(gitDir, s);
}
} catch {
// best-effort; capture still succeeded
}
}
// ---- reconcile (PreToolUse) ----
function subtreeContains(start, target, children) {
const stack = [start];
const seen = new Set();
while (stack.length > 0) {
const n = stack.pop();
if (n === target) return true;
if (seen.has(n)) continue;
seen.add(n);
for (const c of children.get(n) || []) stack.push(c);
}
return false;
}
function reachesLeaf(start, leafSet, children) {
const stack = [start];
const seen = new Set();
while (stack.length > 0) {
const n = stack.pop();
if (leafSet.has(n)) return true;
if (seen.has(n)) continue;
seen.add(n);
for (const c of children.get(n) || []) stack.push(c);
}
return false;
}
// node is a genuine rewind fork: >=2 children, one reaching the active leaf and
// at least one reaching a different (orphaned) leaf.
function isDivergence(node, activeLeaf, leaves, children) {
const kids = children.get(node) || [];
if (kids.length < 2) return false;
const leafSet = new Set(leaves);
const reachesActive = kids.some((k) => subtreeContains(k, activeLeaf, children));
const reachesOther = kids.some(
(k) => !subtreeContains(k, activeLeaf, children) && reachesLeaf(k, leafSet, children)
);
return reachesActive && reachesOther;
}
// Restore the working tree to a commit's tree, saving the current state to a
// safety ref first. Returns the safety ref name, or null on failure.
function restoreToCommit(gitDir, commit) {
try {
const restoreIndex = path.join(gitDir, '.git', 'raccoon-restore.index');
const stashIndex = path.join(gitDir, '.git', 'raccoon-stash.index');
const gitStash = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: stashIndex });
const gitRestore = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: restoreIndex });
const gitPlain = makeGit(gitDir, RACCOON_AUTHOR);
ensureExcludes(gitDir);
disableAutoGc(gitDir);
gitStash('read-tree HEAD');
gitStash('add -A');
const curTree = gitStash('write-tree');
let parent = null;
try {
parent = gitPlain('rev-parse HEAD');
} catch {
parent = null;
}
const curCommit = gitStash(
`commit-tree ${curTree}${parent ? ` -p ${parent}` : ''} -m "raccoon: pre-rewind safety"`
);
const safetyRef = `refs/raccoon/pre-rewind/${Date.now()}`;
gitPlain(`update-ref ${safetyRef} ${curCommit}`);
// Files to delete = present in the current tree but absent from the target
// checkpoint. Computed as a set difference of `ls-tree` listings rather than
// `diff --diff-filter=A`, because git's rename/copy detection reclassifies an
// added path as R/C, which a filter on "A" would miss — leaving the renamed-to
// file stranded in the worktree after a restore.
let added = [];
try {
const inCheckpoint = new Set(
gitPlain(`ls-tree -r --name-only ${commit}`).split('\n').filter(Boolean)
);
const inCurrent = gitPlain(`ls-tree -r --name-only ${curCommit}`).split('\n').filter(Boolean);
added = inCurrent.filter((f) => !inCheckpoint.has(f));
} catch {
added = [];
}
gitRestore(`read-tree ${commit}`);
gitRestore('checkout-index -a -f');
for (const f of added) {
try {
fs.rmSync(path.join(gitDir, f), { force: true });
} catch {
// ignore
}
}
for (const idx of [restoreIndex, stashIndex]) {
try {
fs.unlinkSync(idx);
} catch {
// ignore
}
}
return safetyRef;
} catch {
return null;
}
}
function reconcile(gitDir, data) {
let s;
try {
s = loadState(gitDir);
} catch {
return;
}
if (!s || !s.anchor_map || Object.keys(s.anchor_map).length === 0) return;
const entries = readEntries(data.transcript_path);
if (entries.length === 0) return;
const byId = new Map();
const children = new Map();
const referenced = new Set();
for (const e of entries) byId.set(e.uuid, e);
for (const e of entries) {
// Parentless or dangling-parent nodes hang off the synthetic ROOT.
const p = e.parentUuid && byId.has(e.parentUuid) ? e.parentUuid : ROOT;
if (!children.has(p)) children.set(p, []);
children.get(p).push(e.uuid);
referenced.add(p);
}
const leaves = [...byId.keys()].filter((u) => !referenced.has(u));
if (leaves.length === 0) return;
// active branch = leaf with the newest tip (the branch CC is appending to now)
let activeLeaf = leaves[0];
for (const u of leaves) if (tsOf(byId.get(u)) > tsOf(byId.get(activeLeaf))) activeLeaf = u;
// Walk the active chain (through ROOT) for the newest anchor that is a GENUINE
// rewind divergence: the anchor node has an orphaned child branch (the discarded
// turns) alongside the active branch. We skip anchors that are NOT forks, so:
// - pure forward progress (single child) never triggers a restore;
// - redoing the LAST turn still triggers (the new turn builds on the same
// boundary as the prior turn, but the discarded turn is an orphan sibling);
// - a stray anchor recorded on the active branch itself (e.g. the post-rewind
// capture's tip) can't mask the real divergence further up the chain.
// branchChild = the divergence node's child on the active path (the "branch id").
let cur = activeLeaf;
let prev = null;
let branchChild = null;
const seen = new Set();
let activeAnchor = null;
while (cur && !seen.has(cur)) {
seen.add(cur);
if (
Object.prototype.hasOwnProperty.call(s.anchor_map, cur) &&
isDivergence(cur, activeLeaf, leaves, children)
) {
activeAnchor = cur;
branchChild = prev;
break;
}
prev = cur;
if (cur === ROOT) break;
const e = byId.get(cur);
cur = e && e.parentUuid && byId.has(e.parentUuid) ? e.parentUuid : ROOT;
}
if (!activeAnchor) return;
// Idempotency keyed on (anchor, active branch) — NOT the active leaf. Every
// tool call within a post-rewind turn advances the leaf, but the branch is
// stable, so we restore exactly once per rewind. A fresh re-rewind to the same
// turn forks a NEW child off the anchor, changing the key, so it restores again.
const reconKey = `${activeAnchor}:${branchChild || ''}`;
if (s.reconciled_for === reconKey) return; // this rewind already reconciled
const safety = restoreToCommit(gitDir, s.anchor_map[activeAnchor]);
try {
makeGit(gitDir)(`update-ref ${CHECKPOINT_REF} ${s.anchor_map[activeAnchor]}`);
} catch {
// ignore
}
s.reconciled_for = reconKey;
saveState(gitDir, s);
// Record the restore to a side log ONLY — never stdout/stderr. Hook output on
// PreToolUse is captured into the transcript (as an attachment entry) and the
// transcript is the task data, so any emission here would contaminate it. The
// log lives under .git/, which is never staged, snapshotted, or transcribed.
try {
const line =
`${new Date().toISOString()} rewind reconciled: restored to anchor ${activeAnchor} ` +
`(${s.anchor_map[activeAnchor]})${safety ? `; pre-rewind state saved to ${safety}` : ''}\n`;
fs.appendFileSync(path.join(gitDir, '.git', 'raccoon-rewind.log'), line);
} catch {
// best-effort; the restore itself already succeeded
}
}
function main(data) {
const cwd = data.cwd || process.cwd();
const gitDir = findGitDir(cwd);
if (!gitDir) return;
const event = data.hook_event_name || (data.tool_name ? 'PreToolUse' : 'UserPromptSubmit');
// RECONCILE undoes a /rewind, which only Claude Code has. Other harnesses append
// and never fork, so there is nothing to reconcile and the transcript it would walk
// has no parentUuid DAG.
const canRewind = (process.env.RACCOON_HARNESS || 'claude-code') === 'claude-code';
if (event === 'PreToolUse') {
if (canRewind) reconcile(gitDir, data);
} else capture(gitDir, data);
}
let input = '';
process.stdin.setEncoding('utf8');
process.stdin.on('data', (chunk) => {
input += chunk;
});
process.stdin.on('end', () => {
let data;
try {
data = JSON.parse(input);
} catch {
process.exit(0);
}
try {
main(data);
} catch {
// A failure here must never break the user's session.
}
process.exit(0);
});

View File

@@ -0,0 +1,29 @@
// Types for harness-session.mjs, so TS consumers (its test, snapshot-to-task) see a
// real shape instead of `any`.
export interface Turn {
/** Line index in the native session file. */
index: number;
role: 'user' | 'assistant';
text: string;
/** A slash-command turn, not real conversation. */
isCommand: boolean;
/** This record concluded its turn — the truncation boundary. */
endsTurn: boolean;
}
export interface Session {
harness: string;
rawPath: string;
/** The harness own id for this conversation. */
sessionId: string | null;
lines: string[];
turns: Turn[];
}
export function supportedHarnesses(): string[];
export function readSession(harness: string, recordedPath?: string): Session | null;
export function truncationIndex(turns: Turn[]): number;
export function turnsFromLines(harness: string, lines: string[]): Turn[];
export function linearSnapshotLines(session: Session, startLine?: number): string[];
export function stripAuthoringScaffolding(harness: string, lines: string[]): string[];

View File

@@ -0,0 +1,325 @@
// Locate and read a harness's native conversation, so capture-snapshot can work
// against any harness. Everything else in capture (snapshot.patch, restore.sh,
// annotation, metadata) is harness-agnostic.
//
// The returned session stays in the harness's OWN native format: the seeding design
// hands a native blob back to the same harness, and codex_agent reads the same staged
// /tmp/snapshot-session/session.jsonl path that snapshot_agent does.
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
/**
* @typedef {object} Turn
* @property {number} index line index in the native session file
* @property {'user'|'assistant'} role
* @property {string} text
* @property {boolean} isCommand a slash-command turn, not real conversation
* @property {boolean} endsTurn this record concluded its turn
*/
/**
* @typedef {object} Session
* @property {string} harness
* @property {string} rawPath
* @property {string|null} sessionId the harness's own id for this conversation
* @property {string[]} lines
* @property {Turn[]} turns
*/
// Newest matching file beneath `root`, or null. Ties on mtime break on path so the
// answer is stable — two sessions written in the same millisecond are common.
function newestUnder(root, matches) {
if (!fs.existsSync(root)) return null;
const found = [];
const walk = (dir) => {
let entries;
try {
entries = fs.readdirSync(dir, { withFileTypes: true });
} catch {
return;
}
for (const entry of entries) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) walk(full);
else if (matches(entry.name)) found.push({ full, mtimeMs: fs.statSync(full).mtimeMs });
}
};
walk(root);
if (found.length === 0) return null;
found.sort((a, b) => b.mtimeMs - a.mtimeMs || b.full.localeCompare(a.full));
return found[0].full;
}
// User-role records codex writes that the human did not type: its own environment
// preamble, a `$name` skill invocation, and the SKILL.md body injected in response.
// Matched only at the START of the text, so a turn that merely quotes one is still real
// conversation.
function isCodexCommandText(text) {
const trimmed = (text || '').trimStart();
if (trimmed.startsWith('<skill>') || trimmed.startsWith('<environment_context>')) return true;
return /^\$[\w:.-]+\s*$/.test(trimmed);
}
const HARNESSES = {
'claude-code': {
/** Claude Code records one JSONL per session under ~/.claude/projects/<encoded-cwd>/. */
findSession() {
return newestUnder(path.join(os.homedir(), '.claude', 'projects'), (n) =>
n.endsWith('.jsonl')
);
},
/** Claude names the transcript for its session id. */
sessionId(rawPath) {
return path.basename(rawPath, '.jsonl');
},
/**
* One turn per conversational record. `endsTurn` marks an assistant record that
* concluded its turn — the truncation boundary. Bookkeeping records (attachments,
* file-history, permission-mode) carry no role and are skipped.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let entry;
try {
entry = JSON.parse(line);
} catch {
continue;
}
const role =
entry.type === 'user' ? 'user' : entry.type === 'assistant' ? 'assistant' : null;
if (!role) continue;
const content = entry.message?.content;
const text =
typeof content === 'string'
? content
: Array.isArray(content)
? content
.filter((b) => b && b.type === 'text')
.map((b) => b.text ?? '')
.join('')
: '';
turns.push({
index,
role,
text,
isCommand:
role === 'user' &&
typeof content === 'string' &&
/<command-name>|<command-message>|<local-command-caveat>/.test(content),
endsTurn: role === 'assistant' && entry.message?.stop_reason === 'end_turn',
});
}
return turns;
},
},
codex: {
/** codex writes rollout JSONL under $CODEX_HOME/sessions/<date>/. */
findSession() {
const home = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
return newestUnder(
path.join(home, 'sessions'),
(n) => n.startsWith('rollout-') && n.endsWith('.jsonl')
);
},
/** `codex resume <id>` resolves the id recorded in session_meta, not the filename. */
sessionId(rawPath, lines) {
for (const line of lines) {
try {
const rec = JSON.parse(line);
if (rec.type === 'session_meta' && rec.payload?.id) return rec.payload.id;
} catch {
continue;
}
}
return null;
},
/**
* codex rollouts carry `response_item` records whose payload is a message with a
* role. An assistant message with no following tool activity ends the turn; codex
* records no stop_reason, so a turn ends where the next user message begins —
* resolved after the fact below.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let record;
try {
record = JSON.parse(line);
} catch {
continue;
}
if (record.type !== 'response_item') continue;
const payload = record.payload ?? {};
if (payload.type !== 'message') continue;
const role =
payload.role === 'user' ? 'user' : payload.role === 'assistant' ? 'assistant' : null;
if (!role) continue;
const text = Array.isArray(payload.content)
? payload.content.map((b) => b?.text ?? '').join('')
: typeof payload.content === 'string'
? payload.content
: '';
turns.push({
index,
role,
text,
isCommand: role === 'user' && isCodexCommandText(text),
endsTurn: false,
});
}
// An assistant turn ends where the next user turn starts, or at the end.
for (let i = 0; i < turns.length; i += 1) {
if (turns[i].role !== 'assistant') continue;
const next = turns[i + 1];
turns[i].endsTurn = !next || next.role === 'user';
}
return turns;
},
},
};
/** @returns {string[]} */
export function supportedHarnesses() {
return Object.keys(HARNESSES);
}
/**
* Read the current session for `harness`. Returns null when nothing is found, so the
* caller can report which harness had no conversation to capture.
*/
/**
* @param {string} harness
* @param {string} [recordedPath] transcript recorded by the SessionStart hook; preferred
* over the newest-file scan, which can pick a different session in a busy container.
* @returns {Session | null}
*/
export function readSession(harness, recordedPath) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
const rawPath = recordedPath && fs.existsSync(recordedPath) ? recordedPath : reader.findSession();
if (!rawPath) return null;
const lines = fs.readFileSync(rawPath, 'utf8').trimEnd().split('\n');
return {
harness,
rawPath,
lines,
turns: reader.readTurns(lines),
sessionId: reader.sessionId(rawPath, lines),
};
}
/**
* Index of the last record to keep: the last turn-ending assistant record before the
* final real user turn. Drops the prompt that elicited the failure and the failure
* response, so the test agent inherits context but not the answer.
*
* Returns -1 when there is no such boundary (a one-shot conversation), which callers
* treat as "seed nothing and run cold".
*/
/**
* @param {Turn[]} turns
* @returns {number}
*/
export function truncationIndex(turns) {
let lastUser = -1;
for (const turn of turns) {
if (turn.role === 'user' && !turn.isCommand && turn.text.trim()) lastUser = turn.index;
}
if (lastUser < 0) return -1;
let cut = -1;
for (const turn of turns) {
if (turn.index >= lastUser) break;
if (turn.role === 'assistant' && turn.endsTurn) cut = turn.index;
}
return cut;
}
/**
* Parse already-read lines with a harness's reader, for callers that have the text
* rather than a path.
*
* @param {string} harness
* @param {string[]} lines
* @returns {Turn[]}
*/
export function turnsFromLines(harness, lines) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
return reader.readTurns(lines);
}
/**
* Lines to stage as the captured `session.jsonl` for a linear-transcript harness:
* everything up to the snapshot invocation, matching what Claude Code stages when it
* cuts at its slash-command line. Dropping the failure-eliciting turn happens later,
* in snapshot-to-task — capture keeps the full conversation.
*
* `startLine` is the rollout length recorded when the snapshot was invoked; without it
* the whole session is kept, which would include the snapshot's own Q&A.
*
* @param {Session} session
* @param {number} [startLine]
* @returns {string[]}
*/
export function linearSnapshotLines(session, startLine) {
if (typeof startLine === 'number' && startLine >= 0) {
return session.lines.slice(0, startLine);
}
return session.lines;
}
/**
* Drop records that describe the AUTHORING container rather than the conversation.
*
* codex records both its skill catalogue (a `developer` turn) and the machine it ran on (a
* `user` turn of `<environment_context>`). Native resume replays records byte-identically,
* so without this the test agent inherits a list of skills it does not have — one described
* as "capture the current conversation and repo state as a snapshot" — and a working
* directory that does not exist in the trial. codex re-injects both for the trial, and base
* instructions travel in `session_meta`, so removing them loses nothing. Claude's fork
* already re-records with the trial's own cwd; this brings codex to the same place.
*
* @param {string} harness
* @param {string[]} lines
* @returns {string[]}
*/
export function stripAuthoringScaffolding(harness, lines) {
if (harness === 'claude-code') return lines;
return lines.filter((raw) => {
let rec;
try {
rec = JSON.parse(raw);
} catch {
return true;
}
const payload = rec?.payload;
if (rec?.type !== 'response_item' || payload?.type !== 'message') return true;
const text = (payload.content ?? [])
.map((block) => (typeof block?.text === 'string' ? block.text : ''))
.join('')
.trim();
// Match the machine-generated shape only — a turn that STARTS with the tag — so a
// worker who quotes one of these strings mid-conversation keeps their turn.
if (payload.role === 'developer') return !text.startsWith('<skills_instructions>');
if (payload.role === 'user') return !text.startsWith('<environment_context>');
return true;
});
}

View File

@@ -0,0 +1,5 @@
/**
* Plugin-side re-export, so snapshot-to-task.ts resolves `./lib/copy-tree`
* both here and in the toolkit's flat scripts/ dir.
*/
export * from '../../../../raccoon-worker-toolkit/static/scripts/lib/copy-tree';

View File

@@ -0,0 +1,260 @@
/**
* Strip machine-identifying filesystem paths, and optional keywords, from a session
* transcript. Pure: raw JSONL in, JSONL out, no I/O.
*/
export const DEFAULT_PLACEHOLDER = '~/repo';
export const HOME_DIR_PLACEHOLDER = '~';
export const REDACTION_PLACEHOLDER = '[redacted]';
export interface SanitizeOptions {
/** Replacement for the cwd-prefix. Its dash-encoded form is derived from it. */
placeholder?: string;
/** Keyword regexes to redact. Empty by default, leaving a pure path-scrubber. */
forbiddenMarkers?: readonly RegExp[];
/**
* Exact prefix to strip. An inferred one is only the repo root when some cwd sat
* there, so callers that know the root pass it here.
*/
cwdPrefix?: string;
/** Several roots at once (a session spanning two checkouts). Wins over `cwdPrefix`. */
cwdPrefixes?: readonly string[];
/**
* Also strip home-rooted paths in the CONTENT: a sandbox-recorded session has a
* sandbox `cwd`, so the cwd passes never see the local checkout it still mentions.
*/
scrubEmbeddedHomePaths?: boolean;
}
export interface SanitizeResult {
sanitized: string;
prefixStripped: string | null;
encodedPrefixStripped: string | null;
homeDirStripped: string | null;
encodedHomeDirStripped: string | null;
embeddedPrefixStripped: string | null;
embeddedHomeDirStripped: string | null;
/** Replacement count per marker, keyed by the regex's source string. */
markersScrubbed: Record<string, number>;
}
/** Longest common prefix by path COMPONENT: `/a/bb` and `/a/b` share `/a`, not `/a/b`.
* Returns `''` when only the root `/` is common. */
export function findLongestCommonPathPrefix(paths: Iterable<string>): string {
const arr = Array.from(paths);
if (arr.length === 0) return '';
const splits = arr.map((p) => p.split('/'));
const minLen = Math.min(...splits.map((s) => s.length));
let lastShared = 0;
for (let i = 0; i < minLen; i++) {
const c = splits[0][i];
if (splits.some((s) => s[i] !== c)) break;
lastShared = i + 1;
}
// Only the leading empty piece matched → just the root, not useful.
if (lastShared <= 1) return '';
return splits[0].slice(0, lastShared).join('/');
}
/** The home-dir portion of an absolute path, or `null` for an unrecognized shape —
* better to skip the home pass than strip what may be repo content. */
export function extractHomeDir(cwdPrefix: string): string | null {
if (!cwdPrefix.startsWith('/')) return null;
// Windows-under-WSL shapes first: the generic drive shape below would stop at the
// drive letter and leave the account name in. A volume or drive root carries no
// identity by itself, so those take the directory under it.
const patterns: RegExp[] = [
/^\/mnt\/host\/[^/]+\/Users\/[^/]+/,
/^\/mnt\/[^/]+\/Users\/[^/]+/,
/^\/Users\/[^/]+/,
/^\/home\/[^/]+/,
/^\/Volumes\/[^/]+\/[^/]+/,
/^\/mnt\/[^/]+\/[^/]+/,
/^\/var\/root(?=\/|$)/,
/^\/root(?=\/|$)/,
];
for (const re of patterns) {
const m = cwdPrefix.match(re);
if (m) return m[0];
}
return null;
}
/** Every distinct `cwd` in the transcript. Read at the top level (Claude Code) and
* under `payload` (codex), so both harnesses are covered. Bad lines are skipped. */
export function collectCwds(raw: string): Set<string> {
const out = new Set<string>();
const add = (v: unknown) => {
if (typeof v === 'string' && v.startsWith('/')) out.add(v);
};
for (const line of raw.split('\n')) {
if (!line.trim()) continue;
let parsed: unknown;
try {
parsed = JSON.parse(line);
} catch {
continue;
}
if (typeof parsed !== 'object' || parsed === null) continue;
const rec = parsed as { cwd?: unknown; payload?: unknown };
add(rec.cwd);
if (typeof rec.payload === 'object' && rec.payload !== null) {
add((rec.payload as { cwd?: unknown }).cwd);
}
}
return out;
}
/** One path segment: stops at `/`, whitespace, quotes and JSON punctuation. */
const COMP = String.raw`[^/\s"'\\,:;)\]}<>]+`;
// macOS/Windows display names can contain spaces, but only consume them while
// more path follows, so a bare home-dir mention doesn't swallow trailing prose.
const USER_WITH_SPACES = `${COMP}(?:(?: +${COMP})+(?=/))?`;
const EMBEDDED_HOME_RE = new RegExp(
'(?:' +
String.raw`\/home\/${COMP}` +
'|' +
String.raw`\/Users\/${USER_WITH_SPACES}` +
'|' +
String.raw`\/mnt\/c\/Users\/${USER_WITH_SPACES}` +
'|' +
// Component boundary, so these don't match inside `/rootfs` or `/root_ca.pem`.
String.raw`\/var\/root(?![^/])` +
'|' +
String.raw`\/root(?![^/])` +
')' +
String.raw`(?:\/${COMP})*`,
'g'
);
export function collectEmbeddedHomePaths(raw: string): Set<string> {
const out = new Set<string>();
for (const m of raw.matchAll(EMBEDDED_HOME_RE)) out.add(m[0]);
return out;
}
function literalReplaceAll(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
return haystack.split(needle).join(replacement);
}
/** Can `ch` continue a path component? A `.` counts only mid-component, so `…/repo.git`
* is one component but `…/repo.` ending a sentence is not. */
function continuesComponent(text: string, at: number): boolean {
const ch = text[at];
if (ch === undefined) return false;
if (/[A-Za-z0-9_-]/.test(ch)) return true;
return ch === '.' && at + 1 < text.length && /[A-Za-z0-9_-]/.test(text[at + 1]);
}
/** Replace `needle` only where it ends at a component boundary, so stripping `…/wt/repo`
* can't turn `…/wt/repo-backup` into `<replacement>-backup`. Skipped ones go to the home pass. */
function replacePrefixAtBoundary(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
let out = '';
let from = 0;
for (;;) {
const i = haystack.indexOf(needle, from);
if (i === -1) return out + haystack.slice(from);
const end = i + needle.length;
out += haystack.slice(from, i) + (continuesComponent(haystack, end) ? needle : replacement);
from = end;
}
}
/** Replace a prefix and its dash-encoded form (`.claude/projects/<encoded>/`). */
function stripBothForms(haystack: string, needle: string, replacement: string): string {
const out = literalReplaceAll(haystack, needle, replacement);
return literalReplaceAll(out, needle.replace(/\//g, '-'), replacement.replace(/\//g, '-'));
}
export function sanitizeSessionJsonl(raw: string, opts: SanitizeOptions = {}): SanitizeResult {
const placeholder = opts.placeholder ?? DEFAULT_PLACEHOLDER;
const markers = opts.forbiddenMarkers ?? [];
const cwds = collectCwds(raw);
let working = raw;
let prefixStripped: string | null = null;
let encodedPrefixStripped: string | null = null;
let homeDirStripped: string | null = null;
let encodedHomeDirStripped: string | null = null;
let embeddedPrefixStripped: string | null = null;
let embeddedHomeDirStripped: string | null = null;
const requested = opts.cwdPrefixes?.length
? [...opts.cwdPrefixes]
: opts.cwdPrefix
? [opts.cwdPrefix]
: cwds.size > 0
? [findLongestCommonPathPrefix(cwds)]
: [];
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const prefixes = [...new Set(requested.filter(Boolean))].sort((a, b) => b.length - a.length);
// EVERY root before ANY home dir: a home pass run between roots would rewrite a
// sibling root's own prefix, leaving it unmatched when its turn came.
for (const prefix of prefixes) {
const encodedPrefix = prefix.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, prefix, placeholder);
working = literalReplaceAll(working, encodedPrefix, placeholder.replace(/\//g, '-'));
prefixStripped ??= prefix;
encodedPrefixStripped ??= encodedPrefix;
}
// Only catches what is left outside the roots, e.g. `/home/<user>/.claude/projects/`.
const homeDirs = new Set(
prefixes
.map((p) => extractHomeDir(p))
.filter((h): h is string => h !== null && !prefixes.includes(h))
);
for (const homeDir of homeDirs) {
const encodedHomeDir = homeDir.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, homeDir, HOME_DIR_PLACEHOLDER);
working = literalReplaceAll(working, encodedHomeDir, HOME_DIR_PLACEHOLDER.replace(/\//g, '-'));
homeDirStripped ??= homeDir;
encodedHomeDirStripped ??= encodedHomeDir;
}
if (opts.scrubEmbeddedHomePaths) {
const embedded = collectEmbeddedHomePaths(working);
if (embedded.size > 0) {
// Take each path's own shortest `/repo`-terminated prefix rather than a
// common prefix, which mis-collapses when paths diverge above the root.
const repoRoots = new Set<string>();
const homeDirs = new Set<string>();
for (const p of embedded) {
const h = extractHomeDir(p);
if (h) homeDirs.add(h);
const m = p.match(/^(.*?\/repo)(?:\/|$)/);
if (m) repoRoots.add(m[1]);
}
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const sortedRoots = [...repoRoots].sort((a, b) => b.length - a.length);
for (const root of sortedRoots) working = stripBothForms(working, root, placeholder);
for (const h of homeDirs) working = stripBothForms(working, h, HOME_DIR_PLACEHOLDER);
embeddedPrefixStripped = sortedRoots[0] ?? null;
embeddedHomeDirStripped = [...homeDirs][0] ?? null;
}
}
const markersScrubbed: Record<string, number> = {};
for (const re of markers) {
let count = 0;
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
const global = new RegExp(re.source, flags);
working = working.replace(global, () => {
count++;
return REDACTION_PLACEHOLDER;
});
if (count > 0) markersScrubbed[re.source] = count;
}
return {
sanitized: working,
prefixStripped,
encodedPrefixStripped,
homeDirStripped,
encodedHomeDirStripped,
embeddedPrefixStripped,
embeddedHomeDirStripped,
markersScrubbed,
};
}

View File

@@ -0,0 +1,26 @@
#!/usr/bin/env node
import fs from 'node:fs';
import path from 'node:path';
// Read stdin as a stream — hooks may not have /dev/stdin available
let input = '';
process.stdin.setEncoding('utf8');
process.stdin.on('data', (chunk) => {
input += chunk;
});
process.stdin.on('end', () => {
const { session_id, transcript_path } = JSON.parse(input);
const dataDir =
process.env.RACCOON_SNAPSHOT_DATA ||
process.env.CLAUDE_PLUGIN_DATA ||
(process.env.CLAUDE_PLUGIN_ROOT && path.join(process.env.CLAUDE_PLUGIN_ROOT, '.data')) ||
path.join(process.env.HOME || '/root', '.raccoon', 'snapshot-data');
fs.mkdirSync(dataDir, { recursive: true });
fs.writeFileSync(
path.join(dataDir, 'current-session.json'),
JSON.stringify({ session_id, transcript_path }, null, 2) + '\n'
);
});

View File

@@ -0,0 +1,837 @@
/**
* snapshot-to-task: Create a harbor task scaffold from a snapshot.
*
* Usage:
* npx tsx scripts/snapshot-to-task.ts --snapshot <dir>
*/
import { execFileSync, execSync } from 'child_process';
import {
chmodSync,
copyFileSync,
existsSync,
mkdirSync,
readFileSync,
readdirSync,
statSync,
writeFileSync,
} from 'fs';
import { basename, join, resolve } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { copyTree } from './lib/copy-tree';
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
// --- CLI ---
const argv = yargs(hideBin(process.argv))
.option('snapshot', {
type: 'string',
describe: 'Path to the snapshot directory',
demandOption: true,
})
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.strict()
.help()
.parseSync();
const log = pino(
{ name: 'snapshot-to-task', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
// --- Read snapshot data ---
const snapshotDir = argv.snapshot;
if (!existsSync(snapshotDir)) {
log.fatal(
{ path: snapshotDir },
'Snapshot directory not found. Check that the path points to a directory inside explore/snapshots/.'
);
process.exit(1);
}
interface SnapshotMetadata {
slug: string;
session_uuid: string;
/** Absent on snapshots captured before harness selection existed. */
harness?: string;
original_cwd: string;
commit: string | null;
branch: string | null;
remote_url: string | null;
timestamp: string;
plugin_version: string;
}
interface Annotation {
what_trying: string;
what_hoping: string;
what_happened: string;
[key: string]: string;
}
const metadata = JSON.parse(
readFileSync(join(snapshotDir, 'metadata.json'), 'utf8')
) as SnapshotMetadata;
const annotation = JSON.parse(
readFileSync(join(snapshotDir, 'annotation.json'), 'utf8')
) as Annotation;
if (!metadata.slug) {
log.fatal(
'No slug found in snapshot metadata.json. This snapshot may have been created by an older version of the plugin. Please file a bug.'
);
process.exit(1);
}
const slug = metadata.slug;
// --- Locate harbor infrastructure ---
function findRepoRoot(): string | null {
let dir = process.cwd();
while (dir !== resolve(dir, '..')) {
if (existsSync(join(dir, 'harbor-tasks'))) return dir;
dir = resolve(dir, '..');
}
return null;
}
const maybeRepoRoot = findRepoRoot();
if (!maybeRepoRoot) {
log.fatal(
"Could not find harbor-tasks/ directory. Make sure you're running this from the toolkit root (the Authoring container). Please file a bug if this persists."
);
process.exit(1);
}
const repoRoot: string = maybeRepoRoot;
const harborTasks = join(repoRoot, 'harbor-tasks');
const sharedCandidates = [join(harborTasks, 'raccoon-shared'), join(repoRoot, 'task-shared')];
const sharedDir = sharedCandidates.find((d) => existsSync(d));
const taskDir = join(harborTasks, slug);
if (existsSync(taskDir)) {
log.fatal(
{ path: taskDir },
`Task directory already exists. To recreate it, delete it first: rm -rf ${taskDir}`
);
process.exit(1);
}
if (!sharedDir) {
log.fatal(
'Shared infrastructure (Dockerfile, test.sh, etc.) not found. The toolkit may be corrupted. Please file a bug.'
);
process.exit(1);
}
// --- Detect repo name ---
interface ToolkitConfig {
repo: string;
defaultCommit: string;
/** The packed kit's release version (git describe at pack time). */
version?: string;
}
function readToolkitConfig(): ToolkitConfig | null {
const configPath = join(repoRoot, 'toolkit.json');
if (!existsSync(configPath)) return null;
return JSON.parse(readFileSync(configPath, 'utf8')) as ToolkitConfig;
}
function repoNameFromRemote(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const match = remoteUrl.match(/\/([^/]+?)(?:\.git)?$/);
return match ? match[1] : null;
}
function findSubmoduleDir(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const reposDir = join(repoRoot, 'repos');
if (!existsSync(reposDir)) return null;
const normalize = (url: string) =>
url
.replace(/\.git$/, '')
.replace(/^git@github\.com:/, 'https://github.com/')
.toLowerCase();
for (const entry of readdirSync(reposDir)) {
const repoPath = join(reposDir, entry, 'repo');
if (!existsSync(repoPath)) continue;
try {
const remote = execSync('git remote get-url origin', {
cwd: repoPath,
encoding: 'utf8',
stdio: ['pipe', 'pipe', 'pipe'],
}).trim();
if (normalize(remote) === normalize(remoteUrl)) return entry;
} catch {
continue;
}
}
return null;
}
const toolkitConfig = readToolkitConfig();
// A polyglot toolkit's toolkit.json has repos[] + polyglot:true (no top-level .repo).
// Derive which member this task targets from the snapshot's original_cwd basename,
// validated against the member list.
const polyglotMember = (() => {
const cfg = toolkitConfig as { polyglot?: boolean; repos?: Array<{ repo: string }> } | null;
if (!cfg?.polyglot || !Array.isArray(cfg.repos)) return null;
const base = metadata.original_cwd?.split('/').filter(Boolean).pop() ?? null;
const members = cfg.repos.map((r) => r.repo);
return base && members.includes(base) ? base : null;
})();
const repoName =
polyglotMember ??
toolkitConfig?.repo ??
findSubmoduleDir(metadata.remote_url) ??
repoNameFromRemote(metadata.remote_url);
if (!repoName) {
log.fatal(
'Could not determine repo name. The toolkit may be missing toolkit.json. Please file a bug.'
);
process.exit(1);
}
const commitShort = metadata.commit ? metadata.commit.slice(0, 9) : 'unknown';
const sessionUuid = metadata.session_uuid;
log.info({ slug, repo: repoName, commit: commitShort }, 'Creating harbor task');
// --- Create task directory structure ---
mkdirSync(join(taskDir, 'environment'), { recursive: true });
mkdirSync(join(taskDir, 'tests'), { recursive: true });
mkdirSync(join(taskDir, 'reference-runs'), { recursive: true });
// --- Copy shared infrastructure ---
// The complete grader asset set test.sh depends on: the grader system prompt
// and the renderer (test.sh exits without the renderer). Sources missing from
// task-shared/ are skipped by the existsSync guard below.
const sharedFiles = [
{ src: 'test.sh', dest: 'tests/test.sh' },
{
src: 'grader-system-prompt-consolidated.md',
dest: 'tests/grader-system-prompt-consolidated.md',
},
// test.sh execs this to render the grade; without it the verifier writes no reward
// file and the trial errors out rather than scoring.
{ src: 'render-grade-consolidated.py', dest: 'tests/render-grade-consolidated.py' },
];
for (const { src, dest } of sharedFiles) {
const srcPath = join(sharedDir, src);
const destPath = join(taskDir, dest);
if (existsSync(srcPath)) {
copyFileSync(srcPath, destPath);
if (src === 'test.sh') chmodSync(destPath, 0o755);
log.debug({ src, dest }, 'Copied shared file');
} else {
log.warn({ src }, 'Shared file not found');
}
}
// Deterministic checks (tests/typecheck/lint). test.sh sources these and hands
// their output to the grader as evidence for the CORRECTNESS score, so without
// them a code task's correctness is never signal-backed — the grader falls back
// to reading the diff alone. Same per-member-then-generic resolution as the
// Dockerfile below: a polyglot toolkit ships test-commands.<member>.sh per
// member, a single-repo toolkit ships the lone test-commands.sh.
const perMemberTestCommands = join(sharedDir, `test-commands.${repoName.toLowerCase()}.sh`);
const genericTestCommands = join(sharedDir, 'test-commands.sh');
const testCommandsSrc = existsSync(perMemberTestCommands)
? perMemberTestCommands
: genericTestCommands;
if (existsSync(testCommandsSrc)) {
const testCommandsDest = join(taskDir, 'tests', 'test-commands.sh');
copyFileSync(testCommandsSrc, testCommandsDest);
chmodSync(testCommandsDest, 0o755);
log.debug({ src: testCommandsSrc }, 'Copied deterministic checks');
} else {
// Not fatal: the grader still scores correctness by walking the changed code.
log.info(
'No test-commands.sh for this repo — expected when it has no runnable suite. The grader scores correctness by reading the changed code instead; say so in your holistic rubric.'
);
}
// --- Write Dockerfile with session resume support ---
//
// Read the per-repo task-shared/Dockerfile (Ruby/Postgres/Node for ZenBill,
// TS-Node/Postgres/pnpm for Palolo) from the toolkit and append session-
// staging COPY/RUN steps. Session staging happens after the original CMD —
// COPY and RUN are layer ops independent of CMD, so the original
// `CMD ["sleep", "infinity"]` remains active after the appended layers.
//
// Falls back to a bare debian Dockerfile if no task-shared/Dockerfile is
// present (toolkit corruption, or a repo without a per-repo Dockerfile).
// Polyglot toolkits ship a per-member task-shared/Dockerfile.<member>; a graded task
// targets one member, so prefer its Dockerfile. Single-repo toolkits use the lone
// task-shared/Dockerfile. Fall back to the generic one if the per-member file is absent.
const perMemberDockerfile = join(repoRoot, 'task-shared', `Dockerfile.${repoName.toLowerCase()}`);
const taskSharedDockerfile = existsSync(perMemberDockerfile)
? perMemberDockerfile
: join(repoRoot, 'task-shared', 'Dockerfile');
let baseDockerfile: string;
if (existsSync(taskSharedDockerfile)) {
baseDockerfile = readFileSync(taskSharedDockerfile, 'utf-8');
log.debug({ dockerfile: taskSharedDockerfile }, 'Loaded base Dockerfile');
} else {
log.warn(
'task-shared/Dockerfile not found; falling back to bare debian. The harbor task container will lack any language runtime — agents will not be able to execute code in the repo.'
);
baseDockerfile = `FROM debian:bookworm-slim
RUN apt-get update && apt-get install -y \\
git \\
python3 \\
curl \\
jq \\
&& rm -rf /var/lib/apt/lists/*
# Install Claude Code globally (needed by the grader in test.sh)
RUN curl -fsSL https://claude.ai/install.sh | bash && \\
cp /root/.claude-code/claude /usr/local/bin/claude 2>/dev/null || \\
cp /root/.local/bin/claude /usr/local/bin/claude 2>/dev/null || \\
ln -sf $(find /root -name claude -type f 2>/dev/null | head -1) /usr/local/bin/claude
WORKDIR /workspace
COPY workspace/ .
# Block network tools — agent should only read code and write documents
RUN mkdir -p .claude && \\
echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > .claude/settings.json
RUN git init && \\
git config user.email "dev@agent" && \\
git config user.name "Dev" && \\
git add -A && \\
git commit -m "initial" --quiet
CMD ["sleep", "infinity"]
`;
}
// Wrapped in toolkit-managed sentinels so check-task-infra reads this as the
// toolkit's own append rather than an edit to the Dockerfile.
// Only Claude Code produces the sibling session/ directory (subagents, tool results).
// A COPY of an empty directory fails the build outright — buildkit does not carry empty
// directories in the context, so the layer errors with `"/session": not found`.
// Read the SNAPSHOT, not the task dir: the Dockerfile is generated before the session
// files are copied into environment/, so the task-side copy is not there yet.
const sessionSiblingDir = join(snapshotDir, 'session');
const hasSessionSibling =
existsSync(sessionSiblingDir) && readdirSync(sessionSiblingDir).length > 0;
const sessionStaging = `
# >>> toolkit-managed: snapshot-session >>>
# Stage session files for the snapshot agent adapter to install at runtime.
COPY session.jsonl /tmp/snapshot-session/session.jsonl
${hasSessionSibling ? 'COPY session/ /tmp/snapshot-session/session/\n' : ''}RUN echo '${sessionUuid}' > /tmp/snapshot-session/uuid.txt
# <<< toolkit-managed <<<
`;
const dockerfile = baseDockerfile.trimEnd() + '\n' + sessionStaging;
writeFileSync(join(taskDir, 'environment', 'Dockerfile'), dockerfile);
log.debug('Wrote Dockerfile (per-repo base + session staging)');
// --- Copy snapshot.patch as workspace.patch ---
const snapshotPatch = join(snapshotDir, 'snapshot.patch');
if (existsSync(snapshotPatch)) {
copyFileSync(snapshotPatch, join(taskDir, 'environment', 'workspace.patch'));
log.debug('Copied snapshot.patch -> workspace.patch');
}
// --- Scrub the worker's filesystem layout out of the session ---
// In Explore the recorded `cwd` is the worker's HOST checkout (explore/repo is an absolute
// symlink); rewriting the repo root to /workspace both drops the leak and matches the trial.
const WORKSPACE_MOUNT = '/workspace';
const escapeRegExp = (v: string) => v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
/** Member names when this toolkit is polyglot; empty means single-repo. */
const MEMBER_NAMES: readonly string[] = (() => {
const dir = join(repoRoot, 'repos');
if (!existsSync(dir)) return [];
try {
return readdirSync(dir, { withFileTypes: true })
.filter((e) => e.isDirectory())
.map((e) => e.name);
} catch {
return [];
}
})();
/** The repo root within a cwd — the prefix a trial mounts at /workspace. `/repos/<member>`
* anchors only on a polyglot toolkit, so a personal `~/repos/…` above it can't win. */
function repoRootOf(cwd: string): string | null {
if (MEMBER_NAMES.length > 0) {
// A real member of THIS toolkit wins; the generic shape covers a member whose
// directory the toolkit no longer has (an older snapshot, a renamed member).
for (const name of MEMBER_NAMES) {
const hit = cwd.match(new RegExp(`^(.*?/repos/${escapeRegExp(name)})(?:/|$)`));
if (hit) return hit[1];
}
const generic = cwd.match(/^(.*?\/repos\/[^/]+)(?:\/|$)/);
if (generic) return generic[1];
}
// `/repo` needs a component boundary, so it never matches inside `/repos/`.
const m = cwd.match(/^(.*?\/repo)(?:\/|$)/);
return m ? m[1] : null;
}
/** Rewrite every checkout root to /workspace, and the home dir each sits under to `~`. The
* `repo/` anchor needs no host-root list; the home pass still keys off extractHomeDir. */
function scrubWorkerPaths(raw: string): { text: string; roots: string[] } {
// Each cwd contributes its own root, longest first, so a nested root isn't clobbered
// and a session spanning two checkouts is scrubbed rather than skipped.
const roots = [...new Set([...collectCwds(raw)].map(repoRootOf))]
.filter((r): r is string => r !== null)
.sort((a, b) => b.length - a.length);
const { sanitized } = sanitizeSessionJsonl(raw, {
cwdPrefixes: roots,
placeholder: WORKSPACE_MOUNT,
});
return { text: sanitized, roots };
}
// --- Copy session files for --resume ---
//
// The full session.jsonl (including any post-end_turn entries) goes into the
// task root for reference. A truncated version — keeping everything up to
// and including the last assistant entry with stop_reason="end_turn" — goes
// into environment/ for the container. Stopping on a clean assistant turn
// avoids Claude Code's synthetic "No response requested." injection when
// the session is resumed with --fork-session and a new --print prompt.
const sessionJsonl = join(snapshotDir, 'session.jsonl');
if (existsSync(sessionJsonl)) {
// Fail-open: a session this can't scrub ships exactly as it was, because a
// leaked path is a smaller problem than a task that can't be created.
let sessionText = readFileSync(sessionJsonl, 'utf8');
try {
const { text, roots } = scrubWorkerPaths(sessionText);
if (roots.length > 0) {
sessionText = text;
log.info(
{ roots, mountedAt: WORKSPACE_MOUNT },
'Rewrote the authoring checkout path to the trial mount point'
);
} else {
log.debug('No worker-rooted cwd to rewrite; session used as-is');
}
} catch (err) {
log.warn(
{ err: err instanceof Error ? err.message : String(err) },
'Could not rewrite paths in the session; using it as-is'
);
}
// Full version for reference
writeFileSync(join(taskDir, 'session-full.jsonl'), sessionText);
log.debug('Wrote full session.jsonl to task root');
// Truncated version for the container: strip everything from the last
// user text turn onwards. This drops the failure-eliciting question
// (which `--print` will redeliver to the trial agent as the new prompt)
// AND the failure response itself (so the trial agent doesn't see its
// previous answer), while preserving conversational context up to the
// last clean assistant `end_turn`.
//
// Algorithm (refined Option B):
// 1. Find U = index of the last user-text turn that is NOT a slash
// command (use the same command-marker filter as
// extractLastUserMessage).
// 2. Walk backwards from U - 1 to find the last `assistant` entry
// with stop_reason: "end_turn".
// 3. Truncate slice(0, lastEndTurnIndex + 1).
//
// If U doesn't exist or no end_turn assistant precedes U, write an
// empty session.jsonl — the snapshot agent adapter detects this and
// skips --resume entirely, starting fresh from --print.
const sessionLines = sessionText.trimEnd().split('\n');
// A non-Claude session is not a Claude transcript, so the scan below finds no
// `stop_reason: "end_turn"` and would silently write an empty session. Its reader
// applies the same rule in that harness's own format.
const harness = metadata.harness ?? 'claude-code';
const isClaude = harness === 'claude-code';
let lastUserTextIndex = -1;
for (let i = 0; i < sessionLines.length; i++) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type !== 'user' || typeof entry.message?.content !== 'string') continue;
// Compaction summaries are synthetic user turns whose text often quotes
// earlier /create-snapshot:snapshot runs — never the command turn itself,
// so they must not trip the break below.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
// Mirror extractLastUserMessage: skip the snapshot command itself
// and any slash-command / local-command marker turns.
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserTextIndex = i;
} catch {
continue;
}
}
let lastEndTurnIndex = -1;
if (lastUserTextIndex > 0) {
for (let i = lastUserTextIndex - 1; i >= 0; i--) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
message?: { stop_reason?: unknown };
};
if (entry.type === 'assistant' && entry.message?.stop_reason === 'end_turn') {
lastEndTurnIndex = i;
break;
}
} catch {
continue;
}
}
}
if (!isClaude) {
const cut = truncationIndex(turnsFromLines(harness, sessionLines));
const kept = cut >= 0 ? sessionLines.slice(0, cut + 1) : [];
const truncated = stripAuthoringScaffolding(harness, kept);
writeFileSync(
join(taskDir, 'environment', 'session.jsonl'),
truncated.length ? truncated.join('\n') + '\n' : ''
);
log.debug(
{ harness, fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (harness reader)'
);
} else if (lastEndTurnIndex >= 0) {
const truncated = sessionLines.slice(0, lastEndTurnIndex + 1);
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), truncated.join('\n') + '\n');
log.debug(
{ fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (strips last user turn + failure response, keeps through last clean assistant end_turn)'
);
} else {
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), '');
if (lastUserTextIndex < 0) {
log.warn(
'No user text turn found in session — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
} else {
log.warn(
'No assistant entry with stop_reason="end_turn" found before the last user turn (one-shot snapshot) — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
}
}
}
const sessionDir = join(snapshotDir, 'session');
if (existsSync(sessionDir) && statSync(sessionDir).isDirectory()) {
copyTree(sessionDir, join(taskDir, 'environment', 'session'));
// Claude Code writes subagent files write-only (--w-------). Fix them so
// Harbor's dirhash can read them during environment setup.
execSync(`chmod -R +r "${join(taskDir, 'environment', 'session')}"`, { stdio: 'pipe' });
log.debug('Copied session/');
} else {
mkdirSync(join(taskDir, 'environment', 'session'), { recursive: true });
}
// The harness that captured the snapshot; the trial runs this one.
const harness =
typeof metadata.harness === 'string' && metadata.harness ? metadata.harness : 'claude-code';
/**
* The model and effort this harness defaulted to when the task was authored, recorded
* for reference only — nothing reads these back, and a trial still resolves both from
* the registry at run time. Best-effort: a task is not worth failing over a note.
*/
function authoredDefaults(harnessId: string): { model: string; effort: string } | null {
try {
const resolver = join(repoRoot, 'scripts', 'resolve_harness.py');
// Same interpreter search as `_raccoon_python` in scripts/lib/harness-credentials.sh
// and `pythonWithTomllib` in submit-task.ts: `python3` is not always 3.11+, and the
// registry needs tomllib. Best-effort, so a miss just omits the note.
let python = '';
for (const candidate of [
process.env.RACCOON_PYTHON,
'python3',
'python3.13',
'python3.12',
'python3.11',
]) {
if (!candidate) continue;
try {
execFileSync(candidate, ['-c', 'import tomllib'], { stdio: 'ignore' });
python = candidate;
break;
} catch {
continue;
}
}
if (!python) return null;
const rows = execFileSync(python, [resolver, '--defaults'], {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
});
for (const line of rows.split('\n')) {
const [id, model, effort] = line.split('\t');
if (id === harnessId && model) return { model, effort: effort ?? '' };
}
} catch {
// registry unreadable here — omit the note
}
return null;
}
const authored = authoredDefaults(harness);
// --- Write task.toml ---
// The reference-data corpus is included in every zeta task (build-workspace decides from the repo),
// so there's nothing to set here.
const taskToml = `version = "1.0"
[metadata]
program = "raccoon"
author = "rl-env-coding"
category = "sdlc/technical-writing"
repo = "${repoName}"
commit = "${commitShort}"
# The toolkit release this task was created with. Written by the toolkit —
# leave it in place: task tooling reads it to know which toolkit's assets
# this task grades with.
toolkit_version = "${toolkitConfig?.version ?? 'unknown'}"
snapshot = "${basename(snapshotDir)}"
session_uuid = "${sessionUuid}"
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
browser = false
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
[verifier]
timeout_sec = 7200.0
[agent]
harness = "${harness}"
timeout_sec = 18000.0
[environment]
build_timeout_sec = 6000.0
cpus = 2
memory_mb = 4096
storage_mb = 10240
gpus = 0
allow_internet = true
[verifier.env]
ANTHROPIC_API_KEY = "\${ANTHROPIC_API_KEY}"
ANTHROPIC_BASE_URL = "\${ANTHROPIC_BASE_URL}"
[solution.env]
`;
writeFileSync(join(taskDir, 'task.toml'), taskToml);
log.debug('Wrote task.toml');
// --- Extract instruction from session transcript ---
function extractLastUserMessage(sessionPath: string, harness: string): string | null {
if (!existsSync(sessionPath)) return null;
const lines = readFileSync(sessionPath, 'utf8').trimEnd().split('\n');
// A non-Claude session has no `type: "user"` records, so the scan below finds nothing
// and the worker silently gets a placeholder instruction. Its reader applies the same
// rule — last real user turn, ignoring command invocations — in that harness's format.
if (harness !== 'claude-code') {
const userTurns = turnsFromLines(harness, lines).filter(
(t) => t.role === 'user' && !t.isCommand && t.text.trim()
);
return userTurns.length ? userTurns[userTurns.length - 1].text : null;
}
let lastUserMessage: string | null = null;
for (const line of lines) {
try {
const entry = JSON.parse(line) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type === 'user' && typeof entry.message?.content === 'string') {
// Synthetic compaction summary — not a real user turn, and its text
// often quotes earlier /create-snapshot:snapshot runs.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserMessage = content;
}
} catch {
continue;
}
}
return lastUserMessage;
}
const lastUserMessage = extractLastUserMessage(
join(snapshotDir, 'session.jsonl'),
metadata.harness ?? 'claude-code'
);
const instructionHeader =
'# Replace this with your refined task instruction\n\n' +
"<!-- The text below was auto-extracted from your snapshot's last user message.\n" +
' Refine, condense, or rewrite to focus on the behavior you want to elicit. -->\n\n';
if (lastUserMessage) {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader + lastUserMessage.trimEnd() + '\n'
);
log.info('Wrote instruction.md (from last user message in session)');
} else {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader +
'<!-- Could not extract user message from session. Write the instruction manually. -->\n'
);
log.warn('Could not extract instruction from session — needs manual editing');
}
// --- Scaffold holistic-rubric.md ---
const holisticRubricMd = `<!--
HOLISTIC RUBRIC — the file trials grade against. Run
/write-holistic-rubric
to draft it interactively, or point Claude Code at this file,
session-full.jsonl, and task-shared/grading-standard.md.
Snapshot: ${basename(snapshotDir)}
Session: ${metadata.session_uuid}
Repo: ${metadata.remote_url}
Commit: ${metadata.commit}
## What happened in the snapshot conversation
The worker was trying to: ${annotation.what_trying}
They hoped Claude would: ${annotation.what_hoping}
Instead, Claude: ${annotation.what_happened}
## What this file contains
The eight-criterion Grading Standard
(task-shared/grading-standard.md, embedded in
tests/grader-system-prompt-consolidated.md) defines Integrity, Narrow
Correctness, Broader Correctness / craft, Persistence, Communication,
Verification & Thoroughness, Common Sense, and Thought Partnership. This
file adds the task-specific knowledge the grader cannot infer: full task
context, the ground truth you established, what strong and weak responses
look like per criterion, and any dealbreaker penalties — stated as 0.0-1.0
fraction subtractions with a named criterion target, never points, never
caps. The document must stand alone: the grader sees only it and the
shared standard.
-->
<!-- Replace EVERYTHING in this file with the actual holistic rubric,
including the instructions above. -->
`;
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');
// --- Build workspace ---
const buildScript = join(repoRoot, 'scripts', 'build-workspace.sh');
if (existsSync(buildScript)) {
log.info({ repo: repoName, commit: commitShort }, 'Building workspace');
try {
execSync(`bash "${buildScript}" "${slug}" "${commitShort}"`, {
cwd: repoRoot,
encoding: 'utf8',
stdio: 'inherit',
// build-workspace does a bulk-file write burst (git archive|tar of the
// repo tree + a throwaway git add/commit to apply the patch, and for zeta
// toolkits a hardlink-stage of the ~126k-file reference-data corpus that
// falls back to a full copy across filesystems). On a slow bind mount
// (Docker Desktop non-VirtioFS, or WSL2 with the toolkit on a Windows/9p
// path) that legitimately runs into minutes, so a tight cap false-fails a
// working-but-slow build as "not runnable". Keep this generous — it's only
// a backstop against a true hang; the real Harbor build downstream budgets
// build_timeout_sec = 6000.
timeout: 1_200_000,
});
} catch (e: unknown) {
const msg = e instanceof Error ? e.message : String(e);
log.fatal({ error: msg }, 'Workspace build failed — task is not runnable');
log.fatal(` Retry manually: bash scripts/build-workspace.sh ${slug}`);
log.fatal(` Then: scripts/harbor-run harbor-tasks/${slug}`);
process.exit(1);
}
} else {
log.fatal('scripts/build-workspace.sh not found. Please file a bug.');
process.exit(1);
}
try {
execSync('bash -ic "_ev task_created 2>/dev/null" 2>/dev/null', {
stdio: 'ignore',
timeout: 5000,
});
} catch {
// best-effort
}
// --- Done ---
log.info({ taskDir: resolve(taskDir) }, 'Task scaffolded');
log.info('Next steps:');
log.info(' 1. Review instruction.md');
log.info(' 2. Edit tests/holistic-rubric.md — write the rubric');
log.info(' 3. Run calibration trials to validate scoring tiers');