ren worker folder adding orig, mv new one into root
This commit is contained in:
@@ -0,0 +1,788 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
import { execSync } from 'node:child_process';
|
||||
import crypto from 'node:crypto';
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
|
||||
import { linearSnapshotLines, readSession } from './harness-session.mjs';
|
||||
|
||||
// --- Argument parsing ---
|
||||
|
||||
function parseArgs(argv) {
|
||||
const args = {};
|
||||
for (let i = 2; i < argv.length; i++) {
|
||||
if (argv[i].startsWith('--')) {
|
||||
const key = argv[i].slice(2);
|
||||
const val = argv[i + 1];
|
||||
if (!val || val.startsWith('--')) {
|
||||
args[key] = true;
|
||||
} else {
|
||||
args[key] = val;
|
||||
i++;
|
||||
}
|
||||
}
|
||||
}
|
||||
return args;
|
||||
}
|
||||
|
||||
// Last-resort data dir. Claude Code's plugin runtime always provides one, so this is
|
||||
// what makes the start marker and session record work under any other harness.
|
||||
function defaultDataDir() {
|
||||
const home = process.env.HOME || '/root';
|
||||
return path.join(home, '.raccoon', 'snapshot-data');
|
||||
}
|
||||
|
||||
const args = parseArgs(process.argv);
|
||||
const slug = args.slug;
|
||||
const annotationPath = args.annotation;
|
||||
const outputDir = args['output-dir'];
|
||||
|
||||
if (!args['mark-start'] && (!slug || !annotationPath || !outputDir)) {
|
||||
console.error(
|
||||
'Usage: capture-snapshot.mjs --slug <slug> --annotation <path> --output-dir <dir> [--plugin-data <path>]\n' +
|
||||
' capture-snapshot.mjs --mark-start [--harness <id>]'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// --- Locate session info ---
|
||||
|
||||
// --harness wins over the launcher's RACCOON_HARNESS so a caller that knows which
|
||||
// conversation it is capturing can say so; the default keeps Claude Code's plugin
|
||||
// working unchanged. Claude Code has an exact cut point (the snapshot slash command)
|
||||
// and a message tree to prune, so it keeps the bespoke path below; other harnesses go
|
||||
// through the shared reader.
|
||||
const HARNESS = args.harness || process.env.RACCOON_HARNESS || 'claude-code';
|
||||
const IS_CLAUDE = HARNESS === 'claude-code';
|
||||
// The generated restore.sh writes one of exactly two session layouts, and everything
|
||||
// below branches on IS_CLAUDE — so a third harness would silently be handed codex's
|
||||
// $CODEX_HOME/sessions paths. Refuse instead; adding a harness means adding a layout.
|
||||
if (!IS_CLAUDE && HARNESS !== 'codex') {
|
||||
console.error(
|
||||
`capture-snapshot: no session-restore layout for harness "${HARNESS}". ` +
|
||||
'Add one to capture-snapshot.mjs (and harness-session.mjs) before capturing with it.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// Try multiple strategies to find the current session transcript:
|
||||
// 1. Plugin data dir (from SessionStart hook)
|
||||
// 2. Scan ~/.claude/projects/ for the most recently modified JSONL
|
||||
|
||||
let session_id = null;
|
||||
let transcript_path = null;
|
||||
|
||||
const dataDir =
|
||||
args['plugin-data'] ||
|
||||
process.env.RACCOON_SNAPSHOT_DATA ||
|
||||
process.env.CLAUDE_PLUGIN_DATA ||
|
||||
(process.env.CLAUDE_PLUGIN_ROOT && path.join(process.env.CLAUDE_PLUGIN_ROOT, '.data')) ||
|
||||
defaultDataDir();
|
||||
|
||||
// The SessionStart hook records the live session for every harness, so prefer it over
|
||||
// guessing. `readSession` falls back to the newest file on disk when it is absent.
|
||||
let recordedSession = null;
|
||||
if (dataDir) {
|
||||
const sessionInfoPath = path.join(dataDir, 'current-session.json');
|
||||
if (fs.existsSync(sessionInfoPath)) {
|
||||
try {
|
||||
recordedSession = JSON.parse(fs.readFileSync(sessionInfoPath, 'utf8'));
|
||||
} catch {
|
||||
recordedSession = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const startMarkerPath = dataDir ? path.join(dataDir, 'snapshot-start.json') : null;
|
||||
|
||||
// `--mark-start` runs BEFORE the annotation Q&A and records how long the conversation
|
||||
// was at that moment. It is the linear-harness stand-in for Claude Code's slash-command
|
||||
// line: without it, capture would stage the snapshot's own Q&A as conversation.
|
||||
if (args['mark-start']) {
|
||||
const session = readSession(HARNESS);
|
||||
if (!session) {
|
||||
console.error(`No ${HARNESS} session found to mark.`);
|
||||
process.exit(1);
|
||||
}
|
||||
if (!startMarkerPath) {
|
||||
console.error('No data dir available to record the snapshot start marker.');
|
||||
process.exit(1);
|
||||
}
|
||||
fs.mkdirSync(path.dirname(startMarkerPath), { recursive: true });
|
||||
fs.writeFileSync(
|
||||
startMarkerPath,
|
||||
JSON.stringify(
|
||||
{ harness: HARNESS, transcript_path: session.rawPath, line_count: session.lines.length },
|
||||
null,
|
||||
2
|
||||
) + '\n'
|
||||
);
|
||||
console.log(`Snapshot start marked at ${session.lines.length} records.`);
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
let harnessSession = null;
|
||||
if (!IS_CLAUDE) {
|
||||
harnessSession = readSession(HARNESS, recordedSession?.transcript_path);
|
||||
if (!harnessSession) {
|
||||
console.error(
|
||||
`Could not find a ${HARNESS} session to capture. Capture has to run from inside the ${HARNESS} conversation you want to snapshot.`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
transcript_path = harnessSession.rawPath;
|
||||
// Fall back to a fresh id only if the harness records none — restore.sh names the
|
||||
// installed session by it, so it has to match what `resume` will look up.
|
||||
session_id = recordedSession?.session_id || harnessSession.sessionId || crypto.randomUUID();
|
||||
} else if (recordedSession) {
|
||||
session_id = recordedSession.session_id;
|
||||
transcript_path = recordedSession.transcript_path;
|
||||
}
|
||||
|
||||
// Fallback: find the most recently modified JSONL in ~/.claude/projects/
|
||||
if (!transcript_path) {
|
||||
const homeDir = process.env.HOME || '/root';
|
||||
const projectsDir = path.join(homeDir, '.claude', 'projects');
|
||||
if (fs.existsSync(projectsDir)) {
|
||||
let newest = null;
|
||||
let newestMtime = 0;
|
||||
for (const projEntry of fs.readdirSync(projectsDir)) {
|
||||
const projDir = path.join(projectsDir, projEntry);
|
||||
if (!fs.statSync(projDir).isDirectory()) continue;
|
||||
for (const file of fs.readdirSync(projDir)) {
|
||||
if (!file.endsWith('.jsonl')) continue;
|
||||
const filePath = path.join(projDir, file);
|
||||
const mtime = fs.statSync(filePath).mtimeMs;
|
||||
if (mtime > newestMtime) {
|
||||
newestMtime = mtime;
|
||||
newest = filePath;
|
||||
session_id = file.replace(/\.jsonl$/, '');
|
||||
}
|
||||
}
|
||||
}
|
||||
transcript_path = newest;
|
||||
}
|
||||
}
|
||||
|
||||
if (!transcript_path || !fs.existsSync(transcript_path)) {
|
||||
console.error(
|
||||
"Could not find a Claude Code session transcript. This script should be run from within a Claude Code conversation via the /create-snapshot:snapshot command. Please file a bug if you're seeing this unexpectedly."
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (!transcript_path || !fs.existsSync(transcript_path)) {
|
||||
console.error(`Transcript file not found at ${transcript_path}. Please file a bug.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// --- Require a git repo at capture time ---
|
||||
//
|
||||
// A snapshot is "commit SHA + diff vs HEAD", reconstituted later via
|
||||
// `git archive <SHA> | tar -x` + `git apply workspace.patch`. Without a git
|
||||
// repo here we have no SHA to pin, no patch to record, and no way for
|
||||
// downstream `build-workspace.sh` to reproduce the workspace — the resulting
|
||||
// snapshot would be structurally meaningless. This check runs BEFORE the
|
||||
// snapshot directory is created so a misconfigured invocation leaves no
|
||||
// half-written state behind.
|
||||
|
||||
// Find the git repo by asking git itself — walks up from cwd looking for
|
||||
// `.git`, handling submodules and worktrees correctly. Returns null when
|
||||
// cwd is outside any repo, so the worker gets a clear "cd into your repo"
|
||||
// error instead of silently descending into something they didn't name.
|
||||
function findGitRepo() {
|
||||
try {
|
||||
const top = execSync('git rev-parse --show-toplevel', {
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
})
|
||||
.toString()
|
||||
.trim();
|
||||
return top || null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
const gitRepo = findGitRepo();
|
||||
|
||||
if (!gitRepo) {
|
||||
console.error(
|
||||
"Error: Not running inside a git repo. /create-snapshot needs a git repo so it can pin a commit SHA and record a diff of in-flight changes; without one the snapshot can't be reproduced as a task. cd into the repo you're exploring (the toolkit's repo/ submodule) and re-run /create-snapshot:snapshot."
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// --- Create snapshot directory ---
|
||||
|
||||
const ts = new Date().toISOString().replace(/[-:]/g, '').replace('T', '-').slice(0, 15); // 20260403-225449
|
||||
const snapshotDir = path.join(outputDir, `${ts}-${slug}`);
|
||||
|
||||
if (fs.existsSync(snapshotDir)) {
|
||||
console.error(`Snapshot directory already exists: ${snapshotDir}\nPlease file a bug.`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
fs.mkdirSync(snapshotDir, { recursive: true });
|
||||
|
||||
// --- Copy conversation transcript (trimmed + branch-pruned) ---
|
||||
|
||||
// The JSONL is a tree of messages linked by parentUuid. When the user rewinds
|
||||
// a conversation, old branches remain in the file. We need to:
|
||||
// 1. Cut at the LAST /create-snapshot:snapshot command (later invocations
|
||||
// supersede earlier ones in the same session)
|
||||
// 2. Find the tip of the active branch (last message before the cut)
|
||||
// 3. Walk parentUuid back to the root, collecting only messages on that path
|
||||
// 4. Exclude the Q&A subgraphs of any PRIOR /create-snapshot:snapshot
|
||||
// invocations in this session (their cut points are on the same
|
||||
// conversation branch, so the walk would otherwise pull in the
|
||||
// assistant's annotation questions and the user's answers — a
|
||||
// contamination path that snapshot.patch doesn't show). Boundaries
|
||||
// for prior invocations are recorded in a side file (see end of
|
||||
// this script) so this run can identify them.
|
||||
// 5. Drop bookkeeping entries whose content can leak rewound-branch state.
|
||||
|
||||
const rawLines = fs.readFileSync(transcript_path, 'utf8').trimEnd().split('\n');
|
||||
|
||||
// A user message is a /create-snapshot:snapshot invocation when its content
|
||||
// STARTS with one of Claude Code's slash-command tags AND mentions the
|
||||
// command name. The "starts with" guard distinguishes a real invocation
|
||||
// from prose that quotes the command (a worker reporting a bug, the
|
||||
// command-listing skill output, etc.) — prose doesn't begin with those
|
||||
// tags. The whitespace-tolerant pattern survives minor format drift in
|
||||
// Claude Code's slash-command rendering.
|
||||
const SNAPSHOT_CMD_PATTERN =
|
||||
/<command-(?:name|message)>\s*\/?\s*create-snapshot:snapshot\s*<\/command-(?:name|message)>/;
|
||||
function isSnapshotCommandContent(content) {
|
||||
if (typeof content !== 'string') return false;
|
||||
const trimmed = content.trimStart();
|
||||
if (!trimmed.startsWith('<command-name>') && !trimmed.startsWith('<command-message>')) {
|
||||
return false;
|
||||
}
|
||||
return SNAPSHOT_CMD_PATTERN.test(content);
|
||||
}
|
||||
|
||||
// Find every snapshot-command line index, in order. The LAST one is the
|
||||
// current invocation (cut point); earlier ones bound prior Q&A subgraphs.
|
||||
const snapshotCmdIndexes = [];
|
||||
for (let i = 0; i < rawLines.length; i++) {
|
||||
try {
|
||||
const entry = JSON.parse(rawLines[i]);
|
||||
if (entry.type === 'user' && isSnapshotCommandContent(entry.message?.content)) {
|
||||
snapshotCmdIndexes.push(i);
|
||||
}
|
||||
} catch {
|
||||
// Skip malformed lines
|
||||
}
|
||||
}
|
||||
|
||||
const cutIndex =
|
||||
snapshotCmdIndexes.length > 0
|
||||
? snapshotCmdIndexes[snapshotCmdIndexes.length - 1]
|
||||
: rawLines.length;
|
||||
const priorCmdIndexes = snapshotCmdIndexes.slice(0, -1);
|
||||
|
||||
// Load prior-snapshot boundary records so we know where each earlier
|
||||
// invocation's Q&A subgraph ended. The boundary file is written at the
|
||||
// end of every capture run (see below) and is keyed by session uuid.
|
||||
function loadPriorBoundaries() {
|
||||
if (!dataDir || !session_id) return [];
|
||||
const boundariesPath = path.join(dataDir, 'snapshot-boundaries.jsonl');
|
||||
if (!fs.existsSync(boundariesPath)) return [];
|
||||
const lines = fs.readFileSync(boundariesPath, 'utf8').trimEnd().split('\n');
|
||||
const out = [];
|
||||
for (const line of lines) {
|
||||
if (!line) continue;
|
||||
try {
|
||||
const rec = JSON.parse(line);
|
||||
if (rec.sessionUuid === session_id && rec.snapshotCommandUuid) out.push(rec);
|
||||
} catch {
|
||||
/* skip malformed */
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
const priorBoundaries = loadPriorBoundaries();
|
||||
|
||||
// Compute the line ranges to exclude for each prior snapshot. The Q&A
|
||||
// subgraph starts at the prior snapshot's command line and runs through
|
||||
// the line whose entry uuid matches the boundary record (the last entry
|
||||
// in the JSONL when that prior capture-snapshot completed).
|
||||
//
|
||||
// Fall back to the next snapshot command (or the current cut) when no
|
||||
// matching boundary record exists — better to drop too much than to leak
|
||||
// the Q&A; the visible cost is excluding any "real work" that happened
|
||||
// between snapshots without a recorded boundary, which only occurs if
|
||||
// the boundary log was wiped or the prior capture crashed.
|
||||
function findUuidLineIndex(targetUuid, startLine, endLineExclusive) {
|
||||
for (let i = startLine; i < endLineExclusive; i++) {
|
||||
try {
|
||||
const entry = JSON.parse(rawLines[i]);
|
||||
if (entry.uuid === targetUuid) return i;
|
||||
} catch {
|
||||
/* skip */
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
const priorQAExcludedLines = new Set();
|
||||
for (let i = 0; i < priorCmdIndexes.length; i++) {
|
||||
const startLine = priorCmdIndexes[i];
|
||||
const nextCutLine = i + 1 < priorCmdIndexes.length ? priorCmdIndexes[i + 1] : cutIndex;
|
||||
let snapshotCmdUuid = null;
|
||||
try {
|
||||
snapshotCmdUuid = JSON.parse(rawLines[startLine]).uuid || null;
|
||||
} catch {
|
||||
/* unparseable command line — skip */
|
||||
}
|
||||
let endLine = -1;
|
||||
if (snapshotCmdUuid) {
|
||||
const boundary = priorBoundaries.find((b) => b.snapshotCommandUuid === snapshotCmdUuid);
|
||||
if (boundary && boundary.lastEntryUuid) {
|
||||
endLine = findUuidLineIndex(boundary.lastEntryUuid, startLine, nextCutLine);
|
||||
}
|
||||
}
|
||||
// No matching boundary: bound the exclusion at the next snapshot/current
|
||||
// cut so the Q&A doesn't leak even if state was lost.
|
||||
if (endLine < 0) endLine = nextCutLine - 1;
|
||||
for (let j = startLine; j <= endLine; j++) priorQAExcludedLines.add(j);
|
||||
}
|
||||
|
||||
// Step 2: parse all entries before the cut, build uuid index
|
||||
const preCutEntries = [];
|
||||
const byUuid = {};
|
||||
for (let i = 0; i < cutIndex; i++) {
|
||||
try {
|
||||
const entry = JSON.parse(rawLines[i]);
|
||||
preCutEntries.push({ line: rawLines[i], entry, index: i });
|
||||
if (entry.uuid) {
|
||||
byUuid[entry.uuid] = entry;
|
||||
}
|
||||
} catch {
|
||||
// Keep unparseable lines (they'll be included as non-message entries)
|
||||
preCutEntries.push({ line: rawLines[i], entry: null, index: i });
|
||||
}
|
||||
}
|
||||
|
||||
// Step 3: find the tip of the active branch. The snapshot command's parentUuid
|
||||
// points to the message the user was looking at when they ran the snapshot —
|
||||
// this is authoritative even after rewinds.
|
||||
let tipUuid = null;
|
||||
if (cutIndex < rawLines.length) {
|
||||
try {
|
||||
const snapshotCmd = JSON.parse(rawLines[cutIndex]);
|
||||
tipUuid = snapshotCmd.parentUuid || null;
|
||||
} catch {
|
||||
// not valid JSON — leave tipUuid null
|
||||
}
|
||||
}
|
||||
// If the live tip is itself inside a prior snapshot's Q&A subgraph (e.g.
|
||||
// the user ran /create-snapshot:snapshot a second time WITHOUT typing
|
||||
// anything between the two — there's no "real work" gap), walk back past
|
||||
// the excluded range to find the closest non-excluded ancestor. Otherwise
|
||||
// activeBranchUuids would be empty and we'd produce an empty snapshot.
|
||||
function nearestNonExcludedAncestor(startUuid) {
|
||||
let cur = startUuid;
|
||||
while (cur) {
|
||||
const e = byUuid[cur];
|
||||
if (!e) return cur; // unknown uuid — best effort, keep
|
||||
// Find the line index of this entry to check exclusion.
|
||||
// (Line index isn't stored on the entry; recompute via preCutEntries.)
|
||||
const found = preCutEntries.find((p) => p.entry?.uuid === cur);
|
||||
if (!found || !priorQAExcludedLines.has(found.index)) return cur;
|
||||
cur = e.parentUuid || null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
if (tipUuid) tipUuid = nearestNonExcludedAncestor(tipUuid);
|
||||
|
||||
// Fallback: if no snapshot command found, use the last entry with a uuid
|
||||
if (!tipUuid) {
|
||||
for (let i = preCutEntries.length - 1; i >= 0; i--) {
|
||||
if (preCutEntries[i].entry?.uuid && !priorQAExcludedLines.has(preCutEntries[i].index)) {
|
||||
tipUuid = preCutEntries[i].entry.uuid;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Collect all uuids on the active branch
|
||||
const activeBranchUuids = new Set();
|
||||
let current = tipUuid;
|
||||
while (current) {
|
||||
activeBranchUuids.add(current);
|
||||
current = byUuid[current]?.parentUuid || null;
|
||||
}
|
||||
|
||||
// Step 4: filter — keep entries on the active branch.
|
||||
//
|
||||
// Claude Code writes several bookkeeping entry types alongside the message
|
||||
// tree that don't carry a branch uuid. Their content references whatever
|
||||
// branch was active when they were written, so if the user has rewound,
|
||||
// these will leak rewound-branch state (file backups, prior prompt text,
|
||||
// stale titles, queued prompts, PR links, etc.) into the snapshot — a leak
|
||||
// snapshot.patch doesn't show. Drop the ones we can't attribute to the
|
||||
// active branch.
|
||||
//
|
||||
// `file-history-snapshot` is special-cased: it carries a `messageId`
|
||||
// pointing at the message whose pre-edit state it tracks, so we can
|
||||
// keep only those whose messageId is on the active branch. That
|
||||
// preserves /rewind functionality after a snapshot is restored (rewind
|
||||
// needs the file-backup metadata) while still dropping records from
|
||||
// rewound branches.
|
||||
const BLANKET_DROP_TYPES = new Set([
|
||||
'agent-name',
|
||||
'ai-title',
|
||||
'custom-title',
|
||||
'last-prompt',
|
||||
'permission-mode',
|
||||
'pr-link',
|
||||
'queue-operation',
|
||||
]);
|
||||
|
||||
function prunedClaudeLines() {
|
||||
const kept = [];
|
||||
for (const { line, entry, index } of preCutEntries) {
|
||||
if (priorQAExcludedLines.has(index)) continue;
|
||||
if (!entry) {
|
||||
// Unparseable line — keep as-is so we don't lose data we can't classify.
|
||||
kept.push(line);
|
||||
continue;
|
||||
}
|
||||
if (entry.uuid) {
|
||||
if (activeBranchUuids.has(entry.uuid)) kept.push(line);
|
||||
continue;
|
||||
}
|
||||
// No uuid: bookkeeping entry.
|
||||
if (entry.type === 'file-history-snapshot') {
|
||||
// Keep only if the message it tracks is on the active branch.
|
||||
if (entry.messageId && activeBranchUuids.has(entry.messageId)) {
|
||||
kept.push(line);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (!BLANKET_DROP_TYPES.has(entry.type)) kept.push(line);
|
||||
}
|
||||
return kept;
|
||||
}
|
||||
|
||||
// Rewind branches and prior-Q&A exclusion are Claude-transcript concerns; a linear
|
||||
// harness transcript just truncates at its boundary.
|
||||
let startLine;
|
||||
if (!IS_CLAUDE && startMarkerPath && fs.existsSync(startMarkerPath)) {
|
||||
try {
|
||||
const marker = JSON.parse(fs.readFileSync(startMarkerPath, 'utf8'));
|
||||
if (marker.transcript_path === harnessSession.rawPath) startLine = marker.line_count;
|
||||
} catch {
|
||||
startLine = undefined;
|
||||
}
|
||||
}
|
||||
if (!IS_CLAUDE && startLine === undefined) {
|
||||
console.error(
|
||||
'WARNING: no snapshot start marker for this session — the snapshot Q&A may be captured as conversation. Run capture-snapshot.mjs --mark-start before the annotation questions.'
|
||||
);
|
||||
}
|
||||
|
||||
const outputLines = IS_CLAUDE
|
||||
? prunedClaudeLines()
|
||||
: linearSnapshotLines(harnessSession, startLine);
|
||||
|
||||
if (outputLines.length === 0) {
|
||||
console.error(
|
||||
`WARNING: found no conversation to seed in ${transcript_path}, so this snapshot has no prior turns. The task will run cold from its prompt alone — fine if that is what you want, but if you meant to capture a conversation, check that the exchange you wanted came BEFORE this snapshot.`
|
||||
);
|
||||
}
|
||||
|
||||
// Zero bytes, not a lone newline, when there is nothing to seed: downstream decides
|
||||
// single- vs multi-turn on the file's SIZE, so a 1-byte file would try to resume nothing.
|
||||
fs.writeFileSync(
|
||||
path.join(snapshotDir, 'session.jsonl'),
|
||||
outputLines.length > 0 ? outputLines.join('\n') + '\n' : ''
|
||||
);
|
||||
|
||||
// --- Write boundary record so the NEXT capture-snapshot in this session
|
||||
// can identify and exclude this snapshot's Q&A subgraph ---
|
||||
//
|
||||
// The record pairs the current invocation's command-line uuid with the
|
||||
// uuid of the last entry in the JSONL at this moment (which is whichever
|
||||
// assistant turn invoked us as a tool). A subsequent capture run reads
|
||||
// this file, finds these two uuids in its raw lines, and excludes the
|
||||
// range — a small leak still exists for entries appended AFTER capture
|
||||
// returns (the assistant's "Snapshot saved to: ..." reply), but the
|
||||
// substantive annotation Q&A is fully bounded.
|
||||
if (dataDir && session_id && cutIndex < rawLines.length) {
|
||||
let snapshotCommandUuid = null;
|
||||
try {
|
||||
snapshotCommandUuid = JSON.parse(rawLines[cutIndex]).uuid || null;
|
||||
} catch {
|
||||
/* leave null — we'll skip writing */
|
||||
}
|
||||
// Re-read transcript so we pick up any lines Claude Code has appended
|
||||
// since we read it above (the assistant's tool-use entry, etc.).
|
||||
let lastEntryUuid = null;
|
||||
try {
|
||||
const liveLines = fs.readFileSync(transcript_path, 'utf8').trimEnd().split('\n');
|
||||
for (let i = liveLines.length - 1; i >= 0; i--) {
|
||||
try {
|
||||
const e = JSON.parse(liveLines[i]);
|
||||
if (e.uuid) {
|
||||
lastEntryUuid = e.uuid;
|
||||
break;
|
||||
}
|
||||
} catch {
|
||||
/* skip */
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
/* transcript unreadable now — skip writing */
|
||||
}
|
||||
if (snapshotCommandUuid && lastEntryUuid) {
|
||||
const boundariesPath = path.join(dataDir, 'snapshot-boundaries.jsonl');
|
||||
try {
|
||||
fs.mkdirSync(dataDir, { recursive: true });
|
||||
fs.appendFileSync(
|
||||
boundariesPath,
|
||||
JSON.stringify({
|
||||
sessionUuid: session_id,
|
||||
snapshotCommandUuid,
|
||||
lastEntryUuid,
|
||||
timestamp: new Date().toISOString(),
|
||||
}) + '\n'
|
||||
);
|
||||
} catch {
|
||||
// Best-effort: a missing boundary just means the next run falls back
|
||||
// to the conservative "exclude through next snapshot" heuristic.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Copy subagents and tool-results if they exist
|
||||
const sessionSiblingDir = transcript_path.replace(/\.jsonl$/, '');
|
||||
if (fs.existsSync(sessionSiblingDir) && fs.statSync(sessionSiblingDir).isDirectory()) {
|
||||
fs.cpSync(sessionSiblingDir, path.join(snapshotDir, 'session'), { recursive: true });
|
||||
// Claude Code creates subagent files with write-only permissions (--w-------).
|
||||
// Fix them so downstream tools (cpSync in snapshot-to-task, Harbor's dirhash) can read them.
|
||||
execSync(`chmod -R +r "${path.join(snapshotDir, 'session')}"`, { stdio: 'pipe' });
|
||||
}
|
||||
|
||||
// --- Capture git state as a patch ---
|
||||
|
||||
// Returns raw stdout bytes — callers that want a single-line value must
|
||||
// .trim() themselves. Don't trim here: some callers (git diff) produce
|
||||
// patches where a trailing " \n" blank-context line is load-bearing, and
|
||||
// stripping it corrupts the patch.
|
||||
function git(cmd, opts) {
|
||||
try {
|
||||
return execSync(`git ${cmd}`, {
|
||||
encoding: 'utf8',
|
||||
maxBuffer: 50 * 1024 * 1024,
|
||||
cwd: gitRepo,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
...opts,
|
||||
});
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
const commit = git('rev-parse HEAD')?.trim() ?? null;
|
||||
const branch = git('rev-parse --abbrev-ref HEAD')?.trim() ?? null;
|
||||
const remoteUrl = git('remote get-url origin')?.trim() ?? null;
|
||||
|
||||
// Generate a unified patch representing the workspace state AT THE END OF
|
||||
// THE PRIOR TURN — i.e., everything done up to but not including the turn
|
||||
// being snapshotted. This is the state the trial agent should inherit so
|
||||
// it gets a fresh attempt at the prompt that triggered the snapshot.
|
||||
//
|
||||
// The UserPromptSubmit hook checkpoints the working tree to
|
||||
// `refs/raccoon/turn-checkpoint` at every turn boundary (skipping snapshot
|
||||
// invocations themselves), so the latest checkpoint is exactly the state
|
||||
// at the start of the snapshotted turn. We diff HEAD against that
|
||||
// checkpoint to produce the patch.
|
||||
//
|
||||
// Falls back to the pre-checkpoint behavior (full working-tree diff) when
|
||||
// no checkpoint exists — e.g., the worker took a snapshot before any
|
||||
// non-snapshot user message was sent, or the hook never fired (legacy
|
||||
// session, plugin re-installed mid-session, etc.).
|
||||
if (gitRepo) {
|
||||
try {
|
||||
const tmpIndex = path.join(snapshotDir, '.tmp-git-index');
|
||||
const indexEnv = { ...process.env, GIT_INDEX_FILE: tmpIndex };
|
||||
|
||||
// Prefer the FROZEN ref — this is set by checkpoint-workspace at the
|
||||
// moment the user invokes /create-snapshot:*, before any Q&A turns
|
||||
// have a chance to advance the live checkpoint past the state we
|
||||
// want to capture. Fall back to the live checkpoint (then to
|
||||
// working-tree diff) for backward-compat or if the freeze step failed.
|
||||
let baseline = null;
|
||||
try {
|
||||
baseline = git('rev-parse refs/raccoon/turn-checkpoint-frozen')?.trim() ?? null;
|
||||
} catch {
|
||||
baseline = null;
|
||||
}
|
||||
if (!baseline) {
|
||||
try {
|
||||
baseline = git('rev-parse refs/raccoon/turn-checkpoint')?.trim() ?? null;
|
||||
} catch {
|
||||
baseline = null;
|
||||
}
|
||||
}
|
||||
|
||||
// --binary --full-index, on both branches: a plain `git diff` records a
|
||||
// binary difference as an opaque `Binary files a/x and /dev/null differ`
|
||||
// stub, and `git apply` refuses it ("without full index line"), so
|
||||
// build-workspace.sh can't rebuild the task at all. Nobody has to edit a
|
||||
// binary to hit this — a tracked .DS_Store the toolkit zip strips from the
|
||||
// shipped checkout reads as a binary deletion in every session.
|
||||
//
|
||||
// maxBuffer: inlined binaries make patches far bigger than text diffs, and
|
||||
// exceeding the default cap would throw away the whole patch silently.
|
||||
const diffOpts = { env: indexEnv, maxBuffer: 512 * 1024 * 1024 };
|
||||
let patch;
|
||||
if (baseline) {
|
||||
// Diff HEAD against the prior-turn checkpoint. Untracked files in
|
||||
// the checkpoint have been committed to the checkpoint tree, so
|
||||
// they're included automatically.
|
||||
patch = git(`diff --binary --full-index HEAD ${baseline}`, diffOpts);
|
||||
} else {
|
||||
// No checkpoint — fall back to live working-tree diff (pre-fix
|
||||
// behavior). Captures everything different from HEAD, including
|
||||
// any agent edits during the current turn.
|
||||
git('read-tree HEAD', { env: indexEnv });
|
||||
git('add -A', { env: indexEnv });
|
||||
patch = git('diff --cached --binary --full-index HEAD', diffOpts);
|
||||
}
|
||||
|
||||
try {
|
||||
fs.unlinkSync(tmpIndex);
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
|
||||
if (patch) {
|
||||
fs.writeFileSync(
|
||||
path.join(snapshotDir, 'snapshot.patch'),
|
||||
patch.endsWith('\n') ? patch : patch + '\n'
|
||||
);
|
||||
}
|
||||
} catch {
|
||||
// Read-only repo or other git error — skip patch generation
|
||||
}
|
||||
}
|
||||
|
||||
// --- Copy annotation ---
|
||||
|
||||
const annotation = JSON.parse(fs.readFileSync(annotationPath, 'utf8'));
|
||||
fs.writeFileSync(
|
||||
path.join(snapshotDir, 'annotation.json'),
|
||||
JSON.stringify(annotation, null, 2) + '\n'
|
||||
);
|
||||
|
||||
// Clean up temp file
|
||||
try {
|
||||
fs.unlinkSync(annotationPath);
|
||||
} catch {
|
||||
// Ignore cleanup failures
|
||||
}
|
||||
|
||||
// --- Write metadata ---
|
||||
|
||||
const metadata = {
|
||||
slug: slug,
|
||||
session_uuid: session_id,
|
||||
// The harness the session was actually read as, so it can't disagree with what
|
||||
// was captured.
|
||||
harness: HARNESS,
|
||||
original_cwd: process.cwd(),
|
||||
commit: commit,
|
||||
branch: branch,
|
||||
remote_url: remoteUrl,
|
||||
timestamp: new Date().toISOString(),
|
||||
plugin_version: '0.2.0',
|
||||
};
|
||||
|
||||
fs.writeFileSync(path.join(snapshotDir, 'metadata.json'), JSON.stringify(metadata, null, 2) + '\n');
|
||||
|
||||
// --- Generate restore.sh ---
|
||||
|
||||
const restoreScript = `#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
# Restore a snapshot for resuming a Claude Code conversation.
|
||||
#
|
||||
# Usage: ./restore.sh [target-dir]
|
||||
# target-dir: directory to clone/checkout the repo into (default: ./repo)
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "\${BASH_SOURCE[0]}")" && pwd)"
|
||||
TARGET_DIR="\${1:-./repo}"
|
||||
|
||||
# Read metadata
|
||||
COMMIT=$(jq -r '.commit' "$SCRIPT_DIR/metadata.json")
|
||||
REMOTE=$(jq -r '.remote_url' "$SCRIPT_DIR/metadata.json")
|
||||
SESSION_UUID=$(jq -r '.session_uuid' "$SCRIPT_DIR/metadata.json")
|
||||
|
||||
echo "Cloning $REMOTE at $COMMIT..."
|
||||
git clone "$REMOTE" "$TARGET_DIR"
|
||||
cd "$TARGET_DIR"
|
||||
git checkout "$COMMIT"
|
||||
|
||||
# Apply snapshot patch if present
|
||||
if [ -f "$SCRIPT_DIR/snapshot.patch" ]; then
|
||||
echo "Applying snapshot.patch..."
|
||||
git apply "$SCRIPT_DIR/snapshot.patch"
|
||||
fi
|
||||
|
||||
# Install conversation so the authoring harness can resume it
|
||||
${
|
||||
IS_CLAUDE
|
||||
? `ENCODED_CWD=$(echo "$PWD" | sed 's|/|-|g; s|^-||')
|
||||
DEST_DIR="$HOME/.claude/projects/-$ENCODED_CWD"
|
||||
mkdir -p "$DEST_DIR"
|
||||
cp "$SCRIPT_DIR/session.jsonl" "$DEST_DIR/$SESSION_UUID.jsonl"
|
||||
if [ -d "$SCRIPT_DIR/session" ]; then
|
||||
cp -r "$SCRIPT_DIR/session" "$DEST_DIR/$SESSION_UUID"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "Snapshot restored. To resume the conversation:"
|
||||
echo " cd $TARGET_DIR"
|
||||
echo " claude --resume $SESSION_UUID"`
|
||||
: `DEST_DIR="\${CODEX_HOME:-$HOME/.codex}/sessions/$(date -u +%Y/%m/%d)"
|
||||
mkdir -p "$DEST_DIR"
|
||||
cp "$SCRIPT_DIR/session.jsonl" \\
|
||||
"$DEST_DIR/rollout-$(date -u +%Y-%m-%dT%H-%M-%S).000Z-$SESSION_UUID.jsonl"
|
||||
|
||||
echo ""
|
||||
echo "Snapshot restored. To resume the conversation:"
|
||||
echo " cd $TARGET_DIR"
|
||||
echo " codex resume $SESSION_UUID"`
|
||||
}
|
||||
`;
|
||||
|
||||
fs.writeFileSync(path.join(snapshotDir, 'restore.sh'), restoreScript);
|
||||
fs.chmodSync(path.join(snapshotDir, 'restore.sh'), 0o755);
|
||||
|
||||
try {
|
||||
execSync('bash -ic "_ev snapshot_created 2>/dev/null" 2>/dev/null', {
|
||||
stdio: 'ignore',
|
||||
timeout: 5000,
|
||||
});
|
||||
} catch {
|
||||
// best-effort
|
||||
}
|
||||
|
||||
// --- Done ---
|
||||
|
||||
const fullSnapshotDir = path.resolve(snapshotDir);
|
||||
console.log(`Snapshot saved to: ${fullSnapshotDir}`);
|
||||
console.log(` session.jsonl — conversation transcript`);
|
||||
if (fs.existsSync(sessionSiblingDir) && fs.statSync(sessionSiblingDir).isDirectory()) {
|
||||
console.log(` session/ — subagents + tool results`);
|
||||
}
|
||||
if (fs.existsSync(path.join(snapshotDir, 'snapshot.patch'))) {
|
||||
console.log(` snapshot.patch — working tree changes`);
|
||||
}
|
||||
console.log(` annotation.json — worker annotations`);
|
||||
console.log(` metadata.json — session metadata`);
|
||||
console.log(` restore.sh — restore script for resuming`);
|
||||
@@ -0,0 +1,485 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
// Rewind-aware workspace checkpointing for the reduced-toolset Explore agent.
|
||||
// One script, two hook events (branches on hook_event_name):
|
||||
//
|
||||
// UserPromptSubmit -> CAPTURE
|
||||
// Snapshot the pre-turn working tree into refs/raccoon/turn-checkpoint (the
|
||||
// chain capture-snapshot uses for snapshot.patch) AND record, in
|
||||
// .git/raccoon-state.json, anchor_map[tip] = checkpoint-commit and
|
||||
// last_anchor = tip. `tip` is the conversation node the new prompt attaches
|
||||
// to (the END of the previous turn) — exactly the node a future /rewind to
|
||||
// THIS turn will branch from. Anchoring to the prior tip (not the
|
||||
// just-submitted, maybe-unflushed message) makes capture race-free.
|
||||
//
|
||||
// PreToolUse (first tool call of a turn) -> RECONCILE
|
||||
// Claude Code's /rewind restores the conversation but NOT bash-made edits,
|
||||
// and fires no hook. By the first tool call the post-rewind branch message is
|
||||
// reliably persisted and the agent has not yet read/edited code. We parse the
|
||||
// transcript into a parentUuid DAG, pick the ACTIVE branch (leaf with the
|
||||
// newest tip), and walk it for the newest checkpoint anchor that is a genuine
|
||||
// rewind fork (the anchor still has an orphaned child branch — the discarded
|
||||
// turns). If found, restore the working tree to that checkpoint before the tool
|
||||
// runs. Idempotent per (anchor, branch): restores once per rewind.
|
||||
//
|
||||
// Every failure path is a safe no-op: the hook never aborts the session and
|
||||
// never restores to an unverified tree.
|
||||
|
||||
import { execSync } from 'node:child_process';
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
|
||||
const CHECKPOINT_REF = 'refs/raccoon/turn-checkpoint';
|
||||
const FROZEN_REF = 'refs/raccoon/turn-checkpoint-frozen';
|
||||
// Synthetic parent for every parentless transcript node, so that a rewind to the
|
||||
// VERY FIRST turn (where the new prompt also has parentUuid=null) is detected by
|
||||
// the same divergence machinery as any other turn.
|
||||
const ROOT = '__ROOT__';
|
||||
const RACCOON_AUTHOR = {
|
||||
GIT_AUTHOR_NAME: 'raccoon',
|
||||
GIT_AUTHOR_EMAIL: 'raccoon@local',
|
||||
GIT_COMMITTER_NAME: 'raccoon',
|
||||
GIT_COMMITTER_EMAIL: 'raccoon@local',
|
||||
};
|
||||
|
||||
function makeGit(gitDir, extraEnv) {
|
||||
const env = { ...process.env, ...extraEnv };
|
||||
return (cmd) =>
|
||||
execSync(`git ${cmd}`, {
|
||||
cwd: gitDir,
|
||||
env,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
encoding: 'utf8',
|
||||
}).trim();
|
||||
}
|
||||
|
||||
function findGitDir(cwd) {
|
||||
let gitDir = cwd;
|
||||
for (let i = 0; i < 10; i++) {
|
||||
if (fs.existsSync(path.join(gitDir, '.git'))) return gitDir;
|
||||
const parent = path.dirname(gitDir);
|
||||
if (parent === gitDir) return null;
|
||||
gitDir = parent;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// ---- transcript + state ----
|
||||
|
||||
function readEntries(transcriptPath) {
|
||||
try {
|
||||
if (!transcriptPath || !fs.existsSync(transcriptPath)) return [];
|
||||
const out = [];
|
||||
for (const raw of fs.readFileSync(transcriptPath, 'utf8').split('\n')) {
|
||||
const line = raw.trim();
|
||||
if (!line) continue;
|
||||
let o;
|
||||
try {
|
||||
o = JSON.parse(line);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (o && typeof o.uuid === 'string') out.push(o);
|
||||
}
|
||||
return out;
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
function tsOf(e) {
|
||||
const t = e && e.timestamp ? Date.parse(e.timestamp) : 0;
|
||||
return Number.isFinite(t) ? t : 0;
|
||||
}
|
||||
|
||||
function statePath(gitDir) {
|
||||
return path.join(gitDir, '.git', 'raccoon-state.json');
|
||||
}
|
||||
function loadState(gitDir) {
|
||||
try {
|
||||
const s = JSON.parse(fs.readFileSync(statePath(gitDir), 'utf8'));
|
||||
return { anchor_map: {}, last_anchor: null, reconciled_for: null, ...s };
|
||||
} catch {
|
||||
return { anchor_map: {}, last_anchor: null, reconciled_for: null };
|
||||
}
|
||||
}
|
||||
function saveState(gitDir, s) {
|
||||
try {
|
||||
// Atomic write: a tmp file + rename, so a hook killed mid-write can never
|
||||
// leave a half-written (corrupt) state.json behind.
|
||||
const target = statePath(gitDir);
|
||||
const tmp = `${target}.tmp`;
|
||||
fs.writeFileSync(tmp, JSON.stringify(s));
|
||||
fs.renameSync(tmp, target);
|
||||
} catch {
|
||||
// best-effort
|
||||
}
|
||||
}
|
||||
|
||||
// Loose checkpoint objects must survive: a restore resets the checkpoint chain
|
||||
// ref backward, which can orphan later anchors' commits. Disabling auto-gc keeps
|
||||
// every anchor commit fetchable for a future rewind. The task container is
|
||||
// ephemeral, so accumulating loose objects is harmless.
|
||||
function disableAutoGc(gitDir) {
|
||||
try {
|
||||
makeGit(gitDir)('config gc.auto 0');
|
||||
} catch {
|
||||
// best-effort
|
||||
}
|
||||
}
|
||||
|
||||
// The conversation node the new prompt attaches to = end of the previous turn.
|
||||
// Newest uuid-bearing entry, excluding the just-submitted prompt (which may or
|
||||
// may not be flushed yet — excluding it makes this race-robust).
|
||||
function conversationTip(entries, currentPrompt) {
|
||||
const cp = (currentPrompt || '').trim();
|
||||
for (let i = entries.length - 1; i >= 0; i--) {
|
||||
const e = entries[i];
|
||||
if (!e.uuid) continue;
|
||||
const role = e.type || (e.message && e.message.role);
|
||||
const content = e.message && e.message.content;
|
||||
if (role === 'user' && typeof content === 'string' && cp && content.trim() === cp) continue;
|
||||
return e.uuid;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// ---- capture (UserPromptSubmit) ----
|
||||
|
||||
function ensureExcludes(gitDir) {
|
||||
const localExcludePath = path.join(gitDir, '.git', 'info', 'exclude');
|
||||
const MARKER = '# raccoon-checkpoint excludes (auto-managed):';
|
||||
const excludes = [
|
||||
MARKER,
|
||||
'.pnpm-store/',
|
||||
'.yarn/cache/',
|
||||
'.yarn/install-state.gz',
|
||||
'vendor/bundle/',
|
||||
'.bundle/cache/',
|
||||
'.raccoon-setup-done', // run-app's per-repo first-use setup marker (polyglot toolkits)
|
||||
];
|
||||
try {
|
||||
let existing = '';
|
||||
try {
|
||||
existing = fs.readFileSync(localExcludePath, 'utf8');
|
||||
} catch {
|
||||
existing = '';
|
||||
}
|
||||
if (!existing.includes(MARKER)) {
|
||||
fs.mkdirSync(path.dirname(localExcludePath), { recursive: true });
|
||||
fs.appendFileSync(localExcludePath, '\n' + excludes.join('\n') + '\n');
|
||||
}
|
||||
} catch {
|
||||
// best-effort
|
||||
}
|
||||
}
|
||||
|
||||
function freezeForSnapshot(gitDir) {
|
||||
// Freeze the state at the START of the turn being snapshotted — i.e.,
|
||||
// whatever CHECKPOINT_REF already holds (or HEAD, if no turn has happened
|
||||
// yet this session). This must NOT be the live working tree: the live tree
|
||||
// includes the edits made during the turn that triggered /snapshot, and
|
||||
// capture-snapshot's `diff HEAD <frozen>` is supposed to exclude exactly
|
||||
// that turn so the trial agent gets a fresh attempt at the prompt (see the
|
||||
// comment above baseline selection in capture-snapshot.mjs). Freezing the
|
||||
// live tree instead bakes the agent's just-made edits into the snapshot.
|
||||
try {
|
||||
const git = makeGit(gitDir);
|
||||
let source = null;
|
||||
try {
|
||||
source = git(`rev-parse ${CHECKPOINT_REF}`);
|
||||
} catch {
|
||||
try {
|
||||
source = git('rev-parse HEAD');
|
||||
} catch {
|
||||
source = null;
|
||||
}
|
||||
}
|
||||
if (source) git(`update-ref ${FROZEN_REF} ${source}`);
|
||||
} catch {
|
||||
// best-effort — never break /snapshot
|
||||
}
|
||||
}
|
||||
|
||||
// The snapshot invocation, in whichever form the harness uses: Claude Code takes
|
||||
// `/create-snapshot:snapshot`, codex takes `$create-snapshot:snapshot`. Both send the raw
|
||||
// text as `prompt` on the UserPromptSubmit hook (verified against codex 0.146.1), so the
|
||||
// prefix is the only difference — and missing it means freezing never happens and the
|
||||
// snapshotted turn's own edits get baked into the workspace.
|
||||
const SNAPSHOT_INVOCATION_RE = /^[/$](?:create-snapshot|snapshot)(?![\w-])/;
|
||||
|
||||
function capture(gitDir, data) {
|
||||
const prompt = (data.prompt ?? '').trim();
|
||||
if (SNAPSHOT_INVOCATION_RE.test(prompt)) {
|
||||
freezeForSnapshot(gitDir);
|
||||
return;
|
||||
}
|
||||
let commit = null;
|
||||
try {
|
||||
const tmpIndex = path.join(gitDir, '.git', 'raccoon-checkpoint.index');
|
||||
const git = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: tmpIndex });
|
||||
ensureExcludes(gitDir);
|
||||
disableAutoGc(gitDir);
|
||||
git('read-tree HEAD');
|
||||
git('add -A');
|
||||
const tree = git('write-tree');
|
||||
let parent;
|
||||
try {
|
||||
parent = git(`rev-parse ${CHECKPOINT_REF}`);
|
||||
} catch {
|
||||
parent = git('rev-parse HEAD');
|
||||
}
|
||||
commit = git(`commit-tree ${tree} -p ${parent} -m "raccoon-checkpoint: pre-turn"`);
|
||||
git(`update-ref ${CHECKPOINT_REF} ${commit}`);
|
||||
try {
|
||||
fs.unlinkSync(tmpIndex);
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
} catch {
|
||||
return; // never break the session
|
||||
}
|
||||
// Record the anchor mapping for rewind reconciliation. On the very first turn
|
||||
// there is no prior node, so we anchor to the synthetic ROOT — this is the
|
||||
// pre-turn-1 (initial) state, which a rewind to the first turn restores to.
|
||||
try {
|
||||
const tip = conversationTip(readEntries(data.transcript_path), data.prompt) || ROOT;
|
||||
if (commit) {
|
||||
const s = loadState(gitDir);
|
||||
s.anchor_map[tip] = commit;
|
||||
s.last_anchor = tip;
|
||||
saveState(gitDir, s);
|
||||
}
|
||||
} catch {
|
||||
// best-effort; capture still succeeded
|
||||
}
|
||||
}
|
||||
|
||||
// ---- reconcile (PreToolUse) ----
|
||||
|
||||
function subtreeContains(start, target, children) {
|
||||
const stack = [start];
|
||||
const seen = new Set();
|
||||
while (stack.length > 0) {
|
||||
const n = stack.pop();
|
||||
if (n === target) return true;
|
||||
if (seen.has(n)) continue;
|
||||
seen.add(n);
|
||||
for (const c of children.get(n) || []) stack.push(c);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
function reachesLeaf(start, leafSet, children) {
|
||||
const stack = [start];
|
||||
const seen = new Set();
|
||||
while (stack.length > 0) {
|
||||
const n = stack.pop();
|
||||
if (leafSet.has(n)) return true;
|
||||
if (seen.has(n)) continue;
|
||||
seen.add(n);
|
||||
for (const c of children.get(n) || []) stack.push(c);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// node is a genuine rewind fork: >=2 children, one reaching the active leaf and
|
||||
// at least one reaching a different (orphaned) leaf.
|
||||
function isDivergence(node, activeLeaf, leaves, children) {
|
||||
const kids = children.get(node) || [];
|
||||
if (kids.length < 2) return false;
|
||||
const leafSet = new Set(leaves);
|
||||
const reachesActive = kids.some((k) => subtreeContains(k, activeLeaf, children));
|
||||
const reachesOther = kids.some(
|
||||
(k) => !subtreeContains(k, activeLeaf, children) && reachesLeaf(k, leafSet, children)
|
||||
);
|
||||
return reachesActive && reachesOther;
|
||||
}
|
||||
|
||||
// Restore the working tree to a commit's tree, saving the current state to a
|
||||
// safety ref first. Returns the safety ref name, or null on failure.
|
||||
function restoreToCommit(gitDir, commit) {
|
||||
try {
|
||||
const restoreIndex = path.join(gitDir, '.git', 'raccoon-restore.index');
|
||||
const stashIndex = path.join(gitDir, '.git', 'raccoon-stash.index');
|
||||
const gitStash = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: stashIndex });
|
||||
const gitRestore = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: restoreIndex });
|
||||
const gitPlain = makeGit(gitDir, RACCOON_AUTHOR);
|
||||
|
||||
ensureExcludes(gitDir);
|
||||
disableAutoGc(gitDir);
|
||||
gitStash('read-tree HEAD');
|
||||
gitStash('add -A');
|
||||
const curTree = gitStash('write-tree');
|
||||
let parent = null;
|
||||
try {
|
||||
parent = gitPlain('rev-parse HEAD');
|
||||
} catch {
|
||||
parent = null;
|
||||
}
|
||||
const curCommit = gitStash(
|
||||
`commit-tree ${curTree}${parent ? ` -p ${parent}` : ''} -m "raccoon: pre-rewind safety"`
|
||||
);
|
||||
const safetyRef = `refs/raccoon/pre-rewind/${Date.now()}`;
|
||||
gitPlain(`update-ref ${safetyRef} ${curCommit}`);
|
||||
|
||||
// Files to delete = present in the current tree but absent from the target
|
||||
// checkpoint. Computed as a set difference of `ls-tree` listings rather than
|
||||
// `diff --diff-filter=A`, because git's rename/copy detection reclassifies an
|
||||
// added path as R/C, which a filter on "A" would miss — leaving the renamed-to
|
||||
// file stranded in the worktree after a restore.
|
||||
let added = [];
|
||||
try {
|
||||
const inCheckpoint = new Set(
|
||||
gitPlain(`ls-tree -r --name-only ${commit}`).split('\n').filter(Boolean)
|
||||
);
|
||||
const inCurrent = gitPlain(`ls-tree -r --name-only ${curCommit}`).split('\n').filter(Boolean);
|
||||
added = inCurrent.filter((f) => !inCheckpoint.has(f));
|
||||
} catch {
|
||||
added = [];
|
||||
}
|
||||
gitRestore(`read-tree ${commit}`);
|
||||
gitRestore('checkout-index -a -f');
|
||||
for (const f of added) {
|
||||
try {
|
||||
fs.rmSync(path.join(gitDir, f), { force: true });
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
for (const idx of [restoreIndex, stashIndex]) {
|
||||
try {
|
||||
fs.unlinkSync(idx);
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
return safetyRef;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function reconcile(gitDir, data) {
|
||||
let s;
|
||||
try {
|
||||
s = loadState(gitDir);
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
if (!s || !s.anchor_map || Object.keys(s.anchor_map).length === 0) return;
|
||||
|
||||
const entries = readEntries(data.transcript_path);
|
||||
if (entries.length === 0) return;
|
||||
|
||||
const byId = new Map();
|
||||
const children = new Map();
|
||||
const referenced = new Set();
|
||||
for (const e of entries) byId.set(e.uuid, e);
|
||||
for (const e of entries) {
|
||||
// Parentless or dangling-parent nodes hang off the synthetic ROOT.
|
||||
const p = e.parentUuid && byId.has(e.parentUuid) ? e.parentUuid : ROOT;
|
||||
if (!children.has(p)) children.set(p, []);
|
||||
children.get(p).push(e.uuid);
|
||||
referenced.add(p);
|
||||
}
|
||||
const leaves = [...byId.keys()].filter((u) => !referenced.has(u));
|
||||
if (leaves.length === 0) return;
|
||||
|
||||
// active branch = leaf with the newest tip (the branch CC is appending to now)
|
||||
let activeLeaf = leaves[0];
|
||||
for (const u of leaves) if (tsOf(byId.get(u)) > tsOf(byId.get(activeLeaf))) activeLeaf = u;
|
||||
|
||||
// Walk the active chain (through ROOT) for the newest anchor that is a GENUINE
|
||||
// rewind divergence: the anchor node has an orphaned child branch (the discarded
|
||||
// turns) alongside the active branch. We skip anchors that are NOT forks, so:
|
||||
// - pure forward progress (single child) never triggers a restore;
|
||||
// - redoing the LAST turn still triggers (the new turn builds on the same
|
||||
// boundary as the prior turn, but the discarded turn is an orphan sibling);
|
||||
// - a stray anchor recorded on the active branch itself (e.g. the post-rewind
|
||||
// capture's tip) can't mask the real divergence further up the chain.
|
||||
// branchChild = the divergence node's child on the active path (the "branch id").
|
||||
let cur = activeLeaf;
|
||||
let prev = null;
|
||||
let branchChild = null;
|
||||
const seen = new Set();
|
||||
let activeAnchor = null;
|
||||
while (cur && !seen.has(cur)) {
|
||||
seen.add(cur);
|
||||
if (
|
||||
Object.prototype.hasOwnProperty.call(s.anchor_map, cur) &&
|
||||
isDivergence(cur, activeLeaf, leaves, children)
|
||||
) {
|
||||
activeAnchor = cur;
|
||||
branchChild = prev;
|
||||
break;
|
||||
}
|
||||
prev = cur;
|
||||
if (cur === ROOT) break;
|
||||
const e = byId.get(cur);
|
||||
cur = e && e.parentUuid && byId.has(e.parentUuid) ? e.parentUuid : ROOT;
|
||||
}
|
||||
if (!activeAnchor) return;
|
||||
|
||||
// Idempotency keyed on (anchor, active branch) — NOT the active leaf. Every
|
||||
// tool call within a post-rewind turn advances the leaf, but the branch is
|
||||
// stable, so we restore exactly once per rewind. A fresh re-rewind to the same
|
||||
// turn forks a NEW child off the anchor, changing the key, so it restores again.
|
||||
const reconKey = `${activeAnchor}:${branchChild || ''}`;
|
||||
if (s.reconciled_for === reconKey) return; // this rewind already reconciled
|
||||
|
||||
const safety = restoreToCommit(gitDir, s.anchor_map[activeAnchor]);
|
||||
try {
|
||||
makeGit(gitDir)(`update-ref ${CHECKPOINT_REF} ${s.anchor_map[activeAnchor]}`);
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
s.reconciled_for = reconKey;
|
||||
saveState(gitDir, s);
|
||||
// Record the restore to a side log ONLY — never stdout/stderr. Hook output on
|
||||
// PreToolUse is captured into the transcript (as an attachment entry) and the
|
||||
// transcript is the task data, so any emission here would contaminate it. The
|
||||
// log lives under .git/, which is never staged, snapshotted, or transcribed.
|
||||
try {
|
||||
const line =
|
||||
`${new Date().toISOString()} rewind reconciled: restored to anchor ${activeAnchor} ` +
|
||||
`(${s.anchor_map[activeAnchor]})${safety ? `; pre-rewind state saved to ${safety}` : ''}\n`;
|
||||
fs.appendFileSync(path.join(gitDir, '.git', 'raccoon-rewind.log'), line);
|
||||
} catch {
|
||||
// best-effort; the restore itself already succeeded
|
||||
}
|
||||
}
|
||||
|
||||
function main(data) {
|
||||
const cwd = data.cwd || process.cwd();
|
||||
const gitDir = findGitDir(cwd);
|
||||
if (!gitDir) return;
|
||||
const event = data.hook_event_name || (data.tool_name ? 'PreToolUse' : 'UserPromptSubmit');
|
||||
// RECONCILE undoes a /rewind, which only Claude Code has. Other harnesses append
|
||||
// and never fork, so there is nothing to reconcile and the transcript it would walk
|
||||
// has no parentUuid DAG.
|
||||
const canRewind = (process.env.RACCOON_HARNESS || 'claude-code') === 'claude-code';
|
||||
if (event === 'PreToolUse') {
|
||||
if (canRewind) reconcile(gitDir, data);
|
||||
} else capture(gitDir, data);
|
||||
}
|
||||
|
||||
let input = '';
|
||||
process.stdin.setEncoding('utf8');
|
||||
process.stdin.on('data', (chunk) => {
|
||||
input += chunk;
|
||||
});
|
||||
process.stdin.on('end', () => {
|
||||
let data;
|
||||
try {
|
||||
data = JSON.parse(input);
|
||||
} catch {
|
||||
process.exit(0);
|
||||
}
|
||||
try {
|
||||
main(data);
|
||||
} catch {
|
||||
// A failure here must never break the user's session.
|
||||
}
|
||||
process.exit(0);
|
||||
});
|
||||
@@ -0,0 +1,29 @@
|
||||
// Types for harness-session.mjs, so TS consumers (its test, snapshot-to-task) see a
|
||||
// real shape instead of `any`.
|
||||
|
||||
export interface Turn {
|
||||
/** Line index in the native session file. */
|
||||
index: number;
|
||||
role: 'user' | 'assistant';
|
||||
text: string;
|
||||
/** A slash-command turn, not real conversation. */
|
||||
isCommand: boolean;
|
||||
/** This record concluded its turn — the truncation boundary. */
|
||||
endsTurn: boolean;
|
||||
}
|
||||
|
||||
export interface Session {
|
||||
harness: string;
|
||||
rawPath: string;
|
||||
/** The harness own id for this conversation. */
|
||||
sessionId: string | null;
|
||||
lines: string[];
|
||||
turns: Turn[];
|
||||
}
|
||||
|
||||
export function supportedHarnesses(): string[];
|
||||
export function readSession(harness: string, recordedPath?: string): Session | null;
|
||||
export function truncationIndex(turns: Turn[]): number;
|
||||
export function turnsFromLines(harness: string, lines: string[]): Turn[];
|
||||
export function linearSnapshotLines(session: Session, startLine?: number): string[];
|
||||
export function stripAuthoringScaffolding(harness: string, lines: string[]): string[];
|
||||
@@ -0,0 +1,325 @@
|
||||
// Locate and read a harness's native conversation, so capture-snapshot can work
|
||||
// against any harness. Everything else in capture (snapshot.patch, restore.sh,
|
||||
// annotation, metadata) is harness-agnostic.
|
||||
//
|
||||
// The returned session stays in the harness's OWN native format: the seeding design
|
||||
// hands a native blob back to the same harness, and codex_agent reads the same staged
|
||||
// /tmp/snapshot-session/session.jsonl path that snapshot_agent does.
|
||||
|
||||
import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
|
||||
/**
|
||||
* @typedef {object} Turn
|
||||
* @property {number} index line index in the native session file
|
||||
* @property {'user'|'assistant'} role
|
||||
* @property {string} text
|
||||
* @property {boolean} isCommand a slash-command turn, not real conversation
|
||||
* @property {boolean} endsTurn this record concluded its turn
|
||||
*/
|
||||
|
||||
/**
|
||||
* @typedef {object} Session
|
||||
* @property {string} harness
|
||||
* @property {string} rawPath
|
||||
* @property {string|null} sessionId the harness's own id for this conversation
|
||||
* @property {string[]} lines
|
||||
* @property {Turn[]} turns
|
||||
*/
|
||||
|
||||
// Newest matching file beneath `root`, or null. Ties on mtime break on path so the
|
||||
// answer is stable — two sessions written in the same millisecond are common.
|
||||
function newestUnder(root, matches) {
|
||||
if (!fs.existsSync(root)) return null;
|
||||
const found = [];
|
||||
const walk = (dir) => {
|
||||
let entries;
|
||||
try {
|
||||
entries = fs.readdirSync(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
for (const entry of entries) {
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) walk(full);
|
||||
else if (matches(entry.name)) found.push({ full, mtimeMs: fs.statSync(full).mtimeMs });
|
||||
}
|
||||
};
|
||||
walk(root);
|
||||
if (found.length === 0) return null;
|
||||
found.sort((a, b) => b.mtimeMs - a.mtimeMs || b.full.localeCompare(a.full));
|
||||
return found[0].full;
|
||||
}
|
||||
|
||||
// User-role records codex writes that the human did not type: its own environment
|
||||
// preamble, a `$name` skill invocation, and the SKILL.md body injected in response.
|
||||
// Matched only at the START of the text, so a turn that merely quotes one is still real
|
||||
// conversation.
|
||||
function isCodexCommandText(text) {
|
||||
const trimmed = (text || '').trimStart();
|
||||
if (trimmed.startsWith('<skill>') || trimmed.startsWith('<environment_context>')) return true;
|
||||
return /^\$[\w:.-]+\s*$/.test(trimmed);
|
||||
}
|
||||
|
||||
const HARNESSES = {
|
||||
'claude-code': {
|
||||
/** Claude Code records one JSONL per session under ~/.claude/projects/<encoded-cwd>/. */
|
||||
findSession() {
|
||||
return newestUnder(path.join(os.homedir(), '.claude', 'projects'), (n) =>
|
||||
n.endsWith('.jsonl')
|
||||
);
|
||||
},
|
||||
|
||||
/** Claude names the transcript for its session id. */
|
||||
sessionId(rawPath) {
|
||||
return path.basename(rawPath, '.jsonl');
|
||||
},
|
||||
/**
|
||||
* One turn per conversational record. `endsTurn` marks an assistant record that
|
||||
* concluded its turn — the truncation boundary. Bookkeeping records (attachments,
|
||||
* file-history, permission-mode) carry no role and are skipped.
|
||||
*/
|
||||
/** @param {string[]} lines @returns {Turn[]} */
|
||||
readTurns(lines) {
|
||||
/** @type {Turn[]} */
|
||||
const turns = [];
|
||||
for (const [index, line] of lines.entries()) {
|
||||
let entry;
|
||||
try {
|
||||
entry = JSON.parse(line);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
const role =
|
||||
entry.type === 'user' ? 'user' : entry.type === 'assistant' ? 'assistant' : null;
|
||||
if (!role) continue;
|
||||
const content = entry.message?.content;
|
||||
const text =
|
||||
typeof content === 'string'
|
||||
? content
|
||||
: Array.isArray(content)
|
||||
? content
|
||||
.filter((b) => b && b.type === 'text')
|
||||
.map((b) => b.text ?? '')
|
||||
.join('')
|
||||
: '';
|
||||
turns.push({
|
||||
index,
|
||||
role,
|
||||
text,
|
||||
isCommand:
|
||||
role === 'user' &&
|
||||
typeof content === 'string' &&
|
||||
/<command-name>|<command-message>|<local-command-caveat>/.test(content),
|
||||
endsTurn: role === 'assistant' && entry.message?.stop_reason === 'end_turn',
|
||||
});
|
||||
}
|
||||
return turns;
|
||||
},
|
||||
},
|
||||
|
||||
codex: {
|
||||
/** codex writes rollout JSONL under $CODEX_HOME/sessions/<date>/. */
|
||||
findSession() {
|
||||
const home = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
|
||||
return newestUnder(
|
||||
path.join(home, 'sessions'),
|
||||
(n) => n.startsWith('rollout-') && n.endsWith('.jsonl')
|
||||
);
|
||||
},
|
||||
|
||||
/** `codex resume <id>` resolves the id recorded in session_meta, not the filename. */
|
||||
sessionId(rawPath, lines) {
|
||||
for (const line of lines) {
|
||||
try {
|
||||
const rec = JSON.parse(line);
|
||||
if (rec.type === 'session_meta' && rec.payload?.id) return rec.payload.id;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
},
|
||||
/**
|
||||
* codex rollouts carry `response_item` records whose payload is a message with a
|
||||
* role. An assistant message with no following tool activity ends the turn; codex
|
||||
* records no stop_reason, so a turn ends where the next user message begins —
|
||||
* resolved after the fact below.
|
||||
*/
|
||||
/** @param {string[]} lines @returns {Turn[]} */
|
||||
readTurns(lines) {
|
||||
/** @type {Turn[]} */
|
||||
const turns = [];
|
||||
for (const [index, line] of lines.entries()) {
|
||||
let record;
|
||||
try {
|
||||
record = JSON.parse(line);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (record.type !== 'response_item') continue;
|
||||
const payload = record.payload ?? {};
|
||||
if (payload.type !== 'message') continue;
|
||||
const role =
|
||||
payload.role === 'user' ? 'user' : payload.role === 'assistant' ? 'assistant' : null;
|
||||
if (!role) continue;
|
||||
const text = Array.isArray(payload.content)
|
||||
? payload.content.map((b) => b?.text ?? '').join('')
|
||||
: typeof payload.content === 'string'
|
||||
? payload.content
|
||||
: '';
|
||||
turns.push({
|
||||
index,
|
||||
role,
|
||||
text,
|
||||
isCommand: role === 'user' && isCodexCommandText(text),
|
||||
endsTurn: false,
|
||||
});
|
||||
}
|
||||
// An assistant turn ends where the next user turn starts, or at the end.
|
||||
for (let i = 0; i < turns.length; i += 1) {
|
||||
if (turns[i].role !== 'assistant') continue;
|
||||
const next = turns[i + 1];
|
||||
turns[i].endsTurn = !next || next.role === 'user';
|
||||
}
|
||||
return turns;
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
/** @returns {string[]} */
|
||||
export function supportedHarnesses() {
|
||||
return Object.keys(HARNESSES);
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the current session for `harness`. Returns null when nothing is found, so the
|
||||
* caller can report which harness had no conversation to capture.
|
||||
*/
|
||||
/**
|
||||
* @param {string} harness
|
||||
* @param {string} [recordedPath] transcript recorded by the SessionStart hook; preferred
|
||||
* over the newest-file scan, which can pick a different session in a busy container.
|
||||
* @returns {Session | null}
|
||||
*/
|
||||
export function readSession(harness, recordedPath) {
|
||||
const reader = HARNESSES[harness];
|
||||
if (!reader) {
|
||||
throw new Error(
|
||||
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
|
||||
);
|
||||
}
|
||||
const rawPath = recordedPath && fs.existsSync(recordedPath) ? recordedPath : reader.findSession();
|
||||
if (!rawPath) return null;
|
||||
const lines = fs.readFileSync(rawPath, 'utf8').trimEnd().split('\n');
|
||||
return {
|
||||
harness,
|
||||
rawPath,
|
||||
lines,
|
||||
turns: reader.readTurns(lines),
|
||||
sessionId: reader.sessionId(rawPath, lines),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Index of the last record to keep: the last turn-ending assistant record before the
|
||||
* final real user turn. Drops the prompt that elicited the failure and the failure
|
||||
* response, so the test agent inherits context but not the answer.
|
||||
*
|
||||
* Returns -1 when there is no such boundary (a one-shot conversation), which callers
|
||||
* treat as "seed nothing and run cold".
|
||||
*/
|
||||
/**
|
||||
* @param {Turn[]} turns
|
||||
* @returns {number}
|
||||
*/
|
||||
export function truncationIndex(turns) {
|
||||
let lastUser = -1;
|
||||
for (const turn of turns) {
|
||||
if (turn.role === 'user' && !turn.isCommand && turn.text.trim()) lastUser = turn.index;
|
||||
}
|
||||
if (lastUser < 0) return -1;
|
||||
let cut = -1;
|
||||
for (const turn of turns) {
|
||||
if (turn.index >= lastUser) break;
|
||||
if (turn.role === 'assistant' && turn.endsTurn) cut = turn.index;
|
||||
}
|
||||
return cut;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse already-read lines with a harness's reader, for callers that have the text
|
||||
* rather than a path.
|
||||
*
|
||||
* @param {string} harness
|
||||
* @param {string[]} lines
|
||||
* @returns {Turn[]}
|
||||
*/
|
||||
export function turnsFromLines(harness, lines) {
|
||||
const reader = HARNESSES[harness];
|
||||
if (!reader) {
|
||||
throw new Error(
|
||||
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
|
||||
);
|
||||
}
|
||||
return reader.readTurns(lines);
|
||||
}
|
||||
|
||||
/**
|
||||
* Lines to stage as the captured `session.jsonl` for a linear-transcript harness:
|
||||
* everything up to the snapshot invocation, matching what Claude Code stages when it
|
||||
* cuts at its slash-command line. Dropping the failure-eliciting turn happens later,
|
||||
* in snapshot-to-task — capture keeps the full conversation.
|
||||
*
|
||||
* `startLine` is the rollout length recorded when the snapshot was invoked; without it
|
||||
* the whole session is kept, which would include the snapshot's own Q&A.
|
||||
*
|
||||
* @param {Session} session
|
||||
* @param {number} [startLine]
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function linearSnapshotLines(session, startLine) {
|
||||
if (typeof startLine === 'number' && startLine >= 0) {
|
||||
return session.lines.slice(0, startLine);
|
||||
}
|
||||
return session.lines;
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop records that describe the AUTHORING container rather than the conversation.
|
||||
*
|
||||
* codex records both its skill catalogue (a `developer` turn) and the machine it ran on (a
|
||||
* `user` turn of `<environment_context>`). Native resume replays records byte-identically,
|
||||
* so without this the test agent inherits a list of skills it does not have — one described
|
||||
* as "capture the current conversation and repo state as a snapshot" — and a working
|
||||
* directory that does not exist in the trial. codex re-injects both for the trial, and base
|
||||
* instructions travel in `session_meta`, so removing them loses nothing. Claude's fork
|
||||
* already re-records with the trial's own cwd; this brings codex to the same place.
|
||||
*
|
||||
* @param {string} harness
|
||||
* @param {string[]} lines
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function stripAuthoringScaffolding(harness, lines) {
|
||||
if (harness === 'claude-code') return lines;
|
||||
return lines.filter((raw) => {
|
||||
let rec;
|
||||
try {
|
||||
rec = JSON.parse(raw);
|
||||
} catch {
|
||||
return true;
|
||||
}
|
||||
const payload = rec?.payload;
|
||||
if (rec?.type !== 'response_item' || payload?.type !== 'message') return true;
|
||||
const text = (payload.content ?? [])
|
||||
.map((block) => (typeof block?.text === 'string' ? block.text : ''))
|
||||
.join('')
|
||||
.trim();
|
||||
// Match the machine-generated shape only — a turn that STARTS with the tag — so a
|
||||
// worker who quotes one of these strings mid-conversation keeps their turn.
|
||||
if (payload.role === 'developer') return !text.startsWith('<skills_instructions>');
|
||||
if (payload.role === 'user') return !text.startsWith('<environment_context>');
|
||||
return true;
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
/**
|
||||
* Plugin-side re-export, so snapshot-to-task.ts resolves `./lib/copy-tree`
|
||||
* both here and in the toolkit's flat scripts/ dir.
|
||||
*/
|
||||
export * from '../../../../raccoon-worker-toolkit/static/scripts/lib/copy-tree';
|
||||
@@ -0,0 +1,260 @@
|
||||
/**
|
||||
* Strip machine-identifying filesystem paths, and optional keywords, from a session
|
||||
* transcript. Pure: raw JSONL in, JSONL out, no I/O.
|
||||
*/
|
||||
|
||||
export const DEFAULT_PLACEHOLDER = '~/repo';
|
||||
export const HOME_DIR_PLACEHOLDER = '~';
|
||||
export const REDACTION_PLACEHOLDER = '[redacted]';
|
||||
|
||||
export interface SanitizeOptions {
|
||||
/** Replacement for the cwd-prefix. Its dash-encoded form is derived from it. */
|
||||
placeholder?: string;
|
||||
/** Keyword regexes to redact. Empty by default, leaving a pure path-scrubber. */
|
||||
forbiddenMarkers?: readonly RegExp[];
|
||||
/**
|
||||
* Exact prefix to strip. An inferred one is only the repo root when some cwd sat
|
||||
* there, so callers that know the root pass it here.
|
||||
*/
|
||||
cwdPrefix?: string;
|
||||
/** Several roots at once (a session spanning two checkouts). Wins over `cwdPrefix`. */
|
||||
cwdPrefixes?: readonly string[];
|
||||
/**
|
||||
* Also strip home-rooted paths in the CONTENT: a sandbox-recorded session has a
|
||||
* sandbox `cwd`, so the cwd passes never see the local checkout it still mentions.
|
||||
*/
|
||||
scrubEmbeddedHomePaths?: boolean;
|
||||
}
|
||||
|
||||
export interface SanitizeResult {
|
||||
sanitized: string;
|
||||
prefixStripped: string | null;
|
||||
encodedPrefixStripped: string | null;
|
||||
homeDirStripped: string | null;
|
||||
encodedHomeDirStripped: string | null;
|
||||
embeddedPrefixStripped: string | null;
|
||||
embeddedHomeDirStripped: string | null;
|
||||
/** Replacement count per marker, keyed by the regex's source string. */
|
||||
markersScrubbed: Record<string, number>;
|
||||
}
|
||||
|
||||
/** Longest common prefix by path COMPONENT: `/a/bb` and `/a/b` share `/a`, not `/a/b`.
|
||||
* Returns `''` when only the root `/` is common. */
|
||||
export function findLongestCommonPathPrefix(paths: Iterable<string>): string {
|
||||
const arr = Array.from(paths);
|
||||
if (arr.length === 0) return '';
|
||||
const splits = arr.map((p) => p.split('/'));
|
||||
const minLen = Math.min(...splits.map((s) => s.length));
|
||||
let lastShared = 0;
|
||||
for (let i = 0; i < minLen; i++) {
|
||||
const c = splits[0][i];
|
||||
if (splits.some((s) => s[i] !== c)) break;
|
||||
lastShared = i + 1;
|
||||
}
|
||||
// Only the leading empty piece matched → just the root, not useful.
|
||||
if (lastShared <= 1) return '';
|
||||
return splits[0].slice(0, lastShared).join('/');
|
||||
}
|
||||
|
||||
/** The home-dir portion of an absolute path, or `null` for an unrecognized shape —
|
||||
* better to skip the home pass than strip what may be repo content. */
|
||||
export function extractHomeDir(cwdPrefix: string): string | null {
|
||||
if (!cwdPrefix.startsWith('/')) return null;
|
||||
// Windows-under-WSL shapes first: the generic drive shape below would stop at the
|
||||
// drive letter and leave the account name in. A volume or drive root carries no
|
||||
// identity by itself, so those take the directory under it.
|
||||
const patterns: RegExp[] = [
|
||||
/^\/mnt\/host\/[^/]+\/Users\/[^/]+/,
|
||||
/^\/mnt\/[^/]+\/Users\/[^/]+/,
|
||||
/^\/Users\/[^/]+/,
|
||||
/^\/home\/[^/]+/,
|
||||
/^\/Volumes\/[^/]+\/[^/]+/,
|
||||
/^\/mnt\/[^/]+\/[^/]+/,
|
||||
/^\/var\/root(?=\/|$)/,
|
||||
/^\/root(?=\/|$)/,
|
||||
];
|
||||
for (const re of patterns) {
|
||||
const m = cwdPrefix.match(re);
|
||||
if (m) return m[0];
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Every distinct `cwd` in the transcript. Read at the top level (Claude Code) and
|
||||
* under `payload` (codex), so both harnesses are covered. Bad lines are skipped. */
|
||||
export function collectCwds(raw: string): Set<string> {
|
||||
const out = new Set<string>();
|
||||
const add = (v: unknown) => {
|
||||
if (typeof v === 'string' && v.startsWith('/')) out.add(v);
|
||||
};
|
||||
for (const line of raw.split('\n')) {
|
||||
if (!line.trim()) continue;
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(line);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
if (typeof parsed !== 'object' || parsed === null) continue;
|
||||
const rec = parsed as { cwd?: unknown; payload?: unknown };
|
||||
add(rec.cwd);
|
||||
if (typeof rec.payload === 'object' && rec.payload !== null) {
|
||||
add((rec.payload as { cwd?: unknown }).cwd);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** One path segment: stops at `/`, whitespace, quotes and JSON punctuation. */
|
||||
const COMP = String.raw`[^/\s"'\\,:;)\]}<>]+`;
|
||||
// macOS/Windows display names can contain spaces, but only consume them while
|
||||
// more path follows, so a bare home-dir mention doesn't swallow trailing prose.
|
||||
const USER_WITH_SPACES = `${COMP}(?:(?: +${COMP})+(?=/))?`;
|
||||
const EMBEDDED_HOME_RE = new RegExp(
|
||||
'(?:' +
|
||||
String.raw`\/home\/${COMP}` +
|
||||
'|' +
|
||||
String.raw`\/Users\/${USER_WITH_SPACES}` +
|
||||
'|' +
|
||||
String.raw`\/mnt\/c\/Users\/${USER_WITH_SPACES}` +
|
||||
'|' +
|
||||
// Component boundary, so these don't match inside `/rootfs` or `/root_ca.pem`.
|
||||
String.raw`\/var\/root(?![^/])` +
|
||||
'|' +
|
||||
String.raw`\/root(?![^/])` +
|
||||
')' +
|
||||
String.raw`(?:\/${COMP})*`,
|
||||
'g'
|
||||
);
|
||||
|
||||
export function collectEmbeddedHomePaths(raw: string): Set<string> {
|
||||
const out = new Set<string>();
|
||||
for (const m of raw.matchAll(EMBEDDED_HOME_RE)) out.add(m[0]);
|
||||
return out;
|
||||
}
|
||||
|
||||
function literalReplaceAll(haystack: string, needle: string, replacement: string): string {
|
||||
if (!needle) return haystack;
|
||||
return haystack.split(needle).join(replacement);
|
||||
}
|
||||
|
||||
/** Can `ch` continue a path component? A `.` counts only mid-component, so `…/repo.git`
|
||||
* is one component but `…/repo.` ending a sentence is not. */
|
||||
function continuesComponent(text: string, at: number): boolean {
|
||||
const ch = text[at];
|
||||
if (ch === undefined) return false;
|
||||
if (/[A-Za-z0-9_-]/.test(ch)) return true;
|
||||
return ch === '.' && at + 1 < text.length && /[A-Za-z0-9_-]/.test(text[at + 1]);
|
||||
}
|
||||
|
||||
/** Replace `needle` only where it ends at a component boundary, so stripping `…/wt/repo`
|
||||
* can't turn `…/wt/repo-backup` into `<replacement>-backup`. Skipped ones go to the home pass. */
|
||||
function replacePrefixAtBoundary(haystack: string, needle: string, replacement: string): string {
|
||||
if (!needle) return haystack;
|
||||
let out = '';
|
||||
let from = 0;
|
||||
for (;;) {
|
||||
const i = haystack.indexOf(needle, from);
|
||||
if (i === -1) return out + haystack.slice(from);
|
||||
const end = i + needle.length;
|
||||
out += haystack.slice(from, i) + (continuesComponent(haystack, end) ? needle : replacement);
|
||||
from = end;
|
||||
}
|
||||
}
|
||||
|
||||
/** Replace a prefix and its dash-encoded form (`.claude/projects/<encoded>/`). */
|
||||
function stripBothForms(haystack: string, needle: string, replacement: string): string {
|
||||
const out = literalReplaceAll(haystack, needle, replacement);
|
||||
return literalReplaceAll(out, needle.replace(/\//g, '-'), replacement.replace(/\//g, '-'));
|
||||
}
|
||||
|
||||
export function sanitizeSessionJsonl(raw: string, opts: SanitizeOptions = {}): SanitizeResult {
|
||||
const placeholder = opts.placeholder ?? DEFAULT_PLACEHOLDER;
|
||||
const markers = opts.forbiddenMarkers ?? [];
|
||||
const cwds = collectCwds(raw);
|
||||
let working = raw;
|
||||
let prefixStripped: string | null = null;
|
||||
let encodedPrefixStripped: string | null = null;
|
||||
let homeDirStripped: string | null = null;
|
||||
let encodedHomeDirStripped: string | null = null;
|
||||
let embeddedPrefixStripped: string | null = null;
|
||||
let embeddedHomeDirStripped: string | null = null;
|
||||
|
||||
const requested = opts.cwdPrefixes?.length
|
||||
? [...opts.cwdPrefixes]
|
||||
: opts.cwdPrefix
|
||||
? [opts.cwdPrefix]
|
||||
: cwds.size > 0
|
||||
? [findLongestCommonPathPrefix(cwds)]
|
||||
: [];
|
||||
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
|
||||
const prefixes = [...new Set(requested.filter(Boolean))].sort((a, b) => b.length - a.length);
|
||||
|
||||
// EVERY root before ANY home dir: a home pass run between roots would rewrite a
|
||||
// sibling root's own prefix, leaving it unmatched when its turn came.
|
||||
for (const prefix of prefixes) {
|
||||
const encodedPrefix = prefix.replace(/\//g, '-');
|
||||
working = replacePrefixAtBoundary(working, prefix, placeholder);
|
||||
working = literalReplaceAll(working, encodedPrefix, placeholder.replace(/\//g, '-'));
|
||||
prefixStripped ??= prefix;
|
||||
encodedPrefixStripped ??= encodedPrefix;
|
||||
}
|
||||
// Only catches what is left outside the roots, e.g. `/home/<user>/.claude/projects/`.
|
||||
const homeDirs = new Set(
|
||||
prefixes
|
||||
.map((p) => extractHomeDir(p))
|
||||
.filter((h): h is string => h !== null && !prefixes.includes(h))
|
||||
);
|
||||
for (const homeDir of homeDirs) {
|
||||
const encodedHomeDir = homeDir.replace(/\//g, '-');
|
||||
working = replacePrefixAtBoundary(working, homeDir, HOME_DIR_PLACEHOLDER);
|
||||
working = literalReplaceAll(working, encodedHomeDir, HOME_DIR_PLACEHOLDER.replace(/\//g, '-'));
|
||||
homeDirStripped ??= homeDir;
|
||||
encodedHomeDirStripped ??= encodedHomeDir;
|
||||
}
|
||||
|
||||
if (opts.scrubEmbeddedHomePaths) {
|
||||
const embedded = collectEmbeddedHomePaths(working);
|
||||
if (embedded.size > 0) {
|
||||
// Take each path's own shortest `/repo`-terminated prefix rather than a
|
||||
// common prefix, which mis-collapses when paths diverge above the root.
|
||||
const repoRoots = new Set<string>();
|
||||
const homeDirs = new Set<string>();
|
||||
for (const p of embedded) {
|
||||
const h = extractHomeDir(p);
|
||||
if (h) homeDirs.add(h);
|
||||
const m = p.match(/^(.*?\/repo)(?:\/|$)/);
|
||||
if (m) repoRoots.add(m[1]);
|
||||
}
|
||||
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
|
||||
const sortedRoots = [...repoRoots].sort((a, b) => b.length - a.length);
|
||||
for (const root of sortedRoots) working = stripBothForms(working, root, placeholder);
|
||||
for (const h of homeDirs) working = stripBothForms(working, h, HOME_DIR_PLACEHOLDER);
|
||||
embeddedPrefixStripped = sortedRoots[0] ?? null;
|
||||
embeddedHomeDirStripped = [...homeDirs][0] ?? null;
|
||||
}
|
||||
}
|
||||
|
||||
const markersScrubbed: Record<string, number> = {};
|
||||
for (const re of markers) {
|
||||
let count = 0;
|
||||
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
|
||||
const global = new RegExp(re.source, flags);
|
||||
working = working.replace(global, () => {
|
||||
count++;
|
||||
return REDACTION_PLACEHOLDER;
|
||||
});
|
||||
if (count > 0) markersScrubbed[re.source] = count;
|
||||
}
|
||||
|
||||
return {
|
||||
sanitized: working,
|
||||
prefixStripped,
|
||||
encodedPrefixStripped,
|
||||
homeDirStripped,
|
||||
encodedHomeDirStripped,
|
||||
embeddedPrefixStripped,
|
||||
embeddedHomeDirStripped,
|
||||
markersScrubbed,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
|
||||
// Read stdin as a stream — hooks may not have /dev/stdin available
|
||||
let input = '';
|
||||
process.stdin.setEncoding('utf8');
|
||||
process.stdin.on('data', (chunk) => {
|
||||
input += chunk;
|
||||
});
|
||||
process.stdin.on('end', () => {
|
||||
const { session_id, transcript_path } = JSON.parse(input);
|
||||
|
||||
const dataDir =
|
||||
process.env.RACCOON_SNAPSHOT_DATA ||
|
||||
process.env.CLAUDE_PLUGIN_DATA ||
|
||||
(process.env.CLAUDE_PLUGIN_ROOT && path.join(process.env.CLAUDE_PLUGIN_ROOT, '.data')) ||
|
||||
path.join(process.env.HOME || '/root', '.raccoon', 'snapshot-data');
|
||||
|
||||
fs.mkdirSync(dataDir, { recursive: true });
|
||||
fs.writeFileSync(
|
||||
path.join(dataDir, 'current-session.json'),
|
||||
JSON.stringify({ session_id, transcript_path }, null, 2) + '\n'
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,837 @@
|
||||
/**
|
||||
* snapshot-to-task: Create a harbor task scaffold from a snapshot.
|
||||
*
|
||||
* Usage:
|
||||
* npx tsx scripts/snapshot-to-task.ts --snapshot <dir>
|
||||
*/
|
||||
|
||||
import { execFileSync, execSync } from 'child_process';
|
||||
import {
|
||||
chmodSync,
|
||||
copyFileSync,
|
||||
existsSync,
|
||||
mkdirSync,
|
||||
readFileSync,
|
||||
readdirSync,
|
||||
statSync,
|
||||
writeFileSync,
|
||||
} from 'fs';
|
||||
import { basename, join, resolve } from 'path';
|
||||
import pino from 'pino';
|
||||
import pinoPretty from 'pino-pretty';
|
||||
import yargs from 'yargs';
|
||||
import { hideBin } from 'yargs/helpers';
|
||||
|
||||
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
|
||||
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
|
||||
import { copyTree } from './lib/copy-tree';
|
||||
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
|
||||
|
||||
// --- CLI ---
|
||||
|
||||
const argv = yargs(hideBin(process.argv))
|
||||
.option('snapshot', {
|
||||
type: 'string',
|
||||
describe: 'Path to the snapshot directory',
|
||||
demandOption: true,
|
||||
})
|
||||
.option('json', {
|
||||
type: 'boolean',
|
||||
describe: 'Output structured JSON logs',
|
||||
default: false,
|
||||
})
|
||||
.strict()
|
||||
.help()
|
||||
.parseSync();
|
||||
|
||||
const log = pino(
|
||||
{ name: 'snapshot-to-task', level: 'info' },
|
||||
argv.json
|
||||
? process.stdout
|
||||
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
|
||||
);
|
||||
|
||||
// --- Read snapshot data ---
|
||||
|
||||
const snapshotDir = argv.snapshot;
|
||||
|
||||
if (!existsSync(snapshotDir)) {
|
||||
log.fatal(
|
||||
{ path: snapshotDir },
|
||||
'Snapshot directory not found. Check that the path points to a directory inside explore/snapshots/.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
interface SnapshotMetadata {
|
||||
slug: string;
|
||||
session_uuid: string;
|
||||
/** Absent on snapshots captured before harness selection existed. */
|
||||
harness?: string;
|
||||
original_cwd: string;
|
||||
commit: string | null;
|
||||
branch: string | null;
|
||||
remote_url: string | null;
|
||||
timestamp: string;
|
||||
plugin_version: string;
|
||||
}
|
||||
|
||||
interface Annotation {
|
||||
what_trying: string;
|
||||
what_hoping: string;
|
||||
what_happened: string;
|
||||
[key: string]: string;
|
||||
}
|
||||
|
||||
const metadata = JSON.parse(
|
||||
readFileSync(join(snapshotDir, 'metadata.json'), 'utf8')
|
||||
) as SnapshotMetadata;
|
||||
const annotation = JSON.parse(
|
||||
readFileSync(join(snapshotDir, 'annotation.json'), 'utf8')
|
||||
) as Annotation;
|
||||
|
||||
if (!metadata.slug) {
|
||||
log.fatal(
|
||||
'No slug found in snapshot metadata.json. This snapshot may have been created by an older version of the plugin. Please file a bug.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const slug = metadata.slug;
|
||||
|
||||
// --- Locate harbor infrastructure ---
|
||||
|
||||
function findRepoRoot(): string | null {
|
||||
let dir = process.cwd();
|
||||
while (dir !== resolve(dir, '..')) {
|
||||
if (existsSync(join(dir, 'harbor-tasks'))) return dir;
|
||||
dir = resolve(dir, '..');
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const maybeRepoRoot = findRepoRoot();
|
||||
|
||||
if (!maybeRepoRoot) {
|
||||
log.fatal(
|
||||
"Could not find harbor-tasks/ directory. Make sure you're running this from the toolkit root (the Authoring container). Please file a bug if this persists."
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const repoRoot: string = maybeRepoRoot;
|
||||
|
||||
const harborTasks = join(repoRoot, 'harbor-tasks');
|
||||
const sharedCandidates = [join(harborTasks, 'raccoon-shared'), join(repoRoot, 'task-shared')];
|
||||
const sharedDir = sharedCandidates.find((d) => existsSync(d));
|
||||
const taskDir = join(harborTasks, slug);
|
||||
|
||||
if (existsSync(taskDir)) {
|
||||
log.fatal(
|
||||
{ path: taskDir },
|
||||
`Task directory already exists. To recreate it, delete it first: rm -rf ${taskDir}`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
if (!sharedDir) {
|
||||
log.fatal(
|
||||
'Shared infrastructure (Dockerfile, test.sh, etc.) not found. The toolkit may be corrupted. Please file a bug.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
// --- Detect repo name ---
|
||||
|
||||
interface ToolkitConfig {
|
||||
repo: string;
|
||||
defaultCommit: string;
|
||||
/** The packed kit's release version (git describe at pack time). */
|
||||
version?: string;
|
||||
}
|
||||
|
||||
function readToolkitConfig(): ToolkitConfig | null {
|
||||
const configPath = join(repoRoot, 'toolkit.json');
|
||||
if (!existsSync(configPath)) return null;
|
||||
return JSON.parse(readFileSync(configPath, 'utf8')) as ToolkitConfig;
|
||||
}
|
||||
|
||||
function repoNameFromRemote(remoteUrl: string | null): string | null {
|
||||
if (!remoteUrl) return null;
|
||||
const match = remoteUrl.match(/\/([^/]+?)(?:\.git)?$/);
|
||||
return match ? match[1] : null;
|
||||
}
|
||||
|
||||
function findSubmoduleDir(remoteUrl: string | null): string | null {
|
||||
if (!remoteUrl) return null;
|
||||
const reposDir = join(repoRoot, 'repos');
|
||||
if (!existsSync(reposDir)) return null;
|
||||
|
||||
const normalize = (url: string) =>
|
||||
url
|
||||
.replace(/\.git$/, '')
|
||||
.replace(/^git@github\.com:/, 'https://github.com/')
|
||||
.toLowerCase();
|
||||
|
||||
for (const entry of readdirSync(reposDir)) {
|
||||
const repoPath = join(reposDir, entry, 'repo');
|
||||
if (!existsSync(repoPath)) continue;
|
||||
try {
|
||||
const remote = execSync('git remote get-url origin', {
|
||||
cwd: repoPath,
|
||||
encoding: 'utf8',
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
}).trim();
|
||||
if (normalize(remote) === normalize(remoteUrl)) return entry;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const toolkitConfig = readToolkitConfig();
|
||||
|
||||
// A polyglot toolkit's toolkit.json has repos[] + polyglot:true (no top-level .repo).
|
||||
// Derive which member this task targets from the snapshot's original_cwd basename,
|
||||
// validated against the member list.
|
||||
const polyglotMember = (() => {
|
||||
const cfg = toolkitConfig as { polyglot?: boolean; repos?: Array<{ repo: string }> } | null;
|
||||
if (!cfg?.polyglot || !Array.isArray(cfg.repos)) return null;
|
||||
const base = metadata.original_cwd?.split('/').filter(Boolean).pop() ?? null;
|
||||
const members = cfg.repos.map((r) => r.repo);
|
||||
return base && members.includes(base) ? base : null;
|
||||
})();
|
||||
const repoName =
|
||||
polyglotMember ??
|
||||
toolkitConfig?.repo ??
|
||||
findSubmoduleDir(metadata.remote_url) ??
|
||||
repoNameFromRemote(metadata.remote_url);
|
||||
|
||||
if (!repoName) {
|
||||
log.fatal(
|
||||
'Could not determine repo name. The toolkit may be missing toolkit.json. Please file a bug.'
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const commitShort = metadata.commit ? metadata.commit.slice(0, 9) : 'unknown';
|
||||
const sessionUuid = metadata.session_uuid;
|
||||
|
||||
log.info({ slug, repo: repoName, commit: commitShort }, 'Creating harbor task');
|
||||
|
||||
// --- Create task directory structure ---
|
||||
|
||||
mkdirSync(join(taskDir, 'environment'), { recursive: true });
|
||||
mkdirSync(join(taskDir, 'tests'), { recursive: true });
|
||||
mkdirSync(join(taskDir, 'reference-runs'), { recursive: true });
|
||||
|
||||
// --- Copy shared infrastructure ---
|
||||
|
||||
// The complete grader asset set test.sh depends on: the grader system prompt
|
||||
// and the renderer (test.sh exits without the renderer). Sources missing from
|
||||
// task-shared/ are skipped by the existsSync guard below.
|
||||
const sharedFiles = [
|
||||
{ src: 'test.sh', dest: 'tests/test.sh' },
|
||||
{
|
||||
src: 'grader-system-prompt-consolidated.md',
|
||||
dest: 'tests/grader-system-prompt-consolidated.md',
|
||||
},
|
||||
// test.sh execs this to render the grade; without it the verifier writes no reward
|
||||
// file and the trial errors out rather than scoring.
|
||||
{ src: 'render-grade-consolidated.py', dest: 'tests/render-grade-consolidated.py' },
|
||||
];
|
||||
|
||||
for (const { src, dest } of sharedFiles) {
|
||||
const srcPath = join(sharedDir, src);
|
||||
const destPath = join(taskDir, dest);
|
||||
if (existsSync(srcPath)) {
|
||||
copyFileSync(srcPath, destPath);
|
||||
if (src === 'test.sh') chmodSync(destPath, 0o755);
|
||||
log.debug({ src, dest }, 'Copied shared file');
|
||||
} else {
|
||||
log.warn({ src }, 'Shared file not found');
|
||||
}
|
||||
}
|
||||
|
||||
// Deterministic checks (tests/typecheck/lint). test.sh sources these and hands
|
||||
// their output to the grader as evidence for the CORRECTNESS score, so without
|
||||
// them a code task's correctness is never signal-backed — the grader falls back
|
||||
// to reading the diff alone. Same per-member-then-generic resolution as the
|
||||
// Dockerfile below: a polyglot toolkit ships test-commands.<member>.sh per
|
||||
// member, a single-repo toolkit ships the lone test-commands.sh.
|
||||
const perMemberTestCommands = join(sharedDir, `test-commands.${repoName.toLowerCase()}.sh`);
|
||||
const genericTestCommands = join(sharedDir, 'test-commands.sh');
|
||||
const testCommandsSrc = existsSync(perMemberTestCommands)
|
||||
? perMemberTestCommands
|
||||
: genericTestCommands;
|
||||
if (existsSync(testCommandsSrc)) {
|
||||
const testCommandsDest = join(taskDir, 'tests', 'test-commands.sh');
|
||||
copyFileSync(testCommandsSrc, testCommandsDest);
|
||||
chmodSync(testCommandsDest, 0o755);
|
||||
log.debug({ src: testCommandsSrc }, 'Copied deterministic checks');
|
||||
} else {
|
||||
// Not fatal: the grader still scores correctness by walking the changed code.
|
||||
log.info(
|
||||
'No test-commands.sh for this repo — expected when it has no runnable suite. The grader scores correctness by reading the changed code instead; say so in your holistic rubric.'
|
||||
);
|
||||
}
|
||||
|
||||
// --- Write Dockerfile with session resume support ---
|
||||
//
|
||||
// Read the per-repo task-shared/Dockerfile (Ruby/Postgres/Node for ZenBill,
|
||||
// TS-Node/Postgres/pnpm for Palolo) from the toolkit and append session-
|
||||
// staging COPY/RUN steps. Session staging happens after the original CMD —
|
||||
// COPY and RUN are layer ops independent of CMD, so the original
|
||||
// `CMD ["sleep", "infinity"]` remains active after the appended layers.
|
||||
//
|
||||
// Falls back to a bare debian Dockerfile if no task-shared/Dockerfile is
|
||||
// present (toolkit corruption, or a repo without a per-repo Dockerfile).
|
||||
|
||||
// Polyglot toolkits ship a per-member task-shared/Dockerfile.<member>; a graded task
|
||||
// targets one member, so prefer its Dockerfile. Single-repo toolkits use the lone
|
||||
// task-shared/Dockerfile. Fall back to the generic one if the per-member file is absent.
|
||||
const perMemberDockerfile = join(repoRoot, 'task-shared', `Dockerfile.${repoName.toLowerCase()}`);
|
||||
const taskSharedDockerfile = existsSync(perMemberDockerfile)
|
||||
? perMemberDockerfile
|
||||
: join(repoRoot, 'task-shared', 'Dockerfile');
|
||||
let baseDockerfile: string;
|
||||
if (existsSync(taskSharedDockerfile)) {
|
||||
baseDockerfile = readFileSync(taskSharedDockerfile, 'utf-8');
|
||||
log.debug({ dockerfile: taskSharedDockerfile }, 'Loaded base Dockerfile');
|
||||
} else {
|
||||
log.warn(
|
||||
'task-shared/Dockerfile not found; falling back to bare debian. The harbor task container will lack any language runtime — agents will not be able to execute code in the repo.'
|
||||
);
|
||||
baseDockerfile = `FROM debian:bookworm-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y \\
|
||||
git \\
|
||||
python3 \\
|
||||
curl \\
|
||||
jq \\
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Install Claude Code globally (needed by the grader in test.sh)
|
||||
RUN curl -fsSL https://claude.ai/install.sh | bash && \\
|
||||
cp /root/.claude-code/claude /usr/local/bin/claude 2>/dev/null || \\
|
||||
cp /root/.local/bin/claude /usr/local/bin/claude 2>/dev/null || \\
|
||||
ln -sf $(find /root -name claude -type f 2>/dev/null | head -1) /usr/local/bin/claude
|
||||
|
||||
WORKDIR /workspace
|
||||
COPY workspace/ .
|
||||
|
||||
# Block network tools — agent should only read code and write documents
|
||||
RUN mkdir -p .claude && \\
|
||||
echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > .claude/settings.json
|
||||
|
||||
RUN git init && \\
|
||||
git config user.email "dev@agent" && \\
|
||||
git config user.name "Dev" && \\
|
||||
git add -A && \\
|
||||
git commit -m "initial" --quiet
|
||||
|
||||
CMD ["sleep", "infinity"]
|
||||
`;
|
||||
}
|
||||
|
||||
// Wrapped in toolkit-managed sentinels so check-task-infra reads this as the
|
||||
// toolkit's own append rather than an edit to the Dockerfile.
|
||||
// Only Claude Code produces the sibling session/ directory (subagents, tool results).
|
||||
// A COPY of an empty directory fails the build outright — buildkit does not carry empty
|
||||
// directories in the context, so the layer errors with `"/session": not found`.
|
||||
// Read the SNAPSHOT, not the task dir: the Dockerfile is generated before the session
|
||||
// files are copied into environment/, so the task-side copy is not there yet.
|
||||
const sessionSiblingDir = join(snapshotDir, 'session');
|
||||
const hasSessionSibling =
|
||||
existsSync(sessionSiblingDir) && readdirSync(sessionSiblingDir).length > 0;
|
||||
|
||||
const sessionStaging = `
|
||||
# >>> toolkit-managed: snapshot-session >>>
|
||||
# Stage session files for the snapshot agent adapter to install at runtime.
|
||||
COPY session.jsonl /tmp/snapshot-session/session.jsonl
|
||||
${hasSessionSibling ? 'COPY session/ /tmp/snapshot-session/session/\n' : ''}RUN echo '${sessionUuid}' > /tmp/snapshot-session/uuid.txt
|
||||
# <<< toolkit-managed <<<
|
||||
`;
|
||||
|
||||
const dockerfile = baseDockerfile.trimEnd() + '\n' + sessionStaging;
|
||||
|
||||
writeFileSync(join(taskDir, 'environment', 'Dockerfile'), dockerfile);
|
||||
log.debug('Wrote Dockerfile (per-repo base + session staging)');
|
||||
|
||||
// --- Copy snapshot.patch as workspace.patch ---
|
||||
|
||||
const snapshotPatch = join(snapshotDir, 'snapshot.patch');
|
||||
if (existsSync(snapshotPatch)) {
|
||||
copyFileSync(snapshotPatch, join(taskDir, 'environment', 'workspace.patch'));
|
||||
log.debug('Copied snapshot.patch -> workspace.patch');
|
||||
}
|
||||
|
||||
// --- Scrub the worker's filesystem layout out of the session ---
|
||||
// In Explore the recorded `cwd` is the worker's HOST checkout (explore/repo is an absolute
|
||||
// symlink); rewriting the repo root to /workspace both drops the leak and matches the trial.
|
||||
|
||||
const WORKSPACE_MOUNT = '/workspace';
|
||||
|
||||
const escapeRegExp = (v: string) => v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
|
||||
/** Member names when this toolkit is polyglot; empty means single-repo. */
|
||||
const MEMBER_NAMES: readonly string[] = (() => {
|
||||
const dir = join(repoRoot, 'repos');
|
||||
if (!existsSync(dir)) return [];
|
||||
try {
|
||||
return readdirSync(dir, { withFileTypes: true })
|
||||
.filter((e) => e.isDirectory())
|
||||
.map((e) => e.name);
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
})();
|
||||
|
||||
/** The repo root within a cwd — the prefix a trial mounts at /workspace. `/repos/<member>`
|
||||
* anchors only on a polyglot toolkit, so a personal `~/repos/…` above it can't win. */
|
||||
function repoRootOf(cwd: string): string | null {
|
||||
if (MEMBER_NAMES.length > 0) {
|
||||
// A real member of THIS toolkit wins; the generic shape covers a member whose
|
||||
// directory the toolkit no longer has (an older snapshot, a renamed member).
|
||||
for (const name of MEMBER_NAMES) {
|
||||
const hit = cwd.match(new RegExp(`^(.*?/repos/${escapeRegExp(name)})(?:/|$)`));
|
||||
if (hit) return hit[1];
|
||||
}
|
||||
const generic = cwd.match(/^(.*?\/repos\/[^/]+)(?:\/|$)/);
|
||||
if (generic) return generic[1];
|
||||
}
|
||||
// `/repo` needs a component boundary, so it never matches inside `/repos/`.
|
||||
const m = cwd.match(/^(.*?\/repo)(?:\/|$)/);
|
||||
return m ? m[1] : null;
|
||||
}
|
||||
|
||||
/** Rewrite every checkout root to /workspace, and the home dir each sits under to `~`. The
|
||||
* `repo/` anchor needs no host-root list; the home pass still keys off extractHomeDir. */
|
||||
function scrubWorkerPaths(raw: string): { text: string; roots: string[] } {
|
||||
// Each cwd contributes its own root, longest first, so a nested root isn't clobbered
|
||||
// and a session spanning two checkouts is scrubbed rather than skipped.
|
||||
const roots = [...new Set([...collectCwds(raw)].map(repoRootOf))]
|
||||
.filter((r): r is string => r !== null)
|
||||
.sort((a, b) => b.length - a.length);
|
||||
const { sanitized } = sanitizeSessionJsonl(raw, {
|
||||
cwdPrefixes: roots,
|
||||
placeholder: WORKSPACE_MOUNT,
|
||||
});
|
||||
return { text: sanitized, roots };
|
||||
}
|
||||
|
||||
// --- Copy session files for --resume ---
|
||||
//
|
||||
// The full session.jsonl (including any post-end_turn entries) goes into the
|
||||
// task root for reference. A truncated version — keeping everything up to
|
||||
// and including the last assistant entry with stop_reason="end_turn" — goes
|
||||
// into environment/ for the container. Stopping on a clean assistant turn
|
||||
// avoids Claude Code's synthetic "No response requested." injection when
|
||||
// the session is resumed with --fork-session and a new --print prompt.
|
||||
|
||||
const sessionJsonl = join(snapshotDir, 'session.jsonl');
|
||||
if (existsSync(sessionJsonl)) {
|
||||
// Fail-open: a session this can't scrub ships exactly as it was, because a
|
||||
// leaked path is a smaller problem than a task that can't be created.
|
||||
let sessionText = readFileSync(sessionJsonl, 'utf8');
|
||||
try {
|
||||
const { text, roots } = scrubWorkerPaths(sessionText);
|
||||
if (roots.length > 0) {
|
||||
sessionText = text;
|
||||
log.info(
|
||||
{ roots, mountedAt: WORKSPACE_MOUNT },
|
||||
'Rewrote the authoring checkout path to the trial mount point'
|
||||
);
|
||||
} else {
|
||||
log.debug('No worker-rooted cwd to rewrite; session used as-is');
|
||||
}
|
||||
} catch (err) {
|
||||
log.warn(
|
||||
{ err: err instanceof Error ? err.message : String(err) },
|
||||
'Could not rewrite paths in the session; using it as-is'
|
||||
);
|
||||
}
|
||||
|
||||
// Full version for reference
|
||||
writeFileSync(join(taskDir, 'session-full.jsonl'), sessionText);
|
||||
log.debug('Wrote full session.jsonl to task root');
|
||||
|
||||
// Truncated version for the container: strip everything from the last
|
||||
// user text turn onwards. This drops the failure-eliciting question
|
||||
// (which `--print` will redeliver to the trial agent as the new prompt)
|
||||
// AND the failure response itself (so the trial agent doesn't see its
|
||||
// previous answer), while preserving conversational context up to the
|
||||
// last clean assistant `end_turn`.
|
||||
//
|
||||
// Algorithm (refined Option B):
|
||||
// 1. Find U = index of the last user-text turn that is NOT a slash
|
||||
// command (use the same command-marker filter as
|
||||
// extractLastUserMessage).
|
||||
// 2. Walk backwards from U - 1 to find the last `assistant` entry
|
||||
// with stop_reason: "end_turn".
|
||||
// 3. Truncate slice(0, lastEndTurnIndex + 1).
|
||||
//
|
||||
// If U doesn't exist or no end_turn assistant precedes U, write an
|
||||
// empty session.jsonl — the snapshot agent adapter detects this and
|
||||
// skips --resume entirely, starting fresh from --print.
|
||||
const sessionLines = sessionText.trimEnd().split('\n');
|
||||
|
||||
// A non-Claude session is not a Claude transcript, so the scan below finds no
|
||||
// `stop_reason: "end_turn"` and would silently write an empty session. Its reader
|
||||
// applies the same rule in that harness's own format.
|
||||
const harness = metadata.harness ?? 'claude-code';
|
||||
const isClaude = harness === 'claude-code';
|
||||
|
||||
let lastUserTextIndex = -1;
|
||||
for (let i = 0; i < sessionLines.length; i++) {
|
||||
try {
|
||||
const entry = JSON.parse(sessionLines[i]) as {
|
||||
type?: string;
|
||||
isCompactSummary?: boolean;
|
||||
message?: { content?: unknown };
|
||||
};
|
||||
if (entry.type !== 'user' || typeof entry.message?.content !== 'string') continue;
|
||||
// Compaction summaries are synthetic user turns whose text often quotes
|
||||
// earlier /create-snapshot:snapshot runs — never the command turn itself,
|
||||
// so they must not trip the break below.
|
||||
if (entry.isCompactSummary) continue;
|
||||
const content = entry.message.content;
|
||||
// Mirror extractLastUserMessage: skip the snapshot command itself
|
||||
// and any slash-command / local-command marker turns.
|
||||
if (content.includes('create-snapshot:snapshot')) break;
|
||||
if (
|
||||
content.includes('<command-name>') ||
|
||||
content.includes('<command-message>') ||
|
||||
content.includes('<local-command-caveat>')
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
lastUserTextIndex = i;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
let lastEndTurnIndex = -1;
|
||||
if (lastUserTextIndex > 0) {
|
||||
for (let i = lastUserTextIndex - 1; i >= 0; i--) {
|
||||
try {
|
||||
const entry = JSON.parse(sessionLines[i]) as {
|
||||
type?: string;
|
||||
message?: { stop_reason?: unknown };
|
||||
};
|
||||
if (entry.type === 'assistant' && entry.message?.stop_reason === 'end_turn') {
|
||||
lastEndTurnIndex = i;
|
||||
break;
|
||||
}
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!isClaude) {
|
||||
const cut = truncationIndex(turnsFromLines(harness, sessionLines));
|
||||
const kept = cut >= 0 ? sessionLines.slice(0, cut + 1) : [];
|
||||
const truncated = stripAuthoringScaffolding(harness, kept);
|
||||
writeFileSync(
|
||||
join(taskDir, 'environment', 'session.jsonl'),
|
||||
truncated.length ? truncated.join('\n') + '\n' : ''
|
||||
);
|
||||
log.debug(
|
||||
{ harness, fullLines: sessionLines.length, truncatedLines: truncated.length },
|
||||
'Wrote truncated session.jsonl to environment/ (harness reader)'
|
||||
);
|
||||
} else if (lastEndTurnIndex >= 0) {
|
||||
const truncated = sessionLines.slice(0, lastEndTurnIndex + 1);
|
||||
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), truncated.join('\n') + '\n');
|
||||
log.debug(
|
||||
{ fullLines: sessionLines.length, truncatedLines: truncated.length },
|
||||
'Wrote truncated session.jsonl to environment/ (strips last user turn + failure response, keeps through last clean assistant end_turn)'
|
||||
);
|
||||
} else {
|
||||
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), '');
|
||||
if (lastUserTextIndex < 0) {
|
||||
log.warn(
|
||||
'No user text turn found in session — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
|
||||
);
|
||||
} else {
|
||||
log.warn(
|
||||
'No assistant entry with stop_reason="end_turn" found before the last user turn (one-shot snapshot) — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
const sessionDir = join(snapshotDir, 'session');
|
||||
if (existsSync(sessionDir) && statSync(sessionDir).isDirectory()) {
|
||||
copyTree(sessionDir, join(taskDir, 'environment', 'session'));
|
||||
// Claude Code writes subagent files write-only (--w-------). Fix them so
|
||||
// Harbor's dirhash can read them during environment setup.
|
||||
execSync(`chmod -R +r "${join(taskDir, 'environment', 'session')}"`, { stdio: 'pipe' });
|
||||
log.debug('Copied session/');
|
||||
} else {
|
||||
mkdirSync(join(taskDir, 'environment', 'session'), { recursive: true });
|
||||
}
|
||||
|
||||
// The harness that captured the snapshot; the trial runs this one.
|
||||
const harness =
|
||||
typeof metadata.harness === 'string' && metadata.harness ? metadata.harness : 'claude-code';
|
||||
|
||||
/**
|
||||
* The model and effort this harness defaulted to when the task was authored, recorded
|
||||
* for reference only — nothing reads these back, and a trial still resolves both from
|
||||
* the registry at run time. Best-effort: a task is not worth failing over a note.
|
||||
*/
|
||||
function authoredDefaults(harnessId: string): { model: string; effort: string } | null {
|
||||
try {
|
||||
const resolver = join(repoRoot, 'scripts', 'resolve_harness.py');
|
||||
// Same interpreter search as `_raccoon_python` in scripts/lib/harness-credentials.sh
|
||||
// and `pythonWithTomllib` in submit-task.ts: `python3` is not always 3.11+, and the
|
||||
// registry needs tomllib. Best-effort, so a miss just omits the note.
|
||||
let python = '';
|
||||
for (const candidate of [
|
||||
process.env.RACCOON_PYTHON,
|
||||
'python3',
|
||||
'python3.13',
|
||||
'python3.12',
|
||||
'python3.11',
|
||||
]) {
|
||||
if (!candidate) continue;
|
||||
try {
|
||||
execFileSync(candidate, ['-c', 'import tomllib'], { stdio: 'ignore' });
|
||||
python = candidate;
|
||||
break;
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if (!python) return null;
|
||||
const rows = execFileSync(python, [resolver, '--defaults'], {
|
||||
encoding: 'utf-8',
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
});
|
||||
for (const line of rows.split('\n')) {
|
||||
const [id, model, effort] = line.split('\t');
|
||||
if (id === harnessId && model) return { model, effort: effort ?? '' };
|
||||
}
|
||||
} catch {
|
||||
// registry unreadable here — omit the note
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const authored = authoredDefaults(harness);
|
||||
|
||||
// --- Write task.toml ---
|
||||
|
||||
// The reference-data corpus is included in every zeta task (build-workspace decides from the repo),
|
||||
// so there's nothing to set here.
|
||||
const taskToml = `version = "1.0"
|
||||
|
||||
[metadata]
|
||||
program = "raccoon"
|
||||
author = "rl-env-coding"
|
||||
category = "sdlc/technical-writing"
|
||||
repo = "${repoName}"
|
||||
commit = "${commitShort}"
|
||||
# The toolkit release this task was created with. Written by the toolkit —
|
||||
# leave it in place: task tooling reads it to know which toolkit's assets
|
||||
# this task grades with.
|
||||
toolkit_version = "${toolkitConfig?.version ?? 'unknown'}"
|
||||
snapshot = "${basename(snapshotDir)}"
|
||||
session_uuid = "${sessionUuid}"
|
||||
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
|
||||
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
|
||||
browser = false
|
||||
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
|
||||
|
||||
[verifier]
|
||||
timeout_sec = 7200.0
|
||||
|
||||
[agent]
|
||||
harness = "${harness}"
|
||||
timeout_sec = 18000.0
|
||||
|
||||
[environment]
|
||||
build_timeout_sec = 6000.0
|
||||
cpus = 2
|
||||
memory_mb = 4096
|
||||
storage_mb = 10240
|
||||
gpus = 0
|
||||
allow_internet = true
|
||||
|
||||
[verifier.env]
|
||||
ANTHROPIC_API_KEY = "\${ANTHROPIC_API_KEY}"
|
||||
ANTHROPIC_BASE_URL = "\${ANTHROPIC_BASE_URL}"
|
||||
|
||||
[solution.env]
|
||||
`;
|
||||
|
||||
writeFileSync(join(taskDir, 'task.toml'), taskToml);
|
||||
log.debug('Wrote task.toml');
|
||||
|
||||
// --- Extract instruction from session transcript ---
|
||||
|
||||
function extractLastUserMessage(sessionPath: string, harness: string): string | null {
|
||||
if (!existsSync(sessionPath)) return null;
|
||||
|
||||
const lines = readFileSync(sessionPath, 'utf8').trimEnd().split('\n');
|
||||
|
||||
// A non-Claude session has no `type: "user"` records, so the scan below finds nothing
|
||||
// and the worker silently gets a placeholder instruction. Its reader applies the same
|
||||
// rule — last real user turn, ignoring command invocations — in that harness's format.
|
||||
if (harness !== 'claude-code') {
|
||||
const userTurns = turnsFromLines(harness, lines).filter(
|
||||
(t) => t.role === 'user' && !t.isCommand && t.text.trim()
|
||||
);
|
||||
return userTurns.length ? userTurns[userTurns.length - 1].text : null;
|
||||
}
|
||||
|
||||
let lastUserMessage: string | null = null;
|
||||
|
||||
for (const line of lines) {
|
||||
try {
|
||||
const entry = JSON.parse(line) as {
|
||||
type?: string;
|
||||
isCompactSummary?: boolean;
|
||||
message?: { content?: unknown };
|
||||
};
|
||||
if (entry.type === 'user' && typeof entry.message?.content === 'string') {
|
||||
// Synthetic compaction summary — not a real user turn, and its text
|
||||
// often quotes earlier /create-snapshot:snapshot runs.
|
||||
if (entry.isCompactSummary) continue;
|
||||
const content = entry.message.content;
|
||||
if (content.includes('create-snapshot:snapshot')) break;
|
||||
if (
|
||||
content.includes('<command-name>') ||
|
||||
content.includes('<command-message>') ||
|
||||
content.includes('<local-command-caveat>')
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
lastUserMessage = content;
|
||||
}
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
return lastUserMessage;
|
||||
}
|
||||
|
||||
const lastUserMessage = extractLastUserMessage(
|
||||
join(snapshotDir, 'session.jsonl'),
|
||||
metadata.harness ?? 'claude-code'
|
||||
);
|
||||
|
||||
const instructionHeader =
|
||||
'# Replace this with your refined task instruction\n\n' +
|
||||
"<!-- The text below was auto-extracted from your snapshot's last user message.\n" +
|
||||
' Refine, condense, or rewrite to focus on the behavior you want to elicit. -->\n\n';
|
||||
|
||||
if (lastUserMessage) {
|
||||
writeFileSync(
|
||||
join(taskDir, 'instruction.md'),
|
||||
instructionHeader + lastUserMessage.trimEnd() + '\n'
|
||||
);
|
||||
log.info('Wrote instruction.md (from last user message in session)');
|
||||
} else {
|
||||
writeFileSync(
|
||||
join(taskDir, 'instruction.md'),
|
||||
instructionHeader +
|
||||
'<!-- Could not extract user message from session. Write the instruction manually. -->\n'
|
||||
);
|
||||
log.warn('Could not extract instruction from session — needs manual editing');
|
||||
}
|
||||
|
||||
// --- Scaffold holistic-rubric.md ---
|
||||
|
||||
const holisticRubricMd = `<!--
|
||||
HOLISTIC RUBRIC — the file trials grade against. Run
|
||||
/write-holistic-rubric
|
||||
to draft it interactively, or point Claude Code at this file,
|
||||
session-full.jsonl, and task-shared/grading-standard.md.
|
||||
|
||||
Snapshot: ${basename(snapshotDir)}
|
||||
Session: ${metadata.session_uuid}
|
||||
Repo: ${metadata.remote_url}
|
||||
Commit: ${metadata.commit}
|
||||
|
||||
## What happened in the snapshot conversation
|
||||
|
||||
The worker was trying to: ${annotation.what_trying}
|
||||
They hoped Claude would: ${annotation.what_hoping}
|
||||
Instead, Claude: ${annotation.what_happened}
|
||||
|
||||
## What this file contains
|
||||
|
||||
The eight-criterion Grading Standard
|
||||
(task-shared/grading-standard.md, embedded in
|
||||
tests/grader-system-prompt-consolidated.md) defines Integrity, Narrow
|
||||
Correctness, Broader Correctness / craft, Persistence, Communication,
|
||||
Verification & Thoroughness, Common Sense, and Thought Partnership. This
|
||||
file adds the task-specific knowledge the grader cannot infer: full task
|
||||
context, the ground truth you established, what strong and weak responses
|
||||
look like per criterion, and any dealbreaker penalties — stated as 0.0-1.0
|
||||
fraction subtractions with a named criterion target, never points, never
|
||||
caps. The document must stand alone: the grader sees only it and the
|
||||
shared standard.
|
||||
-->
|
||||
|
||||
<!-- Replace EVERYTHING in this file with the actual holistic rubric,
|
||||
including the instructions above. -->
|
||||
`;
|
||||
|
||||
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
|
||||
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');
|
||||
|
||||
// --- Build workspace ---
|
||||
|
||||
const buildScript = join(repoRoot, 'scripts', 'build-workspace.sh');
|
||||
|
||||
if (existsSync(buildScript)) {
|
||||
log.info({ repo: repoName, commit: commitShort }, 'Building workspace');
|
||||
try {
|
||||
execSync(`bash "${buildScript}" "${slug}" "${commitShort}"`, {
|
||||
cwd: repoRoot,
|
||||
encoding: 'utf8',
|
||||
stdio: 'inherit',
|
||||
// build-workspace does a bulk-file write burst (git archive|tar of the
|
||||
// repo tree + a throwaway git add/commit to apply the patch, and for zeta
|
||||
// toolkits a hardlink-stage of the ~126k-file reference-data corpus that
|
||||
// falls back to a full copy across filesystems). On a slow bind mount
|
||||
// (Docker Desktop non-VirtioFS, or WSL2 with the toolkit on a Windows/9p
|
||||
// path) that legitimately runs into minutes, so a tight cap false-fails a
|
||||
// working-but-slow build as "not runnable". Keep this generous — it's only
|
||||
// a backstop against a true hang; the real Harbor build downstream budgets
|
||||
// build_timeout_sec = 6000.
|
||||
timeout: 1_200_000,
|
||||
});
|
||||
} catch (e: unknown) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
log.fatal({ error: msg }, 'Workspace build failed — task is not runnable');
|
||||
log.fatal(` Retry manually: bash scripts/build-workspace.sh ${slug}`);
|
||||
log.fatal(` Then: scripts/harbor-run harbor-tasks/${slug}`);
|
||||
process.exit(1);
|
||||
}
|
||||
} else {
|
||||
log.fatal('scripts/build-workspace.sh not found. Please file a bug.');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
try {
|
||||
execSync('bash -ic "_ev task_created 2>/dev/null" 2>/dev/null', {
|
||||
stdio: 'ignore',
|
||||
timeout: 5000,
|
||||
});
|
||||
} catch {
|
||||
// best-effort
|
||||
}
|
||||
|
||||
// --- Done ---
|
||||
|
||||
log.info({ taskDir: resolve(taskDir) }, 'Task scaffolded');
|
||||
log.info('Next steps:');
|
||||
log.info(' 1. Review instruction.md');
|
||||
log.info(' 2. Edit tests/holistic-rubric.md — write the rubric');
|
||||
log.info(' 3. Run calibration trials to validate scoring tiers');
|
||||
Reference in New Issue
Block a user