ren worker folder adding orig, mv new one into root

This commit is contained in:
2026-09-25 10:34:29 -04:00
parent 10f0668e32
commit 5b010039d7
1308 changed files with 44597 additions and 1511 deletions

View File

@@ -0,0 +1,156 @@
# Polyglot Explore container for the potion-polyglot toolkit (Potion).
#
# One image hosts every member repo (worker switches with `run-app <repo>`). Runtime union
# across the estate: Node (dominant — 26 members, spanning the Node 14 lambdas to the Node 20
# API), Python (17 — ML pipelines, Flask services, data ETL), Terraform (5), PHP (1).
#
# This image only decides what run-app can BOOT. What makes a member gradable is its
# harbor-tasks/raccoon-shared/Dockerfile.<member>, and a member with no inherited test suite is
# still gradable via the rubric — so a member absent from this image is not "not worth grading".
# Postgres is baked as cheap insurance (no member's verifier requires it).
#
# NOT baked (deliberately):
# - (nothing yet — see the MongoDB note below)
#
# MongoDB IS required, and IS installable here. No member's *verifier* needs it (potion-app is
# jsdom, potion-api's usable suites are sinon-mocked), but `run-app` on potion-app and potion-api
# both do, and those are the two apps a worker is most likely to boot. An earlier note in this
# file claimed MongoDB ships no arm64 debian-bookworm package and skipped it. That is true only of
# MongoDB's *Debian* repo; the **Ubuntu jammy arm64** packages install cleanly on bookworm —
# verified 2026-07-31 on this platform: mongodb-org-server 8.0.28 installs, mongod starts, and a
# write round-trips. Bake it from that repo rather than demoting the estate's flagship app to
# read-only.
# - GPU/CUDA — the potion-ai* members load weights from a now-defunct bucket (never in git),
# so they are read-and-edit here regardless.
# - PHP/MySQL — potion-wp-site's first-party code (its custom theme) IS graded, through its
# own hand-authored harbor image with php-cli + composer; it just doesn't boot in Explore.
#
# Runtimes:
# - Node 14 / 16 / 18 / 20 via nvm (run-app's node selector switches per member)
# - Python 3.10 via uv (agent str_replace_editor needs >=3.10; also the Python members)
# - PostgreSQL baked in
FROM debian:bookworm
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update && apt-get install -y --no-install-recommends \
build-essential git curl ca-certificates gnupg procps sudo xz-utils \
libssl-dev zlib1g-dev \
postgresql postgresql-client \
&& rm -rf /var/lib/apt/lists/*
# --- Node via nvm: 14 / 16 / 18 / 20 (prebuilt). Default 20 symlinked to /usr/local/bin so the
# toolkit's own `node -e` (run-app/welcome read toolkit.json) always works; run-app switches PATH
# per member. yarn into each version. v20.* glob (Docker RUN uses dash; nvm.sh is bash-only). ---
ENV NVM_DIR=/usr/local/nvm
RUN mkdir -p "$NVM_DIR" \
&& curl -fsSL https://raw.githubusercontent.com/nvm-sh/nvm/v0.39.7/install.sh | bash \
&& bash -c '. "$NVM_DIR/nvm.sh" \
&& for v in 14 16 18 20; do nvm install "$v" && nvm use "$v" && npm install -g yarn; done \
&& nvm alias default 20' \
&& for b in node npm npx yarn; do ln -sf "$NVM_DIR"/versions/node/v20.*/bin/"$b" /usr/local/bin/"$b"; done
# --- MongoDB 8.0 (the product DB: potion-app + potion-api both need it to BOOT) ---
# From MongoDB's **Ubuntu jammy** arm64 repo, not the Debian one. MongoDB publishes no arm64
# packages for debian/bookworm (verified: no apt candidate), which is why an earlier revision of
# this image skipped Mongo and left the estate's flagship app unbootable. The jammy arm64 build
# installs and runs fine here — verified on this platform: mongodb-org-server 8.0.28 installs,
# mongod starts, a write round-trips. `mongodb-mongosh` ships the shell so a worker can inspect
# the DB. Data lives in /data/db, created here so mongod can start as root in the sandbox.
RUN curl -fsSL https://pgp.mongodb.com/server-8.0.asc \
| gpg --dearmor -o /usr/share/keyrings/mongodb-8.gpg \
&& echo "deb [ signed-by=/usr/share/keyrings/mongodb-8.gpg ] https://repo.mongodb.org/apt/ubuntu jammy/mongodb-org/8.0 multiverse" \
> /etc/apt/sources.list.d/mongodb-org-8.0.list \
&& apt-get update \
&& apt-get install -y --no-install-recommends mongodb-org-server mongodb-mongosh \
&& rm -rf /var/lib/apt/lists/* \
&& mkdir -p /data/db \
&& mongod --version | head -1
# --- Python via uv ---
# 3.10 stays the default `python3`: it is what this estate's Python members run under.
# 3.11 is installed alongside it because harness setup reads the registry with `tomllib`
# (3.11+), and post-create runs under `set -e` — an image with only 3.10 fails container
# creation. setup-harnesses.sh tries python3, then python3.13/3.12/3.11, so exposing the
# newer one under its versioned name is enough and leaves the members' default untouched.
RUN curl -fsSL https://astral.sh/uv/install.sh | env UV_INSTALL_DIR=/usr/local/bin sh \
&& uv python install 3.10 \
&& ln -sf "$(uv python find 3.10)" /usr/local/bin/python3 \
&& uv python install 3.11 \
&& ln -sf "$(uv python find 3.11)" /usr/local/bin/python3.11 \
&& python3 --version \
&& python3.11 -c "import tomllib; print('tomllib ok on', __import__('sys').version.split()[0])"
# --- PostgreSQL trust auth (OVERWRITE pg_hba; Debian default `local … peer` is first-match) ---
RUN PG_VERSION=$(ls /etc/postgresql) \
&& printf 'local all all trust\nhost all all 127.0.0.1/32 trust\nhost all all ::1/128 trust\nhost all all 0.0.0.0/0 trust\n' > "/etc/postgresql/${PG_VERSION}/main/pg_hba.conf" \
&& echo "listen_addresses='*'" >> "/etc/postgresql/${PG_VERSION}/main/postgresql.conf"
# Startup: start postgres AND mongod. printf, NOT a heredoc (colima's legacy builder writes an
# empty file from a Dockerfile heredoc → ENTRYPOINT "exec format error"). No single quotes in the
# body. mongod is backgrounded with --fork and waited on the same way pg is, so a member's
# setupCmd/startCmd never races an unready DB; its log goes to /var/log/mongod.log for triage.
RUN printf '#!/bin/bash\nset -e\nPG_VERSION=$(ls /etc/postgresql)\nsudo pg_ctlcluster ${PG_VERSION} main start\nuntil pg_isready -h localhost -p 5432 -U postgres >/dev/null 2>&1; do sleep 0.5; done\nmkdir -p /data/db\nmongod --dbpath /data/db --bind_ip 127.0.0.1 --fork --logpath /var/log/mongod.log >/dev/null 2>&1 || echo "warning: mongod failed to start, see /var/log/mongod.log"\nuntil mongosh --quiet --eval "db.runCommand({ping:1})" >/dev/null 2>&1; do sleep 0.5; done\nexec "$@"\n' > /usr/local/bin/start-services.sh \
&& chmod +x /usr/local/bin/start-services.sh
USER root
# --- Playwright + Chromium, for driving the app in a real browser -------------
# Self-contained under /opt — the member's own runtime is untouched.
ENV PLAYWRIGHT_BROWSERS_PATH=/opt/ms-playwright
RUN apt-get update -qq \
&& apt-get install -y -qq --no-install-recommends \
xz-utils \
libxcomposite1 \
libxdamage1 \
libxfixes3 \
libxrandr2 \
libasound2 \
libatk1.0-0 \
libatk-bridge2.0-0 \
libatspi2.0-0 \
libcups2 \
libdbus-1-3 \
libgbm1 \
libnspr4 \
libnss3 \
libxkbcommon0 \
libpango-1.0-0 \
libcairo2 \
libxshmfence1 \
libx11-xcb1 \
libxcb-dri3-0 \
libdrm2 \
&& rm -rf /var/lib/apt/lists/*
RUN set -eux; \
arch="$(dpkg --print-architecture)"; \
case "$arch" in amd64) nodearch=x64;; arm64) nodearch=arm64;; *) echo "unsupported arch: $arch" >&2; exit 1;; esac; \
curl -fsSL "https://nodejs.org/dist/v20.19.5/node-v20.19.5-linux-${nodearch}.tar.xz" -o /tmp/pw-node.tar.xz; \
mkdir -p /opt/pw-node; \
tar -xJf /tmp/pw-node.tar.xz -C /opt/pw-node --strip-components=1; \
rm /tmp/pw-node.tar.xz; \
export npm_config_prefix=/opt/pw-node PATH="/opt/pw-node/bin:$PATH"; \
/opt/pw-node/bin/npm install -g playwright@1.56.0; \
test -d /opt/pw-node/lib/node_modules/playwright; \
/opt/pw-node/bin/node /opt/pw-node/lib/node_modules/playwright/cli.js install chromium
# `pw <script.js>` runs Node with `require("playwright")` resolvable (CommonJS).
RUN printf '#!/bin/sh\nNODE_PATH=/opt/pw-node/lib/node_modules exec /opt/pw-node/bin/node "$@"\n' > /usr/local/bin/pw \
&& chmod +x /usr/local/bin/pw
# Fail the build if Chromium cannot start.
RUN printf 'const{chromium}=require("playwright");(async()=>{const b=await chromium.launch();const p=await b.newPage();await p.setContent("<h1 id=t>ok</h1>");if(await p.textContent("#t")!=="ok")throw new Error("bad render");await b.close();console.log("chromium OK");})()\n' > /tmp/pw-check.js \
&& pw /tmp/pw-check.js \
&& rm -f /tmp/pw-check.js
ENV IS_SANDBOX=1
RUN mkdir -p /root/.claude && echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > /root/.claude/settings.json
WORKDIR /workspace
# Resolver for the DNS jail (.devcontainer/dns-jail-container.sh, applied by
# post-start.sh); if this does not land, Explore just runs unjailed.
RUN (command -v apk >/dev/null 2>&1 && apk add --no-cache dnsmasq bind-tools) \
|| (apt-get update && apt-get install -y --no-install-recommends dnsmasq-base dnsutils \
&& rm -rf /var/lib/apt/lists/*) \
|| true
ENTRYPOINT ["/usr/local/bin/start-services.sh"]
CMD ["sleep", "infinity"]

View File

@@ -0,0 +1,26 @@
{
"name": "Codebase Exploration (potion-polyglot)",
"initializeCommand": "node .devcontainer/initialize.js",
"build": {
"dockerfile": "Dockerfile",
"args": {
"TOOLKIT_BUILD_ID": "1788802488308-63ncdn"
}
},
"appPort": [
"${localEnv:EXPLORE_CLIENT_PORT:4300}:3000"
],
"containerEnv": {
"EXPLORE_INSTANCE": "${localEnv:EXPLORE_INSTANCE:}",
"EXPLORE_CLIENT_PORT": "${localEnv:EXPLORE_CLIENT_PORT:4300}"
},
"remoteUser": "root",
"workspaceMount": "source=${localWorkspaceFolder},target=/workspace,type=bind",
"workspaceFolder": "/workspace",
"mounts": [
"source=${localWorkspaceFolder}/repos${localEnv:EXPLORE_INSTANCE:},target=/workspace/repos,type=bind"
],
"postCreateCommand": "bash /workspace/.devcontainer/post-create.sh",
"postStartCommand": "bash /workspace/.devcontainer/post-start.sh",
"containerUser": "root"
}

View File

@@ -0,0 +1,137 @@
#!/bin/sh
# Restrict this container's DNS to the hosts in DNSJAIL_ALLOW (space-separated), leaving
# every other name unresolvable. Runs as root, inside the container.
#
# Baked into the task images and invoked by the agent (scripts/dnsjail.py); shipped to the
# Explore container by the toolkit packaging. Both surfaces run this same file. Supplied from
# outside: DNSJAIL_ALLOW, the hosts the agent will actually dial -- every one must resolve or
# no jail happens -- and DNSJAIL_ALLOW_EXTRA, nice-to-haves that only warn if they do not.
#
# An unreachable model endpoint is a dead trial or a dead session, so nothing here is
# applied before it is verified, and any doubt leaves the container's DNS untouched.
set -u
STATE=/tmp/.dnsjail
CONTROL=example.com # must NOT resolve through us; proves we reached our own filter
bounded() { if command -v timeout >/dev/null 2>&1; then timeout 5 "$@"; else "$@"; fi; }
# Exact match: docker's own embedded resolver is 127.0.0.11, which a prefix match reads as
# already-jailed — and then resolv.orig is never captured, so unjail has nothing to restore.
jailed_now() { grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null; }
# Stop only the dnsmasq we started, so a declined run leaves nothing bound on :53 that a
# later run could mistake for its own filter.
drop_ours() {
if [ -s "$STATE/dnsmasq.pid" ]; then
pid=$(cat "$STATE/dnsmasq.pid")
# /tmp survives docker stop/start but pids restart at 1, so last boot's pid may now be
# some service's child. Confirm it is dnsmasq before signalling it.
case "$(cat "/proc/$pid/comm" 2>/dev/null)" in
dnsmasq) kill "$pid" 2>/dev/null || true ;;
esac
rm -f "$STATE/dnsmasq.pid" 2>/dev/null || true
fi
}
# Never `exit`: a caller may source this, so bailing out has to fall through rather than
# end the caller's shell.
dnsjail_apply() {
required="${DNSJAIL_ALLOW:-}"
extra="${DNSJAIL_ALLOW_EXTRA:-}"
allow=$(echo $required $extra) # unquoted: collapses to a single-spaced word list
# A blank required list means no model endpoint was found: jailing would strand the agent.
set -- $required
[ $# -gt 0 ] || return 0
# Already jailed by us, with our resolver alive and the same allowlist? Do nothing. Tearing
# down and rebinding :53 races the kernel releasing the socket, and losing that race ends
# in a fail-open restore -- so a second apply (the codex fresh path, rejail, run-app) would
# silently UNjail a working container.
if jailed_now && [ -s "$STATE/dnsmasq.pid" ] &&
[ "$(cat "/proc/$(cat "$STATE/dnsmasq.pid")/comm" 2>/dev/null)" = "dnsmasq" ] &&
[ "$(cat "$STATE/allow" 2>/dev/null)" = "$required" ] &&
[ "$(cat "$STATE/allow-extra" 2>/dev/null)" = "$extra" ]; then
return 0
fi
# The state dir has to work first: it holds what unjail restores, and a failed write here
# is what would otherwise truncate /etc/resolv.conf. Sticky world-writable so run-app,
# running as the container user in Explore, can drop its own lift markers.
mkdir -p "$STATE" 2>/dev/null || return 0
chmod 1777 "$STATE" 2>/dev/null || true
: > "$STATE/.probe" 2>/dev/null || return 0
rm -f "$STATE/.probe" 2>/dev/null || true
# Never forward to ourselves. Re-applying to an already-jailed container would otherwise
# read 127.0.0.1 out of resolv.conf and point dnsmasq at its own socket, blackholing
# every name.
src=/etc/resolv.conf
if jailed_now && [ -s "$STATE/resolv.orig" ]; then src="$STATE/resolv.orig"; fi
up=$(awk '/^nameserver[ \t]+[0-9]+\./{print $2; exit}' "$src" 2>/dev/null)
[ "$up" = "127.0.0.1" ] && up=""
if [ -n "$up" ] && command -v dnsmasq >/dev/null 2>&1; then
srv=""
for h in $allow; do srv="$srv --server=/$h/$up"; done
drop_ours
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
# one would rather than an answer this resolver decided to keep.
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
>/dev/null 2>>"$STATE/dnsmasq.err" || true
fi
# Ask the resolver directly: the model endpoint must answer and the control must not --
# otherwise we are looking at somebody else's resolver, not our filter. Only the FIRST
# host gates the jail: an extra host that CNAMEs outside the allowlist cannot resolve
# through the catch-all, and one of those must not silently disable the whole jail.
live=1
for h in $required; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 || { live=""; break; }
done
if [ -n "$live" ] && bounded nslookup "$CONTROL" 127.0.0.1 >/dev/null 2>&1; then live=""; fi
# The extras are reported, never fatal: one that CNAMEs outside the allowlist cannot
# resolve through the catch-all, and must not take the whole jail down with it.
if [ -n "$live" ]; then
for h in $extra; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 ||
echo "dns-jail: $h does not resolve through the jail (CNAME outside the allowlist?)" >&2
done
fi
if [ -z "$live" ]; then
# Say why. A silent decline is indistinguishable from a jail that worked, and the
# reason is usually one line from dnsmasq (gVisor sandboxes, for instance, have no
# AF_NETLINK, so dnsmasq cannot start there at all).
echo "dns-jail: declined, this container keeps normal network access${DNSJAIL_WHY:-}" >&2
[ -s "$STATE/dnsmasq.err" ] && sed 's/^/dns-jail: /' "$STATE/dnsmasq.err" >&2
drop_ours
# Failing open has to mean actually open, including when an earlier run left this
# container jailed.
if jailed_now && [ -s "$STATE/resolv.orig" ]; then
cat "$STATE/resolv.orig" > /etc/resolv.conf 2>/dev/null || true
fi
return 0
fi
# Capture what unjail restores — but never overwrite it with an already-jailed file, which
# would leave unjail a permanent no-op.
if ! jailed_now; then
cp /etc/resolv.conf "$STATE/resolv.orig" 2>/dev/null || return 0
fi
printf '%s\n' "$required" > "$STATE/allow" 2>/dev/null || true
printf '%s\n' "$extra" > "$STATE/allow-extra" 2>/dev/null || true
# A marker from a run-app that was killed would otherwise keep the jail disarmed forever.
rm -rf "$STATE/lifts" 2>/dev/null || true
# /etc/resolv.conf is a bind mount, so it is truncated in place, never renamed over —
# which means the replacement has to be complete BEFORE the write starts. Keep every
# non-nameserver directive docker set (options, search).
{ printf 'nameserver 127.0.0.1\n'
grep -vE '^[[:space:]]*nameserver' /etc/resolv.conf
} > "$STATE/resolv.jailed" 2>/dev/null
[ -s "$STATE/resolv.jailed" ] || return 0
cat "$STATE/resolv.jailed" > /etc/resolv.conf
}
dnsjail_apply || true

View File

@@ -0,0 +1,78 @@
#!/bin/bash
# Apply the DNS jail to this Explore container, and install `unjail` / `rejail`.
#
# Explore is meant to behave like a trial: the session captured here becomes the trial's
# seed, so an agent that reached the network here would produce a snapshot the trial
# cannot reproduce. Same jail, applied every boot (docker remounts /etc/resolv.conf per
# start, so it cannot be baked into the image).
#
# Live resolution only — no address pinning. An Explore container can run for days, so a
# resolved-at-boot address has far longer to go stale than in a single trial.
set -u
JAIL_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
STATE=/tmp/.dnsjail
[ "${RACCOON_DNS_JAIL:-0}" = "1" ] || exit 0
# Only the model endpoint gates the jail. The toolkit's telemetry hosts go in as extras
# (below): those sends are backgrounded and disowned, so one failing to resolve would fail
# silently rather than visibly -- and must not take the whole jail down with it.
allow_hosts() {
local url="${ANTHROPIC_BASE_URL:-}" host=""
[ -n "$url" ] || return 1
host="${url#*://}"; host="${host%%/*}"; host="${host##*@}"; host="${host%%:*}"
[ -n "$host" ] || return 1
case "$host" in *[!A-Za-z0-9.-]* | -* | .* | *.) return 1 ;; esac
printf '%s' "$host"
}
install_helpers() {
sudo tee /usr/local/bin/unjail >/dev/null <<'EOF'
#!/bin/sh
# Restore this container's DNS. The jail comes back on the next container start, or now
# with `rejail`. Package installs need this; run-app does it for you around its own.
[ -f /tmp/.dnsjail/resolv.orig ] || { echo "unjail: not jailed"; exit 0; }
sudo sh -c 'cat /tmp/.dnsjail/resolv.orig > /etc/resolv.conf'
echo "unjail: DNS restored — run 'rejail' when you are done, or restart the container."
EOF
sudo tee /usr/local/bin/rejail >/dev/null <<EOF
#!/bin/sh
[ -f /tmp/.dnsjail/allow ] || { echo "rejail: nothing to restore"; exit 1; }
sudo env DNSJAIL_ALLOW="\$(cat /tmp/.dnsjail/allow)" \
DNSJAIL_ALLOW_EXTRA="\$(cat /tmp/.dnsjail/allow-extra 2>/dev/null)" \
sh $JAIL_DIR/dns-jail-container.sh
grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf && echo "rejail: jailed" || echo "rejail: could not jail — left as is"
EOF
sudo chmod +x /usr/local/bin/unjail /usr/local/bin/rejail
}
# Not fatal: an Explore container that cannot jail is still a usable Explore container.
dnsjail_off() {
mkdir -p "$STATE" 2>/dev/null || true
printf '%s\n' "$1" > "$STATE/why" 2>/dev/null || true
echo "dns-jail: off for this session — normal network access. Not an error."
exit 0
}
[ -f "$JAIL_DIR/dns-jail-container.sh" ] || dnsjail_off "script not present: $JAIL_DIR/dns-jail-container.sh"
# Jailing without the model endpoint on the allowlist would strand the agent, so a
# missing or unusable ANTHROPIC_BASE_URL means no jail at all.
ALLOW="$(allow_hosts)" || dnsjail_off "no usable host in ANTHROPIC_BASE_URL: ${ANTHROPIC_BASE_URL:-<unset>}"
# Parent domains for the telemetry, not the exact endpoints: both CNAME within their own
# domain, and the catch-all would NXDOMAIN a chain target that is not itself allowed.
sudo env DNSJAIL_ALLOW="$ALLOW" \
DNSJAIL_ALLOW_EXTRA="amplitude.com datadoghq.com ${RACCOON_DNS_JAIL_ALLOW:-}" \
sh "$JAIL_DIR/dns-jail-container.sh" || true
install_helpers
# Report what the script decided, rather than re-probing: it already verified the model
# endpoint against its own resolver and failed open if that did not hold. A second probe
# here has to pick a control host -- and any host the worker allowlists makes that control
# resolve, reading a working jail as a broken one and tearing it down.
if grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf; then
echo "dns-jail: DNS limited to the model endpoint and toolkit telemetry."
echo " Installing packages? \`unjail\` (then \`rejail\`). run-app handles its own."
else
dnsjail_off "the jail did not take; see $STATE/dnsmasq.err if present"
fi

View File

@@ -0,0 +1,195 @@
#!/usr/bin/env node
// Runs on the HOST before the container starts.
// Validates prerequisites and sets up files that the container needs
// without using ../ bind mounts (which break on newer Docker runtimes).
//
// This is Node.js (not bash) so it works on Windows without WSL.
import { execSync } from 'node:child_process';
import fs from 'node:fs';
import path from 'node:path';
// The devcontainer CLI runs initializeCommand from the workspace folder (explore/).
// Use CWD, not __dirname, so this works both in production and in tests.
if (!fs.existsSync('../.env')) {
console.error(`
❌ Missing .env file. Create it first:
Create a file named .env in the toolkit root with:
ANTHROPIC_API_KEY=your-key-here
ANTHROPIC_BASE_URL=the-base-url-you-were-given
`);
process.exit(1);
}
// Copy small files from toolkit root into explore/ so the container
// can access them without ../ bind mounts.
fs.copyFileSync('../.env', '.env');
try {
fs.copyFileSync('../toolkit.json', 'toolkit.json');
} catch {}
// Link repo so the bind mount source stays within explore/.
// Use a junction on Windows (Docker Desktop can't follow symlinks,
// but it can follow junctions). On macOS/Linux, 'junction' is ignored
// and creates a regular symlink.
// Single-repo toolkits have ../repo; polyglot toolkits have ../repos (the member
// clones) instead. Link whichever exists so the matching bind mount resolves.
// Reference-data corpus lives at ../data (zeta toolkits only).
// Remove a link WITHOUT following it: unlink covers POSIX symlinks, rmdir covers
// Windows junctions (which reject unlink). Never recursive — the target is real data.
function removeLink(name) {
try {
fs.unlinkSync(name);
} catch {
fs.rmdirSync(name);
}
}
// Replace a stale entry (a link to a path that no longer exists, an empty dir) rather
// than skipping — skipping left the bind mount resolving to nothing, unfixably.
function linkSibling(name) {
const target = path.resolve('..', name);
if (!fs.existsSync(target)) return;
let current = null;
try {
current = fs.lstatSync(name);
} catch {}
if (current) {
if (current.isSymbolicLink()) {
if (fs.existsSync(name) && fs.realpathSync(name) === fs.realpathSync(target)) return;
removeLink(name);
} else if (current.isDirectory()) {
if (fs.readdirSync(name).length > 0) {
console.error(`⚠️ explore/${name} is a non-empty directory, so it was left as is.`);
console.error(
` Expected a link to the toolkit root's ${name}/. Remove it and re-run 'up'.`
);
return;
}
fs.rmdirSync(name);
} else {
return;
}
}
fs.symlinkSync(target, name, 'junction');
}
linkSibling('repo');
linkSibling('repos');
linkSibling('data');
// A named extra instance (EXPLORE_INSTANCE set, normally by instance.js) gets
// its OWN repo working tree, mounted at /workspace/repo in that container, so a
// `git checkout` in one instance doesn't disturb another. A `git clone --local`
// hardlinks the object store, so this is cheap and fully self-contained — unlike
// a git worktree, whose gitdir lives inside the source repo and so wouldn't
// bind-mount into the container. The devcontainer.json mount derives the dir
// name from EXPLORE_INSTANCE (repo<instance>); create it before that mount binds.
// post-create.sh then checks out the default commit + runs setup in the new
// container, exactly as it does for the primary repo.
const instance = process.env.EXPLORE_INSTANCE || '';
if (instance) {
try {
if (fs.existsSync('../repos')) {
// Polyglot toolkit: give the instance its OWN copy of every member repo at
// repos<instance>/<member>, mounted at /workspace/repos. A git clone --local
// hardlinks each member's object store, so this is cheap and fully isolated —
// a member checkout in one instance never disturbs another.
const dir = `repos${instance}`;
if (!fs.existsSync(dir)) {
fs.mkdirSync(dir, { recursive: true });
for (const member of fs.readdirSync(path.resolve('../repos'))) {
const src = path.resolve('../repos', member);
if (!fs.statSync(src).isDirectory()) continue;
execSync(
`git clone --local ${JSON.stringify(src)} ${JSON.stringify(path.join(dir, member))}`,
{
stdio: 'inherit',
}
);
}
}
} else {
// Single-repo toolkit: clone repo → repo<instance>, mounted at /workspace/repo.
const dir = `repo${instance}`;
if (!fs.existsSync(dir)) {
execSync(
`git clone --local ${JSON.stringify(path.resolve('../repo'))} ${JSON.stringify(dir)}`,
{
stdio: 'inherit',
}
);
}
}
} catch {
console.error(
`\n❌ Couldn't create the repo working tree for instance "${instance}".\n` +
` This needs git on your PATH. Install git, then retry.\n`
);
process.exit(1);
}
}
// Best-effort: warn if an existing container for this folder doesn't publish
// the app ports. Docker fixes -p mappings when a container is CREATED, so a
// container built by an older toolkit (before/with different appPort) keeps its
// old mappings even when you re-run `up`. The only way to pick up new ports is
// to recreate the container — so we point that out here rather than letting the
// worker stare at a dead localhost. Wrapped so it can never block startup: any
// failure (docker missing, odd output) is swallowed and the check is skipped.
//
// Skipped for named instances: they're managed by instance.js (their own ports,
// and they carry an id-label instead of this folder's local_folder label), so
// this folder-scoped check would only ever inspect the primary container.
if (!instance)
try {
// Container ports we expect published. The browsable port is 3000 for every
// repo; Palolo also serves its API on 3001; zeta toolkits serve the corpus
// viewer on 3002. Read from toolkit.json when available, else assume the base pair.
let expected = [3000, 3001];
try {
const tk = JSON.parse(fs.readFileSync('toolkit.json', 'utf-8'));
expected = tk.explorePorts && tk.explorePorts.serverHost ? [3000, 3001] : [3000];
if (tk.explorePorts && tk.explorePorts.corpusHost) expected.push(3002);
} catch {}
const folder = process.cwd();
const ids = execSync(`docker ps -aq --filter "label=devcontainer.local_folder=${folder}"`, {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
})
.trim()
.split('\n')
.filter(Boolean);
for (const id of ids) {
const bindings = execSync(
`docker inspect --format "{{json .HostConfig.PortBindings}}" ${id}`,
{
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
}
).trim();
const missing = expected.filter((p) => !bindings.includes(`${p}/tcp`));
if (missing.length > 0) {
console.error(`
⚠️ An existing container for this folder doesn't publish port(s) ${missing.join(', ')}.
Docker fixes port mappings when a container is created, so re-running 'up'
alone won't add them. To expose the app, recreate the container:
npx @devcontainers/cli up --remove-existing-container
Note: recreating wipes the container's Claude history — run /create-snapshot
first if there's a conversation you want to keep.
`);
break;
}
}
} catch {
// docker unavailable or unexpected output — skip the check.
}

View File

@@ -0,0 +1,438 @@
#!/bin/bash
# Post-create setup for the Explore devcontainer.
set -euo pipefail
# Install every harness a worker can author with, and point each at the LLM proxy.
# Driven by scripts/harness-registry.toml, so adding a harness is a registry entry
# rather than an edit here and in the sibling container's post-create.
set -a; . /workspace/.env 2>/dev/null || true; set +a
. /workspace/scripts/setup-harnesses.sh
# Explore is where capture happens, so it is the only surface that gets the capture
# hooks — their commands ship in explore/plugins/.
RACCOON_SURFACE=explore harness_setup_all
# Allow git operations on bind-mounted repo (owned by different uid on host)
git config --global --add safe.directory '*'
# Check out the default commit from toolkit.json. SINGLE-REPO ONLY: a polyglot toolkit
# has no single /workspace/repo and no top-level defaultCommit — each member repo lives
# at /workspace/repos/<slug> and is checked out + set up lazily by run-app/setup_repo.
IS_POLYGLOT=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').polyglot?'1':'')}catch{}" 2>/dev/null || true)
if [ -z "$IS_POLYGLOT" ]; then
DEFAULT_COMMIT=$(node -e "process.stdout.write(require('/workspace/toolkit.json').defaultCommit)")
git -C /workspace/repo -c advice.detachedHead=false checkout "$DEFAULT_COMMIT"
fi
# Install repo-specific runtime deps against the live-mounted /workspace/repo.
# Bringing postgres up (and creating the role/db) lives in post-start.sh so it
# also runs on every later container start, not just first create; call it here
# so the database is ready before db:create / prisma migrate runs below.
REPO_NAME=$(node -e "process.stdout.write(require('/workspace/toolkit.json').repo)" 2>/dev/null || true)
bash /workspace/.devcontainer/post-start.sh
# Symlink ./node_modules (cwd = the dir being installed) to a container-local tree keyed by
# <key> — see the call sites below for why. The target must itself be named `node_modules`
# (Node resolves the symlink, then walks ancestors for that literal name), and its parent
# needs a stub manifest: postinstall scripts that locate the project by truncating their
# realpath at `node_modules` require() `<parent>/package.json`, and die without it.
_nm_link() {
local root="/opt/raccoon-node-modules/$1"
[ -L node_modules ] || rm -rf node_modules
mkdir -p "$root/node_modules"
[ -f "$root/package.json" ] \
|| printf '{"name":"raccoon-node-modules-root","version":"0.0.0","private":true}\n' > "$root/package.json"
ln -sfn "$root/node_modules" node_modules
}
case "$REPO_NAME" in
ZenBill-006)
# Install deps + create databases
#
# node_modules goes to a CONTAINER-LOCAL path, not the bind-mounted repo dir.
# On macOS Docker Desktop the repo is a host bind mount; writing yarn's huge,
# deeply-nested node_modules tree across the file-sharing layer exhausts the
# host open-file table -> ENFILE "file table overflow", failing the install.
# Keeping node_modules inside the Linux VM confines that churn to the VM; the
# repo stays bind-mounted (worker sees edits) and node_modules is a symlink.
# (ZenBill is yarn-classic with a single root node_modules, so one symlink
# relocates the whole tree cleanly — unlike Palolo's pnpm workspace, which
# uses copy mode instead.)
#
# The symlink TARGET must itself be named `node_modules`: Node resolves the
# symlink to its real path, then walks ancestors looking for a dir literally
# named node_modules. If the target were .../zeta-<x> (not node_modules),
# child processes spawned by postinstall scripts (e.g. cypress's `node
# index.js` requiring minimist) can't resolve hoisted deps -> MODULE_NOT_FOUND.
( cd /workspace/repo \
&& cp .env.sample .env 2>/dev/null \
&& sed -i "s/^ruby '3\.1\.2'/ruby '~> 3.1.0'/" Gemfile \
&& rm -f .ruby-version \
&& bundle install \
&& _nm_link zenbill-006 \
&& yarn install --ignore-engines \
&& bundle update jwt \
&& (bundle exec rails db:create db:migrate || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:migrate || true) )
;;
zeta-heimdall)
# API-only Rails 7; Postgres-only; no JS runtime needed. config/database.yml
# and .env are gitignored, so materialize them from the committed .example
# files. The base image is the exact pinned Ruby (3.2.1), so the Gemfile's
# ruby pin needs no loosening. --full-index works around stale-lockfile
# transitive deps (the masked repo's lockfile omits a few). db:prepare loads
# db/schema.rb into the dev DB; the test DB is created + loaded too (rspec's
# maintain_test_schema! reloads it on first run).
( cd /workspace/repo \
&& cp config/database.yml.example config/database.yml 2>/dev/null \
&& cp .env.example .env 2>/dev/null \
&& bundle install --full-index \
&& (bundle exec rails db:prepare || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:schema:load || true) )
;;
zeta-platform)
# Rails 5.1 / Ruby 2.6.6 banking monorepo; Postgres + Redis. config/database.yml
# is committed (only .env is gitignored → copy from .env.example for dotenv).
# Bundler 1.17.3 matches the lockfile (installed in the image), and the base is
# the exact pinned Ruby (2.6.6), so no Gemfile loosening. db:schema:load loads
# db/schema.rb into the dev + test DBs.
( cd /workspace/repo \
&& cp .env.example .env 2>/dev/null \
&& bundle install \
&& (bundle exec rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:schema:load || true) )
# React client (Create React App, react-scripts 2.1.1). Install its JS deps so
# `run-app` can boot the full UI (dev server proxies /graphql → the Rails API).
# node_modules goes to a CONTAINER-LOCAL path, not the bind-mounted repo dir:
# on macOS Docker Desktop the repo is a host bind mount, and writing CRA's huge
# node_modules tree across the file-sharing layer is slow AND exhausts the host's
# open-file table. Keeping it inside the Linux VM confines that churn; the repo
# stays bind-mounted (worker sees edits) and node_modules is a symlink. The
# symlink TARGET must itself be named `node_modules` (Node's resolver walks
# parents looking for a dir literally named node_modules). yarn is v1 (classic),
# matching the committed yarn.lock.
( cd /workspace/repo \
&& _nm_link zeta-platform \
&& yarn install --frozen-lockfile )
;;
Palolo-031)
# Install deps. packages/server/scripts/prisma greps `.env` for
# PUBLIC_PALOLO_ENV inside an `if [ -t 0 ]` block — designed for
# interactive use where the dev's local .env points at staging/prod
# and the script wants confirmation before destructive ops. In a
# fresh clone the file doesn't exist, so the grep fails and `set -e`
# aborts. We materialize a `local`-pointing stub so the script
# finds what it expects, the safety check skips correctly (env is
# local, no confirmation needed), and downstream interactive worker
# invocations of `pnpm run prisma …` also succeed instead of hitting
# the same failure.
( cd /workspace/repo \
&& git config core.hooksPath /dev/null \
&& echo "PUBLIC_PALOLO_ENV=local" > packages/server/.env \
&& pnpm install --frozen-lockfile \
&& pnpm run --dir packages/server prisma generate \
&& (pnpm run --dir packages/server prisma migrate deploy || true) )
# Seed the dev DB with a superuser, the global/superuser orgs, and a set
# of test users so a worker can actually log in when running the app
# locally. Without this the schema exists but every table is empty, and
# the login screen errors out before you can get into the app. Test
# users are <name>@exhalefi.com with password "test" (e.g. zaniyah@exhalefi.com).
# Convenience only — wrapped in `|| true` so a seed hiccup never blocks
# the explore container from coming up.
#
# `--small` keeps every organization the seed builds but caps each at 10
# members per status. The default size gives the last one 200 per status,
# which opens 200 concurrent Prisma interactive transactions and exhausts
# the connection pool (`P2028`) on a machine with few cores, so the seed
# dies partway and leaves perks un-activated.
( cd /workspace/repo/packages/server \
&& DEFAULT_BAAS_PROVIDER=Liquid PUBLIC_BAAS_ENABLED=yes TESTING_SEED=yes \
pnpm run seed --small ) || true
# Leave a fresh container's `git status` clean. The two artifacts below
# are side effects of bootstrap, not edits anyone made:
#
# 1. .pnpm-store/ — pnpm's content-addressable store. It must sit on the
# same filesystem as node_modules to hardlink; /workspace/repo is a
# bind mount on a different fs than HOME, so pnpm can't use the global
# ~/.pnpm-store and drops a project-local store instead. The repo's
# .gitignore covers it as of commit 3af4366a6, but older commits a
# worker may check out don't. Exclude it locally too (idempotent;
# the create-snapshot checkpoint hook excludes it as well).
# 2. deploy_to_eks.sh — the repo's only symlink (-> ../scripts/...). The
# toolkit's zip/unzip packaging path materializes it as a regular file,
# so git reports a "typechange". Restore the symlink from the index
# (no-op if the filesystem can't represent symlinks).
grep -qxF '.pnpm-store/' /workspace/repo/.git/info/exclude 2>/dev/null \
|| printf '\n# raccoon-explore: in-repo pnpm store (bind-mount hardlink fallback)\n.pnpm-store/\n' >> /workspace/repo/.git/info/exclude
git -C /workspace/repo checkout -- provisioning/kubernetes/palolo-app/deploy_to_eks.sh 2>/dev/null || true
;;
human-essentials)
# Rails 8 / Ruby 3.4; pure importmap (no JS bundler → no node_modules). The
# base image is exact Ruby 3.4.3, so no Gemfile loosening. .env is gitignored;
# copy the committed .env.example (public reCAPTCHA test keys etc.) for dotenv,
# then drop its empty PG_USERNAME/PG_PASSWORD lines so they don't override the
# image ENV (PG_USERNAME=postgres). db:schema:load loads db/schema.rb into the
# dev + test DBs; assets:precompile is needed by the Cuprite system specs.
# db:seed (dev, offline via Faker) gives a working login out of the box — the app
# has no usable self-service signup (a fresh user lands org-less/role-less).
( cd /workspace/repo \
&& cp .env.example .env 2>/dev/null || true; \
sed -i '/^PG_USERNAME=/d; /^PG_PASSWORD=/d' .env 2>/dev/null || true; \
bundle install \
&& (bundle exec rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:schema:load || true) \
&& (bundle exec rails db:seed || true) \
&& (bundle exec rails assets:precompile || true) )
;;
endsideout)
# Rails 8.1 / Ruby 4.0; SQLite + importmap (no Node build — tailwindcss-rails
# ships its own binary). No .env (no .env.example; tests need no secrets). The
# SQLite dev + test DBs are plain files created by db:prepare / db:test:prepare.
# db:seed (dev, offline) creates admin@example.com / password — there is no
# self-service signup route, so seeding is the only way into the UI.
# tailwindcss:build writes the gitignored app/assets/builds/ the layout links;
# run-app starts the server alone, without Procfile.dev's tailwindcss:watch.
( cd /workspace/repo \
&& bundle install \
&& (bin/rails db:prepare || true) \
&& (bin/rails db:test:prepare || true) \
&& (bin/rails db:seed || true) \
&& (bin/rails tailwindcss:build || true) )
;;
community-foundation)
# Rails 8.1 / Ruby 4.0; SQLite + importmap + tailwind (no Node). Encrypted
# credentials aren't needed for tests. SQLite dev + test DBs.
# db:seed (dev, offline) creates the 'arlington' tenant + owner@example.com /
# password. Self-signup is a dead end here (needs a pre-existing org + a working
# mailer for confirmation), so seeding is the only offline way into the UI. The
# app is subdomain-multi-tenant — reach the tenant at arlington.lvh.me, not plain
# localhost (see welcome.sh).
# tailwindcss:build writes the gitignored app/assets/builds/ the layout links;
# run-app starts the server alone, without Procfile.dev's tailwindcss:watch.
( cd /workspace/repo \
&& bundle install \
&& (bin/rails db:prepare || true) \
&& (bin/rails db:test:prepare || true) \
&& (bin/rails db:seed || true) \
&& (bin/rails tailwindcss:build || true) )
;;
stocks-in-the-future)
# Rails 8.1 / Ruby 3.4.4; Postgres + Redis; importmap (no Node build).
# config/database.yml is gitignored — materialize from the committed sample.
# PGHOST/PGUSER (set in the image) point rails at the postgres superuser.
# db:seed (dev, offline) creates login-by-username accounts (Admin / password);
# self-signup is disabled (GET /users/sign_up redirects to /), so seed to get in.
# tailwindcss:build writes the gitignored app/assets/builds/ the layout links;
# run-app starts the server alone, without Procfile.dev's tailwindcss:watch.
( cd /workspace/repo \
&& (cp config/database.yml.sample config/database.yml 2>/dev/null || true) \
&& bundle install \
&& (bin/rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bin/rails db:create db:schema:load || true) \
&& (bin/rails db:seed || true) \
&& (bin/rails tailwindcss:build || true) )
;;
casa)
# Rails 8.0 / Ruby 4.0.3; Postgres + Node 24 (jsbundling: esbuild + sass).
# DB env (POSTGRES_USER/DATABASE_HOST/POSTGRES_PASSWORD) is pinned in the image.
# npm ci installs JS deps; `npm run build` + `build:css` (esbuild + sass) write the
# bundles to app/assets/builds. The Selenium system specs serve from there because
# the test env runs with config.assets.compile=true (Sprockets compiles on demand).
# Deliberately NOT `assets:precompile`: that fingerprints untracked copies into
# public/assets which the specs don't need and which make `npm run lint` (standard)
# report ~197k errors over machine-generated bundles. app/assets/builds is already in
# standard's ignore list, so the dev build leaves the tree lint-clean and faithful.
# db:seed (dev, offline via Faker + local logo) creates casa_admin1@example.com /
# 12345678 — users are admin-invited only (ADR 0002), so seeding is the way in.
( cd /workspace/repo \
&& (cp .env.example .env 2>/dev/null || true) \
&& bundle install \
&& npm ci \
&& (bin/rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bin/rails db:create db:schema:load || true) \
&& (bin/rails db:seed || true) \
&& (npm run build && npm run build:css || true) )
;;
awbw)
# Rails 8.1 / Ruby 4.0.1; MySQL 8 (Percona, Trilogy) + Node 22 (Vite). .env from
# .env.sample; DATABASE_URL (image) points Trilogy at 127.0.0.1 root. npm ci + a
# test-mode Vite build for the Selenium system specs.
# Use db:schema:load (NOT migrate): the committed schema.rb is clean native-MySQL-8
# JSON; running migrate re-dumps schema.rb from the live DB (which corrupts it under
# a non-MySQL-8 engine). tz tables are loaded by post-start.sh (Ahoy charts need them).
# db:seed (dev, offline; the seed disables mailer delivery itself) creates the
# pre-confirmed umberto.user@example.com / password super_user — no self-service
# signup exists and :confirmable would block a hand-made user without a mailer.
( cd /workspace/repo \
&& (cp .env.sample .env 2>/dev/null || true) \
&& bundle install \
&& npm ci \
&& (bin/vite build --mode test || true) \
&& (bin/rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bin/rails db:create db:schema:load || true) \
&& (bin/rails db:seed || true) )
;;
alongwithyou)
# Rails 8.1 / Ruby 4.0.5; SQLite + importmap (no app-side Node). No .env / credentials
# needed to boot. This is a young app (a fresh scaffold with no migrations yet), so
# db:prepare just materializes an empty dev/test DB; db:seed is a no-op on the default
# seeds.rb. All wrapped in `|| true` so an empty schema never blocks container startup.
( cd /workspace/repo \
&& bundle install \
&& (bin/rails db:prepare || true) \
&& (bin/rails db:test:prepare || true) \
&& (bin/rails db:seed || true) )
;;
flaredown)
# Polyglot: backend/ Rails 7.1 (Ruby 3.2.3, Mongoid on MongoDB + Postgres + Redis +
# Sidekiq) and frontend/ Ember (Node 14). Postgres/Mongo/Redis are started by
# post-start.sh (called above). .env is gitignored — materialize from the committed
# backend/env-example (public dev secrets). Mongoid creates collections lazily, so
# there's no Mongo schema to load; Postgres holds a small relational slice with a
# committed db/schema.rb → db:schema:load (NOT db:migrate, which re-dumps schema.rb
# from the live DB on a bind-mounted repo).
# env-example points PG at host `postgresql` (the docker-compose service name); in this
# single container everything is on localhost, so rewrite the PG host. Redis defaults to
# localhost already; Mongoid reads MONGODB_HOST (unset → localhost).
( cd /workspace/repo/backend \
&& (cp -n env-example .env 2>/dev/null || true) \
&& sed -i 's/^PG_DATABASE_HOST=.*/PG_DATABASE_HOST=localhost/' .env 2>/dev/null || true; \
bundle install \
&& (bundle exec rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:schema:load || true) )
# Ember frontend on Node 14 (frontend/.nvmrc = v14.21.3; npm pinned to 6 in the image).
# node_modules to a container-local symlink (bind-mount file-sharing exhausts the host fd
# table on big node_modules trees). OPENSSL_CONF=/dev/null lets the old webpack md4 hashing
# run on bookworm's OpenSSL 3. --unsafe-perm so npm (running as root) actually executes the
# postinstall (patch-package + bower install) instead of skipping it with a "cannot run in
# wd" warning; without it bower_components is never populated and `ember build` fails.
NODE14_BIN=$(ls -d /usr/local/nvm/versions/node/v14.* 2>/dev/null | sort -V | tail -1)/bin
( cd /workspace/repo/frontend \
&& export PATH="$NODE14_BIN:$PATH" OPENSSL_CONF=/dev/null \
&& _nm_link flaredown-frontend \
&& (npm install --unsafe-perm --no-audit --no-fund || echo "WARNING: frontend npm install failed (explore-only)" >&2) ) || true
;;
breezy-complete)
# Monorepo: Rails 7.0 / Ruby 3.2.0 API (backend/) + Next.js 14 frontend
# (frontend/); Postgres + Redis baked in the image. The offline Clerk-bypass
# env is injected by run-app at server start only — the ambient env stays
# upstream-CI-shaped so a worker's `cd backend && bundle exec rspec` runs
# green (ambient DISABLE_CLERK 403s several controller specs, and ambient
# RAILS_ENV leaks through rails_helper's `ENV['RAILS_ENV'] ||= 'test'`).
#
# backend: gems + yarn asset-pipeline deps; db:prepare (retried once — the
# first run can race the just-started postgres) + db:seed (offline-safe demo
# tenant; the only way into the UI, auth is invite-less) + test DB. Fresh-DB
# db:test:prepare trips check_protected_environments → stamp the env first.
# db:prepare seeds the DB it creates and the seeds are not idempotent, so the
# explicit db:seed is for the retry case only — skip it on a seeded DB.
# frontend: npm install (not ci) so platform-specific optional deps resolve
# on arm64 + x64. Both node_modules go to CONTAINER-LOCAL paths via symlink
# (bind-mount ENFILE; see the ZenBill comment above — target must itself be
# named node_modules).
( cd /workspace/repo/backend \
&& bundle install --jobs 4 --retry 3 \
&& _nm_link breezy-backend \
&& yarn install --frozen-lockfile \
&& (bundle exec rails db:prepare || bundle exec rails db:prepare) \
&& ( psql -tAc 'select 1 from breezy_professionals limit 1' socratic_systems_development 2>/dev/null | grep -q 1 \
|| bundle exec rails db:seed || true ) \
&& (RAILS_ENV=test bundle exec rails db:environment:set || true) \
&& (RAILS_ENV=test bundle exec rails db:test:prepare || true) )
( cd /workspace/repo/frontend \
&& _nm_link breezy-frontend \
&& npm install --include=optional )
;;
esac
# Mirror Harbor's reduced toolset in the interactive Explore session. Use
# Harbor's /opt path when available, but fall back to a user-writable path for
# generic devcontainer fixtures that run lifecycle hooks as a non-root user.
AGENT_CLI_DIR="/opt/agent-cli"
if ! mkdir -p "$AGENT_CLI_DIR" 2>/dev/null; then
AGENT_CLI_DIR="$HOME/.agent-cli"
mkdir -p "$AGENT_CLI_DIR"
fi
cp -R /workspace/scripts/str_replace_editor /workspace/scripts/str_replace_editor_vendor "$AGENT_CLI_DIR/"
chmod +x "$AGENT_CLI_DIR/str_replace_editor"
mkdir -p "$HOME/.local/bin"
# Explore launchers (one per authoring harness) come from setup-harnesses.sh,
# which reads harness-registry.toml. AGENT_CLI_DIR is where the reduced-toolset
# editor was staged above, and the launcher rewrites the toolset note to match.
AGENT_CLI_DIR="$AGENT_CLI_DIR" harness_install_launchers
mkdir -p "$HOME/.claude"
# SKIP_FAST_MODE_NETWORK_ERRORS: the LLM proxy doesn't forward claude's fast-mode
# availability probe, and claude reads the failed probe as "no network" and refuses
# /fast. The override makes /fast toggleable; fast serving stays OFF until toggled.
node -e '
const fs = require("fs");
const home = process.env.HOME;
const env = {
CLAUDE_CODE_DISABLE_AUTO_MEMORY: "1",
CLAUDE_CODE_SKIP_FAST_MODE_NETWORK_ERRORS: "1",
};
fs.writeFileSync(
`${home}/.claude/settings.json`,
JSON.stringify({ env }, null, 2) + "\n"
);
'
# Reference-data corpus: expose it at the stable /data/zeta-corpus path (the same path a trial
# uses) by symlinking to the toolkit's bind-mounted copy. No-op if this toolkit ships no corpus.
if [ -d /workspace/data/zeta-corpus ]; then
{ mkdir -p /data || sudo mkdir -p /data; } 2>/dev/null || true
{ ln -sfn /workspace/data/zeta-corpus /data/zeta-corpus \
|| sudo ln -sfn /workspace/data/zeta-corpus /data/zeta-corpus; } 2>/dev/null || true
fi
# Shell setup
cat >> ~/.bashrc <<'BASHRC'
export PATH="$HOME/.local/bin:$PATH"
set -a && source /workspace/.env && set +a
# Everything below this line is for interactive shells only. An agent's shell tool
# sources .bashrc too, so without this guard the welcome banner prints into command
# output and container_start fires once per command instead of once per session.
case $- in
*i*) ;;
*) return ;;
esac
alias run-app="bash /workspace/run-app.sh"
[ -f /workspace/corpus-viewer/view-corpus.sh ] && alias view-corpus="bash /workspace/corpus-viewer/view-corpus.sh"
export PS1="\[\033[1;36m\][raccoon-explore]\[\033[0m\] \w\$ "
bash /workspace/welcome.sh explore 2>/dev/null
_AK="fde503c3bdb6e5cc9c48b1f8e4c2abeb"
_DK="e966e45af5ad1a18005f9fdb831186ea"
_WID="w-mtriw5pe-u8me"
_VER="2f696c53b4"
_CT="explore"
_RP=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').repo)}catch{}" 2>/dev/null)
_SID="$(date +%s)-$$"
_LAT=0
_ev() {
[ -z "$_AK" ] && return
{ curl -s -X POST "https://api2.amplitude.com/2/httpapi" \
-H "Content-Type: application/json" \
-d "{\"api_key\":\"$_AK\",\"events\":[{\"user_id\":\"$_WID\",\"event_type\":\"raccoon.$1\",\"event_properties\":{\"product\":\"raccoon\",\"container\":\"$_CT\",\"repo\":\"$_RP\",\"toolkit_version\":\"$_VER\",\"session_id\":\"$_SID\"},\"session_id\":$(date +%s000)}]}" \
>/dev/null 2>&1 & } 2>/dev/null; disown 2>/dev/null
}
_dl() {
[ -z "$_DK" ] && return
{ curl -s -X POST "https://http-intake.logs.datadoghq.com/api/v2/logs" \
-H "DD-API-KEY: $_DK" -H "Content-Type: application/json" \
-d "[{\"ddsource\":\"raccoon\",\"service\":\"toolkit\",\"hostname\":\"$(hostname)\",\"status\":\"$1\",\"message\":\"$2\",\"ddtags\":\"container:$_CT,worker:$_WID,repo:$_RP,toolkit_version:$_VER\"}]" \
>/dev/null 2>&1 & } 2>/dev/null; disown 2>/dev/null
}
_pc() { local n; n=$(date +%s); if (( n - _LAT >= 300 )); then _LAT=$n; _ev active; fi; }
PROMPT_COMMAND="_pc;${PROMPT_COMMAND:-}"
trap '_ev container_stop; _dl info container_stop; wait' EXIT
_ev container_start
_dl info container_start
BASHRC
# One alias per authoring harness: `claude` runs claude, `codex` runs codex.
harness_alias_lines >> ~/.bashrc

View File

@@ -0,0 +1,247 @@
#!/bin/bash
# Post-start setup for the Explore devcontainer.
#
# This runs on EVERY container start (wired as `postStartCommand` in
# devcontainer.json), unlike post-create.sh which runs only once when the
# container is first created. Its job is the lightweight work that has to
# happen on every boot: bring PostgreSQL back up. The heavy one-time work
# (installing dependencies, creating + migrating the database, seeding) stays
# in post-create.sh.
#
# Why this is needed: the container is started with an entrypoint that bypasses
# the image's own startup script, so nothing restarts postgres for you. After
# you stop the container or reboot your machine, postgres stays down until this
# script runs — previously you had to start it by hand every session.
#
# Safe to run repeatedly: if postgres is already accepting connections, the
# start step is skipped and this is effectively a no-op.
set -euo pipefail
REPO_NAME=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').repo)}catch{}" 2>/dev/null || true)
# Dead-end the deployed hostnames this estate's sources still name, so booting an app with a
# non-local environment setting can't send a login form (or anything else) to a live host. Has to
# happen on every start, not in the image: Docker remounts /etc/hosts per container, so a
# Dockerfile write to it never survives.
BLOCKED_HOSTS=$(node -e "try{process.stdout.write((require('/workspace/toolkit.json').blockedHosts||[]).join(' '))}catch{}" 2>/dev/null || true)
if [ -n "$BLOCKED_HOSTS" ] && ! grep -q "raccoon-blocked-hosts" /etc/hosts 2>/dev/null; then
printf '# raccoon-blocked-hosts\n127.0.0.1 %s\n::1 %s\n' "$BLOCKED_HOSTS" "$BLOCKED_HOSTS" \
| sudo tee -a /etc/hosts >/dev/null 2>&1 \
|| echo "warning: could not pin blocked hosts in /etc/hosts" >&2
fi
# Corpus viewer: when this toolkit ships a corpus search index, serve the viewer on
# container port 3002 (published as EXPLORE_CORPUS_PORT on the host). Only
# corpus-shipping toolkits package the viewer at all; where present, the script
# self-guards (no index / no python3 / already running → quiet no-op) and must never
# block container startup.
[ -f /workspace/corpus-viewer/view-corpus.sh ] &&
bash /workspace/corpus-viewer/view-corpus.sh start --quiet || true
# True when postgres is up and answering queries.
pg_ready() { sudo -u postgres psql -c "SELECT 1" >/dev/null 2>&1; }
# Block until postgres is ready, but never hang the container start forever:
# pg_isready alone races on cluster startup, so we poll an actual query with a
# bounded number of attempts (60s) and move on with a warning if it never comes
# up rather than wedging `devcontainer up`.
wait_for_pg() {
local n=0
until pg_ready; do
sleep 0.5
n=$((n + 1))
if [ "$n" -ge 120 ]; then
echo "warning: postgres did not become ready within 60s" >&2
return 0
fi
done
}
# Polyglot toolkit: REPO_NAME is empty (no single repo). Bring up Postgres + Redis
# (members need them; per-member DB setup is deferred to run-app/setup_repo), then done.
# Trial parity: limit DNS to the model endpoint and the toolkit's telemetry, so a session
# captured here cannot depend on network the trial agent will not have. Opt-in
# (RACCOON_DNS_JAIL=1) and best-effort. postCreate patches .bashrc, but postStart gets no
# login shell, so .env is read directly. Called on BOTH paths: the polyglot branch returns
# before the end of this script.
apply_dns_jail() {
[ -f /workspace/.devcontainer/dns-jail.sh ] || return 0
# Read the one line rather than sourcing: this runs on every boot AND every run-app, and
# with the jail off it must not execute the worker's .env as a side effect.
if [ "${RACCOON_DNS_JAIL:-0}" != "1" ] &&
! grep -qE '^[[:space:]]*(export[[:space:]]+)?RACCOON_DNS_JAIL[[:space:]]*=[[:space:]]*"?1"?[[:space:]]*(#.*)?$' \
/workspace/.env 2>/dev/null; then
return 0
fi
(
set -a
# shellcheck disable=SC1091
. /workspace/.env 2>/dev/null || true
set +a
bash /workspace/.devcontainer/dns-jail.sh
) || true
}
IS_POLYGLOT=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').polyglot?'1':'')}catch{}" 2>/dev/null || true)
if [ -n "$IS_POLYGLOT" ]; then
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
wait_for_pg
sudo -u postgres psql -c "ALTER USER postgres PASSWORD 'secret_password';" >/dev/null 2>&1 || true
# Many members' committed .env / database.yml default the DB username to 'root'
# (dotenv-rails applies it at boot, overriding DEV_DB_USERNAME=postgres). The image's
# start-services.sh creates a root superuser, but devcontainers override the ENTRYPOINT so
# it never runs — create root here too, mirroring the harbor task env.
sudo -u postgres psql -c "CREATE ROLE root SUPERUSER LOGIN PASSWORD 'secret_password';" >/dev/null 2>&1 || true
# MongoDB, for the polyglot images that bake it (potion's flagship member stores everything
# in Mongo). `command -v mongod` is the switch, so the Mongo-less polyglot images skip this
# untouched. It has to happen here for the same reason postgres does — the devcontainer
# overrides the image ENTRYPOINT, so the baked start-services.sh never runs — and it matters
# more than a stopped postgres would: mongoose BUFFERS operations while disconnected instead
# of erroring, so a member whose Mongo is down doesn't fail loudly, it serves requests that
# hang forever and pages that never finish rendering. Readiness is a dependency-free TCP
# probe (no mongosh needed; mongoose connects lazily once the port is open).
if command -v mongod >/dev/null 2>&1; then
mongo_up() { (exec 3<>/dev/tcp/127.0.0.1/27017) 2>/dev/null && { exec 3>&- 3<&-; return 0; }; return 1; }
if ! mongo_up; then
sudo mkdir -p /data/db 2>/dev/null || mkdir -p /data/db 2>/dev/null || true
sudo chown -R "$(id -u)":"$(id -g)" /data/db 2>/dev/null || true
mongod --dbpath /data/db --bind_ip 127.0.0.1 --fork --logpath /tmp/mongod.log >/dev/null 2>&1 \
|| (sudo -b mongod --dbpath /data/db --bind_ip 127.0.0.1 --logpath /var/log/mongod.log >/dev/null 2>&1) || true
for _ in $(seq 1 60); do mongo_up && break; sleep 0.5; done
mongo_up || echo "warning: mongod did not come up within 30s" >&2
fi
fi
# OpenSearch, for the polyglot images that bake it — same reason as mongod (the devcontainer
# overrides the ENTRYPOINT, so the image's start-services.sh never runs). Presence of the
# binary is the switch, so images without it are untouched. Flags mirror the trial image's
# start-services block exactly; it runs as its own user because OpenSearch refuses to boot
# as root. Non-fatal: a member that doesn't use it shouldn't be blocked by a slow JVM.
if [ -x /opt/opensearch/bin/opensearch ]; then
os_up() { curl -s --max-time 2 localhost:9200 >/dev/null 2>&1; }
if ! os_up; then
sudo -u opensearch env OPENSEARCH_JAVA_OPTS='-Xms512m -Xmx512m' \
/opt/opensearch/bin/opensearch -Ediscovery.type=single-node \
-Eplugins.security.disabled=true >/tmp/opensearch.log 2>&1 &
for _ in $(seq 1 90); do os_up && break; sleep 2; done
os_up || echo "warning: opensearch did not come up within 180s (see /tmp/opensearch.log)" >&2
fi
fi
apply_dns_jail
return 0 2>/dev/null || exit 0
fi
case "$REPO_NAME" in
ZenBill-006)
# Start is non-fatal: if it fails outright, wait_for_pg is the single
# gate — it warns and continues rather than aborting `devcontainer up`
# and leaving the worker with no shell.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
wait_for_pg
sudo -u postgres psql -c "ALTER USER postgres PASSWORD 'secret_password';" >/dev/null 2>&1 || true
;;
zeta-heimdall)
# Bookworm base → `service postgresql start` (same as ZenBill). Non-fatal
# start; wait_for_pg is the single gate so a hiccup warns rather than wedging
# `devcontainer up` and leaving the worker with no shell.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
wait_for_pg
sudo -u postgres psql -c "ALTER USER postgres PASSWORD 'secret_password';" >/dev/null 2>&1 || true
;;
zeta-platform)
# Postgres + Redis (sidekiq). Start both; wait_for_pg is the single gate.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
wait_for_pg
sudo -u postgres psql -c "ALTER USER postgres PASSWORD 'secret_password';" >/dev/null 2>&1 || true
;;
Palolo-031)
if ! pg_ready; then
PG_VERSION=$(pg_config --version | grep -oP '\d+' | head -1)
sudo pg_ctlcluster "${PG_VERSION}" main start || echo "warning: 'pg_ctlcluster ${PG_VERSION} main start' failed" >&2
fi
wait_for_pg
sudo -u postgres psql -c "CREATE USER test WITH SUPERUSER PASSWORD 'test';" >/dev/null 2>&1 || true
sudo -u postgres psql -c "CREATE DATABASE palolo OWNER test;" >/dev/null 2>&1 || true
;;
human-essentials)
# Bookworm base → `service postgresql start` (same as ZenBill). Non-fatal
# start; wait_for_pg is the single gate. Trust auth (set in the image), so
# no role password to seed — the app connects as postgres with no password.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
wait_for_pg
;;
stocks-in-the-future)
# Postgres + Redis (background jobs). Start both; wait_for_pg is the gate.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
wait_for_pg
;;
casa)
# Postgres only. Non-fatal start; wait_for_pg is the gate.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
wait_for_pg
;;
awbw)
# MySQL 8 (Percona). Start it (init the data dir first if empty), then ensure root
# is passwordless over TCP (mysql_native_password) for Trilogy. Self-contained —
# the pg_ready/wait_for_pg helpers above are Postgres-specific.
if ! mysqladmin ping >/dev/null 2>&1; then
sudo mkdir -p /var/run/mysqld && sudo chown -R mysql:mysql /var/run/mysqld /var/lib/mysql 2>/dev/null || true
[ -d /var/lib/mysql/mysql ] || sudo mysqld --initialize-insecure --user=mysql --datadir=/var/lib/mysql 2>/dev/null || true
sudo service mysql start >/dev/null 2>&1 || (sudo mysqld_safe --user=mysql >/dev/null 2>&1 &) || echo "warning: mysql start failed" >&2
fi
for i in $(seq 1 120); do mysqladmin ping >/dev/null 2>&1 && break; sleep 0.5; done
mysql -u root -e "ALTER USER 'root'@'localhost' IDENTIFIED WITH mysql_native_password BY ''; CREATE USER IF NOT EXISTS 'root'@'%' IDENTIFIED WITH mysql_native_password BY ''; GRANT ALL PRIVILEGES ON *.* TO 'root'@'localhost' WITH GRANT OPTION; GRANT ALL PRIVILEGES ON *.* TO 'root'@'%' WITH GRANT OPTION; FLUSH PRIVILEGES;" >/dev/null 2>&1 || true
# Load MySQL tz tables (Ahoy charts use Groupdate/CONVERT_TZ). One-time.
[ "$(mysql -u root -N -e 'SELECT COUNT(*) FROM mysql.time_zone_name' 2>/dev/null || echo 0)" -gt 0 ] \
|| mysql_tzinfo_to_sql /usr/share/zoneinfo 2>/dev/null | mysql -u root mysql 2>/dev/null || true
;;
flaredown)
# Three datastores: Postgres (relational slice) + Redis (Sidekiq) + MongoDB (Mongoid,
# the primary store). Start all three; wait_for_pg gates the Postgres readiness, and
# we poll mongod separately. All starts are non-fatal so a hiccup warns rather than
# wedging `devcontainer up`. Trust auth on Postgres (set in the image).
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
# MongoDB (server tarball → bin/mongod on PATH; no service unit). Launch mongod against
# a data dir if nothing is already listening on 27017. Readiness is a dependency-free
# TCP probe (no mongosh needed — Mongoid connects lazily once the port is open).
mongo_up() { (exec 3<>/dev/tcp/127.0.0.1/27017) 2>/dev/null && { exec 3>&- 3<&-; return 0; }; return 1; }
if ! mongo_up; then
sudo mkdir -p /data/db 2>/dev/null || mkdir -p /data/db 2>/dev/null || true
sudo chown -R "$(id -u)":"$(id -g)" /data/db 2>/dev/null || true
mongod --dbpath /data/db --bind_ip 127.0.0.1 --fork --logpath /tmp/mongod.log >/dev/null 2>&1 \
|| (sudo -b mongod --dbpath /data/db --bind_ip 127.0.0.1 --logpath /var/log/mongod.log >/dev/null 2>&1) || true
fi
wait_for_pg
for _ in $(seq 1 60); do mongo_up && break; sleep 0.5; done
;;
breezy-complete)
# Postgres + Redis (Sidekiq). Start both; wait_for_pg is the gate. Trust
# auth (set in the image) — PGPASSWORD is baked but inert, no role seeding.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
wait_for_pg
;;
esac
apply_dns_jail

View File

@@ -0,0 +1,256 @@
#!/usr/bin/env node
// instance.js — run more than one Explore container of THIS repo at once.
//
// The normal single container is still just `npx @devcontainers/cli up`, and if
// you only want several Claude sessions on the SAME repo state you don't need
// this at all — just open more shells into the one container
// (`npx @devcontainers/cli exec bash`). Use this when you want ANOTHER container
// with its OWN separate working tree — e.g. to explore a different commit / repo
// state at the same time — without unzipping the toolkit again.
//
// Each named instance gets:
// - its own container (a distinct id-label, so `up` makes a new one),
// - its own host port(s) (auto-picked free, so nothing collides),
// - its own repo working tree (initialize.js clones repo<name>, so a
// `git checkout` in one instance never disturbs another).
//
// Run on the HOST, from the toolkit's explore/ folder (this drives Docker; the
// Explore devcontainer has no Docker socket):
// node instance.js b # create/start instance "b", print its URL
// node instance.js shell b # open a shell in instance "b"
// node instance.js stop b # stop+remove it (keeps the repo clone)
// node instance.js list # list running/stopped instances
//
// This is Node (not bash) so it works on Windows without WSL, matching
// initialize.js.
import { execSync, spawnSync } from 'node:child_process';
import { readFileSync } from 'node:fs';
import { createServer } from 'node:net';
import { dirname, join, relative, resolve } from 'node:path';
import { fileURLToPath } from 'node:url';
const SCRIPT = fileURLToPath(import.meta.url);
const EXPLORE_DIR = dirname(SCRIPT);
// How the worker invoked us, so the follow-up commands we print match their cwd
// (`node instance.js …` from explore/, or `node explore/instance.js …` from the
// toolkit root) instead of guessing.
const SELF = `node ${relative(process.cwd(), SCRIPT) || 'instance.js'}`;
// Our own id-labels. Passing --id-label REPLACES the devcontainer CLI's default
// identity labels (it drops devcontainer.local_folder / devcontainer.config_file
// entirely), so we set our own and look up by them: ROOT_LABEL scopes to THIS
// toolkit copy (so `list`/`stop` never touch another copy's instances or the
// primary), NAME_LABEL identifies the instance.
const ROOT_LABEL = `raccoon-explore-root=${EXPLORE_DIR}`;
const NAME_LABEL = 'raccoon-explore';
const RESERVED = new Set(['list', 'stop', 'shell', 'help', '--help', '-h']);
function die(msg) {
console.error(msg);
process.exit(1);
}
/** Quiet `docker ...` returning trimmed stdout (empty string on any failure). */
function docker(args) {
try {
return execSync(`docker ${args}`, {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
}).trim();
} catch {
return '';
}
}
function readToolkit() {
for (const p of [join(EXPLORE_DIR, 'toolkit.json'), resolve(EXPLORE_DIR, '..', 'toolkit.json')]) {
try {
return JSON.parse(readFileSync(p, 'utf-8'));
} catch {
/* try next */
}
}
return {};
}
function validName(name) {
return typeof name === 'string' && /^[a-z0-9][a-z0-9-]{0,30}$/.test(name);
}
/**
* The per-instance working-tree dir name. A polyglot toolkit gives each instance
* its own `repos<name>` tree (all member repos); a single-repo toolkit a `repo<name>`.
* initialize.js creates whichever matches, and the devcontainer mount derives the
* same name from EXPLORE_INSTANCE.
*/
function worktreeName(name, tk) {
return (tk && tk.polyglot ? 'repos' : 'repo') + name;
}
/** The container id for instance <name> of THIS toolkit, or '' if none. */
function instanceContainer(name) {
return (
docker(`ps -aq --filter "label=${ROOT_LABEL}" --filter "label=${NAME_LABEL}=${name}"`)
.split('\n')
.filter(Boolean)[0] || ''
);
}
/** Resolve true once we find a free TCP port on the host at/after `start`. */
function freePort(start) {
return new Promise((res, rej) => {
const tryPort = (p) => {
if (p > start + 500) return rej(new Error(`no free host port near ${start}`));
const srv = createServer();
srv.once('error', () => tryPort(p + 1));
srv.once('listening', () => srv.close(() => res(p)));
srv.listen(p, '0.0.0.0');
};
tryPort(start);
});
}
function publishedPort(id, containerPort) {
const out = docker(`port ${id} ${containerPort}/tcp`);
const m = out.match(/:(\d+)\s*$/m);
return m ? m[1] : '';
}
function up(name, env) {
const r = spawnSync(
'npx',
[
'@devcontainers/cli',
'up',
'--workspace-folder',
EXPLORE_DIR,
'--id-label',
ROOT_LABEL,
'--id-label',
`${NAME_LABEL}=${name}`,
],
{ stdio: 'inherit', env }
);
if (r.status !== 0)
die(`\ninstance "${name}" failed to start (devcontainer up exited ${r.status}).`);
}
function reportUp(name, tk) {
const id = instanceContainer(name);
const port = publishedPort(id, 3000);
console.log(`\n✅ instance "${name}" is up`);
if (port) console.log(` open http://localhost:${port}`);
console.log(` shell ${SELF} shell ${name} (then run \`run-app\` inside)`);
if (tk.repo === 'Palolo-031') {
console.log(` note a second Palolo runs fine for exploring, but its browser app calls the`);
console.log(` first container's API (the client build bakes in localhost:3001).`);
}
console.log(
` stop ${SELF} stop ${name} (keeps the ${worktreeName(name, tk)} working tree)`
);
}
async function create(name) {
if (!validName(name))
die(`Invalid instance name "${name}". Use letters/digits/hyphens, e.g. b, two, alt2.`);
const tk = readToolkit();
const ports = tk.explorePorts || {};
const existing = instanceContainer(name);
if (existing) {
const running = docker(`inspect -f "{{.State.Running}}" ${existing}`) === 'true';
// Re-up reuses the existing container (and its baked port mapping); pass
// EXPLORE_INSTANCE so initialize.js's clone step stays a no-op.
if (!running) up(name, { ...process.env, EXPLORE_INSTANCE: name });
else console.log(`instance "${name}" is already running.`);
reportUp(name, tk);
return;
}
// Fresh instance: pick free host port(s) clear of the primary's defaults.
const env = { ...process.env, EXPLORE_INSTANCE: name };
const clientBase = (Number(ports.clientHost) || 3000) + 10;
const clientPort = await freePort(clientBase);
env.EXPLORE_CLIENT_PORT = String(clientPort);
if (ports.serverHost) env.EXPLORE_SERVER_PORT = String(await freePort(clientPort + 1));
// Well clear of the primary's default: this one is published host:container identical,
// so a collision would silently point the client's livereload at the other container.
if (ports.livereloadHost)
env.EXPLORE_LIVERELOAD_PORT = String(await freePort(ports.livereloadHost + 10));
up(name, env);
reportUp(name, tk);
}
function shell(name) {
if (!instanceContainer(name)) die(`No instance "${name}". Create it first: ${SELF} ${name}`);
const r = spawnSync(
'npx',
[
'@devcontainers/cli',
'exec',
'--workspace-folder',
EXPLORE_DIR,
'--id-label',
ROOT_LABEL,
'--id-label',
`${NAME_LABEL}=${name}`,
'bash',
],
{ stdio: 'inherit' }
);
process.exit(r.status ?? 0);
}
function stop(name) {
const id = instanceContainer(name);
if (!id) return console.log(`No instance "${name}" to stop.`);
docker(`rm -f ${id}`);
const wt = worktreeName(name, readToolkit());
console.log(
`Stopped instance "${name}". Its ${wt} working tree is kept (delete it with: rm -rf ${join(EXPLORE_DIR, wt)}).`
);
}
function list() {
const rows = docker(
`ps -a --filter "label=${ROOT_LABEL}" ` +
`--format "{{.Label \\"${NAME_LABEL}\\"}}\\t{{.State}}\\t{{.Ports}}"`
);
if (!rows) return console.log(`No extra instances. Create one with: ${SELF} <name>`);
console.log('INSTANCE\tSTATE\tPORTS');
console.log(rows);
}
function usage() {
console.log(
[
'instance.js — run more than one Explore container of this repo at once.',
'',
` ${SELF} <name> create/start instance <name>, print its URL`,
` ${SELF} shell <name> open a shell inside instance <name>`,
` ${SELF} stop <name> stop + remove instance <name> (keeps its repo clone)`,
` ${SELF} list list extra instances`,
'',
'Run on the host, from the explore/ folder. The normal single container',
'is still just `npx @devcontainers/cli up`.',
].join('\n')
);
}
const [cmd, arg] = process.argv.slice(2);
if (!cmd || cmd === 'help' || cmd === '--help' || cmd === '-h') {
usage();
} else if (cmd === 'list') {
list();
} else if (cmd === 'stop') {
if (!validName(arg)) die(`Usage: ${SELF} stop <name>`);
stop(arg);
} else if (cmd === 'shell') {
if (!validName(arg)) die(`Usage: ${SELF} shell <name>`);
shell(arg);
} else if (RESERVED.has(cmd)) {
usage();
} else {
// `instance.js <name>` shorthand for create/start.
await create(cmd);
}

View File

@@ -0,0 +1,8 @@
{
"name": "create-snapshot",
"description": "Capture conversation context and repo state as a snapshot",
"version": "0.1.0",
"author": {
"name": "raccoon"
}
}

View File

@@ -0,0 +1,788 @@
#!/usr/bin/env node
import { execSync } from 'node:child_process';
import crypto from 'node:crypto';
import fs from 'node:fs';
import path from 'node:path';
import { linearSnapshotLines, readSession } from './harness-session.mjs';
// --- Argument parsing ---
function parseArgs(argv) {
const args = {};
for (let i = 2; i < argv.length; i++) {
if (argv[i].startsWith('--')) {
const key = argv[i].slice(2);
const val = argv[i + 1];
if (!val || val.startsWith('--')) {
args[key] = true;
} else {
args[key] = val;
i++;
}
}
}
return args;
}
// Last-resort data dir. Claude Code's plugin runtime always provides one, so this is
// what makes the start marker and session record work under any other harness.
function defaultDataDir() {
const home = process.env.HOME || '/root';
return path.join(home, '.raccoon', 'snapshot-data');
}
const args = parseArgs(process.argv);
const slug = args.slug;
const annotationPath = args.annotation;
const outputDir = args['output-dir'];
if (!args['mark-start'] && (!slug || !annotationPath || !outputDir)) {
console.error(
'Usage: capture-snapshot.mjs --slug <slug> --annotation <path> --output-dir <dir> [--plugin-data <path>]\n' +
' capture-snapshot.mjs --mark-start [--harness <id>]'
);
process.exit(1);
}
// --- Locate session info ---
// --harness wins over the launcher's RACCOON_HARNESS so a caller that knows which
// conversation it is capturing can say so; the default keeps Claude Code's plugin
// working unchanged. Claude Code has an exact cut point (the snapshot slash command)
// and a message tree to prune, so it keeps the bespoke path below; other harnesses go
// through the shared reader.
const HARNESS = args.harness || process.env.RACCOON_HARNESS || 'claude-code';
const IS_CLAUDE = HARNESS === 'claude-code';
// The generated restore.sh writes one of exactly two session layouts, and everything
// below branches on IS_CLAUDE — so a third harness would silently be handed codex's
// $CODEX_HOME/sessions paths. Refuse instead; adding a harness means adding a layout.
if (!IS_CLAUDE && HARNESS !== 'codex') {
console.error(
`capture-snapshot: no session-restore layout for harness "${HARNESS}". ` +
'Add one to capture-snapshot.mjs (and harness-session.mjs) before capturing with it.'
);
process.exit(1);
}
// Try multiple strategies to find the current session transcript:
// 1. Plugin data dir (from SessionStart hook)
// 2. Scan ~/.claude/projects/ for the most recently modified JSONL
let session_id = null;
let transcript_path = null;
const dataDir =
args['plugin-data'] ||
process.env.RACCOON_SNAPSHOT_DATA ||
process.env.CLAUDE_PLUGIN_DATA ||
(process.env.CLAUDE_PLUGIN_ROOT && path.join(process.env.CLAUDE_PLUGIN_ROOT, '.data')) ||
defaultDataDir();
// The SessionStart hook records the live session for every harness, so prefer it over
// guessing. `readSession` falls back to the newest file on disk when it is absent.
let recordedSession = null;
if (dataDir) {
const sessionInfoPath = path.join(dataDir, 'current-session.json');
if (fs.existsSync(sessionInfoPath)) {
try {
recordedSession = JSON.parse(fs.readFileSync(sessionInfoPath, 'utf8'));
} catch {
recordedSession = null;
}
}
}
const startMarkerPath = dataDir ? path.join(dataDir, 'snapshot-start.json') : null;
// `--mark-start` runs BEFORE the annotation Q&A and records how long the conversation
// was at that moment. It is the linear-harness stand-in for Claude Code's slash-command
// line: without it, capture would stage the snapshot's own Q&A as conversation.
if (args['mark-start']) {
const session = readSession(HARNESS);
if (!session) {
console.error(`No ${HARNESS} session found to mark.`);
process.exit(1);
}
if (!startMarkerPath) {
console.error('No data dir available to record the snapshot start marker.');
process.exit(1);
}
fs.mkdirSync(path.dirname(startMarkerPath), { recursive: true });
fs.writeFileSync(
startMarkerPath,
JSON.stringify(
{ harness: HARNESS, transcript_path: session.rawPath, line_count: session.lines.length },
null,
2
) + '\n'
);
console.log(`Snapshot start marked at ${session.lines.length} records.`);
process.exit(0);
}
let harnessSession = null;
if (!IS_CLAUDE) {
harnessSession = readSession(HARNESS, recordedSession?.transcript_path);
if (!harnessSession) {
console.error(
`Could not find a ${HARNESS} session to capture. Capture has to run from inside the ${HARNESS} conversation you want to snapshot.`
);
process.exit(1);
}
transcript_path = harnessSession.rawPath;
// Fall back to a fresh id only if the harness records none — restore.sh names the
// installed session by it, so it has to match what `resume` will look up.
session_id = recordedSession?.session_id || harnessSession.sessionId || crypto.randomUUID();
} else if (recordedSession) {
session_id = recordedSession.session_id;
transcript_path = recordedSession.transcript_path;
}
// Fallback: find the most recently modified JSONL in ~/.claude/projects/
if (!transcript_path) {
const homeDir = process.env.HOME || '/root';
const projectsDir = path.join(homeDir, '.claude', 'projects');
if (fs.existsSync(projectsDir)) {
let newest = null;
let newestMtime = 0;
for (const projEntry of fs.readdirSync(projectsDir)) {
const projDir = path.join(projectsDir, projEntry);
if (!fs.statSync(projDir).isDirectory()) continue;
for (const file of fs.readdirSync(projDir)) {
if (!file.endsWith('.jsonl')) continue;
const filePath = path.join(projDir, file);
const mtime = fs.statSync(filePath).mtimeMs;
if (mtime > newestMtime) {
newestMtime = mtime;
newest = filePath;
session_id = file.replace(/\.jsonl$/, '');
}
}
}
transcript_path = newest;
}
}
if (!transcript_path || !fs.existsSync(transcript_path)) {
console.error(
"Could not find a Claude Code session transcript. This script should be run from within a Claude Code conversation via the /create-snapshot:snapshot command. Please file a bug if you're seeing this unexpectedly."
);
process.exit(1);
}
if (!transcript_path || !fs.existsSync(transcript_path)) {
console.error(`Transcript file not found at ${transcript_path}. Please file a bug.`);
process.exit(1);
}
// --- Require a git repo at capture time ---
//
// A snapshot is "commit SHA + diff vs HEAD", reconstituted later via
// `git archive <SHA> | tar -x` + `git apply workspace.patch`. Without a git
// repo here we have no SHA to pin, no patch to record, and no way for
// downstream `build-workspace.sh` to reproduce the workspace — the resulting
// snapshot would be structurally meaningless. This check runs BEFORE the
// snapshot directory is created so a misconfigured invocation leaves no
// half-written state behind.
// Find the git repo by asking git itself — walks up from cwd looking for
// `.git`, handling submodules and worktrees correctly. Returns null when
// cwd is outside any repo, so the worker gets a clear "cd into your repo"
// error instead of silently descending into something they didn't name.
function findGitRepo() {
try {
const top = execSync('git rev-parse --show-toplevel', {
stdio: ['ignore', 'pipe', 'ignore'],
})
.toString()
.trim();
return top || null;
} catch {
return null;
}
}
const gitRepo = findGitRepo();
if (!gitRepo) {
console.error(
"Error: Not running inside a git repo. /create-snapshot needs a git repo so it can pin a commit SHA and record a diff of in-flight changes; without one the snapshot can't be reproduced as a task. cd into the repo you're exploring (the toolkit's repo/ submodule) and re-run /create-snapshot:snapshot."
);
process.exit(1);
}
// --- Create snapshot directory ---
const ts = new Date().toISOString().replace(/[-:]/g, '').replace('T', '-').slice(0, 15); // 20260403-225449
const snapshotDir = path.join(outputDir, `${ts}-${slug}`);
if (fs.existsSync(snapshotDir)) {
console.error(`Snapshot directory already exists: ${snapshotDir}\nPlease file a bug.`);
process.exit(1);
}
fs.mkdirSync(snapshotDir, { recursive: true });
// --- Copy conversation transcript (trimmed + branch-pruned) ---
// The JSONL is a tree of messages linked by parentUuid. When the user rewinds
// a conversation, old branches remain in the file. We need to:
// 1. Cut at the LAST /create-snapshot:snapshot command (later invocations
// supersede earlier ones in the same session)
// 2. Find the tip of the active branch (last message before the cut)
// 3. Walk parentUuid back to the root, collecting only messages on that path
// 4. Exclude the Q&A subgraphs of any PRIOR /create-snapshot:snapshot
// invocations in this session (their cut points are on the same
// conversation branch, so the walk would otherwise pull in the
// assistant's annotation questions and the user's answers — a
// contamination path that snapshot.patch doesn't show). Boundaries
// for prior invocations are recorded in a side file (see end of
// this script) so this run can identify them.
// 5. Drop bookkeeping entries whose content can leak rewound-branch state.
const rawLines = fs.readFileSync(transcript_path, 'utf8').trimEnd().split('\n');
// A user message is a /create-snapshot:snapshot invocation when its content
// STARTS with one of Claude Code's slash-command tags AND mentions the
// command name. The "starts with" guard distinguishes a real invocation
// from prose that quotes the command (a worker reporting a bug, the
// command-listing skill output, etc.) — prose doesn't begin with those
// tags. The whitespace-tolerant pattern survives minor format drift in
// Claude Code's slash-command rendering.
const SNAPSHOT_CMD_PATTERN =
/<command-(?:name|message)>\s*\/?\s*create-snapshot:snapshot\s*<\/command-(?:name|message)>/;
function isSnapshotCommandContent(content) {
if (typeof content !== 'string') return false;
const trimmed = content.trimStart();
if (!trimmed.startsWith('<command-name>') && !trimmed.startsWith('<command-message>')) {
return false;
}
return SNAPSHOT_CMD_PATTERN.test(content);
}
// Find every snapshot-command line index, in order. The LAST one is the
// current invocation (cut point); earlier ones bound prior Q&A subgraphs.
const snapshotCmdIndexes = [];
for (let i = 0; i < rawLines.length; i++) {
try {
const entry = JSON.parse(rawLines[i]);
if (entry.type === 'user' && isSnapshotCommandContent(entry.message?.content)) {
snapshotCmdIndexes.push(i);
}
} catch {
// Skip malformed lines
}
}
const cutIndex =
snapshotCmdIndexes.length > 0
? snapshotCmdIndexes[snapshotCmdIndexes.length - 1]
: rawLines.length;
const priorCmdIndexes = snapshotCmdIndexes.slice(0, -1);
// Load prior-snapshot boundary records so we know where each earlier
// invocation's Q&A subgraph ended. The boundary file is written at the
// end of every capture run (see below) and is keyed by session uuid.
function loadPriorBoundaries() {
if (!dataDir || !session_id) return [];
const boundariesPath = path.join(dataDir, 'snapshot-boundaries.jsonl');
if (!fs.existsSync(boundariesPath)) return [];
const lines = fs.readFileSync(boundariesPath, 'utf8').trimEnd().split('\n');
const out = [];
for (const line of lines) {
if (!line) continue;
try {
const rec = JSON.parse(line);
if (rec.sessionUuid === session_id && rec.snapshotCommandUuid) out.push(rec);
} catch {
/* skip malformed */
}
}
return out;
}
const priorBoundaries = loadPriorBoundaries();
// Compute the line ranges to exclude for each prior snapshot. The Q&A
// subgraph starts at the prior snapshot's command line and runs through
// the line whose entry uuid matches the boundary record (the last entry
// in the JSONL when that prior capture-snapshot completed).
//
// Fall back to the next snapshot command (or the current cut) when no
// matching boundary record exists — better to drop too much than to leak
// the Q&A; the visible cost is excluding any "real work" that happened
// between snapshots without a recorded boundary, which only occurs if
// the boundary log was wiped or the prior capture crashed.
function findUuidLineIndex(targetUuid, startLine, endLineExclusive) {
for (let i = startLine; i < endLineExclusive; i++) {
try {
const entry = JSON.parse(rawLines[i]);
if (entry.uuid === targetUuid) return i;
} catch {
/* skip */
}
}
return -1;
}
const priorQAExcludedLines = new Set();
for (let i = 0; i < priorCmdIndexes.length; i++) {
const startLine = priorCmdIndexes[i];
const nextCutLine = i + 1 < priorCmdIndexes.length ? priorCmdIndexes[i + 1] : cutIndex;
let snapshotCmdUuid = null;
try {
snapshotCmdUuid = JSON.parse(rawLines[startLine]).uuid || null;
} catch {
/* unparseable command line — skip */
}
let endLine = -1;
if (snapshotCmdUuid) {
const boundary = priorBoundaries.find((b) => b.snapshotCommandUuid === snapshotCmdUuid);
if (boundary && boundary.lastEntryUuid) {
endLine = findUuidLineIndex(boundary.lastEntryUuid, startLine, nextCutLine);
}
}
// No matching boundary: bound the exclusion at the next snapshot/current
// cut so the Q&A doesn't leak even if state was lost.
if (endLine < 0) endLine = nextCutLine - 1;
for (let j = startLine; j <= endLine; j++) priorQAExcludedLines.add(j);
}
// Step 2: parse all entries before the cut, build uuid index
const preCutEntries = [];
const byUuid = {};
for (let i = 0; i < cutIndex; i++) {
try {
const entry = JSON.parse(rawLines[i]);
preCutEntries.push({ line: rawLines[i], entry, index: i });
if (entry.uuid) {
byUuid[entry.uuid] = entry;
}
} catch {
// Keep unparseable lines (they'll be included as non-message entries)
preCutEntries.push({ line: rawLines[i], entry: null, index: i });
}
}
// Step 3: find the tip of the active branch. The snapshot command's parentUuid
// points to the message the user was looking at when they ran the snapshot —
// this is authoritative even after rewinds.
let tipUuid = null;
if (cutIndex < rawLines.length) {
try {
const snapshotCmd = JSON.parse(rawLines[cutIndex]);
tipUuid = snapshotCmd.parentUuid || null;
} catch {
// not valid JSON — leave tipUuid null
}
}
// If the live tip is itself inside a prior snapshot's Q&A subgraph (e.g.
// the user ran /create-snapshot:snapshot a second time WITHOUT typing
// anything between the two — there's no "real work" gap), walk back past
// the excluded range to find the closest non-excluded ancestor. Otherwise
// activeBranchUuids would be empty and we'd produce an empty snapshot.
function nearestNonExcludedAncestor(startUuid) {
let cur = startUuid;
while (cur) {
const e = byUuid[cur];
if (!e) return cur; // unknown uuid — best effort, keep
// Find the line index of this entry to check exclusion.
// (Line index isn't stored on the entry; recompute via preCutEntries.)
const found = preCutEntries.find((p) => p.entry?.uuid === cur);
if (!found || !priorQAExcludedLines.has(found.index)) return cur;
cur = e.parentUuid || null;
}
return null;
}
if (tipUuid) tipUuid = nearestNonExcludedAncestor(tipUuid);
// Fallback: if no snapshot command found, use the last entry with a uuid
if (!tipUuid) {
for (let i = preCutEntries.length - 1; i >= 0; i--) {
if (preCutEntries[i].entry?.uuid && !priorQAExcludedLines.has(preCutEntries[i].index)) {
tipUuid = preCutEntries[i].entry.uuid;
break;
}
}
}
// Collect all uuids on the active branch
const activeBranchUuids = new Set();
let current = tipUuid;
while (current) {
activeBranchUuids.add(current);
current = byUuid[current]?.parentUuid || null;
}
// Step 4: filter — keep entries on the active branch.
//
// Claude Code writes several bookkeeping entry types alongside the message
// tree that don't carry a branch uuid. Their content references whatever
// branch was active when they were written, so if the user has rewound,
// these will leak rewound-branch state (file backups, prior prompt text,
// stale titles, queued prompts, PR links, etc.) into the snapshot — a leak
// snapshot.patch doesn't show. Drop the ones we can't attribute to the
// active branch.
//
// `file-history-snapshot` is special-cased: it carries a `messageId`
// pointing at the message whose pre-edit state it tracks, so we can
// keep only those whose messageId is on the active branch. That
// preserves /rewind functionality after a snapshot is restored (rewind
// needs the file-backup metadata) while still dropping records from
// rewound branches.
const BLANKET_DROP_TYPES = new Set([
'agent-name',
'ai-title',
'custom-title',
'last-prompt',
'permission-mode',
'pr-link',
'queue-operation',
]);
function prunedClaudeLines() {
const kept = [];
for (const { line, entry, index } of preCutEntries) {
if (priorQAExcludedLines.has(index)) continue;
if (!entry) {
// Unparseable line — keep as-is so we don't lose data we can't classify.
kept.push(line);
continue;
}
if (entry.uuid) {
if (activeBranchUuids.has(entry.uuid)) kept.push(line);
continue;
}
// No uuid: bookkeeping entry.
if (entry.type === 'file-history-snapshot') {
// Keep only if the message it tracks is on the active branch.
if (entry.messageId && activeBranchUuids.has(entry.messageId)) {
kept.push(line);
}
continue;
}
if (!BLANKET_DROP_TYPES.has(entry.type)) kept.push(line);
}
return kept;
}
// Rewind branches and prior-Q&A exclusion are Claude-transcript concerns; a linear
// harness transcript just truncates at its boundary.
let startLine;
if (!IS_CLAUDE && startMarkerPath && fs.existsSync(startMarkerPath)) {
try {
const marker = JSON.parse(fs.readFileSync(startMarkerPath, 'utf8'));
if (marker.transcript_path === harnessSession.rawPath) startLine = marker.line_count;
} catch {
startLine = undefined;
}
}
if (!IS_CLAUDE && startLine === undefined) {
console.error(
'WARNING: no snapshot start marker for this session — the snapshot Q&A may be captured as conversation. Run capture-snapshot.mjs --mark-start before the annotation questions.'
);
}
const outputLines = IS_CLAUDE
? prunedClaudeLines()
: linearSnapshotLines(harnessSession, startLine);
if (outputLines.length === 0) {
console.error(
`WARNING: found no conversation to seed in ${transcript_path}, so this snapshot has no prior turns. The task will run cold from its prompt alone — fine if that is what you want, but if you meant to capture a conversation, check that the exchange you wanted came BEFORE this snapshot.`
);
}
// Zero bytes, not a lone newline, when there is nothing to seed: downstream decides
// single- vs multi-turn on the file's SIZE, so a 1-byte file would try to resume nothing.
fs.writeFileSync(
path.join(snapshotDir, 'session.jsonl'),
outputLines.length > 0 ? outputLines.join('\n') + '\n' : ''
);
// --- Write boundary record so the NEXT capture-snapshot in this session
// can identify and exclude this snapshot's Q&A subgraph ---
//
// The record pairs the current invocation's command-line uuid with the
// uuid of the last entry in the JSONL at this moment (which is whichever
// assistant turn invoked us as a tool). A subsequent capture run reads
// this file, finds these two uuids in its raw lines, and excludes the
// range — a small leak still exists for entries appended AFTER capture
// returns (the assistant's "Snapshot saved to: ..." reply), but the
// substantive annotation Q&A is fully bounded.
if (dataDir && session_id && cutIndex < rawLines.length) {
let snapshotCommandUuid = null;
try {
snapshotCommandUuid = JSON.parse(rawLines[cutIndex]).uuid || null;
} catch {
/* leave null — we'll skip writing */
}
// Re-read transcript so we pick up any lines Claude Code has appended
// since we read it above (the assistant's tool-use entry, etc.).
let lastEntryUuid = null;
try {
const liveLines = fs.readFileSync(transcript_path, 'utf8').trimEnd().split('\n');
for (let i = liveLines.length - 1; i >= 0; i--) {
try {
const e = JSON.parse(liveLines[i]);
if (e.uuid) {
lastEntryUuid = e.uuid;
break;
}
} catch {
/* skip */
}
}
} catch {
/* transcript unreadable now — skip writing */
}
if (snapshotCommandUuid && lastEntryUuid) {
const boundariesPath = path.join(dataDir, 'snapshot-boundaries.jsonl');
try {
fs.mkdirSync(dataDir, { recursive: true });
fs.appendFileSync(
boundariesPath,
JSON.stringify({
sessionUuid: session_id,
snapshotCommandUuid,
lastEntryUuid,
timestamp: new Date().toISOString(),
}) + '\n'
);
} catch {
// Best-effort: a missing boundary just means the next run falls back
// to the conservative "exclude through next snapshot" heuristic.
}
}
}
// Copy subagents and tool-results if they exist
const sessionSiblingDir = transcript_path.replace(/\.jsonl$/, '');
if (fs.existsSync(sessionSiblingDir) && fs.statSync(sessionSiblingDir).isDirectory()) {
fs.cpSync(sessionSiblingDir, path.join(snapshotDir, 'session'), { recursive: true });
// Claude Code creates subagent files with write-only permissions (--w-------).
// Fix them so downstream tools (cpSync in snapshot-to-task, Harbor's dirhash) can read them.
execSync(`chmod -R +r "${path.join(snapshotDir, 'session')}"`, { stdio: 'pipe' });
}
// --- Capture git state as a patch ---
// Returns raw stdout bytes — callers that want a single-line value must
// .trim() themselves. Don't trim here: some callers (git diff) produce
// patches where a trailing " \n" blank-context line is load-bearing, and
// stripping it corrupts the patch.
function git(cmd, opts) {
try {
return execSync(`git ${cmd}`, {
encoding: 'utf8',
maxBuffer: 50 * 1024 * 1024,
cwd: gitRepo,
stdio: ['pipe', 'pipe', 'pipe'],
...opts,
});
} catch {
return null;
}
}
const commit = git('rev-parse HEAD')?.trim() ?? null;
const branch = git('rev-parse --abbrev-ref HEAD')?.trim() ?? null;
const remoteUrl = git('remote get-url origin')?.trim() ?? null;
// Generate a unified patch representing the workspace state AT THE END OF
// THE PRIOR TURN — i.e., everything done up to but not including the turn
// being snapshotted. This is the state the trial agent should inherit so
// it gets a fresh attempt at the prompt that triggered the snapshot.
//
// The UserPromptSubmit hook checkpoints the working tree to
// `refs/raccoon/turn-checkpoint` at every turn boundary (skipping snapshot
// invocations themselves), so the latest checkpoint is exactly the state
// at the start of the snapshotted turn. We diff HEAD against that
// checkpoint to produce the patch.
//
// Falls back to the pre-checkpoint behavior (full working-tree diff) when
// no checkpoint exists — e.g., the worker took a snapshot before any
// non-snapshot user message was sent, or the hook never fired (legacy
// session, plugin re-installed mid-session, etc.).
if (gitRepo) {
try {
const tmpIndex = path.join(snapshotDir, '.tmp-git-index');
const indexEnv = { ...process.env, GIT_INDEX_FILE: tmpIndex };
// Prefer the FROZEN ref — this is set by checkpoint-workspace at the
// moment the user invokes /create-snapshot:*, before any Q&A turns
// have a chance to advance the live checkpoint past the state we
// want to capture. Fall back to the live checkpoint (then to
// working-tree diff) for backward-compat or if the freeze step failed.
let baseline = null;
try {
baseline = git('rev-parse refs/raccoon/turn-checkpoint-frozen')?.trim() ?? null;
} catch {
baseline = null;
}
if (!baseline) {
try {
baseline = git('rev-parse refs/raccoon/turn-checkpoint')?.trim() ?? null;
} catch {
baseline = null;
}
}
// --binary --full-index, on both branches: a plain `git diff` records a
// binary difference as an opaque `Binary files a/x and /dev/null differ`
// stub, and `git apply` refuses it ("without full index line"), so
// build-workspace.sh can't rebuild the task at all. Nobody has to edit a
// binary to hit this — a tracked .DS_Store the toolkit zip strips from the
// shipped checkout reads as a binary deletion in every session.
//
// maxBuffer: inlined binaries make patches far bigger than text diffs, and
// exceeding the default cap would throw away the whole patch silently.
const diffOpts = { env: indexEnv, maxBuffer: 512 * 1024 * 1024 };
let patch;
if (baseline) {
// Diff HEAD against the prior-turn checkpoint. Untracked files in
// the checkpoint have been committed to the checkpoint tree, so
// they're included automatically.
patch = git(`diff --binary --full-index HEAD ${baseline}`, diffOpts);
} else {
// No checkpoint — fall back to live working-tree diff (pre-fix
// behavior). Captures everything different from HEAD, including
// any agent edits during the current turn.
git('read-tree HEAD', { env: indexEnv });
git('add -A', { env: indexEnv });
patch = git('diff --cached --binary --full-index HEAD', diffOpts);
}
try {
fs.unlinkSync(tmpIndex);
} catch {
/* ignore */
}
if (patch) {
fs.writeFileSync(
path.join(snapshotDir, 'snapshot.patch'),
patch.endsWith('\n') ? patch : patch + '\n'
);
}
} catch {
// Read-only repo or other git error — skip patch generation
}
}
// --- Copy annotation ---
const annotation = JSON.parse(fs.readFileSync(annotationPath, 'utf8'));
fs.writeFileSync(
path.join(snapshotDir, 'annotation.json'),
JSON.stringify(annotation, null, 2) + '\n'
);
// Clean up temp file
try {
fs.unlinkSync(annotationPath);
} catch {
// Ignore cleanup failures
}
// --- Write metadata ---
const metadata = {
slug: slug,
session_uuid: session_id,
// The harness the session was actually read as, so it can't disagree with what
// was captured.
harness: HARNESS,
original_cwd: process.cwd(),
commit: commit,
branch: branch,
remote_url: remoteUrl,
timestamp: new Date().toISOString(),
plugin_version: '0.2.0',
};
fs.writeFileSync(path.join(snapshotDir, 'metadata.json'), JSON.stringify(metadata, null, 2) + '\n');
// --- Generate restore.sh ---
const restoreScript = `#!/usr/bin/env bash
set -euo pipefail
# Restore a snapshot for resuming a Claude Code conversation.
#
# Usage: ./restore.sh [target-dir]
# target-dir: directory to clone/checkout the repo into (default: ./repo)
SCRIPT_DIR="$(cd "$(dirname "\${BASH_SOURCE[0]}")" && pwd)"
TARGET_DIR="\${1:-./repo}"
# Read metadata
COMMIT=$(jq -r '.commit' "$SCRIPT_DIR/metadata.json")
REMOTE=$(jq -r '.remote_url' "$SCRIPT_DIR/metadata.json")
SESSION_UUID=$(jq -r '.session_uuid' "$SCRIPT_DIR/metadata.json")
echo "Cloning $REMOTE at $COMMIT..."
git clone "$REMOTE" "$TARGET_DIR"
cd "$TARGET_DIR"
git checkout "$COMMIT"
# Apply snapshot patch if present
if [ -f "$SCRIPT_DIR/snapshot.patch" ]; then
echo "Applying snapshot.patch..."
git apply "$SCRIPT_DIR/snapshot.patch"
fi
# Install conversation so the authoring harness can resume it
${
IS_CLAUDE
? `ENCODED_CWD=$(echo "$PWD" | sed 's|/|-|g; s|^-||')
DEST_DIR="$HOME/.claude/projects/-$ENCODED_CWD"
mkdir -p "$DEST_DIR"
cp "$SCRIPT_DIR/session.jsonl" "$DEST_DIR/$SESSION_UUID.jsonl"
if [ -d "$SCRIPT_DIR/session" ]; then
cp -r "$SCRIPT_DIR/session" "$DEST_DIR/$SESSION_UUID"
fi
echo ""
echo "Snapshot restored. To resume the conversation:"
echo " cd $TARGET_DIR"
echo " claude --resume $SESSION_UUID"`
: `DEST_DIR="\${CODEX_HOME:-$HOME/.codex}/sessions/$(date -u +%Y/%m/%d)"
mkdir -p "$DEST_DIR"
cp "$SCRIPT_DIR/session.jsonl" \\
"$DEST_DIR/rollout-$(date -u +%Y-%m-%dT%H-%M-%S).000Z-$SESSION_UUID.jsonl"
echo ""
echo "Snapshot restored. To resume the conversation:"
echo " cd $TARGET_DIR"
echo " codex resume $SESSION_UUID"`
}
`;
fs.writeFileSync(path.join(snapshotDir, 'restore.sh'), restoreScript);
fs.chmodSync(path.join(snapshotDir, 'restore.sh'), 0o755);
try {
execSync('bash -ic "_ev snapshot_created 2>/dev/null" 2>/dev/null', {
stdio: 'ignore',
timeout: 5000,
});
} catch {
// best-effort
}
// --- Done ---
const fullSnapshotDir = path.resolve(snapshotDir);
console.log(`Snapshot saved to: ${fullSnapshotDir}`);
console.log(` session.jsonl — conversation transcript`);
if (fs.existsSync(sessionSiblingDir) && fs.statSync(sessionSiblingDir).isDirectory()) {
console.log(` session/ — subagents + tool results`);
}
if (fs.existsSync(path.join(snapshotDir, 'snapshot.patch'))) {
console.log(` snapshot.patch — working tree changes`);
}
console.log(` annotation.json — worker annotations`);
console.log(` metadata.json — session metadata`);
console.log(` restore.sh — restore script for resuming`);

View File

@@ -0,0 +1,485 @@
#!/usr/bin/env node
// Rewind-aware workspace checkpointing for the reduced-toolset Explore agent.
// One script, two hook events (branches on hook_event_name):
//
// UserPromptSubmit -> CAPTURE
// Snapshot the pre-turn working tree into refs/raccoon/turn-checkpoint (the
// chain capture-snapshot uses for snapshot.patch) AND record, in
// .git/raccoon-state.json, anchor_map[tip] = checkpoint-commit and
// last_anchor = tip. `tip` is the conversation node the new prompt attaches
// to (the END of the previous turn) — exactly the node a future /rewind to
// THIS turn will branch from. Anchoring to the prior tip (not the
// just-submitted, maybe-unflushed message) makes capture race-free.
//
// PreToolUse (first tool call of a turn) -> RECONCILE
// Claude Code's /rewind restores the conversation but NOT bash-made edits,
// and fires no hook. By the first tool call the post-rewind branch message is
// reliably persisted and the agent has not yet read/edited code. We parse the
// transcript into a parentUuid DAG, pick the ACTIVE branch (leaf with the
// newest tip), and walk it for the newest checkpoint anchor that is a genuine
// rewind fork (the anchor still has an orphaned child branch — the discarded
// turns). If found, restore the working tree to that checkpoint before the tool
// runs. Idempotent per (anchor, branch): restores once per rewind.
//
// Every failure path is a safe no-op: the hook never aborts the session and
// never restores to an unverified tree.
import { execSync } from 'node:child_process';
import fs from 'node:fs';
import path from 'node:path';
const CHECKPOINT_REF = 'refs/raccoon/turn-checkpoint';
const FROZEN_REF = 'refs/raccoon/turn-checkpoint-frozen';
// Synthetic parent for every parentless transcript node, so that a rewind to the
// VERY FIRST turn (where the new prompt also has parentUuid=null) is detected by
// the same divergence machinery as any other turn.
const ROOT = '__ROOT__';
const RACCOON_AUTHOR = {
GIT_AUTHOR_NAME: 'raccoon',
GIT_AUTHOR_EMAIL: 'raccoon@local',
GIT_COMMITTER_NAME: 'raccoon',
GIT_COMMITTER_EMAIL: 'raccoon@local',
};
function makeGit(gitDir, extraEnv) {
const env = { ...process.env, ...extraEnv };
return (cmd) =>
execSync(`git ${cmd}`, {
cwd: gitDir,
env,
stdio: ['pipe', 'pipe', 'pipe'],
encoding: 'utf8',
}).trim();
}
function findGitDir(cwd) {
let gitDir = cwd;
for (let i = 0; i < 10; i++) {
if (fs.existsSync(path.join(gitDir, '.git'))) return gitDir;
const parent = path.dirname(gitDir);
if (parent === gitDir) return null;
gitDir = parent;
}
return null;
}
// ---- transcript + state ----
function readEntries(transcriptPath) {
try {
if (!transcriptPath || !fs.existsSync(transcriptPath)) return [];
const out = [];
for (const raw of fs.readFileSync(transcriptPath, 'utf8').split('\n')) {
const line = raw.trim();
if (!line) continue;
let o;
try {
o = JSON.parse(line);
} catch {
continue;
}
if (o && typeof o.uuid === 'string') out.push(o);
}
return out;
} catch {
return [];
}
}
function tsOf(e) {
const t = e && e.timestamp ? Date.parse(e.timestamp) : 0;
return Number.isFinite(t) ? t : 0;
}
function statePath(gitDir) {
return path.join(gitDir, '.git', 'raccoon-state.json');
}
function loadState(gitDir) {
try {
const s = JSON.parse(fs.readFileSync(statePath(gitDir), 'utf8'));
return { anchor_map: {}, last_anchor: null, reconciled_for: null, ...s };
} catch {
return { anchor_map: {}, last_anchor: null, reconciled_for: null };
}
}
function saveState(gitDir, s) {
try {
// Atomic write: a tmp file + rename, so a hook killed mid-write can never
// leave a half-written (corrupt) state.json behind.
const target = statePath(gitDir);
const tmp = `${target}.tmp`;
fs.writeFileSync(tmp, JSON.stringify(s));
fs.renameSync(tmp, target);
} catch {
// best-effort
}
}
// Loose checkpoint objects must survive: a restore resets the checkpoint chain
// ref backward, which can orphan later anchors' commits. Disabling auto-gc keeps
// every anchor commit fetchable for a future rewind. The task container is
// ephemeral, so accumulating loose objects is harmless.
function disableAutoGc(gitDir) {
try {
makeGit(gitDir)('config gc.auto 0');
} catch {
// best-effort
}
}
// The conversation node the new prompt attaches to = end of the previous turn.
// Newest uuid-bearing entry, excluding the just-submitted prompt (which may or
// may not be flushed yet — excluding it makes this race-robust).
function conversationTip(entries, currentPrompt) {
const cp = (currentPrompt || '').trim();
for (let i = entries.length - 1; i >= 0; i--) {
const e = entries[i];
if (!e.uuid) continue;
const role = e.type || (e.message && e.message.role);
const content = e.message && e.message.content;
if (role === 'user' && typeof content === 'string' && cp && content.trim() === cp) continue;
return e.uuid;
}
return null;
}
// ---- capture (UserPromptSubmit) ----
function ensureExcludes(gitDir) {
const localExcludePath = path.join(gitDir, '.git', 'info', 'exclude');
const MARKER = '# raccoon-checkpoint excludes (auto-managed):';
const excludes = [
MARKER,
'.pnpm-store/',
'.yarn/cache/',
'.yarn/install-state.gz',
'vendor/bundle/',
'.bundle/cache/',
'.raccoon-setup-done', // run-app's per-repo first-use setup marker (polyglot toolkits)
];
try {
let existing = '';
try {
existing = fs.readFileSync(localExcludePath, 'utf8');
} catch {
existing = '';
}
if (!existing.includes(MARKER)) {
fs.mkdirSync(path.dirname(localExcludePath), { recursive: true });
fs.appendFileSync(localExcludePath, '\n' + excludes.join('\n') + '\n');
}
} catch {
// best-effort
}
}
function freezeForSnapshot(gitDir) {
// Freeze the state at the START of the turn being snapshotted — i.e.,
// whatever CHECKPOINT_REF already holds (or HEAD, if no turn has happened
// yet this session). This must NOT be the live working tree: the live tree
// includes the edits made during the turn that triggered /snapshot, and
// capture-snapshot's `diff HEAD <frozen>` is supposed to exclude exactly
// that turn so the trial agent gets a fresh attempt at the prompt (see the
// comment above baseline selection in capture-snapshot.mjs). Freezing the
// live tree instead bakes the agent's just-made edits into the snapshot.
try {
const git = makeGit(gitDir);
let source = null;
try {
source = git(`rev-parse ${CHECKPOINT_REF}`);
} catch {
try {
source = git('rev-parse HEAD');
} catch {
source = null;
}
}
if (source) git(`update-ref ${FROZEN_REF} ${source}`);
} catch {
// best-effort — never break /snapshot
}
}
// The snapshot invocation, in whichever form the harness uses: Claude Code takes
// `/create-snapshot:snapshot`, codex takes `$create-snapshot:snapshot`. Both send the raw
// text as `prompt` on the UserPromptSubmit hook (verified against codex 0.146.1), so the
// prefix is the only difference — and missing it means freezing never happens and the
// snapshotted turn's own edits get baked into the workspace.
const SNAPSHOT_INVOCATION_RE = /^[/$](?:create-snapshot|snapshot)(?![\w-])/;
function capture(gitDir, data) {
const prompt = (data.prompt ?? '').trim();
if (SNAPSHOT_INVOCATION_RE.test(prompt)) {
freezeForSnapshot(gitDir);
return;
}
let commit = null;
try {
const tmpIndex = path.join(gitDir, '.git', 'raccoon-checkpoint.index');
const git = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: tmpIndex });
ensureExcludes(gitDir);
disableAutoGc(gitDir);
git('read-tree HEAD');
git('add -A');
const tree = git('write-tree');
let parent;
try {
parent = git(`rev-parse ${CHECKPOINT_REF}`);
} catch {
parent = git('rev-parse HEAD');
}
commit = git(`commit-tree ${tree} -p ${parent} -m "raccoon-checkpoint: pre-turn"`);
git(`update-ref ${CHECKPOINT_REF} ${commit}`);
try {
fs.unlinkSync(tmpIndex);
} catch {
// ignore
}
} catch {
return; // never break the session
}
// Record the anchor mapping for rewind reconciliation. On the very first turn
// there is no prior node, so we anchor to the synthetic ROOT — this is the
// pre-turn-1 (initial) state, which a rewind to the first turn restores to.
try {
const tip = conversationTip(readEntries(data.transcript_path), data.prompt) || ROOT;
if (commit) {
const s = loadState(gitDir);
s.anchor_map[tip] = commit;
s.last_anchor = tip;
saveState(gitDir, s);
}
} catch {
// best-effort; capture still succeeded
}
}
// ---- reconcile (PreToolUse) ----
function subtreeContains(start, target, children) {
const stack = [start];
const seen = new Set();
while (stack.length > 0) {
const n = stack.pop();
if (n === target) return true;
if (seen.has(n)) continue;
seen.add(n);
for (const c of children.get(n) || []) stack.push(c);
}
return false;
}
function reachesLeaf(start, leafSet, children) {
const stack = [start];
const seen = new Set();
while (stack.length > 0) {
const n = stack.pop();
if (leafSet.has(n)) return true;
if (seen.has(n)) continue;
seen.add(n);
for (const c of children.get(n) || []) stack.push(c);
}
return false;
}
// node is a genuine rewind fork: >=2 children, one reaching the active leaf and
// at least one reaching a different (orphaned) leaf.
function isDivergence(node, activeLeaf, leaves, children) {
const kids = children.get(node) || [];
if (kids.length < 2) return false;
const leafSet = new Set(leaves);
const reachesActive = kids.some((k) => subtreeContains(k, activeLeaf, children));
const reachesOther = kids.some(
(k) => !subtreeContains(k, activeLeaf, children) && reachesLeaf(k, leafSet, children)
);
return reachesActive && reachesOther;
}
// Restore the working tree to a commit's tree, saving the current state to a
// safety ref first. Returns the safety ref name, or null on failure.
function restoreToCommit(gitDir, commit) {
try {
const restoreIndex = path.join(gitDir, '.git', 'raccoon-restore.index');
const stashIndex = path.join(gitDir, '.git', 'raccoon-stash.index');
const gitStash = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: stashIndex });
const gitRestore = makeGit(gitDir, { ...RACCOON_AUTHOR, GIT_INDEX_FILE: restoreIndex });
const gitPlain = makeGit(gitDir, RACCOON_AUTHOR);
ensureExcludes(gitDir);
disableAutoGc(gitDir);
gitStash('read-tree HEAD');
gitStash('add -A');
const curTree = gitStash('write-tree');
let parent = null;
try {
parent = gitPlain('rev-parse HEAD');
} catch {
parent = null;
}
const curCommit = gitStash(
`commit-tree ${curTree}${parent ? ` -p ${parent}` : ''} -m "raccoon: pre-rewind safety"`
);
const safetyRef = `refs/raccoon/pre-rewind/${Date.now()}`;
gitPlain(`update-ref ${safetyRef} ${curCommit}`);
// Files to delete = present in the current tree but absent from the target
// checkpoint. Computed as a set difference of `ls-tree` listings rather than
// `diff --diff-filter=A`, because git's rename/copy detection reclassifies an
// added path as R/C, which a filter on "A" would miss — leaving the renamed-to
// file stranded in the worktree after a restore.
let added = [];
try {
const inCheckpoint = new Set(
gitPlain(`ls-tree -r --name-only ${commit}`).split('\n').filter(Boolean)
);
const inCurrent = gitPlain(`ls-tree -r --name-only ${curCommit}`).split('\n').filter(Boolean);
added = inCurrent.filter((f) => !inCheckpoint.has(f));
} catch {
added = [];
}
gitRestore(`read-tree ${commit}`);
gitRestore('checkout-index -a -f');
for (const f of added) {
try {
fs.rmSync(path.join(gitDir, f), { force: true });
} catch {
// ignore
}
}
for (const idx of [restoreIndex, stashIndex]) {
try {
fs.unlinkSync(idx);
} catch {
// ignore
}
}
return safetyRef;
} catch {
return null;
}
}
function reconcile(gitDir, data) {
let s;
try {
s = loadState(gitDir);
} catch {
return;
}
if (!s || !s.anchor_map || Object.keys(s.anchor_map).length === 0) return;
const entries = readEntries(data.transcript_path);
if (entries.length === 0) return;
const byId = new Map();
const children = new Map();
const referenced = new Set();
for (const e of entries) byId.set(e.uuid, e);
for (const e of entries) {
// Parentless or dangling-parent nodes hang off the synthetic ROOT.
const p = e.parentUuid && byId.has(e.parentUuid) ? e.parentUuid : ROOT;
if (!children.has(p)) children.set(p, []);
children.get(p).push(e.uuid);
referenced.add(p);
}
const leaves = [...byId.keys()].filter((u) => !referenced.has(u));
if (leaves.length === 0) return;
// active branch = leaf with the newest tip (the branch CC is appending to now)
let activeLeaf = leaves[0];
for (const u of leaves) if (tsOf(byId.get(u)) > tsOf(byId.get(activeLeaf))) activeLeaf = u;
// Walk the active chain (through ROOT) for the newest anchor that is a GENUINE
// rewind divergence: the anchor node has an orphaned child branch (the discarded
// turns) alongside the active branch. We skip anchors that are NOT forks, so:
// - pure forward progress (single child) never triggers a restore;
// - redoing the LAST turn still triggers (the new turn builds on the same
// boundary as the prior turn, but the discarded turn is an orphan sibling);
// - a stray anchor recorded on the active branch itself (e.g. the post-rewind
// capture's tip) can't mask the real divergence further up the chain.
// branchChild = the divergence node's child on the active path (the "branch id").
let cur = activeLeaf;
let prev = null;
let branchChild = null;
const seen = new Set();
let activeAnchor = null;
while (cur && !seen.has(cur)) {
seen.add(cur);
if (
Object.prototype.hasOwnProperty.call(s.anchor_map, cur) &&
isDivergence(cur, activeLeaf, leaves, children)
) {
activeAnchor = cur;
branchChild = prev;
break;
}
prev = cur;
if (cur === ROOT) break;
const e = byId.get(cur);
cur = e && e.parentUuid && byId.has(e.parentUuid) ? e.parentUuid : ROOT;
}
if (!activeAnchor) return;
// Idempotency keyed on (anchor, active branch) — NOT the active leaf. Every
// tool call within a post-rewind turn advances the leaf, but the branch is
// stable, so we restore exactly once per rewind. A fresh re-rewind to the same
// turn forks a NEW child off the anchor, changing the key, so it restores again.
const reconKey = `${activeAnchor}:${branchChild || ''}`;
if (s.reconciled_for === reconKey) return; // this rewind already reconciled
const safety = restoreToCommit(gitDir, s.anchor_map[activeAnchor]);
try {
makeGit(gitDir)(`update-ref ${CHECKPOINT_REF} ${s.anchor_map[activeAnchor]}`);
} catch {
// ignore
}
s.reconciled_for = reconKey;
saveState(gitDir, s);
// Record the restore to a side log ONLY — never stdout/stderr. Hook output on
// PreToolUse is captured into the transcript (as an attachment entry) and the
// transcript is the task data, so any emission here would contaminate it. The
// log lives under .git/, which is never staged, snapshotted, or transcribed.
try {
const line =
`${new Date().toISOString()} rewind reconciled: restored to anchor ${activeAnchor} ` +
`(${s.anchor_map[activeAnchor]})${safety ? `; pre-rewind state saved to ${safety}` : ''}\n`;
fs.appendFileSync(path.join(gitDir, '.git', 'raccoon-rewind.log'), line);
} catch {
// best-effort; the restore itself already succeeded
}
}
function main(data) {
const cwd = data.cwd || process.cwd();
const gitDir = findGitDir(cwd);
if (!gitDir) return;
const event = data.hook_event_name || (data.tool_name ? 'PreToolUse' : 'UserPromptSubmit');
// RECONCILE undoes a /rewind, which only Claude Code has. Other harnesses append
// and never fork, so there is nothing to reconcile and the transcript it would walk
// has no parentUuid DAG.
const canRewind = (process.env.RACCOON_HARNESS || 'claude-code') === 'claude-code';
if (event === 'PreToolUse') {
if (canRewind) reconcile(gitDir, data);
} else capture(gitDir, data);
}
let input = '';
process.stdin.setEncoding('utf8');
process.stdin.on('data', (chunk) => {
input += chunk;
});
process.stdin.on('end', () => {
let data;
try {
data = JSON.parse(input);
} catch {
process.exit(0);
}
try {
main(data);
} catch {
// A failure here must never break the user's session.
}
process.exit(0);
});

View File

@@ -0,0 +1,29 @@
// Types for harness-session.mjs, so TS consumers (its test, snapshot-to-task) see a
// real shape instead of `any`.
export interface Turn {
/** Line index in the native session file. */
index: number;
role: 'user' | 'assistant';
text: string;
/** A slash-command turn, not real conversation. */
isCommand: boolean;
/** This record concluded its turn — the truncation boundary. */
endsTurn: boolean;
}
export interface Session {
harness: string;
rawPath: string;
/** The harness own id for this conversation. */
sessionId: string | null;
lines: string[];
turns: Turn[];
}
export function supportedHarnesses(): string[];
export function readSession(harness: string, recordedPath?: string): Session | null;
export function truncationIndex(turns: Turn[]): number;
export function turnsFromLines(harness: string, lines: string[]): Turn[];
export function linearSnapshotLines(session: Session, startLine?: number): string[];
export function stripAuthoringScaffolding(harness: string, lines: string[]): string[];

View File

@@ -0,0 +1,325 @@
// Locate and read a harness's native conversation, so capture-snapshot can work
// against any harness. Everything else in capture (snapshot.patch, restore.sh,
// annotation, metadata) is harness-agnostic.
//
// The returned session stays in the harness's OWN native format: the seeding design
// hands a native blob back to the same harness, and codex_agent reads the same staged
// /tmp/snapshot-session/session.jsonl path that snapshot_agent does.
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
/**
* @typedef {object} Turn
* @property {number} index line index in the native session file
* @property {'user'|'assistant'} role
* @property {string} text
* @property {boolean} isCommand a slash-command turn, not real conversation
* @property {boolean} endsTurn this record concluded its turn
*/
/**
* @typedef {object} Session
* @property {string} harness
* @property {string} rawPath
* @property {string|null} sessionId the harness's own id for this conversation
* @property {string[]} lines
* @property {Turn[]} turns
*/
// Newest matching file beneath `root`, or null. Ties on mtime break on path so the
// answer is stable — two sessions written in the same millisecond are common.
function newestUnder(root, matches) {
if (!fs.existsSync(root)) return null;
const found = [];
const walk = (dir) => {
let entries;
try {
entries = fs.readdirSync(dir, { withFileTypes: true });
} catch {
return;
}
for (const entry of entries) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) walk(full);
else if (matches(entry.name)) found.push({ full, mtimeMs: fs.statSync(full).mtimeMs });
}
};
walk(root);
if (found.length === 0) return null;
found.sort((a, b) => b.mtimeMs - a.mtimeMs || b.full.localeCompare(a.full));
return found[0].full;
}
// User-role records codex writes that the human did not type: its own environment
// preamble, a `$name` skill invocation, and the SKILL.md body injected in response.
// Matched only at the START of the text, so a turn that merely quotes one is still real
// conversation.
function isCodexCommandText(text) {
const trimmed = (text || '').trimStart();
if (trimmed.startsWith('<skill>') || trimmed.startsWith('<environment_context>')) return true;
return /^\$[\w:.-]+\s*$/.test(trimmed);
}
const HARNESSES = {
'claude-code': {
/** Claude Code records one JSONL per session under ~/.claude/projects/<encoded-cwd>/. */
findSession() {
return newestUnder(path.join(os.homedir(), '.claude', 'projects'), (n) =>
n.endsWith('.jsonl')
);
},
/** Claude names the transcript for its session id. */
sessionId(rawPath) {
return path.basename(rawPath, '.jsonl');
},
/**
* One turn per conversational record. `endsTurn` marks an assistant record that
* concluded its turn — the truncation boundary. Bookkeeping records (attachments,
* file-history, permission-mode) carry no role and are skipped.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let entry;
try {
entry = JSON.parse(line);
} catch {
continue;
}
const role =
entry.type === 'user' ? 'user' : entry.type === 'assistant' ? 'assistant' : null;
if (!role) continue;
const content = entry.message?.content;
const text =
typeof content === 'string'
? content
: Array.isArray(content)
? content
.filter((b) => b && b.type === 'text')
.map((b) => b.text ?? '')
.join('')
: '';
turns.push({
index,
role,
text,
isCommand:
role === 'user' &&
typeof content === 'string' &&
/<command-name>|<command-message>|<local-command-caveat>/.test(content),
endsTurn: role === 'assistant' && entry.message?.stop_reason === 'end_turn',
});
}
return turns;
},
},
codex: {
/** codex writes rollout JSONL under $CODEX_HOME/sessions/<date>/. */
findSession() {
const home = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
return newestUnder(
path.join(home, 'sessions'),
(n) => n.startsWith('rollout-') && n.endsWith('.jsonl')
);
},
/** `codex resume <id>` resolves the id recorded in session_meta, not the filename. */
sessionId(rawPath, lines) {
for (const line of lines) {
try {
const rec = JSON.parse(line);
if (rec.type === 'session_meta' && rec.payload?.id) return rec.payload.id;
} catch {
continue;
}
}
return null;
},
/**
* codex rollouts carry `response_item` records whose payload is a message with a
* role. An assistant message with no following tool activity ends the turn; codex
* records no stop_reason, so a turn ends where the next user message begins —
* resolved after the fact below.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let record;
try {
record = JSON.parse(line);
} catch {
continue;
}
if (record.type !== 'response_item') continue;
const payload = record.payload ?? {};
if (payload.type !== 'message') continue;
const role =
payload.role === 'user' ? 'user' : payload.role === 'assistant' ? 'assistant' : null;
if (!role) continue;
const text = Array.isArray(payload.content)
? payload.content.map((b) => b?.text ?? '').join('')
: typeof payload.content === 'string'
? payload.content
: '';
turns.push({
index,
role,
text,
isCommand: role === 'user' && isCodexCommandText(text),
endsTurn: false,
});
}
// An assistant turn ends where the next user turn starts, or at the end.
for (let i = 0; i < turns.length; i += 1) {
if (turns[i].role !== 'assistant') continue;
const next = turns[i + 1];
turns[i].endsTurn = !next || next.role === 'user';
}
return turns;
},
},
};
/** @returns {string[]} */
export function supportedHarnesses() {
return Object.keys(HARNESSES);
}
/**
* Read the current session for `harness`. Returns null when nothing is found, so the
* caller can report which harness had no conversation to capture.
*/
/**
* @param {string} harness
* @param {string} [recordedPath] transcript recorded by the SessionStart hook; preferred
* over the newest-file scan, which can pick a different session in a busy container.
* @returns {Session | null}
*/
export function readSession(harness, recordedPath) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
const rawPath = recordedPath && fs.existsSync(recordedPath) ? recordedPath : reader.findSession();
if (!rawPath) return null;
const lines = fs.readFileSync(rawPath, 'utf8').trimEnd().split('\n');
return {
harness,
rawPath,
lines,
turns: reader.readTurns(lines),
sessionId: reader.sessionId(rawPath, lines),
};
}
/**
* Index of the last record to keep: the last turn-ending assistant record before the
* final real user turn. Drops the prompt that elicited the failure and the failure
* response, so the test agent inherits context but not the answer.
*
* Returns -1 when there is no such boundary (a one-shot conversation), which callers
* treat as "seed nothing and run cold".
*/
/**
* @param {Turn[]} turns
* @returns {number}
*/
export function truncationIndex(turns) {
let lastUser = -1;
for (const turn of turns) {
if (turn.role === 'user' && !turn.isCommand && turn.text.trim()) lastUser = turn.index;
}
if (lastUser < 0) return -1;
let cut = -1;
for (const turn of turns) {
if (turn.index >= lastUser) break;
if (turn.role === 'assistant' && turn.endsTurn) cut = turn.index;
}
return cut;
}
/**
* Parse already-read lines with a harness's reader, for callers that have the text
* rather than a path.
*
* @param {string} harness
* @param {string[]} lines
* @returns {Turn[]}
*/
export function turnsFromLines(harness, lines) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
return reader.readTurns(lines);
}
/**
* Lines to stage as the captured `session.jsonl` for a linear-transcript harness:
* everything up to the snapshot invocation, matching what Claude Code stages when it
* cuts at its slash-command line. Dropping the failure-eliciting turn happens later,
* in snapshot-to-task — capture keeps the full conversation.
*
* `startLine` is the rollout length recorded when the snapshot was invoked; without it
* the whole session is kept, which would include the snapshot's own Q&A.
*
* @param {Session} session
* @param {number} [startLine]
* @returns {string[]}
*/
export function linearSnapshotLines(session, startLine) {
if (typeof startLine === 'number' && startLine >= 0) {
return session.lines.slice(0, startLine);
}
return session.lines;
}
/**
* Drop records that describe the AUTHORING container rather than the conversation.
*
* codex records both its skill catalogue (a `developer` turn) and the machine it ran on (a
* `user` turn of `<environment_context>`). Native resume replays records byte-identically,
* so without this the test agent inherits a list of skills it does not have — one described
* as "capture the current conversation and repo state as a snapshot" — and a working
* directory that does not exist in the trial. codex re-injects both for the trial, and base
* instructions travel in `session_meta`, so removing them loses nothing. Claude's fork
* already re-records with the trial's own cwd; this brings codex to the same place.
*
* @param {string} harness
* @param {string[]} lines
* @returns {string[]}
*/
export function stripAuthoringScaffolding(harness, lines) {
if (harness === 'claude-code') return lines;
return lines.filter((raw) => {
let rec;
try {
rec = JSON.parse(raw);
} catch {
return true;
}
const payload = rec?.payload;
if (rec?.type !== 'response_item' || payload?.type !== 'message') return true;
const text = (payload.content ?? [])
.map((block) => (typeof block?.text === 'string' ? block.text : ''))
.join('')
.trim();
// Match the machine-generated shape only — a turn that STARTS with the tag — so a
// worker who quotes one of these strings mid-conversation keeps their turn.
if (payload.role === 'developer') return !text.startsWith('<skills_instructions>');
if (payload.role === 'user') return !text.startsWith('<environment_context>');
return true;
});
}

View File

@@ -0,0 +1,5 @@
/**
* Plugin-side re-export, so snapshot-to-task.ts resolves `./lib/copy-tree`
* both here and in the toolkit's flat scripts/ dir.
*/
export * from '../../../../raccoon-worker-toolkit/static/scripts/lib/copy-tree';

View File

@@ -0,0 +1,260 @@
/**
* Strip machine-identifying filesystem paths, and optional keywords, from a session
* transcript. Pure: raw JSONL in, JSONL out, no I/O.
*/
export const DEFAULT_PLACEHOLDER = '~/repo';
export const HOME_DIR_PLACEHOLDER = '~';
export const REDACTION_PLACEHOLDER = '[redacted]';
export interface SanitizeOptions {
/** Replacement for the cwd-prefix. Its dash-encoded form is derived from it. */
placeholder?: string;
/** Keyword regexes to redact. Empty by default, leaving a pure path-scrubber. */
forbiddenMarkers?: readonly RegExp[];
/**
* Exact prefix to strip. An inferred one is only the repo root when some cwd sat
* there, so callers that know the root pass it here.
*/
cwdPrefix?: string;
/** Several roots at once (a session spanning two checkouts). Wins over `cwdPrefix`. */
cwdPrefixes?: readonly string[];
/**
* Also strip home-rooted paths in the CONTENT: a sandbox-recorded session has a
* sandbox `cwd`, so the cwd passes never see the local checkout it still mentions.
*/
scrubEmbeddedHomePaths?: boolean;
}
export interface SanitizeResult {
sanitized: string;
prefixStripped: string | null;
encodedPrefixStripped: string | null;
homeDirStripped: string | null;
encodedHomeDirStripped: string | null;
embeddedPrefixStripped: string | null;
embeddedHomeDirStripped: string | null;
/** Replacement count per marker, keyed by the regex's source string. */
markersScrubbed: Record<string, number>;
}
/** Longest common prefix by path COMPONENT: `/a/bb` and `/a/b` share `/a`, not `/a/b`.
* Returns `''` when only the root `/` is common. */
export function findLongestCommonPathPrefix(paths: Iterable<string>): string {
const arr = Array.from(paths);
if (arr.length === 0) return '';
const splits = arr.map((p) => p.split('/'));
const minLen = Math.min(...splits.map((s) => s.length));
let lastShared = 0;
for (let i = 0; i < minLen; i++) {
const c = splits[0][i];
if (splits.some((s) => s[i] !== c)) break;
lastShared = i + 1;
}
// Only the leading empty piece matched → just the root, not useful.
if (lastShared <= 1) return '';
return splits[0].slice(0, lastShared).join('/');
}
/** The home-dir portion of an absolute path, or `null` for an unrecognized shape —
* better to skip the home pass than strip what may be repo content. */
export function extractHomeDir(cwdPrefix: string): string | null {
if (!cwdPrefix.startsWith('/')) return null;
// Windows-under-WSL shapes first: the generic drive shape below would stop at the
// drive letter and leave the account name in. A volume or drive root carries no
// identity by itself, so those take the directory under it.
const patterns: RegExp[] = [
/^\/mnt\/host\/[^/]+\/Users\/[^/]+/,
/^\/mnt\/[^/]+\/Users\/[^/]+/,
/^\/Users\/[^/]+/,
/^\/home\/[^/]+/,
/^\/Volumes\/[^/]+\/[^/]+/,
/^\/mnt\/[^/]+\/[^/]+/,
/^\/var\/root(?=\/|$)/,
/^\/root(?=\/|$)/,
];
for (const re of patterns) {
const m = cwdPrefix.match(re);
if (m) return m[0];
}
return null;
}
/** Every distinct `cwd` in the transcript. Read at the top level (Claude Code) and
* under `payload` (codex), so both harnesses are covered. Bad lines are skipped. */
export function collectCwds(raw: string): Set<string> {
const out = new Set<string>();
const add = (v: unknown) => {
if (typeof v === 'string' && v.startsWith('/')) out.add(v);
};
for (const line of raw.split('\n')) {
if (!line.trim()) continue;
let parsed: unknown;
try {
parsed = JSON.parse(line);
} catch {
continue;
}
if (typeof parsed !== 'object' || parsed === null) continue;
const rec = parsed as { cwd?: unknown; payload?: unknown };
add(rec.cwd);
if (typeof rec.payload === 'object' && rec.payload !== null) {
add((rec.payload as { cwd?: unknown }).cwd);
}
}
return out;
}
/** One path segment: stops at `/`, whitespace, quotes and JSON punctuation. */
const COMP = String.raw`[^/\s"'\\,:;)\]}<>]+`;
// macOS/Windows display names can contain spaces, but only consume them while
// more path follows, so a bare home-dir mention doesn't swallow trailing prose.
const USER_WITH_SPACES = `${COMP}(?:(?: +${COMP})+(?=/))?`;
const EMBEDDED_HOME_RE = new RegExp(
'(?:' +
String.raw`\/home\/${COMP}` +
'|' +
String.raw`\/Users\/${USER_WITH_SPACES}` +
'|' +
String.raw`\/mnt\/c\/Users\/${USER_WITH_SPACES}` +
'|' +
// Component boundary, so these don't match inside `/rootfs` or `/root_ca.pem`.
String.raw`\/var\/root(?![^/])` +
'|' +
String.raw`\/root(?![^/])` +
')' +
String.raw`(?:\/${COMP})*`,
'g'
);
export function collectEmbeddedHomePaths(raw: string): Set<string> {
const out = new Set<string>();
for (const m of raw.matchAll(EMBEDDED_HOME_RE)) out.add(m[0]);
return out;
}
function literalReplaceAll(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
return haystack.split(needle).join(replacement);
}
/** Can `ch` continue a path component? A `.` counts only mid-component, so `…/repo.git`
* is one component but `…/repo.` ending a sentence is not. */
function continuesComponent(text: string, at: number): boolean {
const ch = text[at];
if (ch === undefined) return false;
if (/[A-Za-z0-9_-]/.test(ch)) return true;
return ch === '.' && at + 1 < text.length && /[A-Za-z0-9_-]/.test(text[at + 1]);
}
/** Replace `needle` only where it ends at a component boundary, so stripping `…/wt/repo`
* can't turn `…/wt/repo-backup` into `<replacement>-backup`. Skipped ones go to the home pass. */
function replacePrefixAtBoundary(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
let out = '';
let from = 0;
for (;;) {
const i = haystack.indexOf(needle, from);
if (i === -1) return out + haystack.slice(from);
const end = i + needle.length;
out += haystack.slice(from, i) + (continuesComponent(haystack, end) ? needle : replacement);
from = end;
}
}
/** Replace a prefix and its dash-encoded form (`.claude/projects/<encoded>/`). */
function stripBothForms(haystack: string, needle: string, replacement: string): string {
const out = literalReplaceAll(haystack, needle, replacement);
return literalReplaceAll(out, needle.replace(/\//g, '-'), replacement.replace(/\//g, '-'));
}
export function sanitizeSessionJsonl(raw: string, opts: SanitizeOptions = {}): SanitizeResult {
const placeholder = opts.placeholder ?? DEFAULT_PLACEHOLDER;
const markers = opts.forbiddenMarkers ?? [];
const cwds = collectCwds(raw);
let working = raw;
let prefixStripped: string | null = null;
let encodedPrefixStripped: string | null = null;
let homeDirStripped: string | null = null;
let encodedHomeDirStripped: string | null = null;
let embeddedPrefixStripped: string | null = null;
let embeddedHomeDirStripped: string | null = null;
const requested = opts.cwdPrefixes?.length
? [...opts.cwdPrefixes]
: opts.cwdPrefix
? [opts.cwdPrefix]
: cwds.size > 0
? [findLongestCommonPathPrefix(cwds)]
: [];
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const prefixes = [...new Set(requested.filter(Boolean))].sort((a, b) => b.length - a.length);
// EVERY root before ANY home dir: a home pass run between roots would rewrite a
// sibling root's own prefix, leaving it unmatched when its turn came.
for (const prefix of prefixes) {
const encodedPrefix = prefix.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, prefix, placeholder);
working = literalReplaceAll(working, encodedPrefix, placeholder.replace(/\//g, '-'));
prefixStripped ??= prefix;
encodedPrefixStripped ??= encodedPrefix;
}
// Only catches what is left outside the roots, e.g. `/home/<user>/.claude/projects/`.
const homeDirs = new Set(
prefixes
.map((p) => extractHomeDir(p))
.filter((h): h is string => h !== null && !prefixes.includes(h))
);
for (const homeDir of homeDirs) {
const encodedHomeDir = homeDir.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, homeDir, HOME_DIR_PLACEHOLDER);
working = literalReplaceAll(working, encodedHomeDir, HOME_DIR_PLACEHOLDER.replace(/\//g, '-'));
homeDirStripped ??= homeDir;
encodedHomeDirStripped ??= encodedHomeDir;
}
if (opts.scrubEmbeddedHomePaths) {
const embedded = collectEmbeddedHomePaths(working);
if (embedded.size > 0) {
// Take each path's own shortest `/repo`-terminated prefix rather than a
// common prefix, which mis-collapses when paths diverge above the root.
const repoRoots = new Set<string>();
const homeDirs = new Set<string>();
for (const p of embedded) {
const h = extractHomeDir(p);
if (h) homeDirs.add(h);
const m = p.match(/^(.*?\/repo)(?:\/|$)/);
if (m) repoRoots.add(m[1]);
}
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const sortedRoots = [...repoRoots].sort((a, b) => b.length - a.length);
for (const root of sortedRoots) working = stripBothForms(working, root, placeholder);
for (const h of homeDirs) working = stripBothForms(working, h, HOME_DIR_PLACEHOLDER);
embeddedPrefixStripped = sortedRoots[0] ?? null;
embeddedHomeDirStripped = [...homeDirs][0] ?? null;
}
}
const markersScrubbed: Record<string, number> = {};
for (const re of markers) {
let count = 0;
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
const global = new RegExp(re.source, flags);
working = working.replace(global, () => {
count++;
return REDACTION_PLACEHOLDER;
});
if (count > 0) markersScrubbed[re.source] = count;
}
return {
sanitized: working,
prefixStripped,
encodedPrefixStripped,
homeDirStripped,
encodedHomeDirStripped,
embeddedPrefixStripped,
embeddedHomeDirStripped,
markersScrubbed,
};
}

View File

@@ -0,0 +1,26 @@
#!/usr/bin/env node
import fs from 'node:fs';
import path from 'node:path';
// Read stdin as a stream — hooks may not have /dev/stdin available
let input = '';
process.stdin.setEncoding('utf8');
process.stdin.on('data', (chunk) => {
input += chunk;
});
process.stdin.on('end', () => {
const { session_id, transcript_path } = JSON.parse(input);
const dataDir =
process.env.RACCOON_SNAPSHOT_DATA ||
process.env.CLAUDE_PLUGIN_DATA ||
(process.env.CLAUDE_PLUGIN_ROOT && path.join(process.env.CLAUDE_PLUGIN_ROOT, '.data')) ||
path.join(process.env.HOME || '/root', '.raccoon', 'snapshot-data');
fs.mkdirSync(dataDir, { recursive: true });
fs.writeFileSync(
path.join(dataDir, 'current-session.json'),
JSON.stringify({ session_id, transcript_path }, null, 2) + '\n'
);
});

View File

@@ -0,0 +1,837 @@
/**
* snapshot-to-task: Create a harbor task scaffold from a snapshot.
*
* Usage:
* npx tsx scripts/snapshot-to-task.ts --snapshot <dir>
*/
import { execFileSync, execSync } from 'child_process';
import {
chmodSync,
copyFileSync,
existsSync,
mkdirSync,
readFileSync,
readdirSync,
statSync,
writeFileSync,
} from 'fs';
import { basename, join, resolve } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { copyTree } from './lib/copy-tree';
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
// --- CLI ---
const argv = yargs(hideBin(process.argv))
.option('snapshot', {
type: 'string',
describe: 'Path to the snapshot directory',
demandOption: true,
})
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.strict()
.help()
.parseSync();
const log = pino(
{ name: 'snapshot-to-task', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
// --- Read snapshot data ---
const snapshotDir = argv.snapshot;
if (!existsSync(snapshotDir)) {
log.fatal(
{ path: snapshotDir },
'Snapshot directory not found. Check that the path points to a directory inside explore/snapshots/.'
);
process.exit(1);
}
interface SnapshotMetadata {
slug: string;
session_uuid: string;
/** Absent on snapshots captured before harness selection existed. */
harness?: string;
original_cwd: string;
commit: string | null;
branch: string | null;
remote_url: string | null;
timestamp: string;
plugin_version: string;
}
interface Annotation {
what_trying: string;
what_hoping: string;
what_happened: string;
[key: string]: string;
}
const metadata = JSON.parse(
readFileSync(join(snapshotDir, 'metadata.json'), 'utf8')
) as SnapshotMetadata;
const annotation = JSON.parse(
readFileSync(join(snapshotDir, 'annotation.json'), 'utf8')
) as Annotation;
if (!metadata.slug) {
log.fatal(
'No slug found in snapshot metadata.json. This snapshot may have been created by an older version of the plugin. Please file a bug.'
);
process.exit(1);
}
const slug = metadata.slug;
// --- Locate harbor infrastructure ---
function findRepoRoot(): string | null {
let dir = process.cwd();
while (dir !== resolve(dir, '..')) {
if (existsSync(join(dir, 'harbor-tasks'))) return dir;
dir = resolve(dir, '..');
}
return null;
}
const maybeRepoRoot = findRepoRoot();
if (!maybeRepoRoot) {
log.fatal(
"Could not find harbor-tasks/ directory. Make sure you're running this from the toolkit root (the Authoring container). Please file a bug if this persists."
);
process.exit(1);
}
const repoRoot: string = maybeRepoRoot;
const harborTasks = join(repoRoot, 'harbor-tasks');
const sharedCandidates = [join(harborTasks, 'raccoon-shared'), join(repoRoot, 'task-shared')];
const sharedDir = sharedCandidates.find((d) => existsSync(d));
const taskDir = join(harborTasks, slug);
if (existsSync(taskDir)) {
log.fatal(
{ path: taskDir },
`Task directory already exists. To recreate it, delete it first: rm -rf ${taskDir}`
);
process.exit(1);
}
if (!sharedDir) {
log.fatal(
'Shared infrastructure (Dockerfile, test.sh, etc.) not found. The toolkit may be corrupted. Please file a bug.'
);
process.exit(1);
}
// --- Detect repo name ---
interface ToolkitConfig {
repo: string;
defaultCommit: string;
/** The packed kit's release version (git describe at pack time). */
version?: string;
}
function readToolkitConfig(): ToolkitConfig | null {
const configPath = join(repoRoot, 'toolkit.json');
if (!existsSync(configPath)) return null;
return JSON.parse(readFileSync(configPath, 'utf8')) as ToolkitConfig;
}
function repoNameFromRemote(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const match = remoteUrl.match(/\/([^/]+?)(?:\.git)?$/);
return match ? match[1] : null;
}
function findSubmoduleDir(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const reposDir = join(repoRoot, 'repos');
if (!existsSync(reposDir)) return null;
const normalize = (url: string) =>
url
.replace(/\.git$/, '')
.replace(/^git@github\.com:/, 'https://github.com/')
.toLowerCase();
for (const entry of readdirSync(reposDir)) {
const repoPath = join(reposDir, entry, 'repo');
if (!existsSync(repoPath)) continue;
try {
const remote = execSync('git remote get-url origin', {
cwd: repoPath,
encoding: 'utf8',
stdio: ['pipe', 'pipe', 'pipe'],
}).trim();
if (normalize(remote) === normalize(remoteUrl)) return entry;
} catch {
continue;
}
}
return null;
}
const toolkitConfig = readToolkitConfig();
// A polyglot toolkit's toolkit.json has repos[] + polyglot:true (no top-level .repo).
// Derive which member this task targets from the snapshot's original_cwd basename,
// validated against the member list.
const polyglotMember = (() => {
const cfg = toolkitConfig as { polyglot?: boolean; repos?: Array<{ repo: string }> } | null;
if (!cfg?.polyglot || !Array.isArray(cfg.repos)) return null;
const base = metadata.original_cwd?.split('/').filter(Boolean).pop() ?? null;
const members = cfg.repos.map((r) => r.repo);
return base && members.includes(base) ? base : null;
})();
const repoName =
polyglotMember ??
toolkitConfig?.repo ??
findSubmoduleDir(metadata.remote_url) ??
repoNameFromRemote(metadata.remote_url);
if (!repoName) {
log.fatal(
'Could not determine repo name. The toolkit may be missing toolkit.json. Please file a bug.'
);
process.exit(1);
}
const commitShort = metadata.commit ? metadata.commit.slice(0, 9) : 'unknown';
const sessionUuid = metadata.session_uuid;
log.info({ slug, repo: repoName, commit: commitShort }, 'Creating harbor task');
// --- Create task directory structure ---
mkdirSync(join(taskDir, 'environment'), { recursive: true });
mkdirSync(join(taskDir, 'tests'), { recursive: true });
mkdirSync(join(taskDir, 'reference-runs'), { recursive: true });
// --- Copy shared infrastructure ---
// The complete grader asset set test.sh depends on: the grader system prompt
// and the renderer (test.sh exits without the renderer). Sources missing from
// task-shared/ are skipped by the existsSync guard below.
const sharedFiles = [
{ src: 'test.sh', dest: 'tests/test.sh' },
{
src: 'grader-system-prompt-consolidated.md',
dest: 'tests/grader-system-prompt-consolidated.md',
},
// test.sh execs this to render the grade; without it the verifier writes no reward
// file and the trial errors out rather than scoring.
{ src: 'render-grade-consolidated.py', dest: 'tests/render-grade-consolidated.py' },
];
for (const { src, dest } of sharedFiles) {
const srcPath = join(sharedDir, src);
const destPath = join(taskDir, dest);
if (existsSync(srcPath)) {
copyFileSync(srcPath, destPath);
if (src === 'test.sh') chmodSync(destPath, 0o755);
log.debug({ src, dest }, 'Copied shared file');
} else {
log.warn({ src }, 'Shared file not found');
}
}
// Deterministic checks (tests/typecheck/lint). test.sh sources these and hands
// their output to the grader as evidence for the CORRECTNESS score, so without
// them a code task's correctness is never signal-backed — the grader falls back
// to reading the diff alone. Same per-member-then-generic resolution as the
// Dockerfile below: a polyglot toolkit ships test-commands.<member>.sh per
// member, a single-repo toolkit ships the lone test-commands.sh.
const perMemberTestCommands = join(sharedDir, `test-commands.${repoName.toLowerCase()}.sh`);
const genericTestCommands = join(sharedDir, 'test-commands.sh');
const testCommandsSrc = existsSync(perMemberTestCommands)
? perMemberTestCommands
: genericTestCommands;
if (existsSync(testCommandsSrc)) {
const testCommandsDest = join(taskDir, 'tests', 'test-commands.sh');
copyFileSync(testCommandsSrc, testCommandsDest);
chmodSync(testCommandsDest, 0o755);
log.debug({ src: testCommandsSrc }, 'Copied deterministic checks');
} else {
// Not fatal: the grader still scores correctness by walking the changed code.
log.info(
'No test-commands.sh for this repo — expected when it has no runnable suite. The grader scores correctness by reading the changed code instead; say so in your holistic rubric.'
);
}
// --- Write Dockerfile with session resume support ---
//
// Read the per-repo task-shared/Dockerfile (Ruby/Postgres/Node for ZenBill,
// TS-Node/Postgres/pnpm for Palolo) from the toolkit and append session-
// staging COPY/RUN steps. Session staging happens after the original CMD —
// COPY and RUN are layer ops independent of CMD, so the original
// `CMD ["sleep", "infinity"]` remains active after the appended layers.
//
// Falls back to a bare debian Dockerfile if no task-shared/Dockerfile is
// present (toolkit corruption, or a repo without a per-repo Dockerfile).
// Polyglot toolkits ship a per-member task-shared/Dockerfile.<member>; a graded task
// targets one member, so prefer its Dockerfile. Single-repo toolkits use the lone
// task-shared/Dockerfile. Fall back to the generic one if the per-member file is absent.
const perMemberDockerfile = join(repoRoot, 'task-shared', `Dockerfile.${repoName.toLowerCase()}`);
const taskSharedDockerfile = existsSync(perMemberDockerfile)
? perMemberDockerfile
: join(repoRoot, 'task-shared', 'Dockerfile');
let baseDockerfile: string;
if (existsSync(taskSharedDockerfile)) {
baseDockerfile = readFileSync(taskSharedDockerfile, 'utf-8');
log.debug({ dockerfile: taskSharedDockerfile }, 'Loaded base Dockerfile');
} else {
log.warn(
'task-shared/Dockerfile not found; falling back to bare debian. The harbor task container will lack any language runtime — agents will not be able to execute code in the repo.'
);
baseDockerfile = `FROM debian:bookworm-slim
RUN apt-get update && apt-get install -y \\
git \\
python3 \\
curl \\
jq \\
&& rm -rf /var/lib/apt/lists/*
# Install Claude Code globally (needed by the grader in test.sh)
RUN curl -fsSL https://claude.ai/install.sh | bash && \\
cp /root/.claude-code/claude /usr/local/bin/claude 2>/dev/null || \\
cp /root/.local/bin/claude /usr/local/bin/claude 2>/dev/null || \\
ln -sf $(find /root -name claude -type f 2>/dev/null | head -1) /usr/local/bin/claude
WORKDIR /workspace
COPY workspace/ .
# Block network tools — agent should only read code and write documents
RUN mkdir -p .claude && \\
echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > .claude/settings.json
RUN git init && \\
git config user.email "dev@agent" && \\
git config user.name "Dev" && \\
git add -A && \\
git commit -m "initial" --quiet
CMD ["sleep", "infinity"]
`;
}
// Wrapped in toolkit-managed sentinels so check-task-infra reads this as the
// toolkit's own append rather than an edit to the Dockerfile.
// Only Claude Code produces the sibling session/ directory (subagents, tool results).
// A COPY of an empty directory fails the build outright — buildkit does not carry empty
// directories in the context, so the layer errors with `"/session": not found`.
// Read the SNAPSHOT, not the task dir: the Dockerfile is generated before the session
// files are copied into environment/, so the task-side copy is not there yet.
const sessionSiblingDir = join(snapshotDir, 'session');
const hasSessionSibling =
existsSync(sessionSiblingDir) && readdirSync(sessionSiblingDir).length > 0;
const sessionStaging = `
# >>> toolkit-managed: snapshot-session >>>
# Stage session files for the snapshot agent adapter to install at runtime.
COPY session.jsonl /tmp/snapshot-session/session.jsonl
${hasSessionSibling ? 'COPY session/ /tmp/snapshot-session/session/\n' : ''}RUN echo '${sessionUuid}' > /tmp/snapshot-session/uuid.txt
# <<< toolkit-managed <<<
`;
const dockerfile = baseDockerfile.trimEnd() + '\n' + sessionStaging;
writeFileSync(join(taskDir, 'environment', 'Dockerfile'), dockerfile);
log.debug('Wrote Dockerfile (per-repo base + session staging)');
// --- Copy snapshot.patch as workspace.patch ---
const snapshotPatch = join(snapshotDir, 'snapshot.patch');
if (existsSync(snapshotPatch)) {
copyFileSync(snapshotPatch, join(taskDir, 'environment', 'workspace.patch'));
log.debug('Copied snapshot.patch -> workspace.patch');
}
// --- Scrub the worker's filesystem layout out of the session ---
// In Explore the recorded `cwd` is the worker's HOST checkout (explore/repo is an absolute
// symlink); rewriting the repo root to /workspace both drops the leak and matches the trial.
const WORKSPACE_MOUNT = '/workspace';
const escapeRegExp = (v: string) => v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
/** Member names when this toolkit is polyglot; empty means single-repo. */
const MEMBER_NAMES: readonly string[] = (() => {
const dir = join(repoRoot, 'repos');
if (!existsSync(dir)) return [];
try {
return readdirSync(dir, { withFileTypes: true })
.filter((e) => e.isDirectory())
.map((e) => e.name);
} catch {
return [];
}
})();
/** The repo root within a cwd — the prefix a trial mounts at /workspace. `/repos/<member>`
* anchors only on a polyglot toolkit, so a personal `~/repos/…` above it can't win. */
function repoRootOf(cwd: string): string | null {
if (MEMBER_NAMES.length > 0) {
// A real member of THIS toolkit wins; the generic shape covers a member whose
// directory the toolkit no longer has (an older snapshot, a renamed member).
for (const name of MEMBER_NAMES) {
const hit = cwd.match(new RegExp(`^(.*?/repos/${escapeRegExp(name)})(?:/|$)`));
if (hit) return hit[1];
}
const generic = cwd.match(/^(.*?\/repos\/[^/]+)(?:\/|$)/);
if (generic) return generic[1];
}
// `/repo` needs a component boundary, so it never matches inside `/repos/`.
const m = cwd.match(/^(.*?\/repo)(?:\/|$)/);
return m ? m[1] : null;
}
/** Rewrite every checkout root to /workspace, and the home dir each sits under to `~`. The
* `repo/` anchor needs no host-root list; the home pass still keys off extractHomeDir. */
function scrubWorkerPaths(raw: string): { text: string; roots: string[] } {
// Each cwd contributes its own root, longest first, so a nested root isn't clobbered
// and a session spanning two checkouts is scrubbed rather than skipped.
const roots = [...new Set([...collectCwds(raw)].map(repoRootOf))]
.filter((r): r is string => r !== null)
.sort((a, b) => b.length - a.length);
const { sanitized } = sanitizeSessionJsonl(raw, {
cwdPrefixes: roots,
placeholder: WORKSPACE_MOUNT,
});
return { text: sanitized, roots };
}
// --- Copy session files for --resume ---
//
// The full session.jsonl (including any post-end_turn entries) goes into the
// task root for reference. A truncated version — keeping everything up to
// and including the last assistant entry with stop_reason="end_turn" — goes
// into environment/ for the container. Stopping on a clean assistant turn
// avoids Claude Code's synthetic "No response requested." injection when
// the session is resumed with --fork-session and a new --print prompt.
const sessionJsonl = join(snapshotDir, 'session.jsonl');
if (existsSync(sessionJsonl)) {
// Fail-open: a session this can't scrub ships exactly as it was, because a
// leaked path is a smaller problem than a task that can't be created.
let sessionText = readFileSync(sessionJsonl, 'utf8');
try {
const { text, roots } = scrubWorkerPaths(sessionText);
if (roots.length > 0) {
sessionText = text;
log.info(
{ roots, mountedAt: WORKSPACE_MOUNT },
'Rewrote the authoring checkout path to the trial mount point'
);
} else {
log.debug('No worker-rooted cwd to rewrite; session used as-is');
}
} catch (err) {
log.warn(
{ err: err instanceof Error ? err.message : String(err) },
'Could not rewrite paths in the session; using it as-is'
);
}
// Full version for reference
writeFileSync(join(taskDir, 'session-full.jsonl'), sessionText);
log.debug('Wrote full session.jsonl to task root');
// Truncated version for the container: strip everything from the last
// user text turn onwards. This drops the failure-eliciting question
// (which `--print` will redeliver to the trial agent as the new prompt)
// AND the failure response itself (so the trial agent doesn't see its
// previous answer), while preserving conversational context up to the
// last clean assistant `end_turn`.
//
// Algorithm (refined Option B):
// 1. Find U = index of the last user-text turn that is NOT a slash
// command (use the same command-marker filter as
// extractLastUserMessage).
// 2. Walk backwards from U - 1 to find the last `assistant` entry
// with stop_reason: "end_turn".
// 3. Truncate slice(0, lastEndTurnIndex + 1).
//
// If U doesn't exist or no end_turn assistant precedes U, write an
// empty session.jsonl — the snapshot agent adapter detects this and
// skips --resume entirely, starting fresh from --print.
const sessionLines = sessionText.trimEnd().split('\n');
// A non-Claude session is not a Claude transcript, so the scan below finds no
// `stop_reason: "end_turn"` and would silently write an empty session. Its reader
// applies the same rule in that harness's own format.
const harness = metadata.harness ?? 'claude-code';
const isClaude = harness === 'claude-code';
let lastUserTextIndex = -1;
for (let i = 0; i < sessionLines.length; i++) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type !== 'user' || typeof entry.message?.content !== 'string') continue;
// Compaction summaries are synthetic user turns whose text often quotes
// earlier /create-snapshot:snapshot runs — never the command turn itself,
// so they must not trip the break below.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
// Mirror extractLastUserMessage: skip the snapshot command itself
// and any slash-command / local-command marker turns.
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserTextIndex = i;
} catch {
continue;
}
}
let lastEndTurnIndex = -1;
if (lastUserTextIndex > 0) {
for (let i = lastUserTextIndex - 1; i >= 0; i--) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
message?: { stop_reason?: unknown };
};
if (entry.type === 'assistant' && entry.message?.stop_reason === 'end_turn') {
lastEndTurnIndex = i;
break;
}
} catch {
continue;
}
}
}
if (!isClaude) {
const cut = truncationIndex(turnsFromLines(harness, sessionLines));
const kept = cut >= 0 ? sessionLines.slice(0, cut + 1) : [];
const truncated = stripAuthoringScaffolding(harness, kept);
writeFileSync(
join(taskDir, 'environment', 'session.jsonl'),
truncated.length ? truncated.join('\n') + '\n' : ''
);
log.debug(
{ harness, fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (harness reader)'
);
} else if (lastEndTurnIndex >= 0) {
const truncated = sessionLines.slice(0, lastEndTurnIndex + 1);
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), truncated.join('\n') + '\n');
log.debug(
{ fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (strips last user turn + failure response, keeps through last clean assistant end_turn)'
);
} else {
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), '');
if (lastUserTextIndex < 0) {
log.warn(
'No user text turn found in session — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
} else {
log.warn(
'No assistant entry with stop_reason="end_turn" found before the last user turn (one-shot snapshot) — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
}
}
}
const sessionDir = join(snapshotDir, 'session');
if (existsSync(sessionDir) && statSync(sessionDir).isDirectory()) {
copyTree(sessionDir, join(taskDir, 'environment', 'session'));
// Claude Code writes subagent files write-only (--w-------). Fix them so
// Harbor's dirhash can read them during environment setup.
execSync(`chmod -R +r "${join(taskDir, 'environment', 'session')}"`, { stdio: 'pipe' });
log.debug('Copied session/');
} else {
mkdirSync(join(taskDir, 'environment', 'session'), { recursive: true });
}
// The harness that captured the snapshot; the trial runs this one.
const harness =
typeof metadata.harness === 'string' && metadata.harness ? metadata.harness : 'claude-code';
/**
* The model and effort this harness defaulted to when the task was authored, recorded
* for reference only — nothing reads these back, and a trial still resolves both from
* the registry at run time. Best-effort: a task is not worth failing over a note.
*/
function authoredDefaults(harnessId: string): { model: string; effort: string } | null {
try {
const resolver = join(repoRoot, 'scripts', 'resolve_harness.py');
// Same interpreter search as `_raccoon_python` in scripts/lib/harness-credentials.sh
// and `pythonWithTomllib` in submit-task.ts: `python3` is not always 3.11+, and the
// registry needs tomllib. Best-effort, so a miss just omits the note.
let python = '';
for (const candidate of [
process.env.RACCOON_PYTHON,
'python3',
'python3.13',
'python3.12',
'python3.11',
]) {
if (!candidate) continue;
try {
execFileSync(candidate, ['-c', 'import tomllib'], { stdio: 'ignore' });
python = candidate;
break;
} catch {
continue;
}
}
if (!python) return null;
const rows = execFileSync(python, [resolver, '--defaults'], {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
});
for (const line of rows.split('\n')) {
const [id, model, effort] = line.split('\t');
if (id === harnessId && model) return { model, effort: effort ?? '' };
}
} catch {
// registry unreadable here — omit the note
}
return null;
}
const authored = authoredDefaults(harness);
// --- Write task.toml ---
// The reference-data corpus is included in every zeta task (build-workspace decides from the repo),
// so there's nothing to set here.
const taskToml = `version = "1.0"
[metadata]
program = "raccoon"
author = "rl-env-coding"
category = "sdlc/technical-writing"
repo = "${repoName}"
commit = "${commitShort}"
# The toolkit release this task was created with. Written by the toolkit —
# leave it in place: task tooling reads it to know which toolkit's assets
# this task grades with.
toolkit_version = "${toolkitConfig?.version ?? 'unknown'}"
snapshot = "${basename(snapshotDir)}"
session_uuid = "${sessionUuid}"
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
browser = false
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
[verifier]
timeout_sec = 7200.0
[agent]
harness = "${harness}"
timeout_sec = 18000.0
[environment]
build_timeout_sec = 6000.0
cpus = 2
memory_mb = 4096
storage_mb = 10240
gpus = 0
allow_internet = true
[verifier.env]
ANTHROPIC_API_KEY = "\${ANTHROPIC_API_KEY}"
ANTHROPIC_BASE_URL = "\${ANTHROPIC_BASE_URL}"
[solution.env]
`;
writeFileSync(join(taskDir, 'task.toml'), taskToml);
log.debug('Wrote task.toml');
// --- Extract instruction from session transcript ---
function extractLastUserMessage(sessionPath: string, harness: string): string | null {
if (!existsSync(sessionPath)) return null;
const lines = readFileSync(sessionPath, 'utf8').trimEnd().split('\n');
// A non-Claude session has no `type: "user"` records, so the scan below finds nothing
// and the worker silently gets a placeholder instruction. Its reader applies the same
// rule — last real user turn, ignoring command invocations — in that harness's format.
if (harness !== 'claude-code') {
const userTurns = turnsFromLines(harness, lines).filter(
(t) => t.role === 'user' && !t.isCommand && t.text.trim()
);
return userTurns.length ? userTurns[userTurns.length - 1].text : null;
}
let lastUserMessage: string | null = null;
for (const line of lines) {
try {
const entry = JSON.parse(line) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type === 'user' && typeof entry.message?.content === 'string') {
// Synthetic compaction summary — not a real user turn, and its text
// often quotes earlier /create-snapshot:snapshot runs.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserMessage = content;
}
} catch {
continue;
}
}
return lastUserMessage;
}
const lastUserMessage = extractLastUserMessage(
join(snapshotDir, 'session.jsonl'),
metadata.harness ?? 'claude-code'
);
const instructionHeader =
'# Replace this with your refined task instruction\n\n' +
"<!-- The text below was auto-extracted from your snapshot's last user message.\n" +
' Refine, condense, or rewrite to focus on the behavior you want to elicit. -->\n\n';
if (lastUserMessage) {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader + lastUserMessage.trimEnd() + '\n'
);
log.info('Wrote instruction.md (from last user message in session)');
} else {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader +
'<!-- Could not extract user message from session. Write the instruction manually. -->\n'
);
log.warn('Could not extract instruction from session — needs manual editing');
}
// --- Scaffold holistic-rubric.md ---
const holisticRubricMd = `<!--
HOLISTIC RUBRIC — the file trials grade against. Run
/write-holistic-rubric
to draft it interactively, or point Claude Code at this file,
session-full.jsonl, and task-shared/grading-standard.md.
Snapshot: ${basename(snapshotDir)}
Session: ${metadata.session_uuid}
Repo: ${metadata.remote_url}
Commit: ${metadata.commit}
## What happened in the snapshot conversation
The worker was trying to: ${annotation.what_trying}
They hoped Claude would: ${annotation.what_hoping}
Instead, Claude: ${annotation.what_happened}
## What this file contains
The eight-criterion Grading Standard
(task-shared/grading-standard.md, embedded in
tests/grader-system-prompt-consolidated.md) defines Integrity, Narrow
Correctness, Broader Correctness / craft, Persistence, Communication,
Verification & Thoroughness, Common Sense, and Thought Partnership. This
file adds the task-specific knowledge the grader cannot infer: full task
context, the ground truth you established, what strong and weak responses
look like per criterion, and any dealbreaker penalties — stated as 0.0-1.0
fraction subtractions with a named criterion target, never points, never
caps. The document must stand alone: the grader sees only it and the
shared standard.
-->
<!-- Replace EVERYTHING in this file with the actual holistic rubric,
including the instructions above. -->
`;
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');
// --- Build workspace ---
const buildScript = join(repoRoot, 'scripts', 'build-workspace.sh');
if (existsSync(buildScript)) {
log.info({ repo: repoName, commit: commitShort }, 'Building workspace');
try {
execSync(`bash "${buildScript}" "${slug}" "${commitShort}"`, {
cwd: repoRoot,
encoding: 'utf8',
stdio: 'inherit',
// build-workspace does a bulk-file write burst (git archive|tar of the
// repo tree + a throwaway git add/commit to apply the patch, and for zeta
// toolkits a hardlink-stage of the ~126k-file reference-data corpus that
// falls back to a full copy across filesystems). On a slow bind mount
// (Docker Desktop non-VirtioFS, or WSL2 with the toolkit on a Windows/9p
// path) that legitimately runs into minutes, so a tight cap false-fails a
// working-but-slow build as "not runnable". Keep this generous — it's only
// a backstop against a true hang; the real Harbor build downstream budgets
// build_timeout_sec = 6000.
timeout: 1_200_000,
});
} catch (e: unknown) {
const msg = e instanceof Error ? e.message : String(e);
log.fatal({ error: msg }, 'Workspace build failed — task is not runnable');
log.fatal(` Retry manually: bash scripts/build-workspace.sh ${slug}`);
log.fatal(` Then: scripts/harbor-run harbor-tasks/${slug}`);
process.exit(1);
}
} else {
log.fatal('scripts/build-workspace.sh not found. Please file a bug.');
process.exit(1);
}
try {
execSync('bash -ic "_ev task_created 2>/dev/null" 2>/dev/null', {
stdio: 'ignore',
timeout: 5000,
});
} catch {
// best-effort
}
// --- Done ---
log.info({ taskDir: resolve(taskDir) }, 'Task scaffolded');
log.info('Next steps:');
log.info(' 1. Review instruction.md');
log.info(' 2. Edit tests/holistic-rubric.md — write the rubric');
log.info(' 3. Run calibration trials to validate scoring tiers');

View File

@@ -0,0 +1,57 @@
---
description: Capture a snapshot of the current conversation and repo state.
---
# Create Snapshot
You are capturing a snapshot of the current conversation and repo state so it can be replayed as an RL training task.
## Step 1: Ask annotation questions
**Important — tell the user this first, verbatim:**
> ⚠️ This snapshot captures your entire conversation history with me, not just the most
> recent turn. If you told me the answer earlier in this conversation, or steered me
> toward it, the agent will see that same context when the snapshot replays — and will
> probably solve the task without making the mistake. Your task will be contaminated.
>
> If you've leaked the answer at any point in this conversation: if your agent can rewind
> (Claude Code's `/rewind`), rewind to a point before the contamination and snapshot from
> there. If it can't — codex has no rewind — this snapshot is not salvageable: start a
> fresh session, reproduce the mistake without steering, and snapshot that instead.
Wait for the user to acknowledge before moving on.
Ask the user each of these questions **one at a time** as plain text, waiting for their response before proceeding to the next:
1. "What were you trying to do?"
2. "What were you hoping was going to happen?"
3. "What did the agent actually do instead?"
## Step 2: Propose a slug
Based on the user's answers, generate a **short kebab-case slug** (2-4 words) that captures the essence of the mistake. For example: `bad-refactor`, `wrong-test-strategy`, `missed-edge-case`.
Present your suggestion and ask the user to confirm or provide an alternative.
## Step 3: Write annotation file and run capture
Write the annotation to a temporary JSON file, then run the capture script.
Write this JSON to a temp file (use a path like `/tmp/snapshot-annotation-<timestamp>.json`):
```json
{
"what_trying": "<answer to question 1>",
"what_hoping": "<answer to question 2>",
"what_happened": "<answer to question 3>"
}
```
Then run:
```bash
"${CLAUDE_PLUGIN_ROOT}/bin/capture-snapshot.mjs" --slug <slug> --annotation <temp-file-path> --output-dir "${CLAUDE_PLUGIN_ROOT}/../../snapshots" --plugin-data "${CLAUDE_PLUGIN_DATA:-${CLAUDE_PLUGIN_ROOT}/.data}"
```
Report the script's stdout output verbatim to the user. Do not paraphrase or shorten paths.

View File

@@ -0,0 +1,35 @@
{
"hooks": {
"SessionStart": [
{
"hooks": [
{
"type": "command",
"command": "${CLAUDE_PLUGIN_ROOT}/bin/save-session-info.mjs"
}
]
}
],
"UserPromptSubmit": [
{
"hooks": [
{
"type": "command",
"command": "${CLAUDE_PLUGIN_ROOT}/bin/checkpoint-workspace.mjs"
}
]
}
],
"PreToolUse": [
{
"matcher": "Bash",
"hooks": [
{
"type": "command",
"command": "${CLAUDE_PLUGIN_ROOT}/bin/checkpoint-workspace.mjs"
}
]
}
]
}
}

View File

@@ -0,0 +1,76 @@
---
name: snapshot
description: Capture the current conversation and repo state as a snapshot, to be replayed as a task. Use when the user wants to snapshot a mistake the agent just made.
---
# Create Snapshot
You are capturing a snapshot of the current conversation and repo state so it can be
replayed as an RL training task.
## Step 0: Mark where the snapshot begins
Run this FIRST, before asking anything. It records where the conversation ended so the
questions below aren't captured as part of it:
```bash
"${RACCOON_SNAPSHOT_PLUGIN_ROOT:-/workspace/plugins/create-snapshot}/bin/capture-snapshot.mjs" \
--mark-start --harness "${RACCOON_HARNESS:?not set — start your session through the launcher (the plain agent command, e.g. \`codex\`) so the snapshot records which agent it came from}"
```
## Step 1: Ask annotation questions
**Important — tell the user this first, verbatim:**
> ⚠️ This snapshot captures your entire conversation history with me, not just the most
> recent turn. If you told me the answer earlier in this conversation, or steered me
> toward it, the agent will see that same context when the snapshot replays — and will
> probably solve the task without making the mistake. Your task will be contaminated.
>
> If you've leaked the answer at any point in this conversation: if your agent can rewind
> (Claude Code's `/rewind`), rewind to a point before the contamination and snapshot from
> there. If it can't — codex has no rewind — this snapshot is not salvageable: start a
> fresh session, reproduce the mistake without steering, and snapshot that instead.
Wait for the user to acknowledge before moving on.
Ask the user each of these questions **one at a time** as plain text, waiting for their
response before proceeding to the next:
1. "What were you trying to do?"
2. "What were you hoping was going to happen?"
3. "What did the agent actually do instead?"
## Step 2: Propose a slug
Based on the user's answers, generate a **short kebab-case slug** (2-4 words) that
captures the essence of the mistake. For example: `bad-refactor`,
`wrong-test-strategy`, `missed-edge-case`.
Present your suggestion and ask the user to confirm or provide an alternative.
## Step 3: Write annotation file and run capture
Write the annotation to a temporary JSON file, then run the capture script.
Write this JSON to a temp file (use a path like `/tmp/snapshot-annotation-<timestamp>.json`):
```json
{
"what_trying": "<answer to question 1>",
"what_hoping": "<answer to question 2>",
"what_happened": "<answer to question 3>"
}
```
Then run:
```bash
"${RACCOON_SNAPSHOT_PLUGIN_ROOT:-/workspace/plugins/create-snapshot}/bin/capture-snapshot.mjs" \
--harness "$RACCOON_HARNESS" \
--slug <slug> \
--annotation <temp-file-path> \
--output-dir /workspace/snapshots
```
Report the script's stdout output verbatim to the user. Do not paraphrase or shorten paths.

View File

@@ -0,0 +1 @@
/home/eric/workspaces/dataannotation/current-project/worker-toolkit-potion-polyglot/repos

View File

@@ -0,0 +1,978 @@
#!/bin/bash
# run-app — start the source app inside the Explore container with one command.
#
# Before this existed you had to open two shells into the container and start
# the server and client by hand. This wraps that up: it makes sure postgres is
# running, starts each process in the background, waits until they're actually
# listening, and prints the URL to open plus a login. Logs are written to a
# file so the foreground stays clean.
#
# Usage:
# run-app start the app (no-op if it's already running)
# run-app --restart stop, then start again
# run-app --stop stop the app
# run-app --logs follow the server + client logs (Ctrl-C to stop following)
# run-app --status show whether the app is running
# run-app --help this message
set -uo pipefail
RUN_DIR="/tmp/raccoon-app"
mkdir -p "$RUN_DIR"
CYAN='\033[1;36m'; YELLOW='\033[1;33m'; GRAY='\033[0;90m'; RED='\033[1;31m'; RESET='\033[0m'
REPO_NAME=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').repo)}catch{}" 2>/dev/null || true)
# Host port the browser uses. The app always binds the container ports (3000 /
# 3001); the Explore container publishes them on a host port that defaults per
# repo but can be overridden (so more than one container — even of the same
# repo — can run at once). That live value is exported into the container as
# $EXPLORE_CLIENT_PORT; prefer it, falling back to toolkit.json then 3000 for
# older containers built before this var existed.
CLIENT_HOST_PORT="${EXPLORE_CLIENT_PORT:-$(node -e "try{process.stdout.write(String(require('/workspace/toolkit.json').explorePorts.clientHost))}catch{process.stdout.write('3000')}" 2>/dev/null || echo 3000)}"
# --- process helpers ---------------------------------------------------------
# Is the process recorded in $1 (a pidfile) still alive?
_alive() { local pf="$1"; [ -f "$pf" ] && kill -0 "$(cat "$pf" 2>/dev/null)" 2>/dev/null; }
# Start a backgrounded process group leader so we can later kill the whole
# group (vite/tsx spawn children). setsid makes the started process its own
# session+group leader; we record its pid (== the group id).
_spawn() {
local name="$1" workdir="$2" cmd="$3"
local log="$RUN_DIR/$name.log" pf="$RUN_DIR/$name.pid"
: > "$log"
if command -v setsid >/dev/null 2>&1; then
setsid bash -c "cd '$workdir' && exec $cmd" >"$log" 2>&1 &
else
# Fallback: no setsid (children may outlive a stop; best-effort).
( cd "$workdir" && exec $cmd ) >"$log" 2>&1 &
fi
echo $! > "$pf"
}
# Stop the process recorded in pidfile $1 (and its group, when we have one).
_kill_pidfile() {
local pf="$1"; [ -f "$pf" ] || return 0
local pid; pid=$(cat "$pf" 2>/dev/null || true)
if [ -n "${pid:-}" ] && kill -0 "$pid" 2>/dev/null; then
kill -TERM "-$pid" 2>/dev/null || kill -TERM "$pid" 2>/dev/null || true
for _ in 1 2 3 4 5 6 7 8 9 10; do kill -0 "$pid" 2>/dev/null || break; sleep 0.3; done
kill -KILL "-$pid" 2>/dev/null || kill -KILL "$pid" 2>/dev/null || true
fi
rm -f "$pf"
}
# Wait (bounded) until something is listening on TCP port $1.
_wait_tcp() {
local port="$1" tries="${2:-180}" i
for ((i = 0; i < tries; i++)); do
if (exec 3<>"/dev/tcp/127.0.0.1/$port") 2>/dev/null; then exec 3>&- 3<&-; return 0; fi
sleep 1
done
return 1
}
# --- actions -----------------------------------------------------------------
stop_app() {
local stopped=0
for pf in "$RUN_DIR"/*.pid; do
[ -e "$pf" ] || continue
_kill_pidfile "$pf"
stopped=1
done
if [ "$stopped" = 1 ]; then printf "${GRAY}Stopped the app.${RESET}\n"; else printf "${GRAY}Nothing to stop.${RESET}\n"; fi
}
status_app() {
local any=0
for pf in "$RUN_DIR"/*.pid; do
[ -e "$pf" ] || continue
local name; name=$(basename "$pf" .pid)
if _alive "$pf"; then printf " ${GRAY}%-8s${RESET} running (pid %s)\n" "$name" "$(cat "$pf")"; else printf " ${GRAY}%-8s${RESET} not running\n" "$name"; fi
any=1
done
[ "$any" = 1 ] || printf "${GRAY}App is not running.${RESET}\n"
}
logs_app() {
local logs=()
for lf in "$RUN_DIR"/*.log; do [ -e "$lf" ] && logs+=("$lf"); done
if [ "${#logs[@]}" -eq 0 ]; then printf "${GRAY}No logs yet — start the app first with ${RESET}run-app\n"; return 0; fi
printf "${GRAY}Following %s (Ctrl-C to stop following; the app keeps running):${RESET}\n" "${logs[*]}"
tail -n +1 -f "${logs[@]}"
}
# Start helpers per repo. Each starts the process(es) on their container ports.
start_palolo() {
_spawn server /workspace/repo/packages/server "node --import=tsx src/server.ts"
_spawn client /workspace/repo/packages/client "npx vite --host 0.0.0.0 --port 3000"
printf " ${CYAN}\xe2\x96\xb6${RESET} starting server (packages/server)\xe2\x80\xa6\n"
printf " ${CYAN}\xe2\x96\xb6${RESET} starting client (packages/client)\xe2\x80\xa6\n"
printf " ${GRAY}\xe2\x8f\xb3 waiting for the app to come up\xe2\x80\xa6${RESET}\n"
local ok_server=1 ok_client=1
_wait_tcp 3001 || ok_server=0
_wait_tcp 3000 || ok_client=0
if [ "$ok_server" = 1 ] && [ "$ok_client" = 1 ]; then
printf " ${CYAN}\xe2\x9c\x85 app is up${RESET}\n"
printf " open ${CYAN}http://localhost:%s${RESET}\n" "$CLIENT_HOST_PORT"
printf " login ${GRAY}zaniyah@exhalefi.com${RESET} / ${GRAY}test${RESET}\n"
else
printf " ${RED}\xe2\x9a\xa0 the app didn't come up in time${RESET} (server=%s client=%s)\n" "$ok_server" "$ok_client"
printf " check the logs: ${GRAY}run-app --logs${RESET}\n"
fi
printf " logs ${GRAY}%s/{server,client}.log${RESET}\n" "$RUN_DIR"
printf " stop ${GRAY}run-app --stop${RESET}\n"
}
start_zenbill() {
_spawn app /workspace/repo "bundle exec rails server -b 0.0.0.0 -p 3000"
printf " ${YELLOW}\xe2\x96\xb6${RESET} starting Rails (puma)\xe2\x80\xa6\n"
printf " ${GRAY}\xe2\x8f\xb3 waiting for the app to come up\xe2\x80\xa6${RESET}\n"
if _wait_tcp 3000; then
printf " ${YELLOW}\xe2\x9c\x85 app is up${RESET}\n"
printf " open ${YELLOW}http://localhost:%s${RESET}\n" "$CLIENT_HOST_PORT"
printf " ${GRAY}note: this app routes by subdomain. Plain localhost shows only the${RESET}\n"
printf " ${GRAY}Rails welcome page; the real UI needs /etc/hosts entries for${RESET}\n"
printf " ${GRAY}app.dev.zenbill.com etc. (see README \xe2\x86\x92 Running the app).${RESET}\n"
else
printf " ${RED}\xe2\x9a\xa0 the app didn't come up in time${RESET}\n"
printf " check the logs: ${GRAY}run-app --logs${RESET}\n"
fi
printf " logs ${GRAY}%s/app.log${RESET}\n" "$RUN_DIR"
printf " stop ${GRAY}run-app --stop${RESET}\n"
}
start_zeta_heimdall() {
# API-only Rails app — boots a JSON API on container port 3000 (no separate client).
_spawn app /workspace/repo "bundle exec rails server -b 0.0.0.0 -p 3000"
printf " ${YELLOW}\xe2\x96\xb6${RESET} starting Rails API (puma)\xe2\x80\xa6\n"
printf " ${GRAY}\xe2\x8f\xb3 waiting for the app to come up\xe2\x80\xa6${RESET}\n"
if _wait_tcp 3000; then
printf " ${YELLOW}\xe2\x9c\x85 app is up${RESET}\n"
printf " base ${YELLOW}http://localhost:%s${RESET}\n" "$CLIENT_HOST_PORT"
printf " ${GRAY}note: this is a JSON API, not a UI \xe2\x80\x94 hit an endpoint (e.g. an auth route)${RESET}\n"
printf " ${GRAY}rather than expecting a page in the browser.${RESET}\n"
else
printf " ${RED}\xe2\x9a\xa0 the app didn't come up in time${RESET}\n"
printf " check the logs: ${GRAY}run-app --logs${RESET}\n"
fi
printf " logs ${GRAY}%s/app.log${RESET}\n" "$RUN_DIR"
printf " stop ${GRAY}run-app --stop${RESET}\n"
}
start_zeta_platform() {
# Boots BOTH the Rails API and the React client so the full UI comes up.
# The client (Create React App, react-scripts 2.1.1) serves the UI on container
# :3000 (the published port) and proxies /graphql to the Rails API, which its
# package.json "proxy" hardcodes at localhost:5000. So Rails binds :5000 (reached
# only from inside the container — the browser talks solely to the client) and the
# client binds :3000. rspec doesn't need any of this; it's just the interactive app.
#
# react-scripts 2.1.1 is webpack-4 era: on Node 17+ its build hashing crashes
# without --openssl-legacy-provider. HOST=0.0.0.0 + DANGEROUSLY_DISABLE_HOST_CHECK
# let the dev server answer requests arriving via the published host port.
# node_modules is the container-local symlink post-create.sh set up; yarn is v1.
_spawn server /workspace/repo "env PORT=5000 bundle exec rails server -b 0.0.0.0 -p 5000"
_spawn client /workspace/repo "env NODE_OPTIONS=--openssl-legacy-provider BROWSER=none CI=false PORT=3000 HOST=0.0.0.0 DANGEROUSLY_DISABLE_HOST_CHECK=true NODE_PATH=src:src/components/ ./node_modules/.bin/react-app-rewired start"
printf " ${YELLOW}\xe2\x96\xb6${RESET} starting Rails API (puma) on :5000\xe2\x80\xa6\n"
printf " ${CYAN}\xe2\x96\xb6${RESET} starting React client (react-scripts)\xe2\x80\xa6\n"
printf " ${GRAY}\xe2\x8f\xb3 waiting for the app to come up (first client compile takes a minute)\xe2\x80\xa6${RESET}\n"
local ok_server=1 ok_client=1
_wait_tcp 5000 || ok_server=0
_wait_tcp 3000 || ok_client=0
if [ "$ok_server" = 1 ] && [ "$ok_client" = 1 ]; then
printf " ${CYAN}\xe2\x9c\x85 app is up${RESET}\n"
printf " open ${CYAN}http://localhost:%s${RESET}\n" "$CLIENT_HOST_PORT"
printf " ${GRAY}note: that URL is the React UI. It proxies GraphQL to the Rails API on${RESET}\n"
printf " ${GRAY}:5000 inside the container (reach it directly from a container shell at${RESET}\n"
printf " ${GRAY}http://localhost:5000). The DB is schema-loaded but unseeded \xe2\x80\x94 you may need${RESET}\n"
printf " ${GRAY}to create an account/records to see much in the UI.${RESET}\n"
else
printf " ${RED}\xe2\x9a\xa0 the app didn't come up in time${RESET} (server=%s client=%s)\n" "$ok_server" "$ok_client"
printf " check the logs: ${GRAY}run-app --logs${RESET}\n"
fi
printf " logs ${GRAY}%s/{server,client}.log${RESET}\n" "$RUN_DIR"
printf " stop ${GRAY}run-app --stop${RESET}\n"
}
start_flaredown() {
# Polyglot single-container app: the Rails API (backend/) + the Ember client (frontend/).
# The Ember dev server serves the UI on container :3000 (the published port) and proxies
# API calls to the Rails backend, which docker-compose runs on :3000 too — here the client
# takes :3000, so the API binds :5000 (reached only from inside the container) and the
# client proxies to it. rspec needs neither the client nor the running server. Node 14
# (from nvm) drives ember-cli; Ruby 3.2.3 is the image default. OPENSSL_CONF=/dev/null
# lets the old webpack md4 hashing run on bookworm's OpenSSL 3.
local NODE14_BIN
NODE14_BIN=$(ls -d /usr/local/nvm/versions/node/v14.* 2>/dev/null | sort -V | tail -1)/bin
# Three settings the browser needs, none of which a curl of the page reveals:
# PORT config/environment.js bakes ENV.apiHost from it. Left at the
# compose-era 3000 the browser's API calls are cross-origin and CORS-fail;
# set to the published host port they're same-origin and ride --proxy.
# live-reload-port pinned so it matches the published mapping instead of drifting via
# portfinder — the client injects an absolute livereload.js URL.
# FACEBOOK_APP_ID torii's facebook-connect provider reads appId with no default and
# throws during app boot when it's unset.
local LR_PORT="${EXPLORE_LIVERELOAD_PORT:-7020}"
_spawn server /workspace/repo/backend "env PORT=5000 bundle exec rails server -b 0.0.0.0 -p 5000"
_spawn client /workspace/repo/frontend "env PATH=$NODE14_BIN:\$PATH OPENSSL_CONF=/dev/null FACEBOOK_APP_ID=0 PORT=$CLIENT_HOST_PORT ./node_modules/.bin/ember serve --port 3000 --proxy http://localhost:5000 --live-reload-port $LR_PORT"
printf " ${YELLOW}\xe2\x96\xb6${RESET} starting Rails API (puma) on :5000\xe2\x80\xa6\n"
printf " ${CYAN}\xe2\x96\xb6${RESET} starting Ember client (ember-cli)\xe2\x80\xa6\n"
printf " ${GRAY}\xe2\x8f\xb3 waiting for the app to come up (first Ember build takes a minute)\xe2\x80\xa6${RESET}\n"
local ok_server=1 ok_client=1
_wait_tcp 5000 || ok_server=0
_wait_tcp 3000 || ok_client=0
if [ "$ok_server" = 1 ] && [ "$ok_client" = 1 ]; then
printf " ${CYAN}\xe2\x9c\x85 app is up${RESET}\n"
printf " open ${CYAN}http://localhost:%s${RESET}\n" "$CLIENT_HOST_PORT"
printf " ${GRAY}note: that URL is the Ember UI; it proxies API calls to the Rails backend on${RESET}\n"
printf " ${GRAY}:5000 inside the container. The DBs (Postgres + MongoDB) are migrated but${RESET}\n"
printf " ${GRAY}unseeded \xe2\x80\x94 register a user in the UI to see much.${RESET}\n"
else
printf " ${RED}\xe2\x9a\xa0 the app didn't come up in time${RESET} (server=%s client=%s)\n" "$ok_server" "$ok_client"
printf " check the logs: ${GRAY}run-app --logs${RESET}\n"
fi
printf " logs ${GRAY}%s/{server,client}.log${RESET}\n" "$RUN_DIR"
printf " stop ${GRAY}run-app --stop${RESET}\n"
}
start_breezy_complete() {
# Monorepo: Rails API (backend/, container :3001) + Next.js frontend (frontend/,
# container :3000). The offline Clerk-bypass env (DISABLE_CLERK etc.) is injected
# HERE, not baked into the image, so a worker's bare `bundle exec rspec` keeps
# upstream CI's env (ambient DISABLE_CLERK 403s several controller specs).
# NEXT_PUBLIC_BACKEND_URL must be the HOST-visible backend URL — the browser
# calls it — so derive it from the live published server port. Sidekiq is not
# started (only needed for background-job behavior; LLM-dependent jobs degrade
# keyless anyway).
local server_host_port
server_host_port="${EXPLORE_SERVER_PORT:-$(node -e "try{process.stdout.write(String(require('/workspace/toolkit.json').explorePorts.serverHost))}catch{process.stdout.write('4001')}" 2>/dev/null || echo 4001)}"
_spawn server /workspace/repo/backend "env DISABLE_CLERK=true CLERK_SKIP_RAILTIE=true bundle exec rails server -b 0.0.0.0 -p 3001"
_spawn client /workspace/repo/frontend "env NEXT_PUBLIC_BACKEND_URL=http://localhost:${server_host_port} npm run dev -- -H 0.0.0.0 -p 3000"
printf " ${YELLOW}\xe2\x96\xb6${RESET} starting Rails API (backend/) on :3001\xe2\x80\xa6\n"
printf " ${CYAN}\xe2\x96\xb6${RESET} starting Next.js frontend (frontend/)\xe2\x80\xa6\n"
printf " ${GRAY}\xe2\x8f\xb3 waiting for the app to come up\xe2\x80\xa6${RESET}\n"
local ok_server=1 ok_client=1
_wait_tcp 3001 || ok_server=0
_wait_tcp 3000 || ok_client=0
if [ "$ok_server" = 1 ] && [ "$ok_client" = 1 ]; then
printf " ${CYAN}\xe2\x9c\x85 app is up${RESET}\n"
printf " open ${CYAN}http://localhost:%s/pro_signin${RESET}\n" "$CLIENT_HOST_PORT"
printf " ${GRAY}auth is bypassed offline \xe2\x80\x94 /pro_signin auto-redirects to the seeded${RESET}\n"
printf " ${GRAY}professional's dashboard (no login needed). Enter via /pro_signin, not a${RESET}\n"
printf " ${GRAY}bookmarked dashboard URL \xe2\x80\x94 those embed a token that changes on re-seed.${RESET}\n"
else
printf " ${RED}\xe2\x9a\xa0 the app didn't come up in time${RESET} (server=%s client=%s)\n" "$ok_server" "$ok_client"
printf " check the logs: ${GRAY}run-app --logs${RESET}\n"
fi
printf " logs ${GRAY}%s/{server,client}.log${RESET}\n" "$RUN_DIR"
printf " stop ${GRAY}run-app --stop${RESET}\n"
}
# ---- polyglot mode -----------------------------------------------------------
# A polyglot toolkit (toolkit.json .polyglot=true) hosts many member repos under
# repos/<slug>/. The worker picks one with `run-app <repo>`; its deps + DB install on
# first use (deferred), then its app boots on container port 3000. Dispatch is
# RUNTIME-DRIVEN: each member carries a `runtime` ("ruby:3.2.1" | "node:16" |
# "python:3.10" | "none") and an optional `startCmd` in toolkit.json, so there is no
# per-repo hardcoding (scales to all repos). rbenv/pyenv shims must be on PATH inside
# the backgrounded process (a non-login shell), hence the explicit env prefixes.
RBENV_PATH='/usr/local/rbenv/shims:/usr/local/rbenv/bin'
PYENV_PATH='/usr/local/pyenv/shims:/usr/local/pyenv/bin'
# asdf-based estates (salesform-polyglot: Elixir + Ruby + Node from one manager). asdf shims are
# already on PATH image-wide and ASDF_DIR is exported, so a bare `bundle`/`mix`/`npm` resolves each
# member's own .tool-versions — no per-tool PATH/version juggling. Present ONLY in asdf images: every
# other estate has no /usr/local/asdf, so `_asdf_ok` is false there and the rbenv/pyenv/nvm arms below
# run exactly as before. This is also the only place `elixir` runtimes are handled (asdf-only).
_asdf_ok() { [ -f /usr/local/asdf/asdf.sh ]; }
# Let asdf read legacy .ruby-version/.nvmrc (Rails members ship .ruby-version, not .tool-versions).
_asdf_prep() { grep -qs 'legacy_version_file' "$HOME/.asdfrc" 2>/dev/null || echo 'legacy_version_file = yes' >> "$HOME/.asdfrc"; }
_is_polyglot() { node -e "try{process.exit(require('/workspace/toolkit.json').polyglot?0:1)}catch{process.exit(1)}" 2>/dev/null; }
_poly_repos() { node -e "require('/workspace/toolkit.json').repos.forEach(r=>console.log(r.repo))" 2>/dev/null; }
_poly_default() { node -e "process.stdout.write(require('/workspace/toolkit.json').defaultRepo||'')" 2>/dev/null; }
# _poly_field <repo> <field> → the member's field value ('' if absent). Args passed via
# argv (not interpolated) so a repo name can't break the JS.
_poly_field() { node -e "const r=require('/workspace/toolkit.json').repos.find(x=>x.repo===process.argv[1]);process.stdout.write(r&&r[process.argv[2]]!=null?String(r[process.argv[2]]):'')" "$1" "$2" 2>/dev/null; }
# Is rbenv/pyenv version <ver> installed in this image? (EOL runtimes won't be.)
_rb_have() { [ -d "/usr/local/rbenv/versions/$1" ]; }
_py_have() { [ -d "/usr/local/pyenv/versions/$1" ]; }
# Newer estate images ship Python via uv (a system python3 + `uv`) instead of pyenv.
# True when there's no pyenv build for <ver> but uv can provide it — the python arms
# then fall back to a container-local uv venv per member.
_py_uv_ok() { ! _py_have "$1" && command -v uv >/dev/null 2>&1; }
_uv_venv_dir() { printf '/opt/raccoon-venvs/%s' "$1"; }
# Node is multi-version via nvm. Resolve a member's node spec (e.g. "16" or
# "16.20.2") to that major's installed node bin dir, or '' if that major isn't in
# the image (so the selector can fall back to explore-only). Picks the highest
# installed patch of the requested major.
_node_bin() {
local major="${1%%.*}" nvm_dir="${NVM_DIR:-/usr/local/nvm}" d
d=$(ls -d "$nvm_dir"/versions/node/v"$major".* 2>/dev/null | sort -V | tail -1)
[ -n "$d" ] && printf '%s/bin' "$d"
}
# Symlink ./node_modules (cwd = the dir being installed) to a container-local tree keyed by
# <key> — see the ENFILE rationale at the call site. The target must itself be named
# `node_modules` (Node resolves the symlink, then walks ancestors for that literal name),
# and its parent needs a stub manifest: postinstall scripts that locate the project by
# truncating their realpath at `node_modules` require() `<parent>/package.json`.
_nm_link() {
local root="/opt/raccoon-node-modules/$1"
[ -L node_modules ] || rm -rf node_modules
mkdir -p "$root/node_modules"
[ -f "$root/package.json" ] \
|| printf '{"name":"raccoon-node-modules-root","version":"0.0.0","private":true}\n' > "$root/package.json"
ln -sfn "$root/node_modules" node_modules
}
# Rewrite poetry deps of the form `<pkg> = { git = "ssh://git@github.com/AskZeta/<name>.git", rev=… }`
# in <pyproject.toml> to a local path dep at /workspace/repos/zeta-<name>. The sibling repo is a
# member of this toolkit, so the path resolves offline (no SSH key / network needed).
# NB: uses `|` as the s/// delimiter, NOT `{}` — the pattern has `[^}]` and the replacement has
# `{ … }`, which break perl's brace-balanced delimiter parsing.
_rewrite_askzeta_git_deps() {
perl -i -pe 's|=\s*\{\s*git\s*=\s*"ssh://git\@github\.com/AskZeta/([^"]+?)(?:\.git)?"\s*,[^}]*\}|= { path = "/workspace/repos/zeta-\L$1\E", develop = false }|g' "$1" 2>/dev/null || true
}
# Create + schema-load EVERY database of a multi-DB Rails app for one RAILS_ENV ($1).
#
# Rails only defines the namespaced `db:schema:load:<name>` tasks when more than one config
# is VISIBLE to rake, and a config marked `database_tasks: false` is hidden from
# `configs_for`. zeta-plastic marks `source` hidden in development and BOTH connections
# hidden in test, so it has no namespaced tasks at all: the commands below fail with
# `UnrecognizedCommandError`, and plain `db:schema:load` can't reach the extra DB anyway.
# So: try the namespaced path (px-api has it), else walk the configs ourselves.
#
# NEVER db:migrate — its implicit schema:dump regenerates db/source_schema.rb from the
# near-empty source DB, truncating the real file (3487 -> ~77 lines).
_multidb_setup_env() {
local e="$1"
RAILS_ENV="$e" DISABLE_SPRING=1 bundle exec rails db:create 2>/dev/null
if RAILS_ENV="$e" DISABLE_SPRING=1 bundle exec rails db:schema:load:primary >/dev/null 2>&1; then
RAILS_ENV="$e" DISABLE_SPRING=1 bundle exec rails db:schema:load:source >/dev/null 2>&1 || true
return 0
fi
# Fallback: create and load each config, hidden ones included. Two traps, both hit in
# practice on zeta-plastic: (1) `create` raises DatabaseAlreadyExists once the db:create
# above has made the primary DB, and that path leaves ActiveRecord connected to the
# `postgres` MAINTENANCE database; (2) load_schema does not connect on its own (Rails
# 7.2) — it loads into whatever connection is current. Without the explicit
# establish_connection below, the app's tables get created inside `postgres` and the
# real DB is left empty, with every command still reporting success.
# Ruby goes to a real temp file, not /dev/stdin — `rails runner` Kernel.loads the path,
# which needs a seekable file, and a heredoc is a pipe on some shells.
local rb; rb=$(mktemp /tmp/raccoon-load-schemas.XXXXXX.rb)
cat > "$rb" <<'RUBY'
ActiveRecord::Base.configurations
.configs_for(env_name: Rails.env, include_hidden: true)
.reject(&:replica?).each do |c|
begin
ActiveRecord::Tasks::DatabaseTasks.create(c)
rescue ActiveRecord::DatabaseAlreadyExists, ActiveRecord::StatementInvalid
end
dump = c.schema_dump || "schema.rb"
file = Rails.root.join("db", dump)
next unless File.exist?(file)
ActiveRecord::Base.establish_connection(c)
ActiveRecord::Tasks::DatabaseTasks.load_schema(c, :ruby, file.to_s)
puts "loaded db/#{dump} -> #{c.database}"
end
RUBY
# "already exists" is expected for the DB db:create just made — not worth showing.
RAILS_ENV="$e" DISABLE_SPRING=1 bundle exec rails runner "$rb" 2>&1 \
| grep -v "already exists" | sed 's/^/ /'
rm -f "$rb"
return 0
}
# First-use setup writes to two places with different lifetimes, so it takes two markers:
# host — the commit checkout, in the bind-mounted repo dir; survives any container.
# ctr — deps (node_modules / gems / venv / cargo target), databases and ~/.bashrc; all of
# these live in this container and die with it.
# Tracking both with one host-side marker makes a second or rebuilt container skip an install
# it never ran, leaving the member pointed at a node_modules that isn't there.
CTR_MARKER_DIR="/opt/raccoon-setup"
# Record <repo> as set up in THIS container. Best-effort: if the marker can't be written the
# only consequence is that setup runs again next time, and every step of it is idempotent.
_mark_ctr_setup() { mkdir -p "$CTR_MARKER_DIR" 2>/dev/null && : > "$CTR_MARKER_DIR/$1.done" 2>/dev/null || true; }
# First-use setup for a member repo: checkout its commit, install deps, prepare DB.
# The DNS jail (post-start.sh) blocks package registries, and the setup below installs
# from them. Lift it for the install, then put it back — including on Ctrl-C, or the
# container would silently keep its network until the next start.
_DNSJAIL_LIFTED=""
_dnsjail_lift() {
[ -f /tmp/.dnsjail/resolv.orig ] || return 0
# Already unjailed by hand: leave the worker's choice alone rather than putting the
# jail back under them when this exits.
grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null || return 0
mkdir -p /tmp/.dnsjail/lifts 2>/dev/null || return 0
: > "/tmp/.dnsjail/lifts/$$" 2>/dev/null || true
_DNSJAIL_LIFTED=1
sudo sh -c 'cat /tmp/.dnsjail/resolv.orig > /etc/resolv.conf' 2>/dev/null || true
if grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null; then
echo " (could not unjail DNS for the install — run \`unjail\` and retry)" >&2
else
echo " (DNS unjailed for dependency install)"
fi
}
# A trap handler that merely returns leaves the shell alive, so re-raise: without this an
# armed INT trap swallows the worker's Ctrl-C entirely.
_dnsjail_onsig() { _dnsjail_restore; trap - "$1" EXIT; kill -"$1" $$; }
_dnsjail_restore() {
[ -n "$_DNSJAIL_LIFTED" ] || return 0
rm -f "/tmp/.dnsjail/lifts/$$" 2>/dev/null || true
# A marker from a killed run-app would otherwise keep the jail off indefinitely.
for _m in /tmp/.dnsjail/lifts/*; do
[ -e "$_m" ] || continue
kill -0 "${_m##*/}" 2>/dev/null || rm -f "$_m" 2>/dev/null || true
done
# Another run-app is mid-install: leave the network up for it.
[ -n "$(ls -A /tmp/.dnsjail/lifts 2>/dev/null)" ] && return 0
[ -f /tmp/.dnsjail/allow ] || return 0
sudo env DNSJAIL_ALLOW="$(cat /tmp/.dnsjail/allow)" \
sh /workspace/.devcontainer/dns-jail-container.sh >/dev/null 2>&1 || true
}
# Runtime-driven; each marker is written only once its own half has succeeded.
setup_repo() {
local repo="$1" root="/workspace/repos/$1" dir
local hostmarker="/workspace/repos/$1/.raccoon-setup-done" ctrmarker="$CTR_MARKER_DIR/$1.done"
[ -f "$ctrmarker" ] && return 0
_dnsjail_lift
if [ -n "$_DNSJAIL_LIFTED" ]; then
trap _dnsjail_restore EXIT
trap '_dnsjail_onsig INT' INT
trap '_dnsjail_onsig TERM' TERM
fi
local commit runtime kind ver bootenv setupcmd apppath
commit=$(_poly_field "$repo" defaultCommit)
runtime=$(_poly_field "$repo" runtime); kind=${runtime%%:*}; ver=${runtime#*:}
# A member's optional bootEnv ("KEY=val KEY2=val2") supplies dummy values for vars an
# app reads at class-load that its .env.example omits (e.g. wasabi-platform's IVR_UN/
# IVR_PW). dotenv does NOT reliably load .env into the rspec process for some apps, so
# the load-bearing channel is real shell exports in ~/.bashrc (below) — the worker's
# `bundle exec rspec` then sees them. The .env append (in each runtime case) is belt-
# and-suspenders for dotenv-loading apps. Keeps real secrets out; just unblocks boot.
bootenv=$(_poly_field "$repo" bootEnv)
# A member's optional setupCmd runs ONCE here, after deps are installed, for one-time
# app preparation that isn't boot (schema push, seeding, generating a gitignored asset).
# It belongs here rather than in startCmd: startCmd runs on every `run-app`, so seeding
# from there re-runs on each boot and its output is mixed into the server log. Failure
# is non-fatal (a warning) — a member that can still be explored shouldn't be blocked by
# a seed hiccup, mirroring the `|| true` seeds in post-create.sh for single-repo kits.
setupcmd=$(_poly_field "$repo" setupCmd)
# A member whose manifest sits in a subdirectory (monorepo: app/, py/, backend/) installs and
# boots from there. Git state stays at $root; only dependency install and boot use $dir.
apppath=$(_poly_field "$repo" appPath); dir="$root${apppath:+/$apppath}"
# Only the first container to reach a given repo dir checks it out: the checkout is host-side
# state, so redoing it later would move a worker off a commit they had deliberately chosen.
if [ ! -f "$hostmarker" ] && [ -n "$commit" ] \
&& ! git -C "$root" -c advice.detachedHead=false checkout "$commit" >/dev/null 2>&1; then
printf "${RED}checkout %s failed for %s${RESET}\n" "$commit" "$repo"; return 1
fi
# Keep the setup marker out of `git status` — and out of snapshot patches, which
# capture the worker's repo state (mirrors post-create's .pnpm-store exclude; the
# create-snapshot checkpoint hook excludes it as well).
mkdir -p "$root/.git/info"
grep -qxF '.raccoon-setup-done' "$root/.git/info/exclude" 2>/dev/null \
|| printf '\n# raccoon-explore: run-app first-use setup marker\n.raccoon-setup-done\n' >> "$root/.git/info/exclude"
# Same for the node_modules symlink: a `node_modules/` .gitignore entry doesn't match it.
grep -qxF 'node_modules' "$root/.git/info/exclude" 2>/dev/null \
|| printf '\n# raccoon-explore: run-app node_modules symlink\nnode_modules\n' >> "$root/.git/info/exclude"
# A member with no lockfile (or only a pnpm one) gets `yarn install`, which writes a
# lockfile the worker never authored. Excluding only suppresses it while UNTRACKED, so a
# member that commits its lockfile still reports real changes to it.
for lock in yarn.lock package-lock.json; do
grep -qxF "$lock" "$root/.git/info/exclude" 2>/dev/null \
|| printf '\n# raccoon-explore: lockfile generated by run-app'"'"'s install\n%s\n' "$lock" >> "$root/.git/info/exclude"
done
touch "$hostmarker" 2>/dev/null || true
# Persist bootEnv as real exports for ALL the worker's container shells (deduped per repo).
if [ -n "$bootenv" ] && ! grep -q "raccoon-bootenv:$repo" "$HOME/.bashrc" 2>/dev/null; then
{ echo "# raccoon-bootenv:$repo"; for kv in $bootenv; do echo "export $kv"; done; } >> "$HOME/.bashrc"
fi
printf " ${GRAY}first-time setup for %s (%s) \xe2\x80\x94 runs once\xe2\x80\xa6${RESET}\n" "$repo" "${runtime:-explore-only}"
case "$kind" in
elixir)
# asdf-only (no rbenv/nvm estate has elixir). Version comes from the member's
# .tool-versions; shims are already on PATH. deps + a MIX_ENV=test compile so the
# suite is warm and compile errors surface at setup, not mid-explore. bootEnv covers
# any compile-time env a member reads (e.g. epihub's ZOOM_* module attributes). DB/ecto
# prep is member-specific → leave it to setupCmd; a worker runs `mix test` with it.
( cd "$dir" \
&& for kv in $bootenv; do export "$kv"; done \
&& mix local.hex --force >/dev/null 2>&1 \
&& mix local.rebar --force >/dev/null 2>&1 \
&& { [ -f config/dev.secret.exs.example ] && [ ! -f config/dev.secret.exs ] && cp config/dev.secret.exs.example config/dev.secret.exs; true; } \
&& mix deps.get \
&& MIX_ENV=test mix compile ) || return 1 ;;
ruby)
if _asdf_ok; then
_asdf_prep
# Version from .ruby-version (legacy) / .tool-versions; shims already on PATH.
# Regenerate binstubs when the repo ships an empty bin/ (Rails detects an app via
# bin/rails — without it `bundle exec rails` prints `new` help and won't boot).
( cd "$dir" \
&& { [ -f config/database.yml.example ] && cp -n config/database.yml.example config/database.yml; true; } \
&& { [ -f .env.example ] && cp -n .env.example .env; true; } \
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } \
&& { bundle lock --add-platform x86_64-linux aarch64-linux >/dev/null 2>&1 || true; } \
&& { bundle install || bundle install --full-index; } \
&& { [ -f bin/rails ] || bundle binstubs railties --force --path bin >/dev/null 2>&1 || bundle exec rake app:update:bin >/dev/null 2>&1 || true; } \
&& { bundle exec rails db:prepare 2>/dev/null || bundle exec rails db:create db:schema:load 2>/dev/null || true; \
RAILS_ENV=test bundle exec rails db:create 2>/dev/null; \
RAILS_ENV=test bundle exec rails db:schema:load 2>/dev/null; \
RAILS_ENV=test bundle exec rails db:migrate 2>/dev/null || true; } ) || return 1
else
_rb_have "$ver" || { printf " ${GRAY}(Ruby %s not in this image; skipping deps \xe2\x80\x94 explore-only)${RESET}\n" "$ver"; _mark_ctr_setup "$repo"; return 0; }
( cd "$dir" \
&& export PATH="$RBENV_PATH:$PATH" RBENV_VERSION="$ver" \
&& { [ -f config/database.yml.example ] && cp -n config/database.yml.example config/database.yml; true; } \
&& { [ -f .env.example ] && cp -n .env.example .env; true; } \
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } \
&& { bundle lock --add-platform x86_64-linux aarch64-linux >/dev/null 2>&1 || true; } \
&& { bundle install || bundle install --full-index; } \
&& { if [ -f db/source_schema.rb ]; then \
# MULTI-DATABASE (px-api, plastic): the `users` etc. live in the `source` DB.
# Load EACH db's schema for dev AND test. DISABLE_SPRING so a preloaded
# stale connection doesn't make the source load a silent no-op.
for e in development test; do \
_multidb_setup_env "$e" || true; \
done; \
else \
# Single-DB: prepare the dev DB (rails_helper often needs it present), then
# load + migrate the test DB (migrate is a no-op when schema.rb is current,
# and applies pending migrations when it's stale).
bundle exec rails db:prepare 2>/dev/null || bundle exec rails db:create db:schema:load 2>/dev/null || true; \
RAILS_ENV=test bundle exec rails db:create 2>/dev/null; \
RAILS_ENV=test bundle exec rails db:schema:load 2>/dev/null; \
RAILS_ENV=test bundle exec rails db:migrate 2>/dev/null || true; \
fi; } ) || return 1
fi ;;
node)
if _asdf_ok; then
_asdf_prep
( cd "$dir" \
&& _nm_link "$repo" \
&& { [ -f .env.example ] && cp -n .env.example .env; true; } \
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } \
&& { if [ -f yarn.lock ]; then yarn install; elif [ -f package-lock.json ]; then npm install; else yarn install; fi; } ) || return 1
else
nbin=$(_node_bin "$ver")
[ -z "$nbin" ] && { printf " ${GRAY}(Node %s not in this image; skipping deps \xe2\x80\x94 explore-only)${RESET}\n" "$ver"; _mark_ctr_setup "$repo"; return 0; }
# Install node_modules to a CONTAINER-LOCAL path, not the bind-mounted repo dir. On
# macOS Docker Desktop the repo is a host bind mount; writing a huge node_modules tree
# across the file-sharing layer is slow AND exhausts the HOST's open-file table (ENFILE
# "file table overflow"), which can take the whole machine down — not just the install.
# Keeping node_modules inside the Linux VM confines that churn to the VM. The repo stays
# bind-mounted (the worker sees their edits); node_modules is reached via a symlink.
( cd "$dir" \
&& export PATH="$nbin:$PATH" \
&& _nm_link "$repo" \
&& { [ -f .env.example ] && cp -n .env.example .env; true; } \
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } \
&& { if [ -f pnpm-lock.yaml ] && command -v pnpm >/dev/null 2>&1; then
# pnpm member: install the locked pnpm tree (matching the graded image), not yarn.
if [ "${ver%%.*}" -lt 22 ] 2>/dev/null; then
# pnpm@9 (node<22) can't install through the _nm_link node_modules symlink
# (ENOTDIR on mkdir), so give it a real node_modules but keep pnpm's heavy
# virtual + content stores container-local — the ENFILE protection _nm_link
# provides (node_modules then holds only lightweight symlinks).
rm -rf node_modules
pnpm install --no-frozen-lockfile --config.dangerouslyAllowAllBuilds=true \
--virtual-store-dir="/opt/raccoon-node-modules/$repo/.pnpm-vstore" \
--store-dir=/opt/raccoon-pnpm-store
else
# pnpm>=11 (node>=22) follows the _nm_link symlink; node_modules and its .pnpm
# store are already container-local through it.
pnpm install --no-frozen-lockfile --config.dangerouslyAllowAllBuilds=true
fi
elif [ -f yarn.lock ]; then yarn install
elif [ -f package-lock.json ]; then npm install
else yarn install; fi; } ) || return 1
fi ;;
python)
if _py_uv_ok "$ver"; then
# uv-python image (clockwise-polyglot era): container-local venv per member,
# deps via uv. `uv pip install -e .` handles poetry-backend pyprojects too — but it
# resolves from pyproject CONSTRAINTS and ignores poetry.lock, while the graded image
# runs `poetry install` and gets the locked set. That divergence broke search-api-v2
# outright (Explore resolved pydantic 2.13.4 against a lock pinning 2.9.2, and the
# pinned strawberry cannot import on 2.13). Prefer the lock when there is one.
local vdir; vdir=$(_uv_venv_dir "$repo")
( cd "$dir" \
&& uv venv "$vdir" -p "$ver" -q \
&& . "$vdir/bin/activate" \
&& { [ -f .env.example ] && cp -n .env.example .env; true; } \
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } \
&& { if [ -f poetry.lock ] && command -v poetry >/dev/null 2>&1 \
&& POETRY_VIRTUALENVS_CREATE=false poetry install -q --no-interaction --no-root 2>/dev/null; then true; \
elif [ -f pyproject.toml ]; then uv pip install -q -e . || uv pip install -q -r requirements.txt 2>/dev/null || true; \
elif [ -f requirements.txt ]; then uv pip install -q -r requirements.txt; \
elif [ -f server/requirements.txt ]; then uv pip install -q -r server/requirements.txt; \
elif [ -f setup.py ]; then uv pip install -q -e .; else true; fi; } ) || return 1
_mark_ctr_setup "$repo"; return 0
fi
_py_have "$ver" || { printf " ${GRAY}(Python %s not in this image; skipping deps \xe2\x80\x94 explore-only)${RESET}\n" "$ver"; _mark_ctr_setup "$repo"; return 0; }
# Some poetry repos depend on sibling repos via `git = "ssh://git@github.com/AskZeta/<name>.git"`,
# which can't resolve in the container (no SSH key, no network). The deps are TRANSITIVE
# (cx-chatbot → compiler-agent → agent-tools → leaves), so rewrite the target AND every
# sibling pyproject to local path deps — else poetry shells out to `ssh` for a transitive
# git dep and fails ("No such file or directory: 'ssh'").
for pp in /workspace/repos/*/pyproject.toml; do
[ -f "$pp" ] && _rewrite_askzeta_git_deps "$pp"
done
( cd "$dir" && export PATH="$PYENV_PATH:$PATH" PYENV_VERSION="$ver" \
&& { [ -f .env.example ] && cp -n .env.example .env; true; } \
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } \
&& { if [ -f pyproject.toml ]; then \
# Every member Dockerfile sets this; without it poetry builds a .venv here
# that the trial image has no equivalent of.
poetry config virtualenvs.create false 2>/dev/null || true; \
# The git→path rewrite invalidates poetry.lock ("changed significantly");
# regenerate it before installing. Poetry 2.x `lock` preserves pins by
# default (the old `--no-update` flag was removed in 2.0).
poetry lock 2>/dev/null || true; \
# --no-root: install deps only, not the project package itself. Some members'
# pyproject package name doesn't map to a folder poetry can find ("No file/folder
# found for package <x>"), which fails the whole install. The worker explores +
# runs the code from the repo dir (cwd on path), so the project never needs to be
# pip-installed as a package. Mirrors the harbor build.
poetry install --no-interaction --no-root; \
elif [ -f requirements.txt ]; then pip install -r requirements.txt; \
elif [ -f setup.py ]; then pip install -e .; else true; fi; } ) || return 1 ;;
rust)
command -v cargo >/dev/null 2>&1 || { printf " ${GRAY}(Rust not in this image; skipping build \xe2\x80\x94 explore-only)${RESET}\n"; _mark_ctr_setup "$repo"; return 0; }
# Build to a container-local target dir (same ENFILE/bind-mount rationale as
# node_modules): a Cargo workspace target tree is huge and rebuilds often.
( cd "$dir" \
&& { [ -f .env.example ] && cp -n .env.example .env; true; } \
&& { for kv in $bootenv; do grep -qxF "$kv" .env 2>/dev/null || echo "$kv" >> .env; done; true; } \
&& CARGO_TARGET_DIR="/opt/raccoon-cargo-target/$repo" cargo build --workspace ) || return 1 ;;
none|"") : ;; # no-code / explore-only: nothing to install
*) printf "${YELLOW}unknown runtime '%s' for %s \xe2\x80\x94 explore-only${RESET}\n" "$runtime" "$repo" ;;
esac
# Optional one-time app preparation (see setupCmd above), with the member's runtime on
# PATH and its bootEnv exported — same environment the app boots with.
if [ -n "$setupcmd" ]; then
local spath=""
case "$kind" in
ruby) spath="$RBENV_PATH" ;;
node) spath=$(_node_bin "$ver") ;;
python) spath="$PYENV_PATH" ;;
esac
# Output goes to a log, not the worker's terminal: preparation is chatty (an app's
# own seed can log hundreds of lines about services it can't reach offline, all of
# them harmless), and a wall of red JSON reads as "something is broken". The log
# lands in RUN_DIR so `run-app --logs` picks it up like any other.
mkdir -p "$RUN_DIR"
local slog="$RUN_DIR/setup-$repo.log"
printf " ${GRAY}preparing %s (one-time; details in ${RESET}${GRAY}run-app --logs${RESET}${GRAY})\xe2\x80\xa6${RESET}\n" "$repo"
if ( cd "$dir" \
&& export PATH="${spath:+$spath:}$PATH" \
&& case "$kind" in ruby) export RBENV_VERSION="$ver" ;; python) export PYENV_VERSION="$ver" ;; esac \
&& { for kv in $bootenv; do export "$kv"; done; } \
&& eval "$setupcmd" ) > "$slog" 2>&1; then
printf " ${GRAY}\xe2\x9c\x93 %s prepared${RESET}\n" "$repo"
else
printf " ${YELLOW}setup for %s did not finish cleanly \xe2\x80\x94 the repo is still explorable.${RESET}\n" "$repo"
printf " ${GRAY}what went wrong: %s${RESET}\n" "$slog"
fi
fi
_mark_ctr_setup "$repo"
}
start_poly() {
local repo="${1:-}"; [ -z "$repo" ] && repo="$(_poly_default)"
if ! _poly_repos | grep -qx "$repo"; then
printf "${RED}unknown repo '%s'.${RESET} available: ${GRAY}%s${RESET}\n" "$repo" "$(_poly_repos | tr '\n' ' ')"
return 1
fi
local dir="/workspace/repos/$repo" runtime kind ver startcmd bootenv apppath
runtime=$(_poly_field "$repo" runtime); kind=${runtime%%:*}; ver=${runtime#*:}
# Boot from the member's manifest directory when it isn't the repo root — the node arm reads
# $dir/package.json to pick a dev-server script, and would otherwise find none.
apppath=$(_poly_field "$repo" appPath); dir="$dir${apppath:+/$apppath}"
startcmd=$(_poly_field "$repo" startCmd)
bootenv=$(_poly_field "$repo" bootEnv) # dummy class-load vars (e.g. IVR_UN); see setup_repo
# Explore-only members (no-code repos, or no runtime): nothing to boot.
if [ "$kind" = "none" ] || [ -z "$kind" ]; then
printf " ${CYAN}%s${RESET} is explore-only (no app to run). Read it under ${GRAY}/workspace/repos/%s${RESET}.\n" "$repo" "$repo"
return 0
fi
# Runtime not in this image (EOL Ruby 2.6.6 / Python 3.7 / an uninstalled node major):
# explorable, not runnable here. Python counts as present when EITHER pyenv has the
# version or uv can provide it (uv-python images ship no pyenv at all — without the
# _py_uv_ok check this gate refused every python member before the uv setup arm ran).
if ! _asdf_ok && { { [ "$kind" = ruby ] && ! _rb_have "$ver"; } \
|| { [ "$kind" = python ] && ! _py_have "$ver" && ! _py_uv_ok "$ver"; } \
|| { [ "$kind" = node ] && [ -z "$(_node_bin "$ver")" ]; }; }; then
printf " ${YELLOW}%s needs %s, which isn't in this image.${RESET}\n" "$repo" "$runtime"
printf " Explore the code under ${GRAY}/workspace/repos/%s${RESET}; to RUN it use that repo's dedicated toolkit.\n" "$repo"
return 0
fi
local running=0; for pf in "$RUN_DIR"/*.pid; do [ -e "$pf" ] && _alive "$pf" && running=1; done
if [ "$running" = 1 ]; then
printf "${GRAY}An app is already running.${RESET} Stop it first: ${GRAY}run-app --stop${RESET} (then ${GRAY}run-app %s${RESET}).\n" "$repo"
return 0
fi
bash /workspace/.devcontainer/post-start.sh >/dev/null 2>&1 || true
setup_repo "$repo" || { printf "${RED}setup failed for %s${RESET} \xe2\x80\x94 ${GRAY}run-app --logs${RESET}\n" "$repo"; return 1; }
local cmd=""
case "$kind" in
elixir)
# asdf-only. Boot needs member-specific env/port (Phoenix reads endpoint config), so a
# startCmd is the reliable path; without one, leave it explore-only — the worker runs
# `mix test` / `mix phx.server` directly. Version + shims come from .tool-versions.
if [ -n "$startcmd" ]; then
cmd="env $bootenv $startcmd"
else
printf " ${GRAY}%s: deps compiled. No startCmd wired \xe2\x80\x94 run it directly (${RESET}${GRAY}mix phx.server${RESET}${GRAY}) or its tests (${RESET}${GRAY}mix test${RESET}${GRAY}).${RESET}\n" "$repo"
return 0
fi ;;
ruby)
if _asdf_ok; then
# Version + shims from .tool-versions (no rbenv PATH). setup regenerated bin/rails
# when the repo shipped an empty bin/, so the app-detection below still holds.
if [ -n "$startcmd" ]; then cmd="env $bootenv $startcmd"
elif [ -f "$dir/bin/rails" ]; then cmd="env $bootenv bundle exec rails server -b 0.0.0.0 -p 3000"
elif [ -f "$dir/config.ru" ]; then cmd="env $bootenv bundle exec rackup -o 0.0.0.0 -p 3000"
else
printf " ${GRAY}%s isn't a web app (no bin/rails/config.ru) \xe2\x80\x94 run its tests directly (${RESET}${GRAY}bundle exec rails test${RESET}${GRAY}).${RESET}\n" "$repo"
return 0
fi
elif [ -n "$startcmd" ]; then
cmd="env $bootenv PATH=$RBENV_PATH:\$PATH RBENV_VERSION=$ver $startcmd"
elif [ -f "$dir/bin/rails" ]; then
cmd="env $bootenv PATH=$RBENV_PATH:\$PATH RBENV_VERSION=$ver bundle exec rails server -b 0.0.0.0 -p 3000"
elif [ -f "$dir/config.ru" ]; then
# Rack app that isn't Rails (no bin/rails) — boot via rackup.
cmd="env $bootenv PATH=$RBENV_PATH:\$PATH RBENV_VERSION=$ver bundle exec rackup -o 0.0.0.0 -p 3000"
else
printf " ${GRAY}%s isn't a web app (no bin/rails/config.ru) \xe2\x80\x94 run its tests directly (${RESET}${GRAY}bundle exec rspec${RESET}${GRAY}).${RESET}\n" "$repo"
return 0
fi ;;
node)
local nbin sc=""
nbin=$(_node_bin "$ver")
if _asdf_ok; then
# asdf node: shims already on PATH, version from .tool-versions/.nvmrc. Only a
# startCmd-driven or dev-server boot; RN/static members fall through to explore-only.
if [ -n "$startcmd" ]; then
cmd="env $bootenv PORT=3000 BROWSER=none HOST=0.0.0.0 $startcmd"
else
local s2=""
for s2 in start dev develop serve; do
if node -e "process.exit((((require('$dir/package.json')||{}).scripts)||{})['$s2']?0:1)" 2>/dev/null; then break; else s2=""; fi
done
if [ -z "$s2" ]; then
printf " ${GRAY}%s: deps installed, no dev-server script \xe2\x80\x94 run its tests directly (${RESET}${GRAY}yarn test${RESET}${GRAY}).${RESET}\n" "$repo"
return 0
fi
cmd="env $bootenv PORT=3000 BROWSER=none HOST=0.0.0.0 yarn $s2"
fi
elif [ -n "$startcmd" ]; then
cmd="env $bootenv PATH=$nbin:\$PATH PORT=3000 BROWSER=none HOST=0.0.0.0 $startcmd"
elif [ -f "$dir/metro.config.js" ] || [ -d "$dir/ios" ] || [ -d "$dir/android" ]; then
# React Native app: no web server in a Linux container; tests still run.
printf " ${GRAY}%s is a React Native app (no web server here) \xe2\x80\x94 run its Jest tests directly (${RESET}${GRAY}yarn test${RESET}${GRAY}).${RESET}\n" "$repo"
return 0
else
# CRA / generic: first dev-server script the repo defines, bound to :3000.
local s
for s in start dev develop serve; do
if node -e "process.exit((((require('$dir/package.json')||{}).scripts)||{})['$s']?0:1)" 2>/dev/null; then sc="$s"; break; fi
done
if [ -z "$sc" ]; then
printf " ${GRAY}%s: deps installed, no dev-server script \xe2\x80\x94 run its tests directly (${RESET}${GRAY}yarn test${RESET}${GRAY}).${RESET}\n" "$repo"
return 0
fi
# A Create-React-App dev server (react-scripts / react-app-rewired) needs extra env
# to survive in this non-interactive container. We spawn it with stdout redirected
# to a log, so react-scripts sees a non-TTY and (start.js) registers a stdin-"end"
# handler that closes the dev server the moment stdin ends — which it does at once
# when there's no interactive terminal, so the app appears to "crash on boot". The
# guard is `if (isInteractive || process.env.CI !== 'true')`, so CI=true is what
# skips it and keeps the server up. CI=true does NOT make `start` treat warnings as
# errors — that is `build` only (verified against react-scripts 3.4.1). The others:
# DANGEROUSLY_DISABLE_HOST_CHECK=true let the dev server answer requests arriving
# via the published host port (belt-and-braces;
# wds3 already allows IP/localhost hosts).
# NODE_OPTIONS=--openssl-legacy-provider webpack-4-era CRA crashes on Node 17+
# without it; the flag only EXISTS on Node 17+,
# so gate it on the major — older nodes (e.g.
# Node 16) abort on "bad option".
# Non-CRA dev servers (Next.js, vite, …) don't match the test, so they boot unchanged.
local craenv=""
if node -e "const s=(((require('$dir/package.json')||{}).scripts)||{})['$sc']||'';process.exit(/react-scripts|react-app-rewired/.test(s)?0:1)" 2>/dev/null; then
craenv="CI=true DANGEROUSLY_DISABLE_HOST_CHECK=true"
case "${ver%%.*}" in 1[7-9]|[2-9][0-9]) craenv="NODE_OPTIONS=--openssl-legacy-provider $craenv" ;; esac
fi
cmd="env $bootenv $craenv PATH=$nbin:\$PATH PORT=3000 BROWSER=none HOST=0.0.0.0 yarn $sc"
fi ;;
python)
if [ -z "$startcmd" ]; then
printf " ${GRAY}%s: Python deps installed. No web server is wired \xe2\x80\x94 run its tests/scripts directly (e.g. pytest).${RESET}\n" "$repo"
return 0
fi
if _py_uv_ok "$ver"; then
local vdir; vdir=$(_uv_venv_dir "$repo")
cmd="env $bootenv VIRTUAL_ENV=$vdir PATH=$vdir/bin:\$PATH $startcmd"
else
cmd="env $bootenv PATH=$PYENV_PATH:\$PATH PYENV_VERSION=$ver $startcmd"
fi ;;
rust)
if [ -z "$startcmd" ]; then
printf " ${GRAY}%s: workspace built. No web server is wired \xe2\x80\x94 run its tests directly (${RESET}${GRAY}cargo test${RESET}${GRAY}).${RESET}\n" "$repo"
return 0
fi
cmd="env $bootenv CARGO_TARGET_DIR=/opt/raccoon-cargo-target/$repo $startcmd" ;;
*) printf "${YELLOW}runtime '%s' for %s isn't runnable here \xe2\x80\x94 explore-only.${RESET}\n" "$runtime" "$repo"; return 0 ;;
esac
printf " ${CYAN}\xe2\x96\xb6${RESET} starting %s (%s)\xe2\x80\xa6\n" "$repo" "$runtime"
_spawn app "$dir" "$cmd"
if _wait_tcp 3000; then
printf " ${CYAN}\xe2\x9c\x85 %s is up${RESET} open ${CYAN}http://localhost:%s${RESET}\n" "$repo" "$CLIENT_HOST_PORT"
# Per-member "how do I actually get in" notes. Only members whose landing page needs
# more than the URL need an entry here (e.g. an app whose real sign-in is a hosted
# third-party login that can't be reached offline).
case "$repo" in
strongsuit-app)
printf " ${GRAY}Sign-in normally goes through a hosted Auth0 page, which isn't reachable\n"
printf " offline, so this app ships a local-only dev-login route. Open\n"
printf " ${RESET}${CYAN}http://localhost:%s/dev-login${RESET}${GRAY} to sign in as a seeded admin\n" "$CLIENT_HOST_PORT"
printf " (${RESET}${GRAY}?role=MSS${RESET}${GRAY} or ${RESET}${GRAY}?role=MEMBER${RESET}${GRAY} for the other roles). The DB was seeded during setup.${RESET}\n"
;;
ABDM-FE)
printf " ${GRAY}This app is served under a ${RESET}${GRAY}/app${RESET}${GRAY} basename, so the bare URL above renders\n"
printf " nothing. Open ${RESET}${CYAN}http://localhost:%s/app/login${RESET}${GRAY} instead.\n" "$CLIENT_HOST_PORT"
printf " Sign-in itself calls hosted services that aren't reachable offline, so the\n"
printf " login page is as far as you can get — read and edit the code from there.${RESET}\n"
;;
search-api-v2)
printf " ${GRAY}Browse and try the API at ${RESET}${CYAN}http://localhost:%s/docs${RESET}${GRAY}.\n" "$CLIENT_HOST_PORT"
printf " Sign-in goes through a hosted identity provider that isn't reachable offline,\n"
printf " and this app ships no local login, so ${RESET}${GRAY}/security/login${RESET}${GRAY} returns a 500 and\n"
printf " authenticated routes answer ${RESET}${GRAY}Forbidden access${RESET}${GRAY} — that is expected here, not a\n"
printf " broken setup. To exercise authenticated behaviour, run the test suite.${RESET}\n"
;;
potion-app)
printf " ${GRAY}Sign-in normally goes through Google or LinkedIn, neither reachable offline,\n"
printf " so setup seeded a verified local account. Log in at\n"
printf " ${RESET}${CYAN}http://localhost:%s/auth/login${RESET}${GRAY} with ${RESET}${GRAY}dev@example.com${RESET}${GRAY} / ${RESET}${GRAY}devpassword123${RESET}${GRAY}\n" "$CLIENT_HOST_PORT"
printf " — note ${RESET}${GRAY}/login${RESET}${GRAY} and ${RESET}${GRAY}/auth${RESET}${GRAY} both redirect elsewhere.${RESET}\n"
;;
esac
else
printf " ${RED}\xe2\x9a\xa0 %s didn't come up in time${RESET} \xe2\x80\x94 ${GRAY}run-app --logs${RESET}\n" "$repo"
fi
printf " stop ${GRAY}run-app --stop${RESET} switch ${GRAY}run-app --stop && run-app <repo>${RESET}\n"
printf " focus ${GRAY}cd /workspace/repos/%s && claude${RESET} (so Claude works in this repo without being told the path)\n" "$repo"
}
# Generic Rails boot for the standard-shape apps (the rubyforgood repos): a single
# `bin/rails server` on container :3000, no separate client. The DB is seeded during
# post-create (none of these expose a working self-service signup), so the caller
# passes the demo login to print. Optional $2 is a one-line note printed above the
# login (e.g. a subdomain caveat).
# start_rails <login-hint> [url-note]
start_rails() {
local login_hint="${1:-}" url_note="${2:-}"
_spawn app /workspace/repo "bin/rails server -b 0.0.0.0 -p 3000"
printf " ${YELLOW}\xe2\x96\xb6${RESET} starting Rails (puma)\xe2\x80\xa6\n"
printf " ${GRAY}\xe2\x8f\xb3 waiting for the app to come up\xe2\x80\xa6${RESET}\n"
if _wait_tcp 3000; then
printf " ${YELLOW}\xe2\x9c\x85 app is up${RESET}\n"
printf " open ${YELLOW}http://localhost:%s${RESET}\n" "$CLIENT_HOST_PORT"
[ -n "$url_note" ] && printf " ${GRAY}%s${RESET}\n" "$url_note"
[ -n "$login_hint" ] && printf " login ${GRAY}%s${RESET}\n" "$login_hint"
else
printf " ${RED}\xe2\x9a\xa0 the app didn't come up in time${RESET}\n"
printf " check the logs: ${GRAY}run-app --logs${RESET}\n"
fi
printf " logs ${GRAY}%s/app.log${RESET}\n" "$RUN_DIR"
printf " stop ${GRAY}run-app --stop${RESET}\n"
}
start_app() {
# Already running? Don't double-start.
local running=0
for pf in "$RUN_DIR"/*.pid; do [ -e "$pf" ] && _alive "$pf" && running=1; done
if [ "$running" = 1 ]; then
printf "${GRAY}The app is already running.${RESET} Use ${GRAY}run-app --restart${RESET} to restart, ${GRAY}run-app --status${RESET} to check.\n"
printf " open ${CYAN}http://localhost:%s${RESET}\n" "$CLIENT_HOST_PORT"
return 0
fi
# Make sure the database is up before the server tries to connect.
bash /workspace/.devcontainer/post-start.sh >/dev/null 2>&1 || true
case "$REPO_NAME" in
Palolo-031) start_palolo ;;
ZenBill-006) start_zenbill ;;
zeta-heimdall) start_zeta_heimdall ;;
zeta-platform) start_zeta_platform ;;
human-essentials) start_rails "test@example.com / password! (sign in at /users/sign_in)" ;;
endsideout) start_rails "admin@example.com / password (sign in at /session/new)" ;;
community-foundation)
# Multi-tenant: the org is a subdomain, so plain localhost only shows the
# apex landing page. The seed creates the 'arlington' tenant.
start_rails "owner@example.com / password" \
"this app routes by subdomain — open http://arlington.lvh.me:${CLIENT_HOST_PORT}/ (plain localhost shows only the landing page)" ;;
stocks-in-the-future) start_rails "username admin / password (sign in at /users/sign_in — login is by USERNAME, not email)" ;;
casa) start_rails "casa_admin1@example.com / 12345678 (sign in at /users/sign_in)" ;;
awbw) start_rails "umberto.user@example.com / password (sign in at /users/sign_in)" ;;
flaredown) start_flaredown ;;
alongwithyou)
# Fresh scaffold: no routes/auth yet, so plain localhost shows the default Rails
# welcome page. No login to print. The app grows over time.
start_rails "" "young app — no routes defined yet, so this shows the default Rails welcome page" ;;
breezy-complete) start_breezy_complete ;;
*)
printf "${YELLOW}run-app isn't configured for repo '%s'.${RESET}\n" "${REPO_NAME:-unknown}"
printf "Start the app with the project's own dev command from ${GRAY}/workspace/repo${RESET}.\n"
return 1
;;
esac
}
usage() {
sed -n '2,16p' "$0" | sed 's/^# \{0,1\}//'
}
if _is_polyglot; then
# `run-app [<repo>] [--restart|--stop|--logs|--status]` — order-independent: the repo
# name and the action can appear in either order (e.g. `run-app --restart zeta-hook`),
# and the bare verbs (start/restart/stop/...) are recognized as actions, not repos.
poly_repo=""; poly_action="start"
for a in "$@"; do
case "$a" in
start) poly_action="start" ;;
--restart|restart) poly_action="restart" ;;
--stop|stop) poly_action="stop" ;;
--logs|logs) poly_action="logs" ;;
--status|status) poly_action="status" ;;
-h|--help|help) poly_action="help" ;;
-*) printf "${RED}Unknown option:${RESET} %s\n\n" "$a"; usage; exit 2 ;;
*) poly_repo="$a" ;;
esac
done
case "$poly_action" in
start) start_poly "$poly_repo" ;;
restart) stop_app; start_poly "$poly_repo" ;;
stop) stop_app ;;
logs) logs_app ;;
status) status_app ;;
help) usage ;;
esac
exit $?
fi
case "${1:-}" in
""|start) start_app ;;
--restart|restart) stop_app; start_app ;;
--stop|stop) stop_app ;;
--logs|logs) logs_app ;;
--status|status) status_app ;;
-h|--help|help) usage ;;
*) printf "${RED}Unknown option:${RESET} %s\n\n" "$1"; usage; exit 2 ;;
esac

View File

@@ -0,0 +1,271 @@
version = 1
[[harness]]
id = "claude-code"
label = "Claude Code"
agent_import_path = "snapshot_agent:SnapshotClaudeCode"
# `[metadata] browser = true` swaps in these: same reduced toolset plus `Read`, so an agent
# given a browser can look at the screenshot it just took. Distinct classes with distinct
# names, because a different toolset is a different agent.
agent_import_path_browser = "snapshot_agent:BrowserSnapshotClaudeCode"
agent_import_path_single_turn_browser = "snapshot_agent:BrowserPreinstalledClaudeCode"
agent_import_path_single_turn = "snapshot_agent:PreinstalledClaudeCode"
import_path_aliases = [
"snapshot_agent:FullToolsetSnapshotClaudeCode",
"snapshot_agent:FullToolsetPreinstalledClaudeCode",
"harbor.agents.installed.claude_code:ClaudeCode",
]
legacy_bare_model_rows = true
default_model = "claude-opus-5[1m]"
model_id_shape = "bare"
effort_kwarg = "reasoning_effort"
effort_default = "max"
fast_kwarg = "fast_mode"
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = true
seed_native = true
seed_atif = true
authoring = true
cli = "claude"
install = "for i in 1 2 3; do curl -fsSL https://claude.ai/install.sh | bash && break; echo \"claude install attempt $i failed; retrying in 10s\" >&2; sleep 10; done"
# No agent_config: claude reduces its toolset with `--tools`, not `-c key=value`, so the
# reduction is a launch flag here and `--tools Bash` in snapshot_agent.py for the trial.
# Two expressions of one intent, which the $RACCOON_AGENT_FLAGS guard cannot police —
# unlike model and effort, which are interpolated from this row.
explore_launch = """exec claude --model '$RACCOON_MODEL' --effort $RACCOON_EFFORT --tools "$RACCOON_TOOLS" --append-system-prompt "$RACCOON_TOOLSET_NOTE" --plugin-dir /workspace/plugins/create-snapshot --dangerously-skip-permissions "$@""""
[[harness]]
id = "codex"
label = "OpenAI Codex CLI"
agent_import_path = "codex_agent:NativeSnapshotCodex"
agent_import_path_single_turn = "codex_agent:SystemNodeCodex"
import_path_aliases = [
"codex_agent:InlineSnapshotCodex",
"harbor.agents.installed.codex:Codex",
]
legacy_bare_model_rows = true
default_model = "gpt-5.6-sol"
model_id_shape = "bare"
effort_kwarg = "reasoning_effort"
effort_default = "max"
key_env = "OPENAI_API_KEY"
base_url_env = "OPENAI_BASE_URL"
proxy_path = "openai/v1"
writes_atif = true
capture = true
seed_native = true
seed_atif = true
authoring = true
cli = "codex"
install = "for i in 1 2 3; do curl -fsSL https://chatgpt.com/codex/install.sh | CODEX_NON_INTERACTIVE=1 sh && break; echo \"codex install attempt $i failed; retrying in 10s\" >&2; sleep 10; done"
skills_dir = "$HOME/.agents/skills"
config_path = "${CODEX_HOME:-$HOME/.codex}/config.toml"
auth_path = "${CODEX_HOME:-$HOME/.codex}/auth.json"
auth_key_env = "OPENAI_API_KEY"
agent_config = """
web_search = "disabled"
[agents]
enabled = false
[tools]
update_plan = { enabled = false }
experimental_request_user_input = { enabled = false }
[features]
goals = false
multi_agent = false
multi_agent_v2 = false
memories = false
external_agent_memory_import = false
"""
container_config = """
openai_base_url = "${OPENAI_BASE_URL}"
"""
explore_config = """
[hooks]
SessionStart = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/save-session-info.mjs" } ] } ]
UserPromptSubmit = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/checkpoint-workspace.mjs" } ] } ]
"""
explore_launch = """exec codex $RACCOON_AGENT_FLAGS --model $RACCOON_MODEL -c model_reasoning_effort=$RACCOON_EFFORT ${RACCOON_BROWSER_FLAGS[@]+"${RACCOON_BROWSER_FLAGS[@]}"} --dangerously-bypass-approvals-and-sandbox --dangerously-bypass-hook-trust "$@""""
[[harness]]
id = "gemini-cli"
label = "Gemini CLI"
agent_import_path = "gemini_agent:NativeSnapshotGeminiCli"
agent_import_path_single_turn = "gemini_agent:SystemNodeGeminiCli"
import_path_aliases = ["harbor.agents.installed.gemini_cli:GeminiCli"]
legacy_bare_model_rows = true
default_model = "gemini-3.5-flash"
model_id_shape = "provider/model"
effort_kwarg = "reasoning_effort"
effort_default = "high"
key_env = "GEMINI_API_KEY"
base_url_env = "GEMINI_API_BASE"
proxy_path = "gemini"
writes_atif = true
capture = false
seed_native = true
seed_atif = false
[[harness]]
id = "antigravity-cli"
label = "Antigravity CLI"
agent_import_path = "harness_agents:BenchAntigravity"
import_path_aliases = ["harbor.agents.installed.antigravity_cli:AntigravityCli"]
legacy_bare_model_rows = false
# The prefix is load-bearing: harbor's adapter raises without a "/" in the id.
# agy carries its own model catalogue and DROPS entries between point releases
# (1.1.25 removed gemini-3.5-flash, breaking every run). If trials start failing
# with "not recognized as a known model", run `agy --model bogus --prompt=x` to
# print the current catalogue and update this.
default_model = "google/gemini-3.8-flash"
model_id_shape = "provider/model"
# Not optional: agy refuses a Gemini 3 model with no --effort ("requires --effort
# (available: low, medium, high)"). low/high are safe on pro and flash alike.
effort_kwarg = "reasoning_effort"
effort_default = "high"
key_env = "GEMINI_API_KEY"
base_url_env = "GOOGLE_GEMINI_BASE_URL"
proxy_path = "gemini"
writes_atif = true
capture = false
# agy cannot be handed externally-produced history, so multi-turn tasks must
# hard-fail rather than silently run cold. See work-logs/antigravity-harness.md.
seed_native = false
seed_atif = false
[[harness]]
id = "opencode"
label = "OpenCode"
agent_import_path = "harness_agents:BenchOpenCode"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
flaky_hangs = true
[[harness]]
id = "goose"
label = "Goose"
agent_import_path = "harness_agents:BenchGoose"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "mini-swe-agent"
label = "mini-swe-agent"
agent_import_path = "harness_agents:BenchMiniSweAgent"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "cline-cli"
label = "Cline CLI"
agent_import_path = "harness_agents:BenchCline"
legacy_bare_model_rows = false
model_id_shape = "provider:model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "crush"
label = "Crush"
agent_import_path = "harness_agents:Crush"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
flaky_hangs = true
[[harness]]
id = "amp"
label = "Amp"
agent_import_path = "harness_agents:Amp"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "AMP_API_KEY"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "cursor-cli"
label = "Cursor CLI"
agent_import_path = "harness_agents:BenchCursorCli"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "CURSOR_API_KEY"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "copilot-cli"
label = "GitHub Copilot CLI"
agent_import_path = "harness_agents:BenchCopilotCli"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "GITHUB_TOKEN"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "aider"
label = "Aider"
agent_import_path = "harness_agents:BenchAider"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
writes_atif = false
capture = false
seed_native = false
seed_atif = false
enabled = false

View File

@@ -0,0 +1,263 @@
#!/bin/bash
# Read the harness registry and derive per-harness credentials from it.
#
# Source it — the whole point is exporting into the caller's environment, which a subshell
# would lose:
#
# HARNESS_SCRIPTS_DIR=/workspace/scripts . /workspace/scripts/lib/harness-credentials.sh
# harness_setup_credentials
#
# Three callers: `harbor-run`, which needs only this; `refresh-harness-auth`, which
# re-derives and rewrites the auth files before an interactive launch; and
# `setup-harnesses.sh`, which sources it and adds installs, config writing and launchers
# on top.
#
# No -e here — this file is SOURCED, and shell options belong to the caller's shell (both
# post-creates run with -e). An unguarded failure below therefore aborts container
# creation, which is why every failure site is individually guarded rather than relying on
# this line.
set -uo pipefail
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# The registry is read with tomllib (stdlib from 3.11), and `python3` is not always new
# enough — macOS ships 3.9, and a container may symlink an older managed interpreter. Pick
# the first one that can actually import it rather than assuming.
_raccoon_python() {
local p
for p in "${RACCOON_PYTHON:-}" python3 python3.13 python3.12 python3.11; do
[ -n "$p" ] || continue
command -v "$p" >/dev/null 2>&1 || continue
if "$p" -c "import tomllib" >/dev/null 2>&1; then
printf '%s' "$p"
return 0
fi
done
return 1
}
_harness_query() {
local py
py=$(_raccoon_python) || return 1
"$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" "$@"
}
# Drop every whitespace character from a value read out of .env. A Windows-saved .env leaves a
# \r on each value, which reaches the proxy as a 401; no key or base URL legitimately contains
# whitespace anywhere, so deleting rather than trimming needs no cases.
_harness_trim() {
local out
# Fall back to the raw value: a trim that cannot run must never turn a working key into an
# empty one, which is what an unavailable `tr` would otherwise do to every caller.
out="$(printf '%s' "$1" | tr -d '[:space:]' 2>/dev/null)" || out="$1"
printf '%s' "${out:-$1}"
}
# The proxy root: the worker's ANTHROPIC_BASE_URL minus its provider path.
_harness_proxy_root() {
local base_url
base_url="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
[ -n "$base_url" ] || return 1
base_url="${base_url%"${base_url##*[!/]}"}"
# ".../llm_proxy/projects/<id>/anthropic" -> ".../llm_proxy/projects/<id>", so each
# harness's proxy_path composes onto the project route. Requires a path to strip: a base
# URL that is a bare host with no path — a provider's own API root rather than the
# proxy — would yield "https:/", handed to codex as a base URL and failing obscurely.
case "${base_url#*://}" in
*/*) printf '%s' "${base_url%/*}" ;;
*) return 2 ;;
esac
}
harness_setup_credentials() {
# `|| rc=$?` and not a bare assignment: this is sourced into a `set -e` shell (see the
# note at the top), and a bare failing assignment would exit the caller's post-create
# outright — silently, since the failure paths below are what do the explaining.
local root rc=0
root="$(_harness_proxy_root)" || rc=$?
if [ "$rc" -ne 0 ]; then
if [ "$rc" -eq 2 ]; then
echo "harness-setup: ANTHROPIC_BASE_URL (${ANTHROPIC_BASE_URL:-}) has no provider" >&2
echo "harness-setup: path, so it is not the proxy URL other harnesses derive their" >&2
echo "harness-setup: credentials from. claude will work; codex will not be" >&2
echo "harness-setup: authenticated. Use the base URL you were given." >&2
else
echo "harness-setup: ANTHROPIC_BASE_URL unset — skipping credential derivation" >&2
fi
return 0
fi
ANTHROPIC_BASE_URL="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
export ANTHROPIC_BASE_URL
local key
key="$(_harness_trim "${ANTHROPIC_API_KEY:-}")"
if [ -z "$key" ]; then
echo "harness-setup: ANTHROPIC_API_KEY unset — skipping credential derivation" >&2
return 0
fi
# harbor-run sources .env itself and passes ANTHROPIC_* through to the trial sandbox, so
# cleaning only the derived per-harness copies would leave a claude trial carrying the CR.
export ANTHROPIC_API_KEY="$key"
local id key_env base_url_env proxy_path
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
[ -n "$key_env" ] || continue
# ${!name} is an indirect expansion. Only set when empty: an explicit key wins.
if [ -z "${!key_env:-}" ]; then
export "$key_env=$key"
fi
if [ -n "$base_url_env" ] && [ -n "$proxy_path" ] && [ -z "${!base_url_env:-}" ]; then
export "$base_url_env=$root/$proxy_path"
fi
echo "harness-setup: $id credentials ready ($key_env, ${base_url_env:-no base url})" >&2
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
}
# Write the auth file for harnesses that read credentials from disk rather than $ENV.
harness_write_auth() {
local id auth_path key_env target key py
py=$(_raccoon_python) || {
echo "harness-setup: no python3.11+ with tomllib — skipping auth files" >&2
return 0
}
while IFS=$'\t' read -r id auth_path key_env; do
[ -n "$auth_path" ] && [ -n "$key_env" ] || continue
# Last mile: an explicit OPENAI_API_KEY bypasses the derivation above, so trim here
# too — this is the value that reaches the file the harness authenticates with.
key="$(_harness_trim "${!key_env:-}")"
if [ -z "$key" ]; then
echo "harness-setup: $key_env unset — skipping $id auth file" >&2
continue
fi
target=$(eval "printf '%s' \"$auth_path\"") || {
echo "harness-setup: WARNING $id auth_path could not be expanded — skipping" >&2
continue
}
mkdir -p "$(dirname "$target")" || {
echo "harness-setup: WARNING $id auth dir not creatable — skipping $target" >&2
continue
}
# json.dumps, not printf: a key containing a quote or backslash would otherwise
# produce a file the CLI cannot parse, and the failure would surface as an auth
# error rather than a malformed file.
# 0600 tmp + rename, never a redirect onto the target: a redirect truncates the live
# file first, so a write dying mid-flight leaves codex an EMPTY auth.json.
if ! RACCOON_AUTH_K="$key_env" RACCOON_AUTH_V="$key" RACCOON_AUTH_TARGET="$target" \
"$py" -c 'import json, os
target = os.environ["RACCOON_AUTH_TARGET"]
tmp = target + ".raccoon-tmp." + str(os.getpid())
try:
with os.fdopen(os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600), "w") as fh:
json.dump({os.environ["RACCOON_AUTH_K"]: os.environ["RACCOON_AUTH_V"]}, fh)
fh.write("\n")
os.replace(tmp, target)
except OSError:
try:
os.unlink(tmp)
except OSError:
pass
raise SystemExit(1)
'; then
echo "harness-setup: WARNING $id auth file NOT written — $target unwritable." >&2
echo "harness-setup: the key already on disk (if any) is left untouched." >&2
continue
fi
echo "harness-setup: $id auth -> $target" >&2
done < <(_harness_query --auth-files 2>/dev/null || true)
}
# Re-set just the root keys of a harness's config file (codex's `openai_base_url`),
# leaving every other line — the explore surface's [hooks] table included — untouched.
harness_refresh_config_keys() {
local id config_path blob target py
py=$(_raccoon_python) || return 0
# The surface only decides what a CREATE writes. An update takes the root keys off the
# front of the same blob, so a surface's tables survive byte-for-byte either way.
while IFS=$'\t' read -r id config_path blob; do
[ -n "$config_path" ] && [ -n "$blob" ] || continue
target=$(eval "printf '%s' \"$config_path\"") || continue
mkdir -p "$(dirname "$target")" || continue
if printf '%s' "$blob" | base64 -d |
RACCOON_CONFIG_TARGET="$target" "$py" -c '
import os, re, sys, tomllib
HEADER = "# Generated from harness-registry.toml — edits here are overwritten."
target = os.environ["RACCOON_CONFIG_TARGET"]
text = sys.stdin.read()
# Empty counts as unresolved: writing an empty base URL would break a container whose
# config is currently right, which is the one thing this must never do.
if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1))]:
raise SystemExit(1)
text = os.path.expandvars(text)
wanted = []
for line in text.splitlines():
if line.lstrip().startswith("["):
break
m = re.match(r"\s*([A-Za-z0-9_-]+)\s*=", line)
if m:
wanted.append((m.group(1), line.rstrip()))
if not wanted:
raise SystemExit(0)
mode = None
if os.path.exists(target):
try:
with open(target, encoding="utf-8") as fh:
lines = fh.read().splitlines()
mode = os.stat(target).st_mode & 0o777
except OSError:
raise SystemExit(1)
# Everything from the first table header on belongs to a table. A key appended after
# one is reparented into it, so both the search and the insert stay above the line.
root_end = next((i for i, l in enumerate(lines) if l.lstrip().startswith("[")), len(lines))
changed = False
for key, line in wanted:
# The quoted spelling is the same key: replacing it beats adding a duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(root_end) if pat.match(lines[i])), None)
if at is None:
if root_end < len(lines) and lines[root_end].strip():
lines.insert(root_end, "")
lines.insert(root_end, line)
root_end += 1
changed = True
elif lines[at] != line:
lines[at] = line
changed = True
if not changed:
raise SystemExit(0)
out = "\n".join(lines).rstrip("\n") + "\n"
else:
# No file means container-create could not write one, so write what it would have:
# on the explore surface that is the capture hooks too, not just the root keys.
out = HEADER + "\n" + text
try:
doc = tomllib.loads(out)
except tomllib.TOMLDecodeError:
raise SystemExit(1)
# Parsing is not enough: a line edit can land inside a multi-line value, which still
# parses while leaving the key unset. Require every key to have reached the root.
if doc != {**doc, **tomllib.loads("\n".join(line for _, line in wanted))}:
raise SystemExit(1)
# Pid-suffixed: two launches at once must not write the same scratch path.
tmp = target + ".raccoon-tmp." + str(os.getpid())
try:
with open(tmp, "w", encoding="utf-8") as fh:
fh.write(out)
if mode is not None:
os.chmod(tmp, mode)
os.replace(tmp, target)
except OSError:
try:
os.unlink(tmp)
except OSError:
pass
raise SystemExit(1)
'; then
echo "harness-setup: $id config keys refreshed -> $target" >&2
fi
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
}

View File

@@ -0,0 +1,418 @@
"""harness_registry.py — Python loader for ``scripts/harness-registry.toml``.
The ONE loader for the registry: TS callers shell into ``resolve_harness.py`` rather than
parse the TOML themselves, which is why the toolkit ships no TOML parser for TS (its
package.json has no zod/smol-toml).
This module supersedes ``benchmark_models_lib``'s ``HARNESS_BY_IMPORT_PATH`` and
``LEGACY_BARE_MODEL_AGENTS``; those should read from here rather than keep private
copies.
Harbor-free and dependency-free (stdlib ``tomllib``) so it can be imported from a
sandbox agent, a plain unit test, or the devcontainer python alike.
"""
from __future__ import annotations
import tomllib
from dataclasses import dataclass, field
from pathlib import Path
REGISTRY_PATH = Path(__file__).resolve().parent.parent / "harness-registry.toml"
MODEL_ID_SHAPES = frozenset({"bare", "provider/model", "provider:model"})
@dataclass(frozen=True)
class Harness:
"""One harness, as declared in harness-registry.toml."""
id: str
label: str
agent_import_path: str
model_id_shape: str
writes_atif: bool
capture: bool
seed_native: bool
seed_atif: bool
agent_import_path_single_turn: str | None = None
# Browser-opt-in variants (`[metadata] browser = true`). A harness that has no variant
# keeps its normal class: codex, for instance, gains the browser and its disclosure but
# has no `Read` equivalent to switch toolsets for.
agent_import_path_browser: str | None = None
agent_import_path_single_turn_browser: str | None = None
import_path_aliases: tuple[str, ...] = ()
legacy_bare_model_rows: bool = False
default_model: str | None = None
effort_kwarg: str = ""
effort_default: str | None = None
# Agent kwarg that opts a trial into the harness's fast/priority serving mode
# (claude-code: fast mode). Empty means the harness has none and --fast refuses.
fast_kwarg: str = ""
key_env: str | None = None
base_url_env: str | None = None
proxy_path: str | None = None
flaky_hangs: bool = False
enabled: bool = True
# Worker-container fields; see the registry header.
authoring: bool = False
cli: str | None = None
install: str | None = None
skills_dir: str | None = None
auth_path: str | None = None
auth_key_env: str | None = None
explore_launch: str | None = None
config_path: str | None = None
# Config the harness needs wherever it runs, trial sandbox included.
agent_config: str | None = None
# Config for both worker containers (explore and authoring).
container_config: str | None = None
# Config for the EXPLORE container only — the capture hooks, whose commands ship in
# explore/plugins/. Writing them in authoring would register hooks against files that
# are not there, firing on every prompt.
explore_config: str | None = None
# Fields added for a later phase, kept verbatim so this loader doesn't have to
# be edited in lockstep with the schema.
extra: dict = field(default_factory=dict, compare=False)
def agent_import_path_for(self, *, multi_turn: bool, browser: bool = False) -> str:
"""Agent class to launch. Multi-turn tasks need the resuming class; a
single-turn task given it would try to resume a session that isn't there.
``browser`` selects the opt-in variant, which for claude also carries the ``Read``
built-in — a different toolset is a different agent, so it is a different class with
its own name rather than a flag on the canonical one. Harnesses without a variant fall
through to their normal class."""
if browser:
variant = (
self.agent_import_path_browser
if multi_turn
else (self.agent_import_path_single_turn_browser or self.agent_import_path_browser)
)
if variant:
return variant
if multi_turn:
return self.agent_import_path
return self.agent_import_path_single_turn or self.agent_import_path
def row_label(self, model: str) -> str:
"""Row identity for one trial: bare model for legacy harnesses (so
published manifests keep their labels), else ``<harness>:<model>``."""
return model if self.legacy_bare_model_rows else f"{self.id}:{model}"
def agent_config_overrides(self) -> dict[str, str]:
"""``agent_config`` as flat ``dotted.key -> value`` pairs in CLI-override form.
Values are rendered bare — ``disabled``, not ``"disabled"``. Every consumer
interpolates these into a shell command, which would strip the quotes anyway;
emitting them would only make the result depend on how many shell layers the
string crosses. Bare is what the CLIs document (``-c model="o3"`` reaches the
binary as ``model=o3``).
These settings ride the command line as ``-c dotted.key=value`` everywhere the
harness runs, never a config file. A trial sandbox rules the file out: the
harness's own runner appends root keys to it, and TOML has no way back to the
root scope once a table has opened, so a table we appended would swallow them.
Overrides compose in any order and beat the file, so the same rendering serves
the explore launcher too — one declaration, one mechanism.
"""
if not self.agent_config:
return {}
try:
parsed = tomllib.loads(self.agent_config)
except tomllib.TOMLDecodeError as exc:
raise HarnessRegistryError(
f"{self.id}: agent_config is not valid TOML ({exc})"
) from exc
flat: dict[str, str] = {}
def walk(node: dict, prefix: str) -> None:
for key, value in node.items():
path = f"{prefix}{key}"
if isinstance(value, dict):
walk(value, f"{path}.")
elif isinstance(value, bool):
flat[path] = "true" if value else "false"
elif isinstance(value, (int, float)):
flat[path] = str(value)
elif isinstance(value, str):
if value != value.strip() or any(c in value for c in " \"'\\"):
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has a value needing "
"shell quoting, which the -c override form cannot carry"
)
flat[path] = value
else:
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has type "
f"{type(value).__name__}, which has no -c override form"
)
walk(parsed, "")
return flat
def container_config_text(self, *, surface: str) -> str | None:
"""Config file body for a worker container. `surface` is "explore" or
"authoring"; explore additionally gets `explore_config`. Root keys come from
`container_config` first, so appending a table section stays valid TOML."""
parts = [self.container_config]
if surface == "explore":
parts.append(self.explore_config)
kept = [part.strip("\n") for part in parts if part and part.strip()]
return "\n\n".join(kept) + "\n" if kept else None
def agent_config_flags(self) -> str:
"""``agent_config`` as a ``-c key=value`` command-line string."""
return " ".join(
f"-c {key}={value}"
for key, value in sorted(self.agent_config_overrides().items())
)
def explore_launch_command(self) -> str | None:
"""``explore_launch`` with the registry's own values substituted in.
The worker's Explore session and the trial must run the same agent, so the
model, effort and reductions are declared once here and rendered into both.
A literal in the launch string would be a second declaration, and the two
would drift the first time one of them was updated alone.
Only these three placeholders are substituted; ``$@`` and
``$RACCOON_TOOLSET_NOTE`` stay for the launcher's own shell to expand.
"""
if not self.explore_launch:
return None
return (
self.explore_launch.replace("$RACCOON_AGENT_FLAGS", self.agent_config_flags())
.replace("$RACCOON_MODEL", self.default_model or "")
.replace("$RACCOON_EFFORT", self.effort_default or "")
)
def known_import_paths(self) -> tuple[str, ...]:
paths = [self.agent_import_path, *self.import_path_aliases]
if self.agent_import_path_single_turn:
paths.append(self.agent_import_path_single_turn)
return tuple(paths)
_KNOWN_FIELDS = frozenset(
{
"id",
"label",
"agent_import_path",
"agent_import_path_single_turn",
"agent_import_path_browser",
"agent_import_path_single_turn_browser",
"import_path_aliases",
"legacy_bare_model_rows",
"default_model",
"model_id_shape",
"effort_kwarg",
"effort_default",
"fast_kwarg",
"key_env",
"base_url_env",
"proxy_path",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
"flaky_hangs",
"enabled",
"authoring",
"cli",
"install",
"skills_dir",
"auth_path",
"auth_key_env",
"explore_launch",
"config_path",
"agent_config",
"container_config",
"explore_config",
}
)
_REQUIRED_FIELDS = (
"id",
"label",
"agent_import_path",
"model_id_shape",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
)
class HarnessRegistryError(ValueError):
"""Malformed registry. Raised rather than tolerated: a broken registry is a
broken deployment, and silently defaulting would pick the wrong agent."""
def _references_agent_flags(launch: str) -> bool:
return "$RACCOON_AGENT_FLAGS" in launch or "${RACCOON_AGENT_FLAGS}" in launch
@dataclass(frozen=True)
class HarnessRegistry:
version: int
harnesses: tuple[Harness, ...]
def all(self) -> tuple[Harness, ...]:
return self.harnesses
def enabled(self) -> tuple[Harness, ...]:
return tuple(h for h in self.harnesses if h.enabled)
def authoring(self) -> tuple[Harness, ...]:
"""Harnesses a worker can author with — what the worker containers install.
Narrower than enabled(): a harness can be runnable in a trial without having
an authoring story (no CLI to converse with, or no capture)."""
return tuple(h for h in self.harnesses if h.enabled and h.authoring)
def find(self, harness_id: str) -> Harness | None:
return next((h for h in self.harnesses if h.id == harness_id), None)
def require(self, harness_id: str) -> Harness:
harness = self.find(harness_id)
if harness is not None:
return harness
available = ", ".join(sorted(h.id for h in self.enabled()))
raise HarnessRegistryError(
f'Unknown harness "{harness_id}". Available: {available}'
)
def by_import_path(self, agent: str) -> Harness | None:
"""Resolve an agent identity — a ``name()`` or import path from
``result.json`` ``config.agent``, or a manifest row — to its harness."""
needle = (agent or "").strip()
if not needle:
return None
for harness in self.harnesses:
if needle == harness.id or needle in harness.known_import_paths():
return harness
return None
def _build(entry: dict, index: int) -> Harness:
for name in _REQUIRED_FIELDS:
if name not in entry:
raise HarnessRegistryError(
f"harness[{index}]: missing required field '{name}'"
)
shape = entry["model_id_shape"]
if shape not in MODEL_ID_SHAPES:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): model_id_shape {shape!r} not one of "
f"{sorted(MODEL_ID_SHAPES)}"
)
# These three reach `eval` in setup-harnesses.sh, which is how they support the
# `${CODEX_HOME:-$HOME/.codex}` default-value syntax that python's expandvars cannot
# express. Under eval a backtick or $( would EXECUTE, so refuse them here — the registry
# is ours, but "ours" is not an argument that survives a careless future edit.
for shell_field in ("config_path", "auth_path", "skills_dir"):
value = entry.get(shell_field)
if not isinstance(value, str):
continue
# A backtick or $( executes outright. A double quote closes the string these are
# interpolated into, and a semicolon then starts a new command inside it — same
# outcome, one step removed.
bad = [t for t in ("`", "$(", '"', ";") if t in value]
if bad:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): {shell_field} contains "
f"{', '.join(repr(t) for t in bad)} ({value!r}). This value is shell-"
f"expanded, so that would execute; use plain $VAR or ${{VAR:-default}} only."
)
launch = entry.get("explore_launch")
if entry.get("agent_config") and launch and not _references_agent_flags(launch):
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): declares agent_config but its "
"explore_launch does not pass $RACCOON_AGENT_FLAGS. The worker's session "
"would then run with a different toolset than the trial it is authoring "
"for, which is the drift agent_config exists to prevent."
)
return Harness(
id=entry["id"],
label=entry["label"],
agent_import_path=entry["agent_import_path"],
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
agent_import_path_browser=entry.get("agent_import_path_browser"),
agent_import_path_single_turn_browser=entry.get("agent_import_path_single_turn_browser"),
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
default_model=entry.get("default_model"),
model_id_shape=shape,
effort_kwarg=entry.get("effort_kwarg", ""),
effort_default=entry.get("effort_default"),
fast_kwarg=entry.get("fast_kwarg", ""),
key_env=entry.get("key_env"),
base_url_env=entry.get("base_url_env"),
proxy_path=entry.get("proxy_path"),
writes_atif=bool(entry["writes_atif"]),
capture=bool(entry["capture"]),
seed_native=bool(entry["seed_native"]),
seed_atif=bool(entry["seed_atif"]),
flaky_hangs=bool(entry.get("flaky_hangs", False)),
enabled=bool(entry.get("enabled", True)),
authoring=bool(entry.get("authoring", False)),
cli=entry.get("cli"),
install=entry.get("install"),
skills_dir=entry.get("skills_dir"),
auth_path=entry.get("auth_path"),
auth_key_env=entry.get("auth_key_env"),
explore_launch=entry.get("explore_launch"),
config_path=entry.get("config_path"),
agent_config=entry.get("agent_config"),
container_config=entry.get("container_config"),
explore_config=entry.get("explore_config"),
extra={k: v for k, v in entry.items() if k not in _KNOWN_FIELDS},
)
_cache: dict[Path, HarnessRegistry] = {}
def load_harness_registry(path: Path | str = REGISTRY_PATH) -> HarnessRegistry:
"""Parse and validate the registry. Raises HarnessRegistryError on a malformed
file, a duplicate id, or an import path claimed by two harnesses (which would
make ``by_import_path`` depend on declaration order)."""
resolved = Path(path).resolve()
if resolved in _cache:
return _cache[resolved]
with open(resolved, "rb") as handle:
doc = tomllib.load(handle)
if "version" not in doc:
raise HarnessRegistryError("harness-registry: missing 'version'")
entries = doc.get("harness") or []
if not entries:
raise HarnessRegistryError("harness-registry: no [[harness]] entries")
harnesses = tuple(_build(entry, i) for i, entry in enumerate(entries))
seen_ids: set[str] = set()
for harness in harnesses:
if harness.id in seen_ids:
raise HarnessRegistryError(
f"harness-registry: duplicate harness id: {harness.id}"
)
seen_ids.add(harness.id)
owners: dict[str, str] = {}
for harness in harnesses:
for import_path in harness.known_import_paths():
owner = owners.get(import_path)
if owner is not None and owner != harness.id:
raise HarnessRegistryError(
f'harness-registry: import path "{import_path}" claimed by both '
f'"{owner}" and "{harness.id}"'
)
owners[import_path] = harness.id
registry = HarnessRegistry(version=int(doc["version"]), harnesses=harnesses)
_cache[resolved] = registry
return registry

View File

@@ -0,0 +1,37 @@
#!/bin/bash
# Rewrite the auth FILES harnesses read their key from — and the base URL beside them —
# off the live .env, then exec "$@".
#
# codex reads its key from ${CODEX_HOME:-$HOME/.codex}/auth.json, which container-create
# wrote once from the .env of that moment — so a key rotated afterwards never reached it
# and needed a rebuild. claude needs none of this: it has an apiKeyHelper that re-reads
# .env per request. Interactive launches route through here so each one re-derives first.
#
# The base URL never rotates, so the case that matters is the one where container-create
# could not derive it at all (no .env yet) and wrote no config: the key then refreshes
# fine while codex still has no proxy URL and talks to the provider directly.
#
# Trials are unaffected either way: harbor-run re-derives OPENAI_API_KEY per invocation
# and harbor's codex agent authenticates the sandbox from that env var, not from this file.
set -uo pipefail
_scripts_dir="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# Subshell, and every failure swallowed: a refresh that cannot run must never stop the
# agent from starting. The auth file already on disk is the PREVIOUS key, not nothing, so
# failing open leaves the worker exactly where they were before this wrapper existed.
(
set -a
# shellcheck disable=SC1090
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true
set +a
# shellcheck disable=SC1091
HARNESS_SCRIPTS_DIR="$_scripts_dir" . "$_scripts_dir/lib/harness-credentials.sh" || exit 0
harness_setup_credentials
harness_write_auth
harness_refresh_config_keys
) >/dev/null 2>&1 || true
# No args is a valid call: refresh only, for a lifecycle hook.
[ "$#" -gt 0 ] || exit 0
exec "$@"

View File

@@ -0,0 +1,459 @@
#!/usr/bin/env python3
"""resolve_harness.py — turn a harness id + task dir into the flags a trial needs.
``scripts/harbor-run`` is bash and cannot parse the TOML registry, so it shells in
here and evals the result::
RESOLVED="$(python3 scripts/resolve_harness.py --task-dir "$TASK_DIR")" || exit 1
eval "$RESOLVED"
Python rather than TS on purpose: this ships in the worker toolkit, whose
package.json has no ``zod``/``smol-toml``, and ``tomllib`` is stdlib — so the
toolkit gains a harness-aware harbor-run with zero new dependencies. There is no TS
loader: TS callers (submit-task) shell in here, so both the schema and the selection
policy exist exactly once and there is nothing to drift.
Output is POSIX ``KEY='value'`` assignments (single-quoted, embedded quotes
escaped) on stdout; everything human-facing goes to stderr, so the eval only ever
sees assignments. A non-zero exit means "do not launch" — the point is to fail in a
second rather than burn agent minutes on a trial that cannot produce a usable grade.
Refuses to resolve when:
- the harness id is unknown or disabled
- the harness writes no ATIF trajectory (the grader would have no transcript)
- the task ships a session to resume but the harness cannot resume one. This is
the important one: it is the only failure here that would otherwise look like
SUCCESS, with the agent answering a prompt whose conversation it never saw.
- the harness's credential env var is unset
``--check-model`` additionally asks the proxy whether the model is granted. Opt-in
on purpose: it is a network call, and one in every run's critical path trades a fast
local failure for a new way to hang. The credential check, which is free, always runs.
"""
from __future__ import annotations
import argparse
import json
import os
import re
import shlex
import sys
import tomllib
import urllib.error
import urllib.request
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent / "lib"))
from harness_registry import ( # noqa: E402
Harness,
HarnessRegistryError,
load_harness_registry,
)
# Harness used when nothing selects one. Keeps every existing caller on today's
# behaviour, so adding harness selection changes no current run.
DEFAULT_HARNESS = "claude-code"
MODELS_TIMEOUT_SEC = 20
def warn(message: str) -> None:
print(f"resolve-harness: {message}", file=sys.stderr)
def fail(message: str) -> "None":
print(f"resolve-harness: ERROR: {message}", file=sys.stderr)
raise SystemExit(1)
def is_multi_turn(task_dir: str | None) -> bool:
"""A task is multi-turn when it ships a NON-EMPTY session to resume. Empty is
the documented one-shot-snapshot fallback and must run cold, so size is the
test, not existence."""
if not task_dir:
return False
session = Path(task_dir) / "environment" / "session.jsonl"
return session.is_file() and session.stat().st_size > 0
def wants_browser(task_dir: str | None) -> bool:
"""True when task.toml opts into a browser (`[metadata] browser = true`).
Read straight from the file rather than via tomllib: this must agree with
build-workspace.sh, which decides whether the IMAGE gets Playwright using the same
text match. If the two ever disagree the agent is told about a browser the image
lacks, which is the one failure the disclosure is designed to make impossible.
Accepts the quoted form for the same reason build-workspace.sh does."""
if not task_dir:
return False
toml_path = Path(task_dir) / "task.toml"
if not toml_path.is_file():
return False
try:
text = toml_path.read_text(encoding="utf-8")
except OSError:
return False
return re.search(r'^[ \t]*browser[ \t]*=[ \t]*"?true"?[ \t]*$', text, re.M) is not None
def harness_from_task_toml(task_dir: str | None) -> str | None:
"""The task's own `[agent] harness` — the authoritative record of which harness
this task was authored against.
This is where the worker's choice lands: the snapshot flow stamps it from the CLI
that produced the snapshot, and a manual author writes it themselves. Either way
it is set at task-creation time, BEFORE any trial, so nothing here depends on a
trial's output.
Parsed with tomllib rather than a grep: a regex would happily match a commented
line or the wrong table, and picking the wrong harness is a silent
wrong-agent-runs bug.
Returns None when the field is simply absent — the normal case for every task
finalized before harness selection existed — so the caller falls through to the
toolkit default.
But an UNPARSEABLE task.toml refuses outright rather than falling back. Those are
different situations and treating them alike is how the wrong harness runs
quietly: the most likely way to break this file is adding a second `[agent]`
table instead of a `harness` line inside the existing one (tasks already carry
`[agent] timeout_sec`), and TOML rejects a duplicate table. Falling back there
would run claude against a task its author wrote for codex and grade it as if
nothing were wrong.
"""
if not task_dir:
return None
path = Path(task_dir) / "task.toml"
if not path.is_file():
return None
try:
with open(path, "rb") as handle:
doc = tomllib.load(handle)
except (OSError, tomllib.TOMLDecodeError) as exc:
fail(
f"{path} could not be parsed ({exc}). Refusing to guess a harness — fix "
f'the file. If you were adding a harness, put `harness = "..."` inside '
f"the EXISTING [agent] table rather than starting a second one."
)
harness = (doc.get("agent") or {}).get("harness")
return harness if isinstance(harness, str) and harness else None
def normalize_model(harness: Harness, model: str) -> str:
"""Model id on the wire, per the harness's declared shape."""
if harness.model_id_shape == "provider:model":
return model.replace("/", ":")
return model
def granted_models(harness: Harness) -> list[str] | None:
"""Model ids the key is granted, or None when the check couldn't run."""
base_url = os.environ.get(harness.base_url_env or "")
key = os.environ.get(harness.key_env or "")
if not base_url or not key:
warn("--check-model skipped: base URL or key env is unset")
return None
request = urllib.request.Request(
f"{base_url.rstrip('/')}/models", headers={"Authorization": f"Bearer {key}"}
)
try:
with urllib.request.urlopen(request, timeout=MODELS_TIMEOUT_SEC) as response:
body = json.loads(response.read().decode("utf-8"))
except (urllib.error.URLError, TimeoutError, ValueError, OSError) as exc:
warn(f"--check-model skipped: /models unreachable ({exc})")
return None
return [m["id"] for m in body.get("data", []) if isinstance(m.get("id"), str)]
def assert_model_granted(harness: Harness, model: str) -> None:
granted = granted_models(harness)
if granted is None:
return
# The proxy LISTS ids provider-prefixed ("openai/gpt-5.6-sol") but 400s on that
# form — requests take the bare id. Accept either spelling.
bare = {g.split("/")[-1] for g in granted}
if model not in granted and model not in bare:
shown = ", ".join(granted[:12]) + (", …" if len(granted) > 12 else "")
fail(
f'Model "{model}" is not granted for this key. Granted ({len(granted)}): {shown}'
)
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
parser.add_argument(
"--harness",
help=f"harness id (default: the task's [agent] harness, else {DEFAULT_HARNESS})",
)
parser.add_argument(
"--task-dir",
help="task directory; decides multi-turn from environment/session.jsonl",
)
parser.add_argument("--model", help="override the harness's default model")
parser.add_argument(
"--fast",
action="store_true",
help="run the trial agent in the harness's fast serving mode (higher token "
"rate, faster output). Refuses on a harness that has none.",
)
parser.add_argument(
"--check-model",
action="store_true",
help="also ask the proxy whether the model is granted (network call)",
)
parser.add_argument(
"--authoring-installs",
action="store_true",
help="print '<id>\\t<cli>\\t<install>' for each harness a worker can author "
"with, and exit. Consumed by scripts/setup-harnesses.sh so the worker "
"containers install from the registry rather than from hardcoded lists that "
"drift.",
)
parser.add_argument(
"--container-configs",
action="store_true",
help="print '<id>\\t<config_path>\\t<base64 container_config>' for each "
"authoring harness that declares one, and exit. Base64 because the config is "
"multi-line TOML and these query modes are line-oriented.",
)
parser.add_argument(
"--surface",
choices=("authoring", "explore"),
default="authoring",
help="which worker container --container-configs is for; explore additionally "
"gets the capture hooks, whose commands only ship there.",
)
parser.add_argument(
"--defaults",
action="store_true",
help="print '<id>\\t<default_model>\\t<effort_default>' for every harness, and "
"exit. For recording what a task was authored against; nothing reads it back.",
)
parser.add_argument(
"--explore-launchers",
action="store_true",
help="print '<id>\\t<cli>\\t<launch command>' for each authoring harness, and "
"exit. The launch command has the registry's model, effort and agent_config "
"already substituted, so Explore and a trial cannot disagree about them. "
"Consumed by setup-harnesses.sh to write one launcher per harness.",
)
parser.add_argument(
"--skills-dirs",
action="store_true",
help="print '<id>\t<skills_dir>' for each authoring harness that discovers "
"skills from a directory, and exit. Lets setup-harnesses.sh install the "
"snapshot skill for harnesses that have no plugin system.",
)
parser.add_argument(
"--auth-files",
action="store_true",
help="print '<id>\t<auth_path>\t<auth_key_env>' for each authoring harness that "
"authenticates from a file rather than the environment, and exit.",
)
parser.add_argument(
"--authoring-credentials",
action="store_true",
help="print '<id>\\t<key_env>\\t<base_url_env>\\t<proxy_path>' for each "
"authoring harness, and exit. Lets the containers point every harness at the "
"same proxy key on its own provider path.",
)
parser.add_argument(
"--declared-harness",
action="store_true",
help="print ONLY the harness --task-dir's task.toml declares (empty if it "
"declares none) and exit. Unlike the default mode this applies no fallback, so "
"a caller can tell 'declared' from 'defaulted'. Exists so consumers without a "
"TOML parser never hand-roll one: a regex would match a commented line or the "
"wrong table, and the duplicate-[agent] shape is exactly the likely mistake.",
)
parser.add_argument(
"--resolve-identity",
action="append",
default=None,
metavar="AGENT",
help="resolve agent identities (a result.json config.agent import_path or name) "
"to harness ids and exit; repeatable. Prints one TAB-separated "
"'<identity>\\t<harness-id>' line each, with an empty id when nothing claims it. "
"Lets callers that cannot import the registry (the worker toolkit has no "
"zod/smol-toml) still resolve through the one source of truth.",
)
parser.add_argument(
"--list",
action="store_true",
help="print the selectable harnesses and exit (what task.toml's [agent] harness accepts)",
)
parser.add_argument("--registry", default=None, help="registry path (tests)")
args = parser.parse_args(argv)
try:
registry = (
load_harness_registry(args.registry)
if args.registry
else load_harness_registry()
)
except HarnessRegistryError as exc:
fail(str(exc))
# --- read-only query modes: answer and exit, never emit assignments -------
if args.authoring_installs:
for harness in registry.authoring():
print(f"{harness.id}\t{harness.cli or ''}\t{harness.install or ''}")
return 0
if args.container_configs:
import base64
for harness in registry.authoring():
config = harness.container_config_text(surface=args.surface)
if not (harness.config_path and config):
continue
blob = base64.b64encode(config.encode()).decode()
print(f"{harness.id}\t{harness.config_path}\t{blob}")
return 0
if args.defaults:
for harness in registry.all():
print(
f"{harness.id}\t{harness.default_model or ''}\t"
f"{harness.effort_default or ''}"
)
return 0
if args.explore_launchers:
for harness in registry.authoring():
print(
f"{harness.id}\t{harness.cli or ''}\t"
f"{harness.explore_launch_command() or ''}"
)
return 0
if args.skills_dirs:
for harness in registry.authoring():
if harness.skills_dir:
print(f"{harness.id}\t{harness.skills_dir}")
return 0
if args.auth_files:
for harness in registry.authoring():
if harness.auth_path and harness.auth_key_env:
print(f"{harness.id}\t{harness.auth_path}\t{harness.auth_key_env}")
return 0
if args.authoring_credentials:
for harness in registry.authoring():
print(
f"{harness.id}\t{harness.key_env or ''}\t"
f"{harness.base_url_env or ''}\t{harness.proxy_path or ''}"
)
return 0
if args.declared_harness:
print(harness_from_task_toml(args.task_dir) or "")
return 0
if args.resolve_identity:
for identity in args.resolve_identity:
harness = registry.by_import_path(identity)
print(f"{identity}\t{harness.id if harness else ''}")
return 0
if args.list:
# Printed on stdout because it is the requested output here, not the
# eval-able assignments — this mode is for a human, and never shelled into.
for harness in registry.enabled():
turns = (
"multi-turn + single-turn"
if harness.seed_native
else "single-turn only"
)
model = harness.default_model or "(pass --model)"
print(f"{harness.id:<16} {harness.label:<20} {turns:<24} {model}")
return 0
# --- selection ------------------------------------------------------------
# Precedence: an explicit --harness (a benchmark, or a deliberate override) beats the
# task's own record, which is what its author chose. Everything else — every task
# finalized before harness selection existed — is the default.
requested = args.harness or harness_from_task_toml(args.task_dir) or DEFAULT_HARNESS
try:
harness = registry.require(requested)
except HarnessRegistryError as exc:
fail(str(exc))
if not harness.enabled:
fail(
f'Harness "{harness.id}" is disabled in the registry (never verified here). '
f"Enable it in scripts/harness-registry.toml once a trial has been run with it."
)
if not harness.writes_atif:
fail(
f'Harness "{harness.id}" writes no ATIF trajectory, so the grader would have '
f"no transcript and its rewards would be meaningless."
)
multi_turn = is_multi_turn(args.task_dir)
if multi_turn and not harness.seed_native:
fail(
f'Task ships a session to resume, but harness "{harness.id}" cannot resume '
f"one. Running anyway would look like a success while the agent answered a "
f"prompt whose conversation it never saw."
)
if harness.key_env and not os.environ.get(harness.key_env):
fail(f'{harness.key_env} is unset — required by harness "{harness.id}".')
if args.fast and not harness.fast_kwarg:
fail(
f'Harness "{harness.id}" has no fast serving mode (no fast_kwarg in the '
f"registry). Drop --fast or pick a harness that declares one."
)
model = args.model or harness.default_model
if not model:
fail(
f'Harness "{harness.id}" has no default_model in the registry; pass --model '
f"explicitly."
)
if args.check_model:
assert_model_granted(harness, model)
# Every assignment here becomes a harbor flag. Nothing else: the caller is bash, and
# anything it would only echo back at the worker is said below instead.
browser = wants_browser(args.task_dir)
assignments = {
"AGENT_IMPORT_PATH": harness.agent_import_path_for(multi_turn=multi_turn, browser=browser),
"MODEL": normalize_model(harness, model),
"EFFORT_KWARG": harness.effort_kwarg,
"EFFORT_VALUE": (harness.effort_default or "") if harness.effort_kwarg else "",
"FAST_KWARG": harness.fast_kwarg if args.fast else "",
}
warn(
f"{harness.label} · model={assignments['MODEL']} · "
f"{'multi-turn' if multi_turn else 'single-turn'} · "
f"{'browser · ' if browser else ''}"
f"{'fast · ' if args.fast else ''}"
f"agent={assignments['AGENT_IMPORT_PATH']}"
)
if browser and not harness.agent_import_path_browser:
# Not a failure: the image still gets Playwright and the agent is still told about
# it. Only the Read-enabled toolset swap is claude-specific, and saying so beats
# letting someone infer from a log line that the opt-in was ignored entirely.
warn(
f'"{harness.id}" has no browser-specific agent, so it runs its usual toolset. '
f"The browser and its disclosure are unaffected."
)
if harness.flaky_hangs:
warn(
f"{harness.label} is known to hang with no client-side timeout on a small "
f"fraction of trials. A silent, output-less trial is that, not a task defect."
)
for key, value in assignments.items():
print(f"{key}={shlex.quote(value)}")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,340 @@
#!/bin/bash
# Install the harnesses a worker can author with, from scripts/harness-registry.toml.
#
# Source it, then call unpiped — it exports credentials, which a subshell would lose:
#
# . /workspace/scripts/setup-harnesses.sh
# harness_setup_all
#
# Registry reading and credential derivation live in lib/harness-credentials.sh, sourced
# below, because `harbor-run` needs those and nothing else here.
#
# No -e here — but this file is SOURCED, and shell options belong to the caller's shell:
# both post-creates run with -e, so that is what is in force. An unguarded failure below
# therefore aborts container creation, which is why every failure site is individually
# guarded (`|| true`, `if !`) rather than relying on this line.
set -uo pipefail
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
if [ ! -f "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh" ]; then
echo "harness-setup: FATAL — $_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh is" >&2
echo "harness-setup: missing, so nothing here can read the registry. Every step below" >&2
echo "harness-setup: would report a missing interpreter instead of this." >&2
return 1 2>/dev/null || exit 1
fi
# shellcheck disable=SC1091
. "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh"
# Every setup step reads the registry through _harness_query, and each call suppresses
# stderr so one bad row can't abort the container. That means a BROKEN interpreter turns
# the whole of setup into a silent no-op: no credentials, no CLIs, no config, no
# launchers, and no error anywhere. Check it once, loudly, before any of that.
harness_preflight() {
local err py found=yes
py=$(_raccoon_python) || { py=python3; found=no; }
if ! err=$("$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" --list 2>&1 >/dev/null); then
echo "harness-setup: FATAL — cannot read the harness registry, so no agent CLI" >&2
echo "harness-setup: would be installed. Nothing below will run." >&2
echo "harness-setup: interpreter: $(command -v "$py" || echo MISSING) ($("$py" -V 2>&1))" >&2
if [ "$found" = no ]; then
echo "harness-setup: no python3.11+ with tomllib found; set RACCOON_PYTHON to override" >&2
fi
echo "harness-setup: registry: $_HARNESS_REGISTRY_DIR/harness-registry.toml" >&2
printf 'harness-setup: %s\n' "$err" >&2
return 1
fi
}
# claude installs into $HOME/.local/bin, which is not on PATH during post-create.
case ":$PATH:" in
*":$HOME/.local/bin:"*) ;;
*) export PATH="$HOME/.local/bin:$PATH" ;;
esac
# --- installs ----------------------------------------------------------------
harness_install_clis() {
local id cli install
while IFS=$'\t' read -r id cli install; do
[ -n "$install" ] || continue
if command -v "$cli" >/dev/null 2>&1; then
echo "harness-setup: $cli already installed — skipping" >&2
continue
fi
echo "harness-setup: installing $id ($cli)" >&2
# Reported as unavailable below rather than fatal.
if ! bash -c "$install" >&2; then
echo "harness-setup: WARNING $id failed to install — $cli will be unavailable" >&2
fi
done < <(_harness_query --authoring-installs 2>/dev/null || true)
}
# Report which CLIs are usable. Non-zero when NONE are: one harness missing is survivable
# (a worker uses the other), but zero means the container cannot author anything at all,
# and that must stop setup rather than read as a couple of warnings.
harness_report() {
local id cli install ready=0 missing=0
while IFS=$'\t' read -r id cli install; do
[ -n "$cli" ] || continue
if command -v "$cli" >/dev/null 2>&1; then
echo " $cli — ready" >&2
ready=$((ready + 1))
else
echo " $cli — NOT AVAILABLE (install failed; see above)" >&2
missing=$((missing + 1))
fi
done < <(_harness_query --authoring-installs 2>/dev/null || true)
# A CLI on PATH with no key is worse than a missing one: it starts, then fails at the
# first request with the harness's own auth error, which says nothing about setup.
local id key_env base_url_env proxy_path
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
[ -n "$key_env" ] || continue
if [ -z "${!key_env:-}" ]; then
echo " $id — installed but NO CREDENTIALS: $key_env is unset." >&2
echo " Derived from ANTHROPIC_BASE_URL + ANTHROPIC_API_KEY; set both in .env." >&2
fi
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
if [ "$ready" -eq 0 ]; then
echo "harness-setup: FATAL — no agent CLI installed ($missing attempted)." >&2
echo "harness-setup: This container cannot author a task. Check the install" >&2
echo "harness-setup: output above: the CLIs download over the network, so a" >&2
echo "harness-setup: proxy, DNS or upstream change breaks every one at once." >&2
return 1
fi
[ "$missing" -gt 0 ] && echo "harness-setup: $missing harness(es) unavailable; $ready usable" >&2
return 0
}
# --- Explore launchers -------------------------------------------------------
# One `raccoon-explore-<cli>` per harness, aliased to its `cli`.
harness_install_launchers() {
local bin="$HOME/.local/bin"
mkdir -p "$bin"
# Read at launcher run time so the note stays a file, not a baked-in copy.
local note_src="${HARNESS_TOOLSET_NOTE:-/workspace/scripts/toolset_note.md}"
local browser_note_src="${note_src%.md}_browser.md"
local read_note_src="${note_src%.md}_read.md"
local agent_cli_dir="${AGENT_CLI_DIR:-/opt/agent-cli}"
# Which harnesses keep their key in a file rather than reading $ENV per request. Those
# launchers refresh it first: the file dates from container create, so a key rotated in
# .env since then would otherwise reach the harness only after a rebuild.
local file_auth_ids="" aid apath akey
while IFS=$'\t' read -r aid apath akey; do
[ -n "$apath" ] || continue
file_auth_ids="${file_auth_ids:+$file_auth_ids }$aid"
done < <(_harness_query --auth-files 2>/dev/null || true)
local id cli launch switchable refresh_line
while IFS=$'\t' read -r id cli launch; do
[ -n "$cli" ] && [ -n "$launch" ] || continue
# `|| true` twice over (here and inside the script): the launcher runs under
# `set -e`, and a failed refresh must not cost the worker their agent.
if [[ " $file_auth_ids " == *" $id "* ]]; then
refresh_line="\"$_HARNESS_REGISTRY_DIR/refresh-harness-auth\" || true"
else
refresh_line=""
fi
# Whether RACCOON_BROWSER_TASK can change THIS harness's toolset, read off the
# registry rather than hardcoded: a launch line that interpolates $RACCOON_TOOLS
# can, and one that doesn't cannot. codex is the second case — it ships view_image,
# so a browser task needs nothing added and the flag has nothing to switch.
# Match the whole variable name: a substring test also hits RACCOON_TOOLSET_NOTE,
# which every launch line references, and every harness would look switchable.
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
switchable=1
else
switchable=0
fi
cat > "$bin/raccoon-explore-$cli" <<LAUNCHER
#!/bin/bash
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
set -euo pipefail
if [ -f "$note_src" ]; then
RACCOON_TOOLSET_NOTE="\$(sed "s#/opt/agent-cli#$agent_cli_dir#g" "$note_src")"
else
RACCOON_TOOLSET_NOTE=""
fi
# RACCOON_BROWSER_TASK=1 explores with the toolset a \`browser = true\` task runs under.
# Named for the flag it mirrors: one word, \`browser\`, whether it's set in task.toml or
# here. Per invocation, not per container — authoring a browser task shouldn't need a
# rebuild, and neither should changing your mind. Default off, so ordinary exploring
# still mirrors an ordinary trial.
#
# The correction must be appended AFTER the base note, which says there is no Read tool.
RACCOON_TOOLS="Bash"
if [ "\${RACCOON_BROWSER_TASK:-0}" = "1" ] && [ "$switchable" = "1" ] && [ -f "$read_note_src" ]; then
RACCOON_TOOLS="Bash,Read"
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
\$(cat "$read_note_src")"
fi
export RACCOON_TOOLS
# Only mention the browser on an image that actually has one — most don't. Probed at
# launch, not baked in, so the same launcher is correct in whichever container it runs.
#
# Exported two ways because the harnesses take extra instructions differently: claude
# appends the whole toolset note to --append-system-prompt, while codex has no equivalent
# and takes -c developer_instructions=. codex must NOT get the claude-shaped toolset note
# (it has no str_replace_editor), so the browser part is exported on its own too.
RACCOON_BROWSER_NOTE=""
RACCOON_BROWSER_FLAGS=()
if command -v pw >/dev/null 2>&1 && [ -f "$browser_note_src" ]; then
RACCOON_BROWSER_NOTE="\$(cat "$browser_note_src")"
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
\${RACCOON_BROWSER_NOTE}"
RACCOON_BROWSER_FLAGS=(-c "developer_instructions=\${RACCOON_BROWSER_NOTE}")
fi
export RACCOON_TOOLSET_NOTE RACCOON_BROWSER_NOTE
export RACCOON_HARNESS="$id"
# These launchers exist only in explore, and a refresh that has to CREATE a config
# needs the surface to know the capture hooks belong in it.
export RACCOON_SURFACE=explore
# No RACCOON_SNAPSHOT_DATA here on purpose. capture-snapshot.mjs and save-session-info.mjs
# already share the same default ($HOME/.raccoon/snapshot-data), which is what codex needs
# — it has no CLAUDE_PLUGIN_* to fall back to. Exporting it ALSO overrode the dir for
# claude, whose slash command pins --plugin-data to the plugin dir, so the hook wrote one
# place and capture read another and the recorded session was silently ignored.
$refresh_line
$launch
LAUNCHER
chmod +x "$bin/raccoon-explore-$cli"
echo "harness-setup: launcher raccoon-explore-$cli" >&2
done < <(_harness_query --explore-launchers 2>/dev/null || true)
}
# Alias lines for ~/.bashrc.
harness_alias_lines() {
local id cli launch switchable
local browser_clis=""
while IFS=$'\t' read -r id cli launch; do
[ -n "$cli" ] && [ -n "$launch" ] || continue
echo "alias $cli=\"raccoon-explore-$cli\""
# Same derivation as the launcher: only a harness whose launch line takes
# $RACCOON_TOOLS has a toolset the flag can change.
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
browser_clis="${browser_clis:+$browser_clis }$cli"
fi
done < <(_harness_query --explore-launchers 2>/dev/null || true)
# The browser hint belongs at the shell prompt, not in the launcher. Claude Code takes the
# alternate screen buffer, so anything printed just before exec is hidden for the whole
# session and resurfaces only after quitting — advice arriving exactly too late. Here it
# lands in ordinary scrollback, before any TUI exists, and there is nothing to quit yet.
#
# `pw` is probed at shell start, so one ~/.bashrc is correct in a container with a browser
# and in one without.
[ -n "$browser_clis" ] || return 0
local first="${browser_clis%% *}"
cat <<HINT
if [[ \$- == *i* ]] && [ "\${RACCOON_BROWSER_TASK:-0}" != "1" ] && command -v pw >/dev/null 2>&1; then
echo "browser available (Playwright + Chromium, \\\`pw <script.js>\\\`)."
echo "Authoring a \\\`browser = true\\\` task? Start it with: RACCOON_BROWSER_TASK=1 $first"
fi
HINT
}
# Write each harness's config file from the registry, replacing whatever was there.
#
# The file is OWNED, not merged: TOML has no way to return to the document root after a
# table header, so appending or prepending around foreign content silently reparents
# root-level keys into whichever table happens to precede them. Owning it also means a
# registry change actually reaches a container that was already set up.
harness_write_configs() {
local id config_path blob target tmp
while IFS=$'\t' read -r id config_path blob; do
[ -n "$config_path" ] && [ -n "$blob" ] || continue
# Guarded: a bare failing assignment exits the caller's `set -e` post-create with
# no explanation. A path this cannot expand is one harness's problem, not the
# container's.
target=$(eval "printf '%s' \"$config_path\"") || {
echo "harness-setup: WARNING $id config_path could not be expanded — skipping" >&2
continue
}
mkdir -p "$(dirname "$target")"
tmp="$target.raccoon-tmp"
# Expansion is strict: an unset var would otherwise be written through as the
# literal ${VAR}, which surfaces much later as an unparseable value.
if ! {
echo "# Generated from harness-registry.toml — edits here are overwritten."
printf '%s' "$blob" | base64 -d | python3 -c '
import os, re, sys
text = sys.stdin.read()
missing = sorted(
{m.group(1) for m in re.finditer(r"\$\{(\w+)\}", text) if m.group(1) not in os.environ}
)
if missing:
sys.stderr.write("unset: " + ", ".join(missing) + "\n")
raise SystemExit(1)
sys.stdout.write(os.path.expandvars(text))
'
} > "$tmp"; then
rm -f "$tmp"
echo "harness-setup: WARNING $id config NOT written — a value it needs is unset." >&2
echo "harness-setup: run harness_setup_credentials first (harness_setup_all does)." >&2
continue
fi
mv "$tmp" "$target"
echo "harness-setup: $id config -> $target" >&2
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
}
# Link every available skill into each harness's skills_dir, for harnesses that declare one.
# Both container layouts are covered: the explore container holds the snapshot skill under
# plugins/, the authoring container holds the authoring skills under .claude/skills. Whichever
# directories exist here are the ones this container has.
harness_install_skills() {
local sources="${RACCOON_SKILL_SOURCE_DIRS:-/workspace/plugins/create-snapshot/skills /workspace/.claude/skills}"
local id dir target src skill name installed
while IFS=$'\t' read -r id dir; do
[ -n "$dir" ] || continue
target=$(eval "printf '%s' \"$dir\"") || {
echo "harness-setup: WARNING $id skills_dir could not be expanded — skipping" >&2
continue
}
mkdir -p "$target"
installed=0
for src in $sources; do
[ -d "$src" ] || continue
for skill in "$src"/*/; do
[ -f "$skill/SKILL.md" ] || continue
name=$(basename "$skill")
ln -sfn "${skill%/}" "$target/$name"
installed=$((installed + 1))
done
done
echo "harness-setup: $id skills -> $target ($installed linked)" >&2
done < <(_harness_query --skills-dirs 2>/dev/null || true)
}
# The lines that explain a setup failure are printed as it happens, and the devcontainer
# CLI's own stack trace lands on top of them. Close with a banner so the worker has
# something to look for, and something to send us.
_harness_fatal_banner() {
echo "" >&2
echo " ============================================================" >&2
echo " HARNESS SETUP FAILED — this container has no agent CLI." >&2
echo "" >&2
echo " The harness-setup: lines above say why. Anything the" >&2
echo " devcontainer prints after this is a consequence, not the" >&2
echo " cause; send us the harness-setup: lines." >&2
echo " ============================================================" >&2
echo "" >&2
}
harness_setup_all() {
harness_preflight || { _harness_fatal_banner; return 1; }
harness_setup_credentials
harness_write_auth
harness_install_clis
harness_write_configs
harness_install_skills
# Launchers are NOT installed here. They are an Explore concern (that container aliases
# `claude`/`codex` to them), and it passes its own AGENT_CLI_DIR — installing them here
# too wrote every launcher twice, the first time with the wrong editor path, and left an
# unused one in the authoring container.
echo "harness-setup: authoring harnesses" >&2
harness_report || { _harness_fatal_banner; return 1; }
}

View File

@@ -0,0 +1,93 @@
#!/usr/bin/env python3
"""str_replace_editor — CLI-as-MCP wrapper around the vendored EditTool.
This is the "CLI-as-MCP" delivery of the `str_replace_editor` tool: the agent
(which has ONLY the bash tool) invokes this script and passes the tool's
arguments as one JSON object on stdin. The actual editing logic is the vendored
`EditTool` under str_replace_editor_vendor/ (see VENDORED.md) — we add no
behavior, we only:
* instantiate it with run_command_preexec_fn=None (the class's own documented
way to skip its uid/gid-1000 demotion, which would break writes in our
sandbox where the workspace is owned by the agent user); and
* adapt structured stdin-JSON <-> a bash-invokable CLI.
stdin: one JSON object, e.g.
{"command":"view","path":"/workspace/app/models/x.rb"}
{"command":"view","path":"/workspace/x.rb","view_range":[1,40]}
{"command":"str_replace","path":"/workspace/x.rb","old_str":"a","new_str":"b"}
{"command":"create","path":"/workspace/new.rb","file_text":"..."}
{"command":"insert","path":"/workspace/x.rb","insert_line":10,"insert_text":"..."}
stdout: the tool's result text (exit 0). stderr + exit 1: a tool error message.
"""
import asyncio
import json
import os
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from str_replace_editor_vendor.base import ToolError # noqa: E402
from str_replace_editor_vendor.edit import EditTool # noqa: E402
# The keyword-only params the vendored EditTool.__call__ accepts.
_ACCEPTED = {
"command", "path", "file_text", "view_range",
"old_str", "new_str", "insert_text", "insert_line",
}
async def _run(payload: dict):
# Reject unknown keys instead of silently dropping them: a typo like
# `old_string` (vs `old_str`) should be a clear argument error, not a
# confusing failure deeper inside EditTool with the param silently missing.
unknown = set(payload) - _ACCEPTED
if unknown:
raise ToolError(
f"unknown argument(s): {', '.join(sorted(unknown))}. "
f"accepted keys: {', '.join(sorted(_ACCEPTED))}."
)
kwargs = dict(payload)
if "command" not in kwargs or "path" not in kwargs:
raise ToolError("Both `command` and `path` are required.")
# run_command_preexec_fn=None → no uid/gid demotion (see module docstring).
tool = EditTool(run_command_preexec_fn=None)
return await tool(**kwargs)
def main() -> int:
raw = sys.stdin.read()
if not raw.strip():
sys.stderr.write("str_replace_editor: expected a JSON object on stdin\n")
return 2
try:
payload = json.loads(raw)
except json.JSONDecodeError as e:
sys.stderr.write(f"str_replace_editor: invalid JSON on stdin: {e}\n")
return 2
if not isinstance(payload, dict):
sys.stderr.write("str_replace_editor: stdin JSON must be an object\n")
return 2
try:
result = asyncio.run(_run(payload))
except ToolError as e:
sys.stderr.write((e.message or "tool error") + "\n")
return 1
except TypeError as e:
# e.g. an unexpected/duplicate kwarg shape — surface like a tool error.
sys.stderr.write(f"str_replace_editor: bad arguments: {e}\n")
return 1
# EditTool returns a (CLI)Result with .output / .error / .base64_image / .system
if getattr(result, "error", None):
sys.stderr.write(result.error if result.error.endswith("\n") else result.error + "\n")
if getattr(result, "system", None):
sys.stderr.write(f"[system] {result.system}\n")
out = getattr(result, "output", None) or ""
if getattr(result, "base64_image", None):
out += "\n(image content omitted in CLI mode)"
if out:
sys.stdout.write(out if out.endswith("\n") else out + "\n")
return 1 if getattr(result, "error", None) else 0
if __name__ == "__main__":
sys.exit(main())

View File

@@ -0,0 +1 @@
"""Vendored verbatim — do not edit. See VENDORED.md for provenance."""

View File

@@ -0,0 +1,49 @@
from dataclasses import dataclass, fields, replace
@dataclass(kw_only=True, frozen=True)
class ToolResult:
"""Represents the result of a tool execution."""
output: str | None = None
error: str | None = None
base64_image: str | None = None
system: str | None = None
def __bool__(self):
return any(getattr(self, field.name) for field in fields(self))
def __add__(self, other: "ToolResult"):
def combine_fields(field: str | None, other_field: str | None, concatenate: bool = True):
if field and other_field:
if concatenate:
return field + other_field
raise ValueError("Cannot combine tool results")
return field or other_field
return ToolResult(
output=combine_fields(self.output, other.output),
error=combine_fields(self.error, other.error),
base64_image=combine_fields(self.base64_image, other.base64_image, False),
system=combine_fields(self.system, other.system),
)
def replace(self, **kwargs):
"""Returns a new ToolResult with the given fields replaced."""
return replace(self, **kwargs)
# QUESTION(simon): What's our intent behind differentiating here?
class CLIResult(ToolResult):
"""A ToolResult that can be rendered as a CLI output."""
class ToolFailure(ToolResult):
"""A ToolResult that represents a failure."""
class ToolError(Exception):
"""Raised when a tool encounters an error."""
def __init__(self, message):
self.message = message

View File

@@ -0,0 +1,476 @@
import asyncio
import base64
import shlex
from collections import deque
from dataclasses import dataclass
from pathlib import Path
from typing import Literal, get_args
from .base import CLIResult, ToolError, ToolResult
from .run import demote, maybe_truncate, run
TRUNCATED_MESSAGE: str = "<response clipped><NOTE>To save on context only part of this file has been shown to you. You should retry this tool after you have searched inside the file with `grep -n` in order to find the line numbers of what you are looking for.</NOTE>"
Command = Literal[
"view",
"create",
"str_replace",
"insert",
]
SNIPPET_LINES: int = 4
MAX_RESPONSE_LEN: int = 16000
class EditTool:
"""
An filesystem editor tool that allows the agent to view, create, and edit files.
The tool parameters are defined by Anthropic and are not editable.
"""
def __init__(self, run_command_preexec_fn=demote):
"""
Initialize the EditTool.
Args:
run_command_preexec_fn: Function to run in child process before executing
shell commands via the run() utility.
Defaults to demote() which drops privileges to uid/gid 1000.
Pass None to skip preexec, or any callable for custom behavior.
"""
self._run_command_preexec_fn = run_command_preexec_fn
async def __call__(
self,
*,
command: Command,
path: str,
file_text: str | None = None,
view_range: list[int] | None = None,
old_str: str | None = None,
new_str: str | None = None,
insert_text: str | None = None,
insert_line: int | None = None,
):
_path = Path(path)
self.validate_path(command, _path)
if command == "view":
return await self.view(_path, view_range)
elif command == "create":
if file_text is None:
raise ToolError("Parameter `file_text` is required for command: create")
await self.write_file(_path, file_text)
return ToolResult(output=f"File created successfully at: {_path}")
elif command == "str_replace":
if old_str is None:
raise ToolError("Parameter `old_str` is required for command: str_replace")
return await self.str_replace(_path, old_str, new_str)
elif command == "insert":
if insert_line is None:
raise ToolError("Parameter `insert_line` is required for command: insert")
if insert_text is None:
raise ToolError("Parameter `insert_text` is required for command: insert")
return await self.insert(_path, insert_line, insert_text)
raise ToolError(
f"Unrecognized command {command}. The allowed commands for the {self.name} tool are: {', '.join(get_args(Command))}"
)
def validate_path(self, command: str, path: Path):
"""
Check that the path/command combination is valid.
"""
# Check if its an absolute path
if not path.is_absolute():
suggested_path = Path("") / path
raise ToolError(
f"The path {path} is not an absolute path, it should start with `/`. Maybe you meant {suggested_path}?"
)
# Check if path exists
if not path.exists() and command != "create":
raise ToolError(f"The path {path} does not exist. Please provide a valid path.")
if path.exists() and command == "create":
raise ToolError(f"File already exists at: {path}. Cannot overwrite files using command `create`.")
# Check if the path points to a directory
if path.is_dir():
if command != "view":
raise ToolError(
f"The path {path} is a directory and only the `view` command can be used on directories"
)
async def view(self, path: Path, view_range: list[int] | None = None):
"""Implement the view command"""
if path.is_dir():
if view_range:
raise ToolError("The `view_range` parameter is not allowed when `path` points to a directory.")
_, stdout, stderr = await run(
rf"find {path} -maxdepth 2 -not -path '*/\.*'", preexec_fn=self._run_command_preexec_fn
)
if not stderr:
stdout = f"Here's the files and directories up to 2 levels deep in {path}, excluding hidden items:\n{stdout}\n"
return CLIResult(output=stdout, error=stderr)
image_extensions = {'.png', '.jpg', '.jpeg', '.gif', '.bmp', '.tiff', '.tif', '.webp', '.svg', '.ico'}
if path.suffix.lower() in image_extensions:
if view_range:
raise ToolError("The `view_range` parameter is not allowed when `path` points to an image file.")
try:
image_bytes = path.read_bytes()
base64_encoded = base64.b64encode(image_bytes).decode()
return CLIResult(
output=f"Displaying image file: {path}",
base64_image=base64_encoded
)
except Exception as e:
raise ToolError(f"Failed to read image file {path}: {e}") from None
file_content = await self.read_file(path, truncate_after=None)
file_text_lines = file_content.splitlines(keepends=True)
n_lines_file = len(file_text_lines) + (1 if file_content.endswith(("\n", "\r\n", "\r")) else 0)
if view_range:
if len(view_range) != 2 or not all(isinstance(i, int) for i in view_range):
raise ToolError("Invalid `view_range`. It should be a list of two integers.")
init_line, final_line = view_range
if init_line < 1 or init_line > n_lines_file:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its first element `{init_line}` should be within the range of lines of the file: {[1, n_lines_file]}"
)
if final_line > n_lines_file:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its second element `{final_line}` should be smaller than the number of lines in the file: `{n_lines_file}`"
)
if final_line != -1 and final_line < init_line:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its second element `{final_line}` should be larger or equal than its first `{init_line}`"
)
# Extract only the requested lines
if final_line != -1:
selected_lines = file_text_lines[max(view_range[0] - 1, 0) : view_range[1]]
else:
selected_lines = file_text_lines[max(view_range[0] - 1, 0) :]
# Join without modifying the original line endings
file_content = "".join(selected_lines)
file_content = process_view_output_str(
file_text=file_content,
path=str(path),
total_path_lines=n_lines_file,
max_resp_ln=MAX_RESPONSE_LEN,
view_range=(view_range[0], view_range[1]) if view_range else None,
)
return CLIResult(output=file_content)
async def str_replace(self, path: Path, old_str: str, new_str: str | None):
"""Implement the str_replace command, which replaces old_str with new_str in the file content"""
# Read the file content
file_content = await self.read_file(path, truncate_after=None)
new_str = new_str if new_str is not None else ""
# Check if old_str is unique in the file
occurrences = file_content.count(old_str)
if occurrences == 0:
raise ToolError(f"No replacement was performed, old_str `{old_str}` did not appear verbatim in {path}.")
elif occurrences > 1:
file_content_lines = file_content.split("\n")
lines = [idx + 1 for idx, line in enumerate(file_content_lines) if old_str in line]
raise ToolError(
f"No replacement was performed. Multiple occurrences of old_str `{old_str}` in lines {lines}. Please ensure it is unique"
)
# Replace old_str with new_str
new_file_content = file_content.replace(old_str, new_str)
# Write the new content to the file
await self.write_file(path, new_file_content)
# Create a snippet of the edited section
replacement_line = file_content.split(old_str)[0].count("\n")
start_line = max(0, replacement_line - SNIPPET_LINES)
end_line = replacement_line + SNIPPET_LINES + new_str.count("\n")
snippet = "\n".join(new_file_content.split("\n")[start_line : end_line + 1])
# Prepare the success message
success_msg = f"The file {path} has been edited. "
success_msg += self._make_output(snippet, f"a snippet of {path}", start_line + 1)
success_msg += "Review the changes and make sure they are as expected. Edit the file again if necessary."
return CLIResult(output=success_msg)
async def insert(self, path: Path, insert_line: int, new_str: str):
"""Implement the insert command, which inserts new_str at the specified line in the file content."""
file_text = await self.read_file(path, truncate_after=None)
file_text_lines = file_text.split("\n")
n_lines_file = len(file_text_lines)
if insert_line < 0 or insert_line > n_lines_file:
raise ToolError(
f"Invalid `insert_line` parameter: {insert_line}. It should be within the range of lines of the file: {[0, n_lines_file]}"
)
new_str_lines = new_str.split("\n")
new_file_text_lines = file_text_lines[:insert_line] + new_str_lines + file_text_lines[insert_line:]
snippet_lines = (
file_text_lines[max(0, insert_line - SNIPPET_LINES) : insert_line]
+ new_str_lines
+ file_text_lines[insert_line : insert_line + SNIPPET_LINES]
)
new_file_text = "\n".join(new_file_text_lines)
snippet = "\n".join(snippet_lines)
await self.write_file(path, new_file_text)
success_msg = f"The file {path} has been edited. "
success_msg += self._make_output(
snippet,
"a snippet of the edited file",
max(1, insert_line - SNIPPET_LINES + 1),
)
success_msg += "Review the changes and make sure they are as expected (correct indentation, no duplicate lines, etc). Edit the file again if necessary."
return CLIResult(output=success_msg)
async def read_file(self, path: Path, truncate_after: int | None = MAX_RESPONSE_LEN):
"""Read the content of a file from a given path; raise a ToolError if an error occurs."""
try:
code, out, err = await run(
f"cat {shlex.quote(str(path))}", truncate_after=truncate_after, preexec_fn=self._run_command_preexec_fn
)
if code != 0:
raise ToolError(f"Ran into {err} while trying to read {path}")
return out
except Exception as e:
print(e)
raise ToolError(f"Ran into {e} while trying to read {path}") from None
async def write_file(self, path: Path, file: str):
"""Write the content of a file to a given path; raise a ToolError if an error occurs."""
try:
# Write using stdin to avoid argument size limits
process = await asyncio.create_subprocess_shell(
f"cat > {shlex.quote(str(path))}",
stdin=asyncio.subprocess.PIPE,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
preexec_fn=self._run_command_preexec_fn,
)
stdout, stderr = await asyncio.wait_for(
process.communicate(input=file.encode('utf-8')),
timeout=120.0
)
if process.returncode != 0:
raise ToolError(f"Ran into {stderr.decode()} while trying to write to {path}")
except asyncio.TimeoutError:
raise ToolError(f"Timed out while trying to write to {path}")
except Exception as e:
raise ToolError(f"Ran into {e} while trying to write to {path}") from None
def _make_output(
self,
file_content: str,
file_descriptor: str,
init_line: int = 1,
expand_tabs: bool = True,
):
"""Generate output for the CLI based on the content of a file."""
file_content = maybe_truncate(file_content)
if expand_tabs:
file_content = file_content.expandtabs()
file_content = "\n".join([f"{i + init_line:6}\t{line}" for i, line in enumerate(file_content.split("\n"))])
return f"Here's the result of running `cat -n` on {file_descriptor}:\n" + file_content + "\n"
### AUX utilities
def add_line_numbers(text: str, includes_final_line: bool, n_first_line: int = 1) -> str:
"""
Given a string, returns the string with line numbers prepended to each line.
This function:
- Preserves the original line endings (CR, LF, or CRLF) of each line
- Adds a tab-separated line number prefix to each line
- If the text ends with any newline character (\n, \r\n, or \r), adds an
additional empty numbered line to represent the terminal empty line
"""
lines_with_endings = text.splitlines(keepends=True)
result = [f"{ind + n_first_line:6}\t{line_with_ending}" for ind, line_with_ending in enumerate(lines_with_endings)]
# Add an extra empty line with line number if original text ends with newline
if includes_final_line and text.endswith(("\n", "\r\n", "\r")):
result.append(f"{len(lines_with_endings) + n_first_line:6}\t")
return "".join(result)
def process_view_output_str(
file_text: str,
path: str,
total_path_lines: int,
max_resp_ln: int,
view_range: tuple[int, int] | None = None,
) -> str:
# Get header
header = f"Here's the content of {path} with line numbers"
if total_path_lines is not None and view_range is not None:
header += f" (which has a total of {total_path_lines} lines) with view_range={list(view_range)}"
# See if final line is included in the view_range
if view_range is None or view_range[1] == -1 or view_range[1] == total_path_lines:
includes_final_line = True
else:
includes_final_line = False
n_first_line = view_range[0] if view_range is not None else 1
# Truncate if needed
maybe_truncated_str = truncate_from_middle_v2(ss=file_text, max_len=max_resp_ln, n_line_offset=n_first_line - 1)
if isinstance(maybe_truncated_str, str):
# No truncation
file_text_with_line_numbers = add_line_numbers(
file_text,
includes_final_line=includes_final_line,
n_first_line=n_first_line,
)
else:
# Truncation occurred
before_with_line_numbers = add_line_numbers(
text="".join(maybe_truncated_str.as_str(maybe_truncated_str.before_lines)),
includes_final_line=False,
n_first_line=n_first_line,
)
if maybe_truncated_str.single_line:
file_text_with_line_numbers = before_with_line_numbers
else:
after_with_line_numbers = add_line_numbers(
text="".join(maybe_truncated_str.as_str(maybe_truncated_str.after_lines)),
includes_final_line=includes_final_line,
n_first_line=1 + maybe_truncated_str.truncated_end_line,
)
file_text_with_line_numbers = (
before_with_line_numbers + f"\t{maybe_truncated_str.truncation_msg}" + after_with_line_numbers
)
# Add context-aware truncation message
if view_range is not None:
# User already using view_range, suggest adjusting it
truncation_note = "\n<response clipped><NOTE>To save on context only part of the view range has been shown. You can adjust the view_range parameters or use `grep -n` to find specific content.</NOTE>"
else:
# User viewing whole file, suggest view_range or grep
truncation_note = "\n<response clipped><NOTE>To save on context only part of this file has been shown to you. You can use view_range=[start_line, end_line] to see specific sections, or use `grep -n` to find what you're looking for.</NOTE>"
file_text_with_line_numbers += truncation_note
return f"{header}:\n{file_text_with_line_numbers}"
@dataclass
class TruncatedString:
# Blocks
before_lines: list[str]
middle_lines: list[str]
after_lines: list[str]
# Line numbers (starting from 1)
truncated_start_line: int
truncated_end_line: int
# Truncation msg
truncation_msg: str
single_line: bool
def as_str(self, lines: list[str]) -> str:
return "".join(lines)
@property
def full_truncated_str(self) -> str:
return "".join(self.before_lines + [self.truncation_msg] + self.after_lines)
def truncate_from_middle_v2(ss: str, max_len: int, n_line_offset: int = 0) -> "str | TruncatedString":
"""
If no truncation is needed, returns the original string.
If truncation is needed, returns TruncatedString
"""
# No truncation needed
if len(ss) <= max_len:
return ss
# Single line
lines_with_endings = ss.splitlines(True)
if len(lines_with_endings) == 1:
chars_per_side = max(1, max_len // 2)
truncated_char_count = len(ss) - (chars_per_side * 2)
truncation_msg = f"...< truncated {truncated_char_count} characters >..."
before_lines = [ss[:chars_per_side] + truncation_msg + ss[-chars_per_side:]]
return TruncatedString(
before_lines=before_lines,
middle_lines=[],
after_lines=[],
truncated_start_line=1 + n_line_offset,
truncated_end_line=1 + n_line_offset,
truncation_msg=truncation_msg,
single_line=True,
)
# Line truncation
current_len = 0
before_lines = []
middle_lines = deque(lines_with_endings)
after_lines = deque([])
while current_len < max_len and len(middle_lines) > 1:
# Before
before_candidate_line = middle_lines[0]
if len(before_candidate_line) + current_len <= max_len:
before_lines.append(middle_lines.popleft())
current_len += len(before_candidate_line)
else:
break
# After
if len(middle_lines) > 1:
after_candidate_line = middle_lines[-1]
if len(after_candidate_line) + current_len <= max_len:
after_lines.appendleft(middle_lines.pop())
current_len += len(after_candidate_line)
else:
break
# Find truncated lines
first_truncated_line = 1 + len(before_lines) + n_line_offset
last_truncated_line = first_truncated_line + len(middle_lines) - 1
if ss.endswith(("\n", "\r", "\r\n")) and len(after_lines) == 0:
last_truncated_line += 1
# Create truncation msg
if first_truncated_line == last_truncated_line:
truncation_msg = f"< truncated line {first_truncated_line} >"
else:
truncation_msg = f"< truncated lines {first_truncated_line}-{last_truncated_line} >"
if len(after_lines) != 0:
if before_lines[0].endswith("\r\n"):
truncation_msg += "\r\n"
elif before_lines[0].endswith("\r"):
truncation_msg += "\r"
else:
truncation_msg += "\n"
return TruncatedString(
# Blocks
before_lines=before_lines,
middle_lines=list(middle_lines),
after_lines=list(after_lines),
# Line numbers (starting from 1)
truncated_start_line=first_truncated_line,
truncated_end_line=last_truncated_line,
# Truncation msg
truncation_msg=truncation_msg,
single_line=False,
)

View File

@@ -0,0 +1,66 @@
"""Utility to run shell commands asynchronously with a timeout."""
import asyncio # noqa -- swapping to trio would be beneficial, but not blocking atm
import os
TRUNCATED_MESSAGE: str = "<response clipped><NOTE>To save on context only part of this file has been shown to you. You should retry this tool after you have searched inside the file with `grep -n` in order to find the line numbers of what you are looking for.</NOTE>"
MAX_RESPONSE_LEN: int = 16000
def maybe_truncate(content: str, truncate_after: int | None = MAX_RESPONSE_LEN):
"""Truncate content and append a notice if content exceeds the specified length."""
return (
content
if not truncate_after or len(content) <= truncate_after
else content[:truncate_after] + TRUNCATED_MESSAGE
)
def demote():
"""Drop privileges to uid/gid 1000 for security.
This function is intended to be used as a preexec_fn in subprocess calls
to ensure commands run with reduced privileges.
"""
os.setgid(1000)
os.setuid(1000)
async def run(
cmd: str,
timeout: float | None = 120.0, # seconds # noqa: ASYNC109
truncate_after: int | None = MAX_RESPONSE_LEN,
preexec_fn=demote,
):
"""Run a shell command asynchronously with a timeout.
Args:
cmd: Command to execute
timeout: Command timeout in seconds
truncate_after: Maximum response length before truncation
preexec_fn: Function to run in child process before exec (default: demote).
Pass None to skip preexec, or any callable for custom behavior.
Returns:
Tuple of (return_code, stdout, stderr)
"""
process = await asyncio.create_subprocess_shell(
cmd,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
preexec_fn=preexec_fn,
)
try:
stdout, stderr = await asyncio.wait_for(process.communicate(), timeout=timeout)
return (
process.returncode or 0,
maybe_truncate(stdout.decode(), truncate_after=truncate_after),
maybe_truncate(stderr.decode(), truncate_after=truncate_after),
)
except TimeoutError as exc:
try:
process.kill()
except ProcessLookupError:
pass
raise TimeoutError(f"Command '{cmd}' timed out after {timeout} seconds") from exc

View File

@@ -0,0 +1,20 @@
# Your actual toolset (this overrides any earlier tool guidance above)
This harness gives you exactly two ways to act, both through the `Bash` tool:
1. **Shell commands** for everything read-only and for running things: view and search files with `cat`, `sed -n`, `grep -rn`, `find`, `ls`; run tests; run `git`; etc.
2. **A `str_replace_editor` file editor**, which you invoke from Bash by piping ONE JSON object on stdin to `/opt/agent-cli/str_replace_editor`. Use a quoted heredoc so backslashes and quotes survive:
`/opt/agent-cli/str_replace_editor <<'EDITOR'` then a line of JSON then `EDITOR`
The JSON `"command"` field selects the operation:
- `view` — view a file (optionally `"view_range":[start,end]`) or list a directory: `{"command":"view","path":"/abs/file.rb"}`
- `create` — create a NEW file (fails if it exists): `{"command":"create","path":"/abs/new.rb","file_text":"..."}`
- `str_replace` — replace a UNIQUE substring: `{"command":"str_replace","path":"/abs/file.rb","old_str":"...","new_str":"..."}`
- `insert` — insert text after a line: `{"command":"insert","path":"/abs/file.rb","insert_line":N,"insert_text":"..."}`
Paths must be absolute. Inside JSON strings, escape newlines as `\n` and double-quotes as `\"`.
There are **no** `Read`, `Grep`, `Glob`, `Edit`, `Write`, `MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite`, or `AskUserQuestion` tools — `Bash` is your only built-in tool. So disregard the earlier "Prefer the dedicated file/search tools over shell commands" guidance and the Memory section's "use the Write tool" instruction: those tools are not available in this harness. Search and read with shell commands; view, create, and edit files with `str_replace_editor`.
There is also no tool for asking the user an interactive question. If you need to ask the user something, or raise a concern about the request before acting on it, put it in your normal text response.

View File

@@ -0,0 +1,4 @@
## Browser
Chromium is available in this environment via Playwright. `pw <script.js>` runs Node with
`require("playwright")` resolvable (CommonJS — `import` will not find it).

View File

@@ -0,0 +1,7 @@
## Correction to the toolset above: you also have `Read`
This task runs with `Read` in addition to `Bash`, so the statement above that there is no `Read`
tool does not apply here. `Read` renders images — use it to look at a screenshot you have
written to disk. Everything else above still holds: no `Grep`, `Glob`, `Edit`, `Write`,
`MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite` or `AskUserQuestion`, and you still create and
edit files with `str_replace_editor`.

View File

@@ -0,0 +1,298 @@
{
"polyglot": true,
"repos": [
{
"repo": "lambda-cloudwatch-logs-to-loggly",
"defaultCommit": "f17e2d3",
"runtime": "node:14"
},
{
"repo": "lambda-potion-engagement",
"defaultCommit": "c64365b",
"runtime": "node:14"
},
{
"repo": "lambda-potion-schedular",
"defaultCommit": "0843570",
"runtime": "node:14"
},
{
"repo": "lambda-potion-transcription-scheduler",
"defaultCommit": "1a2e3d5",
"runtime": "node:14"
},
{
"repo": "lambda-video-processing",
"defaultCommit": "0e4a9b5",
"runtime": "node:18"
},
{
"repo": "microservice-dynamic-screen-recording",
"defaultCommit": "31e142b",
"runtime": "node:18"
},
{
"repo": "microservice-potion-voice",
"defaultCommit": "b65ca17",
"runtime": "node:14"
},
{
"repo": "potion-dynamic-screen-recording-lambda",
"defaultCommit": "57ed9e6",
"runtime": "node:14"
},
{
"repo": "potion-job-consumer",
"defaultCommit": "93f8a10",
"runtime": "node:18"
},
{
"repo": "potion-job-producer",
"defaultCommit": "04663d1",
"runtime": "node:18"
},
{
"repo": "potion-video-processing",
"defaultCommit": "59c6af9",
"runtime": "node:14"
},
{
"repo": "potion-voice",
"defaultCommit": "fcd8a9d",
"runtime": "node:14"
},
{
"repo": "potion-watcher",
"defaultCommit": "0e5973b",
"runtime": "node:18"
},
{
"repo": "potion-website-recording-handler",
"defaultCommit": "c58a9bb",
"runtime": "node:18"
},
{
"repo": "potion-app",
"defaultCommit": "6b4fee0c",
"runtime": "node:16",
"startCmd": "bash -c \"cp -n .env.client.development .env.local 2>/dev/null || true; export POTION_APP_ENV=local; [ -f .nuxt/store.js ] || npx nuxt build; node scripts/seed-dev-user.js || true; node server/index.js\"",
"setupCmd": "bash -c \"cp -n .env.client.development .env.local 2>/dev/null || true; export POTION_APP_ENV=local; [ -f .nuxt/store.js ] || npx nuxt build\""
},
{
"repo": "potion-custom-domain-app",
"defaultCommit": "01a7034",
"runtime": "none"
},
{
"repo": "potion-website",
"defaultCommit": "27995f8",
"runtime": "node:16"
},
{
"repo": "browser-extensions",
"defaultCommit": "b5e75d4",
"runtime": "node:18"
},
{
"repo": "gcp-application",
"defaultCommit": "469056f",
"runtime": "node:18"
},
{
"repo": "lambda-text-to-speech",
"defaultCommit": "99054ac",
"runtime": "node:18"
},
{
"repo": "potion-multi-dsr-watcher",
"defaultCommit": "c275d7f",
"runtime": "node:18",
"startCmd": "npx @google-cloud/functions-framework --target=potion-multi-dsr-watcher",
"bootEnv": "MONGODB_URI=mongodb://127.0.0.1:27017/potion_dev"
},
{
"repo": "potion-qa",
"defaultCommit": "3920e6c",
"runtime": "node:18"
},
{
"repo": "potion-snapshot-testing",
"defaultCommit": "a80eb8d",
"runtime": "node:18"
},
{
"repo": "potion-web",
"defaultCommit": "0a7e699",
"runtime": "node:18",
"startCmd": "npx nuxt dev --host 0.0.0.0 --port 3000",
"bootEnv": "POTION_APP_ENV=development BUGSNAG_FRONTEND_KEY=00000000000000000000000000000000 API_BASE_URL=http://localhost:4300 POTION_BASE_URL=http://localhost:4300"
},
{
"repo": "potion-analytics",
"defaultCommit": "43a7d23",
"runtime": "node:20"
},
{
"repo": "potion-api",
"defaultCommit": "5abe18f",
"runtime": "node:20"
},
{
"repo": "MODNet-with-training",
"defaultCommit": "dace325",
"runtime": "python:3.10"
},
{
"repo": "avds-cleaner",
"defaultCommit": "bd3a503",
"runtime": "python:3.10"
},
{
"repo": "avspeech",
"defaultCommit": "ca0f90d",
"runtime": "python:3.10"
},
{
"repo": "lambda-datadog-forwarder",
"defaultCommit": "a57ae74",
"runtime": "python:3.10"
},
{
"repo": "potion-ai",
"defaultCommit": "0e454d8",
"runtime": "python:3.10"
},
{
"repo": "potion-ai-cpu",
"defaultCommit": "ad61fa7",
"runtime": "python:3.10"
},
{
"repo": "potion-ai-gpu",
"defaultCommit": "8413d71",
"runtime": "python:3.10"
},
{
"repo": "potion-stitch",
"defaultCommit": "cfaed2f",
"runtime": "python:3.10"
},
{
"repo": "potion-tryon",
"defaultCommit": "b7da6a2",
"runtime": "python:3.10"
},
{
"repo": "potion-video-background-change",
"defaultCommit": "e6f2ea4",
"runtime": "python:3.10"
},
{
"repo": "potion-voice-dataset",
"defaultCommit": "f3d79d6",
"runtime": "python:3.10"
},
{
"repo": "potion-voice-utils",
"defaultCommit": "eadc48b",
"runtime": "python:3.10"
},
{
"repo": "sentence-split-service",
"defaultCommit": "32356d2",
"runtime": "python:3.10"
},
{
"repo": "urlbox-experiments",
"defaultCommit": "141fe18",
"runtime": "python:3.10"
},
{
"repo": "video-synth-api",
"defaultCommit": "167fcd7",
"runtime": "python:3.10"
},
{
"repo": "wav2lip-fa",
"defaultCommit": "8448ef0",
"runtime": "python:3.10"
},
{
"repo": "yeahsure-tryon",
"defaultCommit": "c8dee39",
"runtime": "python:3.10"
},
{
"repo": "gcp-infrastructure",
"defaultCommit": "a7dc5cc",
"runtime": "none"
},
{
"repo": "potion-ai-pretrained-models-infra",
"defaultCommit": "8a88770",
"runtime": "none"
},
{
"repo": "potion-app-infra",
"defaultCommit": "2107464",
"runtime": "none"
},
{
"repo": "potion-bastion",
"defaultCommit": "062af16",
"runtime": "none"
},
{
"repo": "potion-video-processing-devops",
"defaultCommit": "566286d",
"runtime": "none"
},
{
"repo": "elasticmq-container",
"defaultCommit": "de8acb5",
"runtime": "none"
},
{
"repo": "gcp-cloud-infrastructure",
"defaultCommit": "aa033c8",
"runtime": "none"
},
{
"repo": "potion-devops",
"defaultCommit": "84a4532",
"runtime": "none"
},
{
"repo": "potion-wp-site",
"defaultCommit": "cb71e3a",
"runtime": "none"
}
],
"defaultRepo": "potion-app",
"version": "2f696c53b4",
"blockedHosts": [
"sendpotion.com",
"www.sendpotion.com",
"app.sendpotion.com",
"staging.sendpotion.com",
"development.sendpotion.com",
"devleopment.sendpotion.com",
"meawww.sendpotion.com",
"blog.sendpotion.com",
"help.sendpotion.com",
"terms.sendpotion.com",
"pricing.sendpotion.com",
"videoassets.sendpotion.com",
"subtitleassets.sendpotion.com",
"audioassets.sendpotion.com",
"videoassets.staging.sendpotion.com",
"subtitleassets.staging.sendpotion.com",
"audioassets.staging.sendpotion.com"
],
"explorePorts": {
"clientHost": 4300,
"serverHost": null,
"livereloadHost": null,
"corpusHost": null
}
}

View File

@@ -0,0 +1,222 @@
#!/bin/bash
# Welcome banner for raccoon dev containers
CYAN='\033[1;36m'
YELLOW='\033[1;33m'
GRAY='\033[0;90m'
RESET='\033[0m'
CONTAINER_TYPE="${1:-explore}"
if [ "$CONTAINER_TYPE" = "explore" ]; then
COLOR="$CYAN"
else
COLOR="$YELLOW"
fi
cat << 'RACCOON'
.----------------. .----------------. .----------------. .----------------. .----------------. .----------------. .-----------------.
| .--------------. || .--------------. || .--------------. || .--------------. || .--------------. || .--------------. || .--------------. |
| | _______ | || | __ | || | ______ | || | ______ | || | ____ | || | ____ | || | ____ _____ | |
| | |_ __ \ | || | / \ | || | .' ___ | | || | .' ___ | | || | .' `. | || | .' `. | || ||_ \|_ _| | |
| | | |__) | | || | / /\ \ | || | / .' \_| | || | / .' \_| | || | / .--. \ | || | / .--. \ | || | | \ | | | |
| | | __ / | || | / ____ \ | || | | | | || | | | | || | | | | | | || | | | | | | || | | |\ \| | | |
| | _| | \ \_ | || | _/ / \ \_ | || | \ `.___.'\ | || | \ `.___.'\ | || | \ `--' / | || | \ `--' / | || | _| |_\ |_ | |
| | |____| |___| | || ||____| |____|| || | `._____.' | || | `._____.' | || | `.____.' | || | `.____.' | || ||_____|\____| | |
| | | || | | || | | || | | || | | || | | || | | |
| '--------------' || '--------------' || '--------------' || '--------------' || '--------------' || '--------------' || '--------------' |
'----------------' '----------------' '----------------' '----------------' '----------------' '----------------' '----------------'
__ .-.
.-"` .`'. /\\|
_(\-/)_" , . ,\ /\\\/
{(#b^d#)} . ./, |/\\\/
`-.(Y).-` , | , |\.-`
/~/,_/~~~\,__.-`
////~ // ~\\
==`==` ==` ==`
------------------------------------------------
RACCOON
# Per-repo notes. Two kinds of thing surface here:
#
# 1. Setup side effects — some source repos need their toolchain adapted to the
# container at setup time (e.g. a pinned language version the base image
# doesn't ship, or a dependency incompatible with the base image's OpenSSL).
# Those adjustments touch tracked files, so a fresh container can show a
# non-empty `git status` even though the worker hasn't changed anything.
# Calling it out here keeps it reading as expected setup, not the worker's
# own edits.
# 2. How to run the app locally — the commands to bring the app up in the
# browser so the worker can click through the real workflows while they
# explore. Ports are published by the explore devcontainer.json, and the
# dev DB is seeded during post-create so login works out of the box.
REPO=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').repo)}catch{}" 2>/dev/null)
# Prefer the live host port exported into the container ($EXPLORE_CLIENT_PORT),
# falling back to the toolkit.json default then 3000 for older containers.
CLIENT_PORT="${EXPLORE_CLIENT_PORT:-$(node -e "try{process.stdout.write(String(require('/workspace/toolkit.json').explorePorts.clientHost))}catch{process.stdout.write('3000')}" 2>/dev/null || echo 3000)}"
IS_POLYGLOT=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').polyglot?'1':'')}catch{}" 2>/dev/null)
# Reference-data corpus viewer (zeta toolkits): a small always-on web UI + search index
# over the shipped corpus. Only mentioned when this build actually carries the index.
CORPUS_PORT="${EXPLORE_CORPUS_PORT:-$(node -e "try{const p=require('/workspace/toolkit.json').explorePorts.corpusHost;if(p)process.stdout.write(String(p))}catch{}" 2>/dev/null)}"
corpus_banner() {
if [ -f /workspace/data/corpus-index/corpus.db ]; then
printf "${COLOR}Reference-data corpus:${RESET} real company slack/jira/email/support data at ${GRAY}/data/zeta-corpus${RESET}.\n"
printf "Browse + search it at ${GRAY}http://localhost:${CORPUS_PORT:-3002}${RESET} (auto-started; ${GRAY}view-corpus --help${RESET} to manage),\n"
printf "or query the index directly — see ${GRAY}/workspace/corpus-viewer/README.md${RESET}. Great for anchoring\n"
printf "a task in a real incident, ticket, or support thread.\n\n"
fi
}
if [ -n "$IS_POLYGLOT" ]; then
DEF=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').defaultRepo||'')}catch{}" 2>/dev/null)
printf "${COLOR}This toolkit hosts several repos.${RESET} Pick one to explore and run:\n"
node -e "require('/workspace/toolkit.json').repos.forEach(r=>console.log(' • '+r.repo))" 2>/dev/null
printf "\n${COLOR}To work on one repo:${RESET}\n"
printf " 1. ${GRAY}run-app ${DEF}${RESET} installs deps + prepares the DB on first use, then boots the app\n"
printf " ${GRAY}(it prints the URL to open, and how to sign in when the app needs a login)${RESET}\n"
printf " 2. ${GRAY}cd /workspace/repos/${DEF}${RESET} focus your shell on that repo\n"
printf " 3. ${GRAY}codex${RESET} launch it FROM the repo dir, so it works there without being told the path (${GRAY}claude${RESET} also available)\n"
printf "Then open ${GRAY}http://localhost:${CLIENT_PORT}${RESET}. Switch repos: ${GRAY}run-app --stop${RESET}, then repeat for another.\n"
printf "Some repos ship a runnable test suite, many don't — ${GRAY}run-app${RESET} tells you what each one has.\n"
printf "Where a suite exists the grader runs it for the deterministic checks behind the correctness\n"
printf "score; where none does, correctness is judged from the code alone.\n\n"
corpus_banner
printf "${GRAY}Want another container in parallel (its own copy of every member repo, e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
return 0 2>/dev/null || exit 0
fi
case "$REPO" in
ZenBill-006)
printf "${YELLOW}Heads up:${RESET} first-time setup adapts this app to the container's Ruby/OpenSSL,\n"
printf "modifying a few tracked files — ${GRAY}Gemfile${RESET}, ${GRAY}Gemfile.lock${RESET}, ${GRAY}db/schema.rb${RESET}.\n"
printf "They show in ${GRAY}git status${RESET}, but that's expected setup — not your changes.\n\n"
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} (this app routes by subdomain —\n"
printf "plain ${GRAY}localhost${RESET} shows only the Rails welcome page; see the README for /etc/hosts setup).\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
zeta-heimdall)
printf "${YELLOW}Heads up:${RESET} first-time setup installs gems and prepares the DB, which can\n"
printf "touch tracked files (${GRAY}Gemfile${RESET}, ${GRAY}Gemfile.lock${RESET}, ${GRAY}db/schema.rb${RESET}, ${GRAY}config/database.yml${RESET}).\n"
printf "They show in ${GRAY}git status${RESET}, but that's expected setup — not your changes.\n\n"
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET}. It's a JSON API (no UI) on\n"
printf "${GRAY}http://localhost:${CLIENT_PORT}${RESET} — hit an endpoint rather than expecting a page.\n"
printf "Runnable test suite: ${GRAY}bundle exec rspec${RESET} — the grader draws on it for the\n"
printf "deterministic checks behind the correctness score.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
zeta-platform)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the React UI on ${GRAY}http://localhost:${CLIENT_PORT}${RESET}\n"
printf "plus the Rails API it proxies to (on :5000 inside the container).\n"
printf "Runnable test suite: ${GRAY}bundle exec rspec${RESET} (large suite; Postgres + Redis are baked in;\n"
printf "rspec needs neither the client nor the running server).\n\n"
printf "${YELLOW}About this codebase:${RESET} a handful of specs are red here for reasons unrelated to\n"
printf "any task (timestamp precision, one stale model-method reference, and specs needing\n"
printf "third-party credentials this copy doesn't carry). They're skipped in\n"
printf "${GRAY}spec/support/known_failing_specs.rb${RESET}, so ${GRAY}bundle exec rspec${RESET} is green out of the box.\n"
printf "Scope your task's deterministic checks to the specs relevant to your task rather than\n"
printf "the whole suite.\n\n"
corpus_banner
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
Palolo-031)
printf "${COLOR}Run the app${RESET} in one command:\n"
printf " ${GRAY}run-app${RESET} (starts the server + client, waits until ready, prints the URL)\n"
printf "Then open ${GRAY}http://localhost:${CLIENT_PORT}${RESET} and log in as ${GRAY}zaniyah@exhalefi.com${RESET} / ${GRAY}test${RESET}.\n"
printf "${GRAY}Stop it with ${RESET}${GRAY}run-app --stop${RESET}${GRAY}; follow logs with ${RESET}${GRAY}run-app --logs${RESET}${GRAY}.${RESET}\n"
printf "${GRAY}(The dev DB is seeded automatically during setup — re-run the seed with${RESET}\n"
printf "${GRAY} DEFAULT_BAAS_PROVIDER=Liquid PUBLIC_BAAS_ENABLED=yes TESTING_SEED=yes pnpm run seed --small.)${RESET}\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n"
printf "${GRAY} (it runs for exploring, but its browser app calls the first container's API.)${RESET}\n\n"
;;
human-essentials)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Rails UI on\n"
printf "${GRAY}http://localhost:${CLIENT_PORT}${RESET}. The dev DB is seeded during setup; log in at\n"
printf "${GRAY}/users/sign_in${RESET} as ${GRAY}test@example.com${RESET} / ${GRAY}password!${RESET} (there's no self-service\n"
printf "signup — re-seed with ${GRAY}bin/rails db:seed${RESET}). Verifier: ${GRAY}bundle exec rspec${RESET}.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
endsideout)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Rails UI on\n"
printf "${GRAY}http://localhost:${CLIENT_PORT}${RESET}. The dev DB (SQLite) is seeded during setup; log in at\n"
printf "${GRAY}/session/new${RESET} as ${GRAY}admin@example.com${RESET} / ${GRAY}password${RESET} (there's no self-service\n"
printf "signup — re-seed with ${GRAY}bin/rails db:seed${RESET}). Verifier: ${GRAY}bin/rails test${RESET}.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
community-foundation)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Rails UI. This app is\n"
printf "${YELLOW}multi-tenant by subdomain${RESET}: plain ${GRAY}localhost${RESET} shows only the apex landing page.\n"
printf "Open the seeded tenant at ${GRAY}http://arlington.lvh.me:${CLIENT_PORT}/${RESET} and log in as\n"
printf "${GRAY}owner@example.com${RESET} / ${GRAY}password${RESET} (seeded during setup; no self-service signup —\n"
printf "re-seed with ${GRAY}bin/rails db:seed${RESET}). Verifier: ${GRAY}bin/rails test${RESET}.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
stocks-in-the-future)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Rails UI on\n"
printf "${GRAY}http://localhost:${CLIENT_PORT}${RESET}. The dev DB is seeded during setup; log in at\n"
printf "${GRAY}/users/sign_in${RESET} as username ${GRAY}admin${RESET} / ${GRAY}password${RESET} (login is by ${YELLOW}username${RESET}, not\n"
printf "email; no self-service signup — re-seed with ${GRAY}bin/rails db:seed${RESET}). Verifier: ${GRAY}bin/rails test${RESET}.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
casa)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Rails UI on\n"
printf "${GRAY}http://localhost:${CLIENT_PORT}${RESET}. The dev DB is seeded during setup; log in at\n"
printf "${GRAY}/users/sign_in${RESET} as ${GRAY}casa_admin1@example.com${RESET} / ${GRAY}12345678${RESET} (users are admin-invited,\n"
printf "no self-service signup — re-seed with ${GRAY}bin/rails db:seed${RESET}). Verifier: ${GRAY}bundle exec rspec${RESET}.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
awbw)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Rails UI on\n"
printf "${GRAY}http://localhost:${CLIENT_PORT}${RESET}. The dev DB is seeded during setup; log in at\n"
printf "${GRAY}/users/sign_in${RESET} as ${GRAY}umberto.user@example.com${RESET} / ${GRAY}password${RESET} (there's no self-service\n"
printf "signup — re-seed with ${GRAY}bin/rails db:seed${RESET}). Verifier: ${GRAY}bundle exec rspec${RESET}.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
flaredown)
printf "${YELLOW}Heads up:${RESET} this app has two parts — a Rails API in ${GRAY}backend/${RESET} and an Ember\n"
printf "client in ${GRAY}frontend/${RESET} (not at the repo root). First-time setup installs gems +\n"
printf "JS deps and migrates the databases, which can touch tracked files.\n\n"
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Ember UI on\n"
printf "${GRAY}http://localhost:${CLIENT_PORT}${RESET} plus the Rails API it proxies to (on :5000 inside the\n"
printf "container). Three datastores are baked in: ${GRAY}Postgres${RESET} + ${GRAY}Redis${RESET} (Sidekiq) + ${GRAY}MongoDB${RESET}\n"
printf "(Mongoid, the primary store). The backend test suite is the API verifier:\n"
printf "${GRAY}cd backend && bundle exec rspec${RESET} (needs neither the client nor the running server).\n\n"
printf "The repo's own ${GRAY}CLAUDE.md${RESET} / ${GRAY}README${RESET} describe running it with ${GRAY}make${RESET} + ${GRAY}docker compose${RESET}.\n"
printf "That's the upstream workflow, for your host — ${YELLOW}there's no Docker daemon in here${RESET}, so\n"
printf "use ${GRAY}run-app${RESET} and ${GRAY}bundle exec rspec${RESET} instead. Everything is already installed.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
alongwithyou)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Rails UI on\n"
printf "${GRAY}http://localhost:${CLIENT_PORT}${RESET}. This is a ${YELLOW}young app${RESET} (a fresh Rails 8.1 scaffold\n"
printf "being built with the Dewberry Cancer Center) — no routes or auth exist yet, so\n"
printf "the browser shows the default Rails welcome page. It grows over time.\n"
printf "Verifier: ${GRAY}bin/rails test${RESET} (SQLite; no external services).\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
breezy-complete)
printf "${COLOR}Run the app${RESET} in one command: ${GRAY}run-app${RESET} — boots the Rails API (${GRAY}backend/${RESET}) and the\n"
printf "Next.js frontend (${GRAY}frontend/${RESET}). Open ${GRAY}http://localhost:${CLIENT_PORT}/pro_signin${RESET} — auth is\n"
printf "bypassed offline and it auto-redirects to the seeded professional's dashboard\n"
printf "(no login needed; re-seed with ${GRAY}cd backend && bundle exec rails db:seed${RESET}).\n"
printf "Verifier: ${GRAY}cd backend && RAILS_ENV=test bundle exec rspec${RESET} (RSpec is the suite of record).\n"
printf "${YELLOW}Don't${RESET} export ${GRAY}DISABLE_CLERK${RESET}/${GRAY}CLERK_SKIP_RAILTIE${RESET}/${GRAY}DATABASE_URL${RESET} into your shell — several\n"
printf "controller specs 403 under the Clerk bypass; run-app scopes it to the servers.\n\n"
printf "${GRAY}Want another container with its own separate working tree (e.g. a different commit / repo state)?${RESET}\n"
printf "${GRAY}On the host, from explore/: node instance.js b then: node instance.js shell b${RESET}\n\n"
;;
esac