lots of change - all to start my 3rd redo

This commit is contained in:
2026-09-26 14:31:52 -04:00
parent 7f4d388e19
commit bceb52e8ee
1046 changed files with 4476 additions and 0 deletions

View File

@@ -0,0 +1,156 @@
# Polyglot Explore container for the potion-polyglot toolkit (Potion).
#
# One image hosts every member repo (worker switches with `run-app <repo>`). Runtime union
# across the estate: Node (dominant — 26 members, spanning the Node 14 lambdas to the Node 20
# API), Python (17 — ML pipelines, Flask services, data ETL), Terraform (5), PHP (1).
#
# This image only decides what run-app can BOOT. What makes a member gradable is its
# harbor-tasks/raccoon-shared/Dockerfile.<member>, and a member with no inherited test suite is
# still gradable via the rubric — so a member absent from this image is not "not worth grading".
# Postgres is baked as cheap insurance (no member's verifier requires it).
#
# NOT baked (deliberately):
# - (nothing yet — see the MongoDB note below)
#
# MongoDB IS required, and IS installable here. No member's *verifier* needs it (potion-app is
# jsdom, potion-api's usable suites are sinon-mocked), but `run-app` on potion-app and potion-api
# both do, and those are the two apps a worker is most likely to boot. An earlier note in this
# file claimed MongoDB ships no arm64 debian-bookworm package and skipped it. That is true only of
# MongoDB's *Debian* repo; the **Ubuntu jammy arm64** packages install cleanly on bookworm —
# verified 2026-07-31 on this platform: mongodb-org-server 8.0.28 installs, mongod starts, and a
# write round-trips. Bake it from that repo rather than demoting the estate's flagship app to
# read-only.
# - GPU/CUDA — the potion-ai* members load weights from a now-defunct bucket (never in git),
# so they are read-and-edit here regardless.
# - PHP/MySQL — potion-wp-site's first-party code (its custom theme) IS graded, through its
# own hand-authored harbor image with php-cli + composer; it just doesn't boot in Explore.
#
# Runtimes:
# - Node 14 / 16 / 18 / 20 via nvm (run-app's node selector switches per member)
# - Python 3.10 via uv (agent str_replace_editor needs >=3.10; also the Python members)
# - PostgreSQL baked in
FROM debian:bookworm
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update && apt-get install -y --no-install-recommends \
build-essential git curl ca-certificates gnupg procps sudo xz-utils \
libssl-dev zlib1g-dev \
postgresql postgresql-client \
&& rm -rf /var/lib/apt/lists/*
# --- Node via nvm: 14 / 16 / 18 / 20 (prebuilt). Default 20 symlinked to /usr/local/bin so the
# toolkit's own `node -e` (run-app/welcome read toolkit.json) always works; run-app switches PATH
# per member. yarn into each version. v20.* glob (Docker RUN uses dash; nvm.sh is bash-only). ---
ENV NVM_DIR=/usr/local/nvm
RUN mkdir -p "$NVM_DIR" \
&& curl -fsSL https://raw.githubusercontent.com/nvm-sh/nvm/v0.39.7/install.sh | bash \
&& bash -c '. "$NVM_DIR/nvm.sh" \
&& for v in 14 16 18 20; do nvm install "$v" && nvm use "$v" && npm install -g yarn; done \
&& nvm alias default 20' \
&& for b in node npm npx yarn; do ln -sf "$NVM_DIR"/versions/node/v20.*/bin/"$b" /usr/local/bin/"$b"; done
# --- MongoDB 8.0 (the product DB: potion-app + potion-api both need it to BOOT) ---
# From MongoDB's **Ubuntu jammy** arm64 repo, not the Debian one. MongoDB publishes no arm64
# packages for debian/bookworm (verified: no apt candidate), which is why an earlier revision of
# this image skipped Mongo and left the estate's flagship app unbootable. The jammy arm64 build
# installs and runs fine here — verified on this platform: mongodb-org-server 8.0.28 installs,
# mongod starts, a write round-trips. `mongodb-mongosh` ships the shell so a worker can inspect
# the DB. Data lives in /data/db, created here so mongod can start as root in the sandbox.
RUN curl -fsSL https://pgp.mongodb.com/server-8.0.asc \
| gpg --dearmor -o /usr/share/keyrings/mongodb-8.gpg \
&& echo "deb [ signed-by=/usr/share/keyrings/mongodb-8.gpg ] https://repo.mongodb.org/apt/ubuntu jammy/mongodb-org/8.0 multiverse" \
> /etc/apt/sources.list.d/mongodb-org-8.0.list \
&& apt-get update \
&& apt-get install -y --no-install-recommends mongodb-org-server mongodb-mongosh \
&& rm -rf /var/lib/apt/lists/* \
&& mkdir -p /data/db \
&& mongod --version | head -1
# --- Python via uv ---
# 3.10 stays the default `python3`: it is what this estate's Python members run under.
# 3.11 is installed alongside it because harness setup reads the registry with `tomllib`
# (3.11+), and post-create runs under `set -e` — an image with only 3.10 fails container
# creation. setup-harnesses.sh tries python3, then python3.13/3.12/3.11, so exposing the
# newer one under its versioned name is enough and leaves the members' default untouched.
RUN curl -fsSL https://astral.sh/uv/0.12.10/install.sh | env UV_INSTALL_DIR=/usr/local/bin sh \
&& uv python install 3.10 \
&& ln -sf "$(uv python find 3.10)" /usr/local/bin/python3 \
&& uv python install 3.11 \
&& ln -sf "$(uv python find 3.11)" /usr/local/bin/python3.11 \
&& python3 --version \
&& python3.11 -c "import tomllib; print('tomllib ok on', __import__('sys').version.split()[0])"
# --- PostgreSQL trust auth (OVERWRITE pg_hba; Debian default `local … peer` is first-match) ---
RUN PG_VERSION=$(ls /etc/postgresql) \
&& printf 'local all all trust\nhost all all 127.0.0.1/32 trust\nhost all all ::1/128 trust\nhost all all 0.0.0.0/0 trust\n' > "/etc/postgresql/${PG_VERSION}/main/pg_hba.conf" \
&& echo "listen_addresses='*'" >> "/etc/postgresql/${PG_VERSION}/main/postgresql.conf"
# Startup: start postgres AND mongod. printf, NOT a heredoc (colima's legacy builder writes an
# empty file from a Dockerfile heredoc → ENTRYPOINT "exec format error"). No single quotes in the
# body. mongod is backgrounded with --fork and waited on the same way pg is, so a member's
# setupCmd/startCmd never races an unready DB; its log goes to /var/log/mongod.log for triage.
RUN printf '#!/bin/bash\nset -e\nPG_VERSION=$(ls /etc/postgresql)\nsudo pg_ctlcluster ${PG_VERSION} main start\nuntil pg_isready -h localhost -p 5432 -U postgres >/dev/null 2>&1; do sleep 0.5; done\nmkdir -p /data/db\nmongod --dbpath /data/db --bind_ip 127.0.0.1 --fork --logpath /var/log/mongod.log >/dev/null 2>&1 || echo "warning: mongod failed to start, see /var/log/mongod.log"\nuntil mongosh --quiet --eval "db.runCommand({ping:1})" >/dev/null 2>&1; do sleep 0.5; done\nexec "$@"\n' > /usr/local/bin/start-services.sh \
&& chmod +x /usr/local/bin/start-services.sh
USER root
# --- Playwright + Chromium, for driving the app in a real browser -------------
# Self-contained under /opt — the member's own runtime is untouched.
ENV PLAYWRIGHT_BROWSERS_PATH=/opt/ms-playwright
RUN apt-get update -qq \
&& apt-get install -y -qq --no-install-recommends \
xz-utils \
libxcomposite1 \
libxdamage1 \
libxfixes3 \
libxrandr2 \
libasound2 \
libatk1.0-0 \
libatk-bridge2.0-0 \
libatspi2.0-0 \
libcups2 \
libdbus-1-3 \
libgbm1 \
libnspr4 \
libnss3 \
libxkbcommon0 \
libpango-1.0-0 \
libcairo2 \
libxshmfence1 \
libx11-xcb1 \
libxcb-dri3-0 \
libdrm2 \
&& rm -rf /var/lib/apt/lists/*
RUN set -eux; \
arch="$(dpkg --print-architecture)"; \
case "$arch" in amd64) nodearch=x64;; arm64) nodearch=arm64;; *) echo "unsupported arch: $arch" >&2; exit 1;; esac; \
curl -fsSL "https://nodejs.org/dist/v20.19.5/node-v20.19.5-linux-${nodearch}.tar.xz" -o /tmp/pw-node.tar.xz; \
mkdir -p /opt/pw-node; \
tar -xJf /tmp/pw-node.tar.xz -C /opt/pw-node --strip-components=1; \
rm /tmp/pw-node.tar.xz; \
export npm_config_prefix=/opt/pw-node PATH="/opt/pw-node/bin:$PATH"; \
/opt/pw-node/bin/npm install -g playwright@1.56.0; \
test -d /opt/pw-node/lib/node_modules/playwright; \
/opt/pw-node/bin/node /opt/pw-node/lib/node_modules/playwright/cli.js install chromium
# `pw <script.js>` runs Node with `require("playwright")` resolvable (CommonJS).
RUN printf '#!/bin/sh\nNODE_PATH=/opt/pw-node/lib/node_modules exec /opt/pw-node/bin/node "$@"\n' > /usr/local/bin/pw \
&& chmod +x /usr/local/bin/pw
# Fail the build if Chromium cannot start.
RUN printf 'const{chromium}=require("playwright");(async()=>{const b=await chromium.launch();const p=await b.newPage();await p.setContent("<h1 id=t>ok</h1>");if(await p.textContent("#t")!=="ok")throw new Error("bad render");await b.close();console.log("chromium OK");})()\n' > /tmp/pw-check.js \
&& pw /tmp/pw-check.js \
&& rm -f /tmp/pw-check.js
ENV IS_SANDBOX=1
RUN mkdir -p /root/.claude && echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > /root/.claude/settings.json
WORKDIR /workspace
# Resolver for the DNS jail (.devcontainer/dns-jail-container.sh, applied by
# post-start.sh); if this does not land, Explore just runs unjailed.
RUN (command -v apk >/dev/null 2>&1 && apk add --no-cache dnsmasq bind-tools) \
|| (apt-get update && apt-get install -y --no-install-recommends dnsmasq-base dnsutils \
&& rm -rf /var/lib/apt/lists/*) \
|| true
ENTRYPOINT ["/usr/local/bin/start-services.sh"]
CMD ["sleep", "infinity"]

View File

@@ -0,0 +1,24 @@
#!/bin/bash
# Trial parity: limit DNS to the model endpoint and the toolkit's telemetry, so a session
# captured here cannot depend on network the trial agent will not have. Opt-in, best-effort.
#
# Its own lifecycle step, NOT part of post-start.sh: post-create.sh calls post-start to get
# postgres up before it installs the repo's dependencies, so a jail applied there breaks
# `bundle install` on every fresh container. postStartCommand runs after postCreateCommand.
set -u
[ -f /workspace/.devcontainer/dns-jail.sh ] || exit 0
# Read the one line rather than sourcing: with the jail off this must not execute the
# worker's .env as a side effect.
if [ "${RACCOON_DNS_JAIL:-0}" != "1" ] &&
! grep -qE '^[[:space:]]*(export[[:space:]]+)?RACCOON_DNS_JAIL[[:space:]]*=[[:space:]]*"?1"?[[:space:]]*(#.*)?$' \
/workspace/.env 2>/dev/null; then
exit 0
fi
(
set -a
# shellcheck disable=SC1091
. /workspace/.env 2>/dev/null || true
set +a
bash /workspace/.devcontainer/dns-jail.sh
) || true

View File

@@ -0,0 +1,26 @@
{
"name": "Codebase Exploration (potion-polyglot)",
"initializeCommand": "node .devcontainer/initialize.js",
"build": {
"dockerfile": "Dockerfile",
"args": {
"TOOLKIT_BUILD_ID": "1789430760274-eddbpv"
}
},
"appPort": [
"${localEnv:EXPLORE_CLIENT_PORT:4300}:3000"
],
"containerEnv": {
"EXPLORE_INSTANCE": "${localEnv:EXPLORE_INSTANCE:}",
"EXPLORE_CLIENT_PORT": "${localEnv:EXPLORE_CLIENT_PORT:4300}"
},
"remoteUser": "root",
"workspaceMount": "source=${localWorkspaceFolder},target=/workspace,type=bind",
"workspaceFolder": "/workspace",
"mounts": [
"source=${localWorkspaceFolder}/repos${localEnv:EXPLORE_INSTANCE:},target=/workspace/repos,type=bind"
],
"postCreateCommand": "bash /workspace/.devcontainer/post-create.sh",
"postStartCommand": "bash /workspace/.devcontainer/post-start.sh; bash /workspace/.devcontainer/apply-dns-jail.sh",
"containerUser": "root"
}

View File

@@ -0,0 +1,137 @@
#!/bin/sh
# Restrict this container's DNS to the hosts in DNSJAIL_ALLOW (space-separated), leaving
# every other name unresolvable. Runs as root, inside the container.
#
# Baked into the task images and invoked by the agent (scripts/dnsjail.py); shipped to the
# Explore container by the toolkit packaging. Both surfaces run this same file. Supplied from
# outside: DNSJAIL_ALLOW, the hosts the agent will actually dial -- every one must resolve or
# no jail happens -- and DNSJAIL_ALLOW_EXTRA, nice-to-haves that only warn if they do not.
#
# An unreachable model endpoint is a dead trial or a dead session, so nothing here is
# applied before it is verified, and any doubt leaves the container's DNS untouched.
set -u
STATE=/tmp/.dnsjail
CONTROL=example.com # must NOT resolve through us; proves we reached our own filter
bounded() { if command -v timeout >/dev/null 2>&1; then timeout 5 "$@"; else "$@"; fi; }
# Exact match: docker's own embedded resolver is 127.0.0.11, which a prefix match reads as
# already-jailed — and then resolv.orig is never captured, so unjail has nothing to restore.
jailed_now() { grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null; }
# Stop only the dnsmasq we started, so a declined run leaves nothing bound on :53 that a
# later run could mistake for its own filter.
drop_ours() {
if [ -s "$STATE/dnsmasq.pid" ]; then
pid=$(cat "$STATE/dnsmasq.pid")
# /tmp survives docker stop/start but pids restart at 1, so last boot's pid may now be
# some service's child. Confirm it is dnsmasq before signalling it.
case "$(cat "/proc/$pid/comm" 2>/dev/null)" in
dnsmasq) kill "$pid" 2>/dev/null || true ;;
esac
rm -f "$STATE/dnsmasq.pid" 2>/dev/null || true
fi
}
# Never `exit`: a caller may source this, so bailing out has to fall through rather than
# end the caller's shell.
dnsjail_apply() {
required="${DNSJAIL_ALLOW:-}"
extra="${DNSJAIL_ALLOW_EXTRA:-}"
allow=$(echo $required $extra) # unquoted: collapses to a single-spaced word list
# A blank required list means no model endpoint was found: jailing would strand the agent.
set -- $required
[ $# -gt 0 ] || return 0
# Already jailed by us, with our resolver alive and the same allowlist? Do nothing. Tearing
# down and rebinding :53 races the kernel releasing the socket, and losing that race ends
# in a fail-open restore -- so a second apply (the codex fresh path, rejail, run-app) would
# silently UNjail a working container.
if jailed_now && [ -s "$STATE/dnsmasq.pid" ] &&
[ "$(cat "/proc/$(cat "$STATE/dnsmasq.pid")/comm" 2>/dev/null)" = "dnsmasq" ] &&
[ "$(cat "$STATE/allow" 2>/dev/null)" = "$required" ] &&
[ "$(cat "$STATE/allow-extra" 2>/dev/null)" = "$extra" ]; then
return 0
fi
# The state dir has to work first: it holds what unjail restores, and a failed write here
# is what would otherwise truncate /etc/resolv.conf. Sticky world-writable so run-app,
# running as the container user in Explore, can drop its own lift markers.
mkdir -p "$STATE" 2>/dev/null || return 0
chmod 1777 "$STATE" 2>/dev/null || true
: > "$STATE/.probe" 2>/dev/null || return 0
rm -f "$STATE/.probe" 2>/dev/null || true
# Never forward to ourselves. Re-applying to an already-jailed container would otherwise
# read 127.0.0.1 out of resolv.conf and point dnsmasq at its own socket, blackholing
# every name.
src=/etc/resolv.conf
if jailed_now && [ -s "$STATE/resolv.orig" ]; then src="$STATE/resolv.orig"; fi
up=$(awk '/^nameserver[ \t]+[0-9]+\./{print $2; exit}' "$src" 2>/dev/null)
[ "$up" = "127.0.0.1" ] && up=""
if [ -n "$up" ] && command -v dnsmasq >/dev/null 2>&1; then
srv=""
for h in $allow; do srv="$srv --server=/$h/$up"; done
drop_ours
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
# one would rather than an answer this resolver decided to keep.
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
>/dev/null 2>>"$STATE/dnsmasq.err" || true
fi
# Ask the resolver directly: the model endpoint must answer and the control must not --
# otherwise we are looking at somebody else's resolver, not our filter. Only the FIRST
# host gates the jail: an extra host that CNAMEs outside the allowlist cannot resolve
# through the catch-all, and one of those must not silently disable the whole jail.
live=1
for h in $required; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 || { live=""; break; }
done
if [ -n "$live" ] && bounded nslookup "$CONTROL" 127.0.0.1 >/dev/null 2>&1; then live=""; fi
# The extras are reported, never fatal: one that CNAMEs outside the allowlist cannot
# resolve through the catch-all, and must not take the whole jail down with it.
if [ -n "$live" ]; then
for h in $extra; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 ||
echo "dns-jail: $h does not resolve through the jail (CNAME outside the allowlist?)" >&2
done
fi
if [ -z "$live" ]; then
# Say why. A silent decline is indistinguishable from a jail that worked, and the
# reason is usually one line from dnsmasq (gVisor sandboxes, for instance, have no
# AF_NETLINK, so dnsmasq cannot start there at all).
echo "dns-jail: declined, this container keeps normal network access${DNSJAIL_WHY:-}" >&2
[ -s "$STATE/dnsmasq.err" ] && sed 's/^/dns-jail: /' "$STATE/dnsmasq.err" >&2
drop_ours
# Failing open has to mean actually open, including when an earlier run left this
# container jailed.
if jailed_now && [ -s "$STATE/resolv.orig" ]; then
cat "$STATE/resolv.orig" > /etc/resolv.conf 2>/dev/null || true
fi
return 0
fi
# Capture what unjail restores — but never overwrite it with an already-jailed file, which
# would leave unjail a permanent no-op.
if ! jailed_now; then
cp /etc/resolv.conf "$STATE/resolv.orig" 2>/dev/null || return 0
fi
printf '%s\n' "$required" > "$STATE/allow" 2>/dev/null || true
printf '%s\n' "$extra" > "$STATE/allow-extra" 2>/dev/null || true
# A marker from a run-app that was killed would otherwise keep the jail disarmed forever.
rm -rf "$STATE/lifts" 2>/dev/null || true
# /etc/resolv.conf is a bind mount, so it is truncated in place, never renamed over —
# which means the replacement has to be complete BEFORE the write starts. Keep every
# non-nameserver directive docker set (options, search).
{ printf 'nameserver 127.0.0.1\n'
grep -vE '^[[:space:]]*nameserver' /etc/resolv.conf
} > "$STATE/resolv.jailed" 2>/dev/null
[ -s "$STATE/resolv.jailed" ] || return 0
cat "$STATE/resolv.jailed" > /etc/resolv.conf
}
dnsjail_apply || true

View File

@@ -0,0 +1,78 @@
#!/bin/bash
# Apply the DNS jail to this Explore container, and install `unjail` / `rejail`.
#
# Explore is meant to behave like a trial: the session captured here becomes the trial's
# seed, so an agent that reached the network here would produce a snapshot the trial
# cannot reproduce. Same jail, applied every boot (docker remounts /etc/resolv.conf per
# start, so it cannot be baked into the image).
#
# Live resolution only — no address pinning. An Explore container can run for days, so a
# resolved-at-boot address has far longer to go stale than in a single trial.
set -u
JAIL_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
STATE=/tmp/.dnsjail
[ "${RACCOON_DNS_JAIL:-0}" = "1" ] || exit 0
# Only the model endpoint gates the jail. The toolkit's telemetry hosts go in as extras
# (below): those sends are backgrounded and disowned, so one failing to resolve would fail
# silently rather than visibly -- and must not take the whole jail down with it.
allow_hosts() {
local url="${ANTHROPIC_BASE_URL:-}" host=""
[ -n "$url" ] || return 1
host="${url#*://}"; host="${host%%/*}"; host="${host##*@}"; host="${host%%:*}"
[ -n "$host" ] || return 1
case "$host" in *[!A-Za-z0-9.-]* | -* | .* | *.) return 1 ;; esac
printf '%s' "$host"
}
install_helpers() {
sudo tee /usr/local/bin/unjail >/dev/null <<'EOF'
#!/bin/sh
# Restore this container's DNS. The jail comes back on the next container start, or now
# with `rejail`. Package installs need this; run-app does it for you around its own.
[ -f /tmp/.dnsjail/resolv.orig ] || { echo "unjail: not jailed"; exit 0; }
sudo sh -c 'cat /tmp/.dnsjail/resolv.orig > /etc/resolv.conf'
echo "unjail: DNS restored — run 'rejail' when you are done, or restart the container."
EOF
sudo tee /usr/local/bin/rejail >/dev/null <<EOF
#!/bin/sh
[ -f /tmp/.dnsjail/allow ] || { echo "rejail: nothing to restore"; exit 1; }
sudo env DNSJAIL_ALLOW="\$(cat /tmp/.dnsjail/allow)" \
DNSJAIL_ALLOW_EXTRA="\$(cat /tmp/.dnsjail/allow-extra 2>/dev/null)" \
sh $JAIL_DIR/dns-jail-container.sh
grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf && echo "rejail: jailed" || echo "rejail: could not jail — left as is"
EOF
sudo chmod +x /usr/local/bin/unjail /usr/local/bin/rejail
}
# Not fatal: an Explore container that cannot jail is still a usable Explore container.
dnsjail_off() {
mkdir -p "$STATE" 2>/dev/null || true
printf '%s\n' "$1" > "$STATE/why" 2>/dev/null || true
echo "dns-jail: off for this session — normal network access. Not an error."
exit 0
}
[ -f "$JAIL_DIR/dns-jail-container.sh" ] || dnsjail_off "script not present: $JAIL_DIR/dns-jail-container.sh"
# Jailing without the model endpoint on the allowlist would strand the agent, so a
# missing or unusable ANTHROPIC_BASE_URL means no jail at all.
ALLOW="$(allow_hosts)" || dnsjail_off "no usable host in ANTHROPIC_BASE_URL: ${ANTHROPIC_BASE_URL:-<unset>}"
# Parent domains for the telemetry, not the exact endpoints: both CNAME within their own
# domain, and the catch-all would NXDOMAIN a chain target that is not itself allowed.
sudo env DNSJAIL_ALLOW="$ALLOW" \
DNSJAIL_ALLOW_EXTRA="amplitude.com datadoghq.com ${RACCOON_DNS_JAIL_ALLOW:-}" \
sh "$JAIL_DIR/dns-jail-container.sh" || true
install_helpers
# Report what the script decided, rather than re-probing: it already verified the model
# endpoint against its own resolver and failed open if that did not hold. A second probe
# here has to pick a control host -- and any host the worker allowlists makes that control
# resolve, reading a working jail as a broken one and tearing it down.
if grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf; then
echo "dns-jail: DNS limited to the model endpoint and toolkit telemetry."
echo " Installing packages? \`unjail\` (then \`rejail\`). run-app handles its own."
else
dnsjail_off "the jail did not take; see $STATE/dnsmasq.err if present"
fi

View File

@@ -0,0 +1,195 @@
#!/usr/bin/env node
// Runs on the HOST before the container starts.
// Validates prerequisites and sets up files that the container needs
// without using ../ bind mounts (which break on newer Docker runtimes).
//
// This is Node.js (not bash) so it works on Windows without WSL.
import { execSync } from 'node:child_process';
import fs from 'node:fs';
import path from 'node:path';
// The devcontainer CLI runs initializeCommand from the workspace folder (explore/).
// Use CWD, not __dirname, so this works both in production and in tests.
if (!fs.existsSync('../.env')) {
console.error(`
❌ Missing .env file. Create it first:
Create a file named .env in the toolkit root with:
ANTHROPIC_API_KEY=your-key-here
ANTHROPIC_BASE_URL=the-base-url-you-were-given
`);
process.exit(1);
}
// Copy small files from toolkit root into explore/ so the container
// can access them without ../ bind mounts.
fs.copyFileSync('../.env', '.env');
try {
fs.copyFileSync('../toolkit.json', 'toolkit.json');
} catch {}
// Link repo so the bind mount source stays within explore/.
// Use a junction on Windows (Docker Desktop can't follow symlinks,
// but it can follow junctions). On macOS/Linux, 'junction' is ignored
// and creates a regular symlink.
// Single-repo toolkits have ../repo; polyglot toolkits have ../repos (the member
// clones) instead. Link whichever exists so the matching bind mount resolves.
// Reference-data corpus lives at ../data (zeta toolkits only).
// Remove a link WITHOUT following it: unlink covers POSIX symlinks, rmdir covers
// Windows junctions (which reject unlink). Never recursive — the target is real data.
function removeLink(name) {
try {
fs.unlinkSync(name);
} catch {
fs.rmdirSync(name);
}
}
// Replace a stale entry (a link to a path that no longer exists, an empty dir) rather
// than skipping — skipping left the bind mount resolving to nothing, unfixably.
function linkSibling(name) {
const target = path.resolve('..', name);
if (!fs.existsSync(target)) return;
let current = null;
try {
current = fs.lstatSync(name);
} catch {}
if (current) {
if (current.isSymbolicLink()) {
if (fs.existsSync(name) && fs.realpathSync(name) === fs.realpathSync(target)) return;
removeLink(name);
} else if (current.isDirectory()) {
if (fs.readdirSync(name).length > 0) {
console.error(`⚠️ explore/${name} is a non-empty directory, so it was left as is.`);
console.error(
` Expected a link to the toolkit root's ${name}/. Remove it and re-run 'up'.`
);
return;
}
fs.rmdirSync(name);
} else {
return;
}
}
fs.symlinkSync(target, name, 'junction');
}
linkSibling('repo');
linkSibling('repos');
linkSibling('data');
// A named extra instance (EXPLORE_INSTANCE set, normally by instance.js) gets
// its OWN repo working tree, mounted at /workspace/repo in that container, so a
// `git checkout` in one instance doesn't disturb another. A `git clone --local`
// hardlinks the object store, so this is cheap and fully self-contained — unlike
// a git worktree, whose gitdir lives inside the source repo and so wouldn't
// bind-mount into the container. The devcontainer.json mount derives the dir
// name from EXPLORE_INSTANCE (repo<instance>); create it before that mount binds.
// post-create.sh then checks out the default commit + runs setup in the new
// container, exactly as it does for the primary repo.
const instance = process.env.EXPLORE_INSTANCE || '';
if (instance) {
try {
if (fs.existsSync('../repos')) {
// Polyglot toolkit: give the instance its OWN copy of every member repo at
// repos<instance>/<member>, mounted at /workspace/repos. A git clone --local
// hardlinks each member's object store, so this is cheap and fully isolated —
// a member checkout in one instance never disturbs another.
const dir = `repos${instance}`;
if (!fs.existsSync(dir)) {
fs.mkdirSync(dir, { recursive: true });
for (const member of fs.readdirSync(path.resolve('../repos'))) {
const src = path.resolve('../repos', member);
if (!fs.statSync(src).isDirectory()) continue;
execSync(
`git clone --local ${JSON.stringify(src)} ${JSON.stringify(path.join(dir, member))}`,
{
stdio: 'inherit',
}
);
}
}
} else {
// Single-repo toolkit: clone repo → repo<instance>, mounted at /workspace/repo.
const dir = `repo${instance}`;
if (!fs.existsSync(dir)) {
execSync(
`git clone --local ${JSON.stringify(path.resolve('../repo'))} ${JSON.stringify(dir)}`,
{
stdio: 'inherit',
}
);
}
}
} catch {
console.error(
`\n❌ Couldn't create the repo working tree for instance "${instance}".\n` +
` This needs git on your PATH. Install git, then retry.\n`
);
process.exit(1);
}
}
// Best-effort: warn if an existing container for this folder doesn't publish
// the app ports. Docker fixes -p mappings when a container is CREATED, so a
// container built by an older toolkit (before/with different appPort) keeps its
// old mappings even when you re-run `up`. The only way to pick up new ports is
// to recreate the container — so we point that out here rather than letting the
// worker stare at a dead localhost. Wrapped so it can never block startup: any
// failure (docker missing, odd output) is swallowed and the check is skipped.
//
// Skipped for named instances: they're managed by instance.js (their own ports,
// and they carry an id-label instead of this folder's local_folder label), so
// this folder-scoped check would only ever inspect the primary container.
if (!instance)
try {
// Container ports we expect published. The browsable port is 3000 for every
// repo; Palolo also serves its API on 3001; zeta toolkits serve the corpus
// viewer on 3002. Read from toolkit.json when available, else assume the base pair.
let expected = [3000, 3001];
try {
const tk = JSON.parse(fs.readFileSync('toolkit.json', 'utf-8'));
expected = tk.explorePorts && tk.explorePorts.serverHost ? [3000, 3001] : [3000];
if (tk.explorePorts && tk.explorePorts.corpusHost) expected.push(3002);
} catch {}
const folder = process.cwd();
const ids = execSync(`docker ps -aq --filter "label=devcontainer.local_folder=${folder}"`, {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
})
.trim()
.split('\n')
.filter(Boolean);
for (const id of ids) {
const bindings = execSync(
`docker inspect --format "{{json .HostConfig.PortBindings}}" ${id}`,
{
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
}
).trim();
const missing = expected.filter((p) => !bindings.includes(`${p}/tcp`));
if (missing.length > 0) {
console.error(`
⚠️ An existing container for this folder doesn't publish port(s) ${missing.join(', ')}.
Docker fixes port mappings when a container is created, so re-running 'up'
alone won't add them. To expose the app, recreate the container:
npx @devcontainers/cli up --remove-existing-container
Note: recreating wipes the container's Claude history — run /create-snapshot
first if there's a conversation you want to keep.
`);
break;
}
}
} catch {
// docker unavailable or unexpected output — skip the check.
}

View File

@@ -0,0 +1,457 @@
#!/bin/bash
# Post-create setup for the Explore devcontainer.
set -euo pipefail
# Install every harness a worker can author with, and point each at the LLM proxy.
# Driven by scripts/harness-registry.toml, so adding a harness is a registry entry
# rather than an edit here and in the sibling container's post-create.
set -a; . /workspace/.env 2>/dev/null || true; set +a
. /workspace/scripts/setup-harnesses.sh
# Explore is where capture happens, so it is the only surface that gets the capture
# hooks — their commands ship in explore/plugins/.
RACCOON_SURFACE=explore harness_setup_all
# Allow git operations on bind-mounted repo (owned by different uid on host)
git config --global --add safe.directory '*'
# Check out the default commit from toolkit.json. SINGLE-REPO ONLY: a polyglot toolkit
# has no single /workspace/repo and no top-level defaultCommit — each member repo lives
# at /workspace/repos/<slug> and is checked out + set up lazily by run-app/setup_repo.
IS_POLYGLOT=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').polyglot?'1':'')}catch{}" 2>/dev/null || true)
if [ -z "$IS_POLYGLOT" ]; then
DEFAULT_COMMIT=$(node -e "process.stdout.write(require('/workspace/toolkit.json').defaultCommit)")
git -C /workspace/repo -c advice.detachedHead=false checkout "$DEFAULT_COMMIT"
fi
# Install repo-specific runtime deps against the live-mounted /workspace/repo.
# Bringing postgres up (and creating the role/db) lives in post-start.sh so it
# also runs on every later container start, not just first create; call it here
# so the database is ready before db:create / prisma migrate runs below.
REPO_NAME=$(node -e "process.stdout.write(require('/workspace/toolkit.json').repo)" 2>/dev/null || true)
bash /workspace/.devcontainer/post-start.sh
# Symlink ./node_modules (cwd = the dir being installed) to a container-local tree keyed by
# <key> — see the call sites below for why. The target must itself be named `node_modules`
# (Node resolves the symlink, then walks ancestors for that literal name), and its parent
# needs a stub manifest: postinstall scripts that locate the project by truncating their
# realpath at `node_modules` require() `<parent>/package.json`, and die without it.
_nm_link() {
local root="/opt/raccoon-node-modules/$1"
[ -L node_modules ] || rm -rf node_modules
mkdir -p "$root/node_modules"
[ -f "$root/package.json" ] \
|| printf '{"name":"raccoon-node-modules-root","version":"0.0.0","private":true}\n' > "$root/package.json"
ln -sfn "$root/node_modules" node_modules
}
case "$REPO_NAME" in
ZenBill-006)
# Install deps + create databases
#
# node_modules goes to a CONTAINER-LOCAL path, not the bind-mounted repo dir.
# On macOS Docker Desktop the repo is a host bind mount; writing yarn's huge,
# deeply-nested node_modules tree across the file-sharing layer exhausts the
# host open-file table -> ENFILE "file table overflow", failing the install.
# Keeping node_modules inside the Linux VM confines that churn to the VM; the
# repo stays bind-mounted (worker sees edits) and node_modules is a symlink.
# (ZenBill is yarn-classic with a single root node_modules, so one symlink
# relocates the whole tree cleanly — unlike Palolo's pnpm workspace, which
# uses copy mode instead.)
#
# The symlink TARGET must itself be named `node_modules`: Node resolves the
# symlink to its real path, then walks ancestors looking for a dir literally
# named node_modules. If the target were .../zeta-<x> (not node_modules),
# child processes spawned by postinstall scripts (e.g. cypress's `node
# index.js` requiring minimist) can't resolve hoisted deps -> MODULE_NOT_FOUND.
( cd /workspace/repo \
&& cp .env.sample .env 2>/dev/null \
&& sed -i "s/^ruby '3\.1\.2'/ruby '~> 3.1.0'/" Gemfile \
&& rm -f .ruby-version \
&& bundle install \
&& _nm_link zenbill-006 \
&& yarn install --ignore-engines \
&& bundle update jwt \
&& (bundle exec rails db:create db:migrate || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:migrate || true) )
;;
zeta-heimdall)
# API-only Rails 7; Postgres-only; no JS runtime needed. config/database.yml
# and .env are gitignored, so materialize them from the committed .example
# files. The base image is the exact pinned Ruby (3.2.1), so the Gemfile's
# ruby pin needs no loosening. --full-index works around stale-lockfile
# transitive deps (the masked repo's lockfile omits a few). db:prepare loads
# db/schema.rb into the dev DB; the test DB is created + loaded too (rspec's
# maintain_test_schema! reloads it on first run).
( cd /workspace/repo \
&& cp config/database.yml.example config/database.yml 2>/dev/null \
&& cp .env.example .env 2>/dev/null \
&& bundle install --full-index \
&& (bundle exec rails db:prepare || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:schema:load || true) )
;;
zeta-platform)
# Rails 5.1 / Ruby 2.6.6 banking monorepo; Postgres + Redis. config/database.yml
# is committed (only .env is gitignored → copy from .env.example for dotenv).
# Bundler 1.17.3 matches the lockfile (installed in the image), and the base is
# the exact pinned Ruby (2.6.6), so no Gemfile loosening. db:schema:load loads
# db/schema.rb into the dev + test DBs.
( cd /workspace/repo \
&& cp .env.example .env 2>/dev/null \
&& bundle install \
&& (bundle exec rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:schema:load || true) )
# React client (Create React App, react-scripts 2.1.1). Install its JS deps so
# `run-app` can boot the full UI (dev server proxies /graphql → the Rails API).
# node_modules goes to a CONTAINER-LOCAL path, not the bind-mounted repo dir:
# on macOS Docker Desktop the repo is a host bind mount, and writing CRA's huge
# node_modules tree across the file-sharing layer is slow AND exhausts the host's
# open-file table. Keeping it inside the Linux VM confines that churn; the repo
# stays bind-mounted (worker sees edits) and node_modules is a symlink. The
# symlink TARGET must itself be named `node_modules` (Node's resolver walks
# parents looking for a dir literally named node_modules). yarn is v1 (classic),
# matching the committed yarn.lock.
( cd /workspace/repo \
&& _nm_link zeta-platform \
&& yarn install --frozen-lockfile )
;;
Palolo-031)
# Install deps. packages/server/scripts/prisma greps `.env` for
# PUBLIC_PALOLO_ENV inside an `if [ -t 0 ]` block — designed for
# interactive use where the dev's local .env points at staging/prod
# and the script wants confirmation before destructive ops. In a
# fresh clone the file doesn't exist, so the grep fails and `set -e`
# aborts. We materialize a `local`-pointing stub so the script
# finds what it expects, the safety check skips correctly (env is
# local, no confirmation needed), and downstream interactive worker
# invocations of `pnpm run prisma …` also succeed instead of hitting
# the same failure.
( cd /workspace/repo \
&& git config core.hooksPath /dev/null \
&& echo "PUBLIC_PALOLO_ENV=local" > packages/server/.env \
&& pnpm install --frozen-lockfile \
&& pnpm run --dir packages/server prisma generate \
&& (pnpm run --dir packages/server prisma migrate deploy || true) )
# Seed the dev DB with a superuser, the global/superuser orgs, and a set
# of test users so a worker can actually log in when running the app
# locally. Without this the schema exists but every table is empty, and
# the login screen errors out before you can get into the app. Test
# users are <name>@exhalefi.com with password "test" (e.g. zaniyah@exhalefi.com).
# Convenience only — wrapped in `|| true` so a seed hiccup never blocks
# the explore container from coming up.
#
# `--small` keeps every organization the seed builds but caps each at 10
# members per status. The default size gives the last one 200 per status,
# which opens 200 concurrent Prisma interactive transactions and exhausts
# the connection pool (`P2028`) on a machine with few cores, so the seed
# dies partway and leaves perks un-activated.
( cd /workspace/repo/packages/server \
&& DEFAULT_BAAS_PROVIDER=Liquid PUBLIC_BAAS_ENABLED=yes TESTING_SEED=yes \
pnpm run seed --small ) || true
# Leave a fresh container's `git status` clean. The two artifacts below
# are side effects of bootstrap, not edits anyone made:
#
# 1. .pnpm-store/ — pnpm's content-addressable store. It must sit on the
# same filesystem as node_modules to hardlink; /workspace/repo is a
# bind mount on a different fs than HOME, so pnpm can't use the global
# ~/.pnpm-store and drops a project-local store instead. The repo's
# .gitignore covers it as of commit 3af4366a6, but older commits a
# worker may check out don't. Exclude it locally too (idempotent;
# the create-snapshot checkpoint hook excludes it as well).
# 2. deploy_to_eks.sh — the repo's only symlink (-> ../scripts/...). The
# toolkit's zip/unzip packaging path materializes it as a regular file,
# so git reports a "typechange". Restore the symlink from the index
# (no-op if the filesystem can't represent symlinks).
grep -qxF '.pnpm-store/' /workspace/repo/.git/info/exclude 2>/dev/null \
|| printf '\n# raccoon-explore: in-repo pnpm store (bind-mount hardlink fallback)\n.pnpm-store/\n' >> /workspace/repo/.git/info/exclude
git -C /workspace/repo checkout -- provisioning/kubernetes/palolo-app/deploy_to_eks.sh 2>/dev/null || true
;;
human-essentials)
# Rails 8 / Ruby 3.4; pure importmap (no JS bundler → no node_modules). The
# base image is exact Ruby 3.4.3, so no Gemfile loosening. .env is gitignored;
# copy the committed .env.example (public reCAPTCHA test keys etc.) for dotenv,
# then drop its empty PG_USERNAME/PG_PASSWORD lines so they don't override the
# image ENV (PG_USERNAME=postgres). db:schema:load loads db/schema.rb into the
# dev + test DBs; assets:precompile is needed by the Cuprite system specs.
# db:seed (dev, offline via Faker) gives a working login out of the box — the app
# has no usable self-service signup (a fresh user lands org-less/role-less).
( cd /workspace/repo \
&& cp .env.example .env 2>/dev/null || true; \
sed -i '/^PG_USERNAME=/d; /^PG_PASSWORD=/d' .env 2>/dev/null || true; \
bundle install \
&& (bundle exec rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:schema:load || true) \
&& (bundle exec rails db:seed || true) \
&& (bundle exec rails assets:precompile || true) )
;;
endsideout)
# Rails 8.1 / Ruby 4.0; SQLite + importmap (no Node build — tailwindcss-rails
# ships its own binary). No .env (no .env.example; tests need no secrets). The
# SQLite dev + test DBs are plain files created by db:prepare / db:test:prepare.
# db:seed (dev, offline) creates admin@example.com / password — there is no
# self-service signup route, so seeding is the only way into the UI.
# tailwindcss:build writes the gitignored app/assets/builds/ the layout links;
# run-app starts the server alone, without Procfile.dev's tailwindcss:watch.
( cd /workspace/repo \
&& bundle install \
&& (bin/rails db:prepare || true) \
&& (bin/rails db:test:prepare || true) \
&& (bin/rails db:seed || true) \
&& (bin/rails tailwindcss:build || true) )
;;
community-foundation)
# Rails 8.1 / Ruby 4.0; SQLite + importmap + tailwind (no Node). Encrypted
# credentials aren't needed for tests. SQLite dev + test DBs.
# db:seed (dev, offline) creates the 'arlington' tenant + owner@example.com /
# password. Self-signup is a dead end here (needs a pre-existing org + a working
# mailer for confirmation), so seeding is the only offline way into the UI. The
# app is subdomain-multi-tenant — reach the tenant at arlington.lvh.me, not plain
# localhost (see welcome.sh).
# tailwindcss:build writes the gitignored app/assets/builds/ the layout links;
# run-app starts the server alone, without Procfile.dev's tailwindcss:watch.
( cd /workspace/repo \
&& bundle install \
&& (bin/rails db:prepare || true) \
&& (bin/rails db:test:prepare || true) \
&& (bin/rails db:seed || true) \
&& (bin/rails tailwindcss:build || true) )
;;
stocks-in-the-future)
# Rails 8.1 / Ruby 3.4.4; Postgres + Redis; importmap (no Node build).
# config/database.yml is gitignored — materialize from the committed sample.
# PGHOST/PGUSER (set in the image) point rails at the postgres superuser.
# db:seed (dev, offline) creates login-by-username accounts (Admin / password);
# self-signup is disabled (GET /users/sign_up redirects to /), so seed to get in.
# tailwindcss:build writes the gitignored app/assets/builds/ the layout links;
# run-app starts the server alone, without Procfile.dev's tailwindcss:watch.
( cd /workspace/repo \
&& (cp config/database.yml.sample config/database.yml 2>/dev/null || true) \
&& bundle install \
&& (bin/rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bin/rails db:create db:schema:load || true) \
&& (bin/rails db:seed || true) \
&& (bin/rails tailwindcss:build || true) )
;;
casa)
# Rails 8.0 / Ruby 4.0.3; Postgres + Node 24 (jsbundling: esbuild + sass).
# DB env (POSTGRES_USER/DATABASE_HOST/POSTGRES_PASSWORD) is pinned in the image.
# npm ci installs JS deps; `npm run build` + `build:css` (esbuild + sass) write the
# bundles to app/assets/builds. The Selenium system specs serve from there because
# the test env runs with config.assets.compile=true (Sprockets compiles on demand).
# Deliberately NOT `assets:precompile`: that fingerprints untracked copies into
# public/assets which the specs don't need and which make `npm run lint` (standard)
# report ~197k errors over machine-generated bundles. app/assets/builds is already in
# standard's ignore list, so the dev build leaves the tree lint-clean and faithful.
# db:seed (dev, offline via Faker + local logo) creates casa_admin1@example.com /
# 12345678 — users are admin-invited only (ADR 0002), so seeding is the way in.
( cd /workspace/repo \
&& (cp .env.example .env 2>/dev/null || true) \
&& bundle install \
&& npm ci \
&& (bin/rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bin/rails db:create db:schema:load || true) \
&& (bin/rails db:seed || true) \
&& (npm run build && npm run build:css || true) )
;;
awbw)
# Rails 8.1 / Ruby 4.0.1; MySQL 8 (Percona, Trilogy) + Node 22 (Vite). .env from
# .env.sample; DATABASE_URL (image) points Trilogy at 127.0.0.1 root. npm ci + a
# test-mode Vite build for the Selenium system specs.
# Use db:schema:load (NOT migrate): the committed schema.rb is clean native-MySQL-8
# JSON; running migrate re-dumps schema.rb from the live DB (which corrupts it under
# a non-MySQL-8 engine). tz tables are loaded by post-start.sh (Ahoy charts need them).
# db:seed (dev, offline; the seed disables mailer delivery itself) creates the
# pre-confirmed umberto.user@example.com / password super_user — no self-service
# signup exists and :confirmable would block a hand-made user without a mailer.
( cd /workspace/repo \
&& (cp .env.sample .env 2>/dev/null || true) \
&& bundle install \
&& npm ci \
&& (bin/vite build --mode test || true) \
&& (bin/rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bin/rails db:create db:schema:load || true) \
&& (bin/rails db:seed || true) )
;;
alongwithyou)
# Rails 8.1 / Ruby 4.0.5; SQLite + importmap (no app-side Node). No .env / credentials
# needed to boot. This is a young app (a fresh scaffold with no migrations yet), so
# db:prepare just materializes an empty dev/test DB; db:seed is a no-op on the default
# seeds.rb. All wrapped in `|| true` so an empty schema never blocks container startup.
( cd /workspace/repo \
&& bundle install \
&& (bin/rails db:prepare || true) \
&& (bin/rails db:test:prepare || true) \
&& (bin/rails db:seed || true) )
;;
flaredown)
# Polyglot: backend/ Rails 7.1 (Ruby 3.2.3, Mongoid on MongoDB + Postgres + Redis +
# Sidekiq) and frontend/ Ember (Node 14). Postgres/Mongo/Redis are started by
# post-start.sh (called above). .env is gitignored — materialize from the committed
# backend/env-example (public dev secrets). Mongoid creates collections lazily, so
# there's no Mongo schema to load; Postgres holds a small relational slice with a
# committed db/schema.rb → db:schema:load (NOT db:migrate, which re-dumps schema.rb
# from the live DB on a bind-mounted repo).
# env-example points PG at host `postgresql` (the docker-compose service name); in this
# single container everything is on localhost, so rewrite the PG host. Redis defaults to
# localhost already; Mongoid reads MONGODB_HOST (unset → localhost).
( cd /workspace/repo/backend \
&& (cp -n env-example .env 2>/dev/null || true) \
&& sed -i 's/^PG_DATABASE_HOST=.*/PG_DATABASE_HOST=localhost/' .env 2>/dev/null || true; \
bundle install \
&& (bundle exec rails db:create db:schema:load || true) \
&& (RAILS_ENV=test bundle exec rails db:create db:schema:load || true) )
# Ember frontend on Node 14 (frontend/.nvmrc = v14.21.3; npm pinned to 6 in the image).
# node_modules to a container-local symlink (bind-mount file-sharing exhausts the host fd
# table on big node_modules trees). OPENSSL_CONF=/dev/null lets the old webpack md4 hashing
# run on bookworm's OpenSSL 3. --unsafe-perm so npm (running as root) actually executes the
# postinstall (patch-package + bower install) instead of skipping it with a "cannot run in
# wd" warning; without it bower_components is never populated and `ember build` fails.
NODE14_BIN=$(ls -d /usr/local/nvm/versions/node/v14.* 2>/dev/null | sort -V | tail -1)/bin
( cd /workspace/repo/frontend \
&& export PATH="$NODE14_BIN:$PATH" OPENSSL_CONF=/dev/null \
&& _nm_link flaredown-frontend \
&& (npm install --unsafe-perm --no-audit --no-fund || echo "WARNING: frontend npm install failed (explore-only)" >&2) ) || true
;;
breezy-complete)
# Monorepo: Rails 7.0 / Ruby 3.2.0 API (backend/) + Next.js 14 frontend
# (frontend/); Postgres + Redis baked in the image. The offline Clerk-bypass
# env is injected by run-app at server start only — the ambient env stays
# upstream-CI-shaped so a worker's `cd backend && bundle exec rspec` runs
# green (ambient DISABLE_CLERK 403s several controller specs, and ambient
# RAILS_ENV leaks through rails_helper's `ENV['RAILS_ENV'] ||= 'test'`).
#
# backend: gems + yarn asset-pipeline deps; db:prepare (retried once — the
# first run can race the just-started postgres) + db:seed (offline-safe demo
# tenant; the only way into the UI, auth is invite-less) + test DB. Fresh-DB
# db:test:prepare trips check_protected_environments → stamp the env first.
# db:prepare seeds the DB it creates and the seeds are not idempotent, so the
# explicit db:seed is for the retry case only — skip it on a seeded DB.
# frontend: npm install (not ci) so platform-specific optional deps resolve
# on arm64 + x64. Both node_modules go to CONTAINER-LOCAL paths via symlink
# (bind-mount ENFILE; see the ZenBill comment above — target must itself be
# named node_modules).
( cd /workspace/repo/backend \
&& bundle install --jobs 4 --retry 3 \
&& _nm_link breezy-backend \
&& yarn install --frozen-lockfile \
&& (bundle exec rails db:prepare || bundle exec rails db:prepare) \
&& ( psql -tAc 'select 1 from breezy_professionals limit 1' socratic_systems_development 2>/dev/null | grep -q 1 \
|| bundle exec rails db:seed || true ) \
&& (RAILS_ENV=test bundle exec rails db:environment:set || true) \
&& (RAILS_ENV=test bundle exec rails db:test:prepare || true) )
( cd /workspace/repo/frontend \
&& _nm_link breezy-frontend \
&& npm install --include=optional )
;;
esac
# Mirror Harbor's reduced toolset in the interactive Explore session. Use
# Harbor's /opt path when available, but fall back to a user-writable path for
# generic devcontainer fixtures that run lifecycle hooks as a non-root user.
AGENT_CLI_DIR="/opt/agent-cli"
if ! mkdir -p "$AGENT_CLI_DIR" 2>/dev/null; then
AGENT_CLI_DIR="$HOME/.agent-cli"
mkdir -p "$AGENT_CLI_DIR"
fi
cp -R /workspace/scripts/str_replace_editor /workspace/scripts/str_replace_editor_vendor "$AGENT_CLI_DIR/"
chmod +x "$AGENT_CLI_DIR/str_replace_editor"
mkdir -p "$HOME/.local/bin"
# Explore launchers (one per authoring harness) come from setup-harnesses.sh,
# which reads harness-registry.toml. AGENT_CLI_DIR is where the reduced-toolset
# editor was staged above, and the launcher rewrites the toolset note to match.
AGENT_CLI_DIR="$AGENT_CLI_DIR" harness_install_launchers
mkdir -p "$HOME/.claude"
# SKIP_FAST_MODE_NETWORK_ERRORS: the LLM proxy doesn't forward claude's fast-mode
# availability probe, and claude reads the failed probe as "no network" and refuses
# /fast. The override makes /fast toggleable; fast serving stays OFF until toggled.
# Names this container as the surface a proxy call came from: claude reads it from the
# settings env below, codex from the shell (its config maps the header to this var name).
# Only on the proxy — a provider endpoint must not carry this attribution.
CALL_METADATA=""
case "$(set -a; . /workspace/.env 2>/dev/null || true; set +a; printf '%s' "${ANTHROPIC_BASE_URL:-}")" in
*/llm_proxy/*) CALL_METADATA='{"origin":"explore"}' ;;
esac
CALL_METADATA="$CALL_METADATA" node -e '
const fs = require("fs");
const home = process.env.HOME;
const env = {
CLAUDE_CODE_DISABLE_AUTO_MEMORY: "1",
CLAUDE_CODE_SKIP_FAST_MODE_NETWORK_ERRORS: "1",
};
if (process.env.CALL_METADATA) {
env.ANTHROPIC_CUSTOM_HEADERS =
`X-Surge-Client-Metadata: ${process.env.CALL_METADATA}`;
}
fs.writeFileSync(
`${home}/.claude/settings.json`,
JSON.stringify({ env }, null, 2) + "\n"
);
'
# Reference-data corpus: expose it at the stable /data/zeta-corpus path (the same path a trial
# uses) by symlinking to the toolkit's bind-mounted copy. No-op if this toolkit ships no corpus.
if [ -d /workspace/data/zeta-corpus ]; then
{ mkdir -p /data || sudo mkdir -p /data; } 2>/dev/null || true
{ ln -sfn /workspace/data/zeta-corpus /data/zeta-corpus \
|| sudo ln -sfn /workspace/data/zeta-corpus /data/zeta-corpus; } 2>/dev/null || true
fi
# Shell setup
# Ahead of the block below so every shell exports it, interactive or not.
if [ -n "$CALL_METADATA" ]; then
# A file rather than a .bashrc line alone: the launcher and refresh-harness-auth
# read it too, so a non-login shell does not silently lose the attribution.
printf 'export LLM_CALL_METADATA=%s\n' "'$CALL_METADATA'" > ~/.raccoon-call-origin
printf '. "$HOME/.raccoon-call-origin"\n' >> ~/.bashrc
fi
cat >> ~/.bashrc <<'BASHRC'
export PATH="$HOME/.local/bin:$PATH"
set -a && source /workspace/.env && set +a
# Everything below this line is for interactive shells only. An agent's shell tool
# sources .bashrc too, so without this guard the welcome banner prints into command
# output and container_start fires once per command instead of once per session.
case $- in
*i*) ;;
*) return ;;
esac
alias run-app="bash /workspace/run-app.sh"
[ -f /workspace/corpus-viewer/view-corpus.sh ] && alias view-corpus="bash /workspace/corpus-viewer/view-corpus.sh"
export PS1="\[\033[1;36m\][raccoon-explore]\[\033[0m\] \w\$ "
bash /workspace/welcome.sh explore 2>/dev/null
_AK="fde503c3bdb6e5cc9c48b1f8e4c2abeb"
_DK="e966e45af5ad1a18005f9fdb831186ea"
_WID="w-mu1wvj9k-gtnk"
_VER="037cfcf94b"
_CT="explore"
_RP=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').repo)}catch{}" 2>/dev/null)
_SID="$(date +%s)-$$"
_LAT=0
_ev() {
[ -z "$_AK" ] && return
{ curl -s -X POST "https://api2.amplitude.com/2/httpapi" \
-H "Content-Type: application/json" \
-d "{\"api_key\":\"$_AK\",\"events\":[{\"user_id\":\"$_WID\",\"event_type\":\"raccoon.$1\",\"event_properties\":{\"product\":\"raccoon\",\"container\":\"$_CT\",\"repo\":\"$_RP\",\"toolkit_version\":\"$_VER\",\"session_id\":\"$_SID\"},\"session_id\":$(date +%s000)}]}" \
>/dev/null 2>&1 & } 2>/dev/null; disown 2>/dev/null
}
_dl() {
[ -z "$_DK" ] && return
{ curl -s -X POST "https://http-intake.logs.datadoghq.com/api/v2/logs" \
-H "DD-API-KEY: $_DK" -H "Content-Type: application/json" \
-d "[{\"ddsource\":\"raccoon\",\"service\":\"toolkit\",\"hostname\":\"$(hostname)\",\"status\":\"$1\",\"message\":\"$2\",\"ddtags\":\"container:$_CT,worker:$_WID,repo:$_RP,toolkit_version:$_VER\"}]" \
>/dev/null 2>&1 & } 2>/dev/null; disown 2>/dev/null
}
_pc() { local n; n=$(date +%s); if (( n - _LAT >= 300 )); then _LAT=$n; _ev active; fi; }
PROMPT_COMMAND="_pc;${PROMPT_COMMAND:-}"
trap '_ev container_stop; _dl info container_stop; wait' EXIT
_ev container_start
_dl info container_start
BASHRC
# One alias per authoring harness: `claude` runs claude, `codex` runs codex.
harness_alias_lines >> ~/.bashrc

View File

@@ -0,0 +1,222 @@
#!/bin/bash
# Post-start setup for the Explore devcontainer.
#
# This runs on EVERY container start (wired as `postStartCommand` in
# devcontainer.json), unlike post-create.sh which runs only once when the
# container is first created. Its job is the lightweight work that has to
# happen on every boot: bring PostgreSQL back up. The heavy one-time work
# (installing dependencies, creating + migrating the database, seeding) stays
# in post-create.sh.
#
# Why this is needed: the container is started with an entrypoint that bypasses
# the image's own startup script, so nothing restarts postgres for you. After
# you stop the container or reboot your machine, postgres stays down until this
# script runs — previously you had to start it by hand every session.
#
# Safe to run repeatedly: if postgres is already accepting connections, the
# start step is skipped and this is effectively a no-op.
set -euo pipefail
REPO_NAME=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').repo)}catch{}" 2>/dev/null || true)
# Dead-end the deployed hostnames this estate's sources still name, so booting an app with a
# non-local environment setting can't send a login form (or anything else) to a live host. Has to
# happen on every start, not in the image: Docker remounts /etc/hosts per container, so a
# Dockerfile write to it never survives.
BLOCKED_HOSTS=$(node -e "try{process.stdout.write((require('/workspace/toolkit.json').blockedHosts||[]).join(' '))}catch{}" 2>/dev/null || true)
if [ -n "$BLOCKED_HOSTS" ] && ! grep -q "raccoon-blocked-hosts" /etc/hosts 2>/dev/null; then
printf '# raccoon-blocked-hosts\n127.0.0.1 %s\n::1 %s\n' "$BLOCKED_HOSTS" "$BLOCKED_HOSTS" \
| sudo tee -a /etc/hosts >/dev/null 2>&1 \
|| echo "warning: could not pin blocked hosts in /etc/hosts" >&2
fi
# Corpus viewer: when this toolkit ships a corpus search index, serve the viewer on
# container port 3002 (published as EXPLORE_CORPUS_PORT on the host). Only
# corpus-shipping toolkits package the viewer at all; where present, the script
# self-guards (no index / no python3 / already running → quiet no-op) and must never
# block container startup.
[ -f /workspace/corpus-viewer/view-corpus.sh ] &&
bash /workspace/corpus-viewer/view-corpus.sh start --quiet || true
# True when postgres is up and answering queries.
pg_ready() { sudo -u postgres psql -c "SELECT 1" >/dev/null 2>&1; }
# Block until postgres is ready, but never hang the container start forever:
# pg_isready alone races on cluster startup, so we poll an actual query with a
# bounded number of attempts (60s) and move on with a warning if it never comes
# up rather than wedging `devcontainer up`.
wait_for_pg() {
local n=0
until pg_ready; do
sleep 0.5
n=$((n + 1))
if [ "$n" -ge 120 ]; then
echo "warning: postgres did not become ready within 60s" >&2
return 0
fi
done
}
# Polyglot toolkit: REPO_NAME is empty (no single repo). Bring up Postgres + Redis
# (members need them; per-member DB setup is deferred to run-app/setup_repo), then done.
IS_POLYGLOT=$(node -e "try{process.stdout.write(require('/workspace/toolkit.json').polyglot?'1':'')}catch{}" 2>/dev/null || true)
if [ -n "$IS_POLYGLOT" ]; then
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
wait_for_pg
sudo -u postgres psql -c "ALTER USER postgres PASSWORD 'secret_password';" >/dev/null 2>&1 || true
# Many members' committed .env / database.yml default the DB username to 'root'
# (dotenv-rails applies it at boot, overriding DEV_DB_USERNAME=postgres). The image's
# start-services.sh creates a root superuser, but devcontainers override the ENTRYPOINT so
# it never runs — create root here too, mirroring the harbor task env.
sudo -u postgres psql -c "CREATE ROLE root SUPERUSER LOGIN PASSWORD 'secret_password';" >/dev/null 2>&1 || true
# MongoDB, for the polyglot images that bake it (potion's flagship member stores everything
# in Mongo). `command -v mongod` is the switch, so the Mongo-less polyglot images skip this
# untouched. It has to happen here for the same reason postgres does — the devcontainer
# overrides the image ENTRYPOINT, so the baked start-services.sh never runs — and it matters
# more than a stopped postgres would: mongoose BUFFERS operations while disconnected instead
# of erroring, so a member whose Mongo is down doesn't fail loudly, it serves requests that
# hang forever and pages that never finish rendering. Readiness is a dependency-free TCP
# probe (no mongosh needed; mongoose connects lazily once the port is open).
if command -v mongod >/dev/null 2>&1; then
mongo_up() { (exec 3<>/dev/tcp/127.0.0.1/27017) 2>/dev/null && { exec 3>&- 3<&-; return 0; }; return 1; }
if ! mongo_up; then
sudo mkdir -p /data/db 2>/dev/null || mkdir -p /data/db 2>/dev/null || true
sudo chown -R "$(id -u)":"$(id -g)" /data/db 2>/dev/null || true
mongod --dbpath /data/db --bind_ip 127.0.0.1 --fork --logpath /tmp/mongod.log >/dev/null 2>&1 \
|| (sudo -b mongod --dbpath /data/db --bind_ip 127.0.0.1 --logpath /var/log/mongod.log >/dev/null 2>&1) || true
for _ in $(seq 1 60); do mongo_up && break; sleep 0.5; done
mongo_up || echo "warning: mongod did not come up within 30s" >&2
fi
fi
# OpenSearch, for the polyglot images that bake it — same reason as mongod (the devcontainer
# overrides the ENTRYPOINT, so the image's start-services.sh never runs). Presence of the
# binary is the switch, so images without it are untouched. Flags mirror the trial image's
# start-services block exactly; it runs as its own user because OpenSearch refuses to boot
# as root. Non-fatal: a member that doesn't use it shouldn't be blocked by a slow JVM.
if [ -x /opt/opensearch/bin/opensearch ]; then
os_up() { curl -s --max-time 2 localhost:9200 >/dev/null 2>&1; }
if ! os_up; then
sudo -u opensearch env OPENSEARCH_JAVA_OPTS='-Xms512m -Xmx512m' \
/opt/opensearch/bin/opensearch -Ediscovery.type=single-node \
-Eplugins.security.disabled=true >/tmp/opensearch.log 2>&1 &
for _ in $(seq 1 90); do os_up && break; sleep 2; done
os_up || echo "warning: opensearch did not come up within 180s (see /tmp/opensearch.log)" >&2
fi
fi
return 0 2>/dev/null || exit 0
fi
case "$REPO_NAME" in
ZenBill-006)
# Start is non-fatal: if it fails outright, wait_for_pg is the single
# gate — it warns and continues rather than aborting `devcontainer up`
# and leaving the worker with no shell.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
wait_for_pg
sudo -u postgres psql -c "ALTER USER postgres PASSWORD 'secret_password';" >/dev/null 2>&1 || true
;;
zeta-heimdall)
# Bookworm base → `service postgresql start` (same as ZenBill). Non-fatal
# start; wait_for_pg is the single gate so a hiccup warns rather than wedging
# `devcontainer up` and leaving the worker with no shell.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
wait_for_pg
sudo -u postgres psql -c "ALTER USER postgres PASSWORD 'secret_password';" >/dev/null 2>&1 || true
;;
zeta-platform)
# Postgres + Redis (sidekiq). Start both; wait_for_pg is the single gate.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
wait_for_pg
sudo -u postgres psql -c "ALTER USER postgres PASSWORD 'secret_password';" >/dev/null 2>&1 || true
;;
Palolo-031)
if ! pg_ready; then
PG_VERSION=$(pg_config --version | grep -oP '\d+' | head -1)
sudo pg_ctlcluster "${PG_VERSION}" main start || echo "warning: 'pg_ctlcluster ${PG_VERSION} main start' failed" >&2
fi
wait_for_pg
sudo -u postgres psql -c "CREATE USER test WITH SUPERUSER PASSWORD 'test';" >/dev/null 2>&1 || true
sudo -u postgres psql -c "CREATE DATABASE palolo OWNER test;" >/dev/null 2>&1 || true
;;
human-essentials)
# Bookworm base → `service postgresql start` (same as ZenBill). Non-fatal
# start; wait_for_pg is the single gate. Trust auth (set in the image), so
# no role password to seed — the app connects as postgres with no password.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
wait_for_pg
;;
stocks-in-the-future)
# Postgres + Redis (background jobs). Start both; wait_for_pg is the gate.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
wait_for_pg
;;
casa)
# Postgres only. Non-fatal start; wait_for_pg is the gate.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
wait_for_pg
;;
awbw)
# MySQL 8 (Percona). Start it (init the data dir first if empty), then ensure root
# is passwordless over TCP (mysql_native_password) for Trilogy. Self-contained —
# the pg_ready/wait_for_pg helpers above are Postgres-specific.
if ! mysqladmin ping >/dev/null 2>&1; then
sudo mkdir -p /var/run/mysqld && sudo chown -R mysql:mysql /var/run/mysqld /var/lib/mysql 2>/dev/null || true
[ -d /var/lib/mysql/mysql ] || sudo mysqld --initialize-insecure --user=mysql --datadir=/var/lib/mysql 2>/dev/null || true
sudo service mysql start >/dev/null 2>&1 || (sudo mysqld_safe --user=mysql >/dev/null 2>&1 &) || echo "warning: mysql start failed" >&2
fi
for i in $(seq 1 120); do mysqladmin ping >/dev/null 2>&1 && break; sleep 0.5; done
mysql -u root -e "ALTER USER 'root'@'localhost' IDENTIFIED WITH mysql_native_password BY ''; CREATE USER IF NOT EXISTS 'root'@'%' IDENTIFIED WITH mysql_native_password BY ''; GRANT ALL PRIVILEGES ON *.* TO 'root'@'localhost' WITH GRANT OPTION; GRANT ALL PRIVILEGES ON *.* TO 'root'@'%' WITH GRANT OPTION; FLUSH PRIVILEGES;" >/dev/null 2>&1 || true
# Load MySQL tz tables (Ahoy charts use Groupdate/CONVERT_TZ). One-time.
[ "$(mysql -u root -N -e 'SELECT COUNT(*) FROM mysql.time_zone_name' 2>/dev/null || echo 0)" -gt 0 ] \
|| mysql_tzinfo_to_sql /usr/share/zoneinfo 2>/dev/null | mysql -u root mysql 2>/dev/null || true
;;
flaredown)
# Three datastores: Postgres (relational slice) + Redis (Sidekiq) + MongoDB (Mongoid,
# the primary store). Start all three; wait_for_pg gates the Postgres readiness, and
# we poll mongod separately. All starts are non-fatal so a hiccup warns rather than
# wedging `devcontainer up`. Trust auth on Postgres (set in the image).
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
# MongoDB (server tarball → bin/mongod on PATH; no service unit). Launch mongod against
# a data dir if nothing is already listening on 27017. Readiness is a dependency-free
# TCP probe (no mongosh needed — Mongoid connects lazily once the port is open).
mongo_up() { (exec 3<>/dev/tcp/127.0.0.1/27017) 2>/dev/null && { exec 3>&- 3<&-; return 0; }; return 1; }
if ! mongo_up; then
sudo mkdir -p /data/db 2>/dev/null || mkdir -p /data/db 2>/dev/null || true
sudo chown -R "$(id -u)":"$(id -g)" /data/db 2>/dev/null || true
mongod --dbpath /data/db --bind_ip 127.0.0.1 --fork --logpath /tmp/mongod.log >/dev/null 2>&1 \
|| (sudo -b mongod --dbpath /data/db --bind_ip 127.0.0.1 --logpath /var/log/mongod.log >/dev/null 2>&1) || true
fi
wait_for_pg
for _ in $(seq 1 60); do mongo_up && break; sleep 0.5; done
;;
breezy-complete)
# Postgres + Redis (Sidekiq). Start both; wait_for_pg is the gate. Trust
# auth (set in the image) — PGPASSWORD is baked but inert, no role seeding.
if ! pg_ready; then
sudo service postgresql start || echo "warning: 'service postgresql start' failed" >&2
fi
sudo service redis-server start >/dev/null 2>&1 || sudo redis-server --daemonize yes >/dev/null 2>&1 || true
wait_for_pg
;;
esac