lots of change - all to start my 3rd redo

This commit is contained in:
2026-09-26 14:31:52 -04:00
parent 7f4d388e19
commit bceb52e8ee
1046 changed files with 4476 additions and 0 deletions

View File

@@ -1,34 +0,0 @@
# shellcheck shell=bash
# call-origin.sh — build the X-Surge-Client-Metadata header value.
#
# Which surface a proxy call came from (a trial agent, the grader, Explore, a
# dev box). Separate from llm-proxy-env.sh, which carries the project id and the
# proxy routes: those are platform-internal, this is not, so this file is the
# half that ships in the worker toolkit — worker runs go through the same proxy
# and are attributed the same way.
#
# . scripts/lib/call-origin.sh
# meta="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
#
# The proxy rejects the WHOLE CALL over a malformed metadata header (400
# invalid_client_metadata), so an origin that is not a plain slug yields an
# empty string and the caller sends no header at all: losing attribution beats
# failing the call.
CALL_ORIGIN_HEADER="X-Surge-Client-Metadata"
# An unlabelled call is still a real call, so it gets a bucket rather than no
# header: a missing origin in the audit log then means an unplumbed surface.
DEFAULT_CALL_ORIGIN="local"
# Compact JSON for the header, or empty when LLM_CALL_ORIGIN is unusable.
# Only a slug matching this pattern is ever interpolated, so nothing needs
# JSON-escaping and this stays dependency-free (it is sourced in worker
# containers, which have no python).
call_origin_metadata() {
local origin="${LLM_CALL_ORIGIN:-$DEFAULT_CALL_ORIGIN}"
case "$origin" in
"" | *[!a-z0-9._-]* | [!a-z0-9]*) return 0 ;;
esac
[ "${#origin}" -le 64 ] || return 0
printf '{"origin":"%s"}' "$origin"
}

View File

@@ -1,19 +0,0 @@
import { existsSync } from 'node:fs';
(function checkDevcontainer() {
if (process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE === '1') return;
process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE = '1';
const inContainer = process.env.IN_DEVCONTAINER === '1' || existsSync('/.dockerenv');
if (inContainer) return;
if (process.env.SUPPRESS_DEVCONTAINER_WARNING === '1') return;
if (process.env.CI === 'true' || process.env.CI === '1') return;
const yellow = '\x1b[33m';
const reset = '\x1b[0m';
process.stderr.write(
`${yellow}Warning: this script is meant to run inside the toolkit devcontainer.${reset}\n` +
` Reopen this toolkit folder in its devcontainer and run the command again.\n` +
` (suppress with SUPPRESS_DEVCONTAINER_WARNING=1)\n`
);
})();

View File

@@ -1,36 +0,0 @@
"""How codex is handed its API key, kept out of codex_agent so it is testable without
harbor (whose venv has no pytest, so anything importing it SKIPs in CI).
codex reads its key from `$CODEX_HOME/auth.json` and its proxy URL from config.toml —
`OPENAI_API_KEY` / `OPENAI_BASE_URL` in the environment are both ignored, verified against
0.146.0 and 0.152.0 (an env-var-only run sends no `authorization` header at all).
"""
from __future__ import annotations
import json
import shlex
# Characters harbor's own auth.json writer cannot survive: it interpolates the key into a
# shell heredoc, so `"` closes the JSON string and `\` starts an escape.
_UNESCAPABLE = '"\\\n\r'
AUTH_JSON_ENV_VAR = "RACCOON_CODEX_AUTH_JSON"
def auth_json_setup(key: str, remote_auth_path: str) -> tuple[dict[str, str], str]:
"""The one extra env var — returned separately so it reaches ONLY the setup exec — plus
shell writing a parseable auth.json. Subshell: the umask must not outlive this write."""
env = {AUTH_JSON_ENV_VAR: json.dumps({"OPENAI_API_KEY": key})}
command = (
f"(umask 077; printf '%s\\n' \"${AUTH_JSON_ENV_VAR}\" "
f">{shlex.quote(remote_auth_path)})\n"
)
return env, command
def unescapable_chars(key: str) -> list[str]:
"""Which characters in `key` harbor's stock heredoc writer would corrupt — empty for
every ordinary key, so the caller can refuse instead of 401ing three layers down."""
return sorted({c for c in _UNESCAPABLE if c in key})

View File

@@ -1,43 +0,0 @@
/**
* Recursive copy for scripts that must not call `cpSync`: it fails EACCES
* against a macOS docker bind mount, where the toolkit's job dirs live.
*/
import {
chmodSync,
copyFileSync,
lstatSync,
mkdirSync,
readdirSync,
readlinkSync,
rmSync,
statSync,
symlinkSync,
} from 'fs';
import { join } from 'path';
/** Copy one entry — symlink, directory or file — preserving its mode. */
export function copyPath(src: string, dest: string) {
const st = lstatSync(src);
if (st.isSymbolicLink()) {
rmSync(dest, { force: true });
symlinkSync(readlinkSync(src), dest);
return;
}
if (st.isDirectory()) {
copyTree(src, dest);
return;
}
// Unlink first: copyFileSync onto an existing file keeps that file's mode.
rmSync(dest, { force: true });
copyFileSync(src, dest);
chmodSync(dest, statSync(src).mode & 0o777);
}
/** Copy `src`'s contents into `dest`, creating `dest` if it doesn't exist. */
export function copyTree(src: string, dest: string) {
mkdirSync(dest, { recursive: true });
for (const entry of readdirSync(src, { withFileTypes: true })) {
copyPath(join(src, entry.name), join(dest, entry.name));
}
}

View File

@@ -1,137 +0,0 @@
#!/bin/sh
# Restrict this container's DNS to the hosts in DNSJAIL_ALLOW (space-separated), leaving
# every other name unresolvable. Runs as root, inside the container.
#
# Baked into the task images and invoked by the agent (scripts/dnsjail.py); shipped to the
# Explore container by the toolkit packaging. Both surfaces run this same file. Supplied from
# outside: DNSJAIL_ALLOW, the hosts the agent will actually dial -- every one must resolve or
# no jail happens -- and DNSJAIL_ALLOW_EXTRA, nice-to-haves that only warn if they do not.
#
# An unreachable model endpoint is a dead trial or a dead session, so nothing here is
# applied before it is verified, and any doubt leaves the container's DNS untouched.
set -u
STATE=/tmp/.dnsjail
CONTROL=example.com # must NOT resolve through us; proves we reached our own filter
bounded() { if command -v timeout >/dev/null 2>&1; then timeout 5 "$@"; else "$@"; fi; }
# Exact match: docker's own embedded resolver is 127.0.0.11, which a prefix match reads as
# already-jailed — and then resolv.orig is never captured, so unjail has nothing to restore.
jailed_now() { grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null; }
# Stop only the dnsmasq we started, so a declined run leaves nothing bound on :53 that a
# later run could mistake for its own filter.
drop_ours() {
if [ -s "$STATE/dnsmasq.pid" ]; then
pid=$(cat "$STATE/dnsmasq.pid")
# /tmp survives docker stop/start but pids restart at 1, so last boot's pid may now be
# some service's child. Confirm it is dnsmasq before signalling it.
case "$(cat "/proc/$pid/comm" 2>/dev/null)" in
dnsmasq) kill "$pid" 2>/dev/null || true ;;
esac
rm -f "$STATE/dnsmasq.pid" 2>/dev/null || true
fi
}
# Never `exit`: a caller may source this, so bailing out has to fall through rather than
# end the caller's shell.
dnsjail_apply() {
required="${DNSJAIL_ALLOW:-}"
extra="${DNSJAIL_ALLOW_EXTRA:-}"
allow=$(echo $required $extra) # unquoted: collapses to a single-spaced word list
# A blank required list means no model endpoint was found: jailing would strand the agent.
set -- $required
[ $# -gt 0 ] || return 0
# Already jailed by us, with our resolver alive and the same allowlist? Do nothing. Tearing
# down and rebinding :53 races the kernel releasing the socket, and losing that race ends
# in a fail-open restore -- so a second apply (the codex fresh path, rejail, run-app) would
# silently UNjail a working container.
if jailed_now && [ -s "$STATE/dnsmasq.pid" ] &&
[ "$(cat "/proc/$(cat "$STATE/dnsmasq.pid")/comm" 2>/dev/null)" = "dnsmasq" ] &&
[ "$(cat "$STATE/allow" 2>/dev/null)" = "$required" ] &&
[ "$(cat "$STATE/allow-extra" 2>/dev/null)" = "$extra" ]; then
return 0
fi
# The state dir has to work first: it holds what unjail restores, and a failed write here
# is what would otherwise truncate /etc/resolv.conf. Sticky world-writable so run-app,
# running as the container user in Explore, can drop its own lift markers.
mkdir -p "$STATE" 2>/dev/null || return 0
chmod 1777 "$STATE" 2>/dev/null || true
: > "$STATE/.probe" 2>/dev/null || return 0
rm -f "$STATE/.probe" 2>/dev/null || true
# Never forward to ourselves. Re-applying to an already-jailed container would otherwise
# read 127.0.0.1 out of resolv.conf and point dnsmasq at its own socket, blackholing
# every name.
src=/etc/resolv.conf
if jailed_now && [ -s "$STATE/resolv.orig" ]; then src="$STATE/resolv.orig"; fi
up=$(awk '/^nameserver[ \t]+[0-9]+\./{print $2; exit}' "$src" 2>/dev/null)
[ "$up" = "127.0.0.1" ] && up=""
if [ -n "$up" ] && command -v dnsmasq >/dev/null 2>&1; then
srv=""
for h in $allow; do srv="$srv --server=/$h/$up"; done
drop_ours
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
# one would rather than an answer this resolver decided to keep.
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
>/dev/null 2>>"$STATE/dnsmasq.err" || true
fi
# Ask the resolver directly: the model endpoint must answer and the control must not --
# otherwise we are looking at somebody else's resolver, not our filter. Only the FIRST
# host gates the jail: an extra host that CNAMEs outside the allowlist cannot resolve
# through the catch-all, and one of those must not silently disable the whole jail.
live=1
for h in $required; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 || { live=""; break; }
done
if [ -n "$live" ] && bounded nslookup "$CONTROL" 127.0.0.1 >/dev/null 2>&1; then live=""; fi
# The extras are reported, never fatal: one that CNAMEs outside the allowlist cannot
# resolve through the catch-all, and must not take the whole jail down with it.
if [ -n "$live" ]; then
for h in $extra; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 ||
echo "dns-jail: $h does not resolve through the jail (CNAME outside the allowlist?)" >&2
done
fi
if [ -z "$live" ]; then
# Say why. A silent decline is indistinguishable from a jail that worked, and the
# reason is usually one line from dnsmasq (gVisor sandboxes, for instance, have no
# AF_NETLINK, so dnsmasq cannot start there at all).
echo "dns-jail: declined, this container keeps normal network access${DNSJAIL_WHY:-}" >&2
[ -s "$STATE/dnsmasq.err" ] && sed 's/^/dns-jail: /' "$STATE/dnsmasq.err" >&2
drop_ours
# Failing open has to mean actually open, including when an earlier run left this
# container jailed.
if jailed_now && [ -s "$STATE/resolv.orig" ]; then
cat "$STATE/resolv.orig" > /etc/resolv.conf 2>/dev/null || true
fi
return 0
fi
# Capture what unjail restores — but never overwrite it with an already-jailed file, which
# would leave unjail a permanent no-op.
if ! jailed_now; then
cp /etc/resolv.conf "$STATE/resolv.orig" 2>/dev/null || return 0
fi
printf '%s\n' "$required" > "$STATE/allow" 2>/dev/null || true
printf '%s\n' "$extra" > "$STATE/allow-extra" 2>/dev/null || true
# A marker from a run-app that was killed would otherwise keep the jail disarmed forever.
rm -rf "$STATE/lifts" 2>/dev/null || true
# /etc/resolv.conf is a bind mount, so it is truncated in place, never renamed over —
# which means the replacement has to be complete BEFORE the write starts. Keep every
# non-nameserver directive docker set (options, search).
{ printf 'nameserver 127.0.0.1\n'
grep -vE '^[[:space:]]*nameserver' /etc/resolv.conf
} > "$STATE/resolv.jailed" 2>/dev/null
[ -s "$STATE/resolv.jailed" ] || return 0
cat "$STATE/resolv.jailed" > /etc/resolv.conf
}
dnsjail_apply || true

View File

@@ -1,328 +0,0 @@
#!/bin/bash
# Read the harness registry and derive per-harness credentials from it.
#
# Source it — the whole point is exporting into the caller's environment, which a subshell
# would lose:
#
# HARNESS_SCRIPTS_DIR=/workspace/scripts . /workspace/scripts/lib/harness-credentials.sh
# harness_setup_credentials
#
# Three callers: `harbor-run`, which needs only this; `refresh-harness-auth`, which
# re-derives and rewrites the auth files before an interactive launch; and
# `setup-harnesses.sh`, which sources it and adds installs, config writing and launchers
# on top.
#
# No -e here — this file is SOURCED, and shell options belong to the caller's shell (both
# post-creates run with -e). An unguarded failure below therefore aborts container
# creation, which is why every failure site is individually guarded rather than relying on
# this line.
set -uo pipefail
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# The registry is read with tomllib (stdlib from 3.11), and `python3` is not always new
# enough — macOS ships 3.9, and a container may symlink an older managed interpreter. Pick
# the first one that can actually import it rather than assuming.
_raccoon_python() {
local p
for p in "${RACCOON_PYTHON:-}" python3 python3.13 python3.12 python3.11; do
[ -n "$p" ] || continue
command -v "$p" >/dev/null 2>&1 || continue
if "$p" -c "import tomllib" >/dev/null 2>&1; then
printf '%s' "$p"
return 0
fi
done
return 1
}
_harness_query() {
local py
py=$(_raccoon_python) || return 1
"$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" "$@"
}
# Drop every whitespace character from a value read out of .env. A Windows-saved .env leaves a
# \r on each value, which reaches the proxy as a 401; no key or base URL legitimately contains
# whitespace anywhere, so deleting rather than trimming needs no cases.
_harness_trim() {
local out
# Fall back to the raw value: a trim that cannot run must never turn a working key into an
# empty one, which is what an unavailable `tr` would otherwise do to every caller.
out="$(printf '%s' "$1" | tr -d '[:space:]' 2>/dev/null)" || out="$1"
printf '%s' "${out:-$1}"
}
# The proxy root: the worker's ANTHROPIC_BASE_URL minus its provider path.
_harness_proxy_root() {
local base_url
base_url="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
[ -n "$base_url" ] || return 1
base_url="${base_url%"${base_url##*[!/]}"}"
# ".../llm_proxy/projects/<id>/anthropic" -> ".../llm_proxy/projects/<id>", so each
# harness's proxy_path composes onto the project route. Requires a path to strip: a base
# URL that is a bare host with no path — a provider's own API root rather than the
# proxy — would yield "https:/", handed to codex as a base URL and failing obscurely.
case "${base_url#*://}" in
*/*) printf '%s' "${base_url%/*}" ;;
*) return 2 ;;
esac
}
harness_setup_credentials() {
# `|| rc=$?` and not a bare assignment: this is sourced into a `set -e` shell (see the
# note at the top), and a bare failing assignment would exit the caller's post-create
# outright — silently, since the failure paths below are what do the explaining.
local root rc=0
root="$(_harness_proxy_root)" || rc=$?
if [ "$rc" -ne 0 ]; then
if [ "$rc" -eq 2 ]; then
echo "harness-setup: ANTHROPIC_BASE_URL (${ANTHROPIC_BASE_URL:-}) has no provider" >&2
echo "harness-setup: path, so it is not the proxy URL other harnesses derive their" >&2
echo "harness-setup: credentials from. claude will work; codex will not be" >&2
echo "harness-setup: authenticated. Use the base URL you were given." >&2
else
echo "harness-setup: ANTHROPIC_BASE_URL unset — skipping credential derivation" >&2
fi
return 0
fi
ANTHROPIC_BASE_URL="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
export ANTHROPIC_BASE_URL
local key
key="$(_harness_trim "${ANTHROPIC_API_KEY:-}")"
if [ -z "$key" ]; then
echo "harness-setup: ANTHROPIC_API_KEY unset — skipping credential derivation" >&2
return 0
fi
# harbor-run sources .env itself and passes ANTHROPIC_* through to the trial sandbox, so
# cleaning only the derived per-harness copies would leave a claude trial carrying the CR.
export ANTHROPIC_API_KEY="$key"
local id key_env base_url_env proxy_path
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
[ -n "$key_env" ] || continue
# ${!name} is an indirect expansion. Only set when empty: an explicit key wins.
if [ -z "${!key_env:-}" ]; then
export "$key_env=$key"
fi
if [ -n "$base_url_env" ] && [ -n "$proxy_path" ] && [ -z "${!base_url_env:-}" ]; then
export "$base_url_env=$root/$proxy_path"
fi
echo "harness-setup: $id credentials ready ($key_env, ${base_url_env:-no base url})" >&2
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
}
# Write the auth file for harnesses that read credentials from disk rather than $ENV.
harness_write_auth() {
local id auth_path key_env target key py
py=$(_raccoon_python) || {
echo "harness-setup: no python3.11+ with tomllib — skipping auth files" >&2
return 0
}
while IFS=$'\t' read -r id auth_path key_env; do
[ -n "$auth_path" ] && [ -n "$key_env" ] || continue
# Last mile: an explicit OPENAI_API_KEY bypasses the derivation above, so trim here
# too — this is the value that reaches the file the harness authenticates with.
key="$(_harness_trim "${!key_env:-}")"
if [ -z "$key" ]; then
echo "harness-setup: $key_env unset — skipping $id auth file" >&2
continue
fi
target=$(eval "printf '%s' \"$auth_path\"") || {
echo "harness-setup: WARNING $id auth_path could not be expanded — skipping" >&2
continue
}
mkdir -p "$(dirname "$target")" || {
echo "harness-setup: WARNING $id auth dir not creatable — skipping $target" >&2
continue
}
# json.dumps, not printf: a key containing a quote or backslash would otherwise
# produce a file the CLI cannot parse, and the failure would surface as an auth
# error rather than a malformed file.
# 0600 tmp + rename, never a redirect onto the target: a redirect truncates the live
# file first, so a write dying mid-flight leaves codex an EMPTY auth.json.
if ! RACCOON_AUTH_K="$key_env" RACCOON_AUTH_V="$key" RACCOON_AUTH_TARGET="$target" \
"$py" -c 'import json, os
target = os.environ["RACCOON_AUTH_TARGET"]
tmp = target + ".raccoon-tmp." + str(os.getpid())
try:
with os.fdopen(os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600), "w") as fh:
json.dump({os.environ["RACCOON_AUTH_K"]: os.environ["RACCOON_AUTH_V"]}, fh)
fh.write("\n")
os.replace(tmp, target)
except OSError:
try:
os.unlink(tmp)
except OSError:
pass
raise SystemExit(1)
'; then
echo "harness-setup: WARNING $id auth file NOT written — $target unwritable." >&2
echo "harness-setup: the key already on disk (if any) is left untouched." >&2
continue
fi
echo "harness-setup: $id auth -> $target" >&2
done < <(_harness_query --auth-files 2>/dev/null || true)
}
# Re-set just the root keys of a harness's config file (codex's `openai_base_url`),
# leaving every other line — the explore surface's [hooks] table included — untouched.
harness_refresh_config_keys() {
local id config_path blob target py
py=$(_raccoon_python) || return 0
# The surface only decides what a CREATE writes. An update takes the root keys off the
# front of the same blob, so a surface's tables survive byte-for-byte either way.
while IFS=$'\t' read -r id config_path blob; do
[ -n "$config_path" ] && [ -n "$blob" ] || continue
target=$(eval "printf '%s' \"$config_path\"") || continue
mkdir -p "$(dirname "$target")" || continue
if printf '%s' "$blob" | base64 -d |
RACCOON_CONFIG_TARGET="$target" "$py" -c '
import os, re, sys, tomllib
HEADER = "# Generated from harness-registry.toml — edits here are overwritten."
target = os.environ["RACCOON_CONFIG_TARGET"]
text = sys.stdin.read()
# Empty counts as unresolved: writing an empty base URL would break a container whose
# config is currently right, which is the one thing this must never do.
if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1))]:
raise SystemExit(1)
text = os.path.expandvars(text)
# Root keys, plus keys inside a [model_providers.*] table: codex reserves its built-in
# provider ids, so the proxy URL it must follow lives in a provider table, not at the
# root. Every other table, [hooks] on the explore surface included, is left alone.
REFRESHABLE_TABLE = re.compile(r"\[model_providers\.[^]]+\]$")
wanted = []
section = None
for line in text.splitlines():
stripped = line.strip()
if stripped.startswith("["):
section = stripped if REFRESHABLE_TABLE.match(stripped) else False
continue
if section is False:
continue
m = re.match(r"\s*\"?([A-Za-z0-9_.-]+)\"?\s*=", line)
if m:
wanted.append((section, m.group(1), line.rstrip()))
if not wanted:
raise SystemExit(0)
def section_path(header):
"""[model_providers.llm-proxy] -> ("model_providers", "llm-proxy")."""
return tuple(header.strip("[]").split("."))
def lookup(doc, header, key):
"""The value a parsed config holds for a wanted key, or KeyError."""
node = doc
if header:
for part in section_path(header):
node = node[part]
return node[key]
mode = None
if os.path.exists(target):
try:
with open(target, encoding="utf-8") as fh:
lines = fh.read().splitlines()
mode = os.stat(target).st_mode & 0o777
except OSError:
raise SystemExit(1)
def span(header):
"""The line range a section owns, or None when the file has no such section.
Root is everything above the first table header: a key appended below one
would be reparented into it, so searches and inserts stay inside the span.
"""
heads = [i for i, l in enumerate(lines) if l.lstrip().startswith("[")]
if header is None:
return 0, (heads[0] if heads else len(lines))
at = next((i for i in heads if lines[i].strip() == header), None)
if at is None:
return None
after = next((i for i in heads if i > at), len(lines))
return at + 1, after
# Grouped, root first, so a section this file lacks can be written whole.
grouped = {}
for header, key, line in wanted:
grouped.setdefault(header, []).append((key, line))
ordered = sorted(grouped, key=lambda h: (h is not None, h or ""))
changed = False
for header in ordered:
if span(header) is None:
# A config written before this section existed. Write the whole table
# rather than leave a root key naming a provider that is not there.
if lines and lines[-1].strip():
lines.append("")
lines.append(header)
lines.extend(line for _, line in grouped[header])
changed = True
continue
for key, line in grouped[header]:
# Re-read the span: an insert for an earlier key moved it.
start, end = span(header)
# The quoted spelling is the same key: replace rather than duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(start, end) if pat.match(lines[i])), None)
if at is None:
if end < len(lines) and lines[end].strip():
lines.insert(end, "")
lines.insert(end, line)
changed = True
elif lines[at] != line:
lines[at] = line
changed = True
if not changed:
raise SystemExit(0)
out = "\n".join(lines).rstrip("\n") + "\n"
else:
# No file means container-create could not write one, so write what it would have:
# on the explore surface that is the capture hooks too, not just the root keys.
out = HEADER + "\n" + text
try:
doc = tomllib.loads(out)
except tomllib.TOMLDecodeError:
raise SystemExit(1)
# Parsing is not enough: a line edit can land inside a multi-line value, which still
# parses while leaving the key unset. Require every key to have landed on the value the
# blob asks for, in its own section — skipping sections this file does not carry.
blob_doc = tomllib.loads(text)
for header, key, _ in wanted:
try:
expected = lookup(blob_doc, header, key)
except (KeyError, TypeError):
raise SystemExit(1)
try:
got = lookup(doc, header, key)
except (KeyError, TypeError):
if header is None:
raise SystemExit(1)
continue
if got != expected:
raise SystemExit(1)
# Pid-suffixed: two launches at once must not write the same scratch path.
tmp = target + ".raccoon-tmp." + str(os.getpid())
try:
with open(tmp, "w", encoding="utf-8") as fh:
fh.write(out)
if mode is not None:
os.chmod(tmp, mode)
os.replace(tmp, target)
except OSError:
try:
os.unlink(tmp)
except OSError:
pass
raise SystemExit(1)
'; then
echo "harness-setup: $id config keys refreshed -> $target" >&2
fi
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
}

View File

@@ -1,418 +0,0 @@
"""harness_registry.py — Python loader for ``scripts/harness-registry.toml``.
The ONE loader for the registry: TS callers shell into ``resolve_harness.py`` rather than
parse the TOML themselves, which is why the toolkit ships no TOML parser for TS (its
package.json has no zod/smol-toml).
This module supersedes ``benchmark_models_lib``'s ``HARNESS_BY_IMPORT_PATH`` and
``LEGACY_BARE_MODEL_AGENTS``; those should read from here rather than keep private
copies.
Harbor-free and dependency-free (stdlib ``tomllib``) so it can be imported from a
sandbox agent, a plain unit test, or the devcontainer python alike.
"""
from __future__ import annotations
import tomllib
from dataclasses import dataclass, field
from pathlib import Path
REGISTRY_PATH = Path(__file__).resolve().parent.parent / "harness-registry.toml"
MODEL_ID_SHAPES = frozenset({"bare", "provider/model", "provider:model"})
@dataclass(frozen=True)
class Harness:
"""One harness, as declared in harness-registry.toml."""
id: str
label: str
agent_import_path: str
model_id_shape: str
writes_atif: bool
capture: bool
seed_native: bool
seed_atif: bool
agent_import_path_single_turn: str | None = None
# Browser-opt-in variants (`[metadata] browser = true`). A harness that has no variant
# keeps its normal class: codex, for instance, gains the browser and its disclosure but
# has no `Read` equivalent to switch toolsets for.
agent_import_path_browser: str | None = None
agent_import_path_single_turn_browser: str | None = None
import_path_aliases: tuple[str, ...] = ()
legacy_bare_model_rows: bool = False
default_model: str | None = None
effort_kwarg: str = ""
effort_default: str | None = None
# Agent kwarg that opts a trial into the harness's fast/priority serving mode
# (claude-code: fast mode). Empty means the harness has none and --fast refuses.
fast_kwarg: str = ""
key_env: str | None = None
base_url_env: str | None = None
proxy_path: str | None = None
flaky_hangs: bool = False
enabled: bool = True
# Worker-container fields; see the registry header.
authoring: bool = False
cli: str | None = None
install: str | None = None
skills_dir: str | None = None
auth_path: str | None = None
auth_key_env: str | None = None
explore_launch: str | None = None
config_path: str | None = None
# Config the harness needs wherever it runs, trial sandbox included.
agent_config: str | None = None
# Config for both worker containers (explore and authoring).
container_config: str | None = None
# Config for the EXPLORE container only — the capture hooks, whose commands ship in
# explore/plugins/. Writing them in authoring would register hooks against files that
# are not there, firing on every prompt.
explore_config: str | None = None
# Fields added for a later phase, kept verbatim so this loader doesn't have to
# be edited in lockstep with the schema.
extra: dict = field(default_factory=dict, compare=False)
def agent_import_path_for(self, *, multi_turn: bool, browser: bool = False) -> str:
"""Agent class to launch. Multi-turn tasks need the resuming class; a
single-turn task given it would try to resume a session that isn't there.
``browser`` selects the opt-in variant, which for claude also carries the ``Read``
built-in — a different toolset is a different agent, so it is a different class with
its own name rather than a flag on the canonical one. Harnesses without a variant fall
through to their normal class."""
if browser:
variant = (
self.agent_import_path_browser
if multi_turn
else (self.agent_import_path_single_turn_browser or self.agent_import_path_browser)
)
if variant:
return variant
if multi_turn:
return self.agent_import_path
return self.agent_import_path_single_turn or self.agent_import_path
def row_label(self, model: str) -> str:
"""Row identity for one trial: bare model for legacy harnesses (so
published manifests keep their labels), else ``<harness>:<model>``."""
return model if self.legacy_bare_model_rows else f"{self.id}:{model}"
def agent_config_overrides(self) -> dict[str, str]:
"""``agent_config`` as flat ``dotted.key -> value`` pairs in CLI-override form.
Values are rendered bare — ``disabled``, not ``"disabled"``. Every consumer
interpolates these into a shell command, which would strip the quotes anyway;
emitting them would only make the result depend on how many shell layers the
string crosses. Bare is what the CLIs document (``-c model="o3"`` reaches the
binary as ``model=o3``).
These settings ride the command line as ``-c dotted.key=value`` everywhere the
harness runs, never a config file. A trial sandbox rules the file out: the
harness's own runner appends root keys to it, and TOML has no way back to the
root scope once a table has opened, so a table we appended would swallow them.
Overrides compose in any order and beat the file, so the same rendering serves
the explore launcher too — one declaration, one mechanism.
"""
if not self.agent_config:
return {}
try:
parsed = tomllib.loads(self.agent_config)
except tomllib.TOMLDecodeError as exc:
raise HarnessRegistryError(
f"{self.id}: agent_config is not valid TOML ({exc})"
) from exc
flat: dict[str, str] = {}
def walk(node: dict, prefix: str) -> None:
for key, value in node.items():
path = f"{prefix}{key}"
if isinstance(value, dict):
walk(value, f"{path}.")
elif isinstance(value, bool):
flat[path] = "true" if value else "false"
elif isinstance(value, (int, float)):
flat[path] = str(value)
elif isinstance(value, str):
if value != value.strip() or any(c in value for c in " \"'\\"):
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has a value needing "
"shell quoting, which the -c override form cannot carry"
)
flat[path] = value
else:
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has type "
f"{type(value).__name__}, which has no -c override form"
)
walk(parsed, "")
return flat
def container_config_text(self, *, surface: str) -> str | None:
"""Config file body for a worker container. `surface` is "explore" or
"authoring"; explore additionally gets `explore_config`. Root keys come from
`container_config` first, so appending a table section stays valid TOML."""
parts = [self.container_config]
if surface == "explore":
parts.append(self.explore_config)
kept = [part.strip("\n") for part in parts if part and part.strip()]
return "\n\n".join(kept) + "\n" if kept else None
def agent_config_flags(self) -> str:
"""``agent_config`` as a ``-c key=value`` command-line string."""
return " ".join(
f"-c {key}={value}"
for key, value in sorted(self.agent_config_overrides().items())
)
def explore_launch_command(self) -> str | None:
"""``explore_launch`` with the registry's own values substituted in.
The worker's Explore session and the trial must run the same agent, so the
model, effort and reductions are declared once here and rendered into both.
A literal in the launch string would be a second declaration, and the two
would drift the first time one of them was updated alone.
Only these three placeholders are substituted; ``$@`` and
``$RACCOON_TOOLSET_NOTE`` stay for the launcher's own shell to expand.
"""
if not self.explore_launch:
return None
return (
self.explore_launch.replace("$RACCOON_AGENT_FLAGS", self.agent_config_flags())
.replace("$RACCOON_MODEL", self.default_model or "")
.replace("$RACCOON_EFFORT", self.effort_default or "")
)
def known_import_paths(self) -> tuple[str, ...]:
paths = [self.agent_import_path, *self.import_path_aliases]
if self.agent_import_path_single_turn:
paths.append(self.agent_import_path_single_turn)
return tuple(paths)
_KNOWN_FIELDS = frozenset(
{
"id",
"label",
"agent_import_path",
"agent_import_path_single_turn",
"agent_import_path_browser",
"agent_import_path_single_turn_browser",
"import_path_aliases",
"legacy_bare_model_rows",
"default_model",
"model_id_shape",
"effort_kwarg",
"effort_default",
"fast_kwarg",
"key_env",
"base_url_env",
"proxy_path",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
"flaky_hangs",
"enabled",
"authoring",
"cli",
"install",
"skills_dir",
"auth_path",
"auth_key_env",
"explore_launch",
"config_path",
"agent_config",
"container_config",
"explore_config",
}
)
_REQUIRED_FIELDS = (
"id",
"label",
"agent_import_path",
"model_id_shape",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
)
class HarnessRegistryError(ValueError):
"""Malformed registry. Raised rather than tolerated: a broken registry is a
broken deployment, and silently defaulting would pick the wrong agent."""
def _references_agent_flags(launch: str) -> bool:
return "$RACCOON_AGENT_FLAGS" in launch or "${RACCOON_AGENT_FLAGS}" in launch
@dataclass(frozen=True)
class HarnessRegistry:
version: int
harnesses: tuple[Harness, ...]
def all(self) -> tuple[Harness, ...]:
return self.harnesses
def enabled(self) -> tuple[Harness, ...]:
return tuple(h for h in self.harnesses if h.enabled)
def authoring(self) -> tuple[Harness, ...]:
"""Harnesses a worker can author with — what the worker containers install.
Narrower than enabled(): a harness can be runnable in a trial without having
an authoring story (no CLI to converse with, or no capture)."""
return tuple(h for h in self.harnesses if h.enabled and h.authoring)
def find(self, harness_id: str) -> Harness | None:
return next((h for h in self.harnesses if h.id == harness_id), None)
def require(self, harness_id: str) -> Harness:
harness = self.find(harness_id)
if harness is not None:
return harness
available = ", ".join(sorted(h.id for h in self.enabled()))
raise HarnessRegistryError(
f'Unknown harness "{harness_id}". Available: {available}'
)
def by_import_path(self, agent: str) -> Harness | None:
"""Resolve an agent identity — a ``name()`` or import path from
``result.json`` ``config.agent``, or a manifest row — to its harness."""
needle = (agent or "").strip()
if not needle:
return None
for harness in self.harnesses:
if needle == harness.id or needle in harness.known_import_paths():
return harness
return None
def _build(entry: dict, index: int) -> Harness:
for name in _REQUIRED_FIELDS:
if name not in entry:
raise HarnessRegistryError(
f"harness[{index}]: missing required field '{name}'"
)
shape = entry["model_id_shape"]
if shape not in MODEL_ID_SHAPES:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): model_id_shape {shape!r} not one of "
f"{sorted(MODEL_ID_SHAPES)}"
)
# These three reach `eval` in setup-harnesses.sh, which is how they support the
# `${CODEX_HOME:-$HOME/.codex}` default-value syntax that python's expandvars cannot
# express. Under eval a backtick or $( would EXECUTE, so refuse them here — the registry
# is ours, but "ours" is not an argument that survives a careless future edit.
for shell_field in ("config_path", "auth_path", "skills_dir"):
value = entry.get(shell_field)
if not isinstance(value, str):
continue
# A backtick or $( executes outright. A double quote closes the string these are
# interpolated into, and a semicolon then starts a new command inside it — same
# outcome, one step removed.
bad = [t for t in ("`", "$(", '"', ";") if t in value]
if bad:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): {shell_field} contains "
f"{', '.join(repr(t) for t in bad)} ({value!r}). This value is shell-"
f"expanded, so that would execute; use plain $VAR or ${{VAR:-default}} only."
)
launch = entry.get("explore_launch")
if entry.get("agent_config") and launch and not _references_agent_flags(launch):
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): declares agent_config but its "
"explore_launch does not pass $RACCOON_AGENT_FLAGS. The worker's session "
"would then run with a different toolset than the trial it is authoring "
"for, which is the drift agent_config exists to prevent."
)
return Harness(
id=entry["id"],
label=entry["label"],
agent_import_path=entry["agent_import_path"],
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
agent_import_path_browser=entry.get("agent_import_path_browser"),
agent_import_path_single_turn_browser=entry.get("agent_import_path_single_turn_browser"),
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
default_model=entry.get("default_model"),
model_id_shape=shape,
effort_kwarg=entry.get("effort_kwarg", ""),
effort_default=entry.get("effort_default"),
fast_kwarg=entry.get("fast_kwarg", ""),
key_env=entry.get("key_env"),
base_url_env=entry.get("base_url_env"),
proxy_path=entry.get("proxy_path"),
writes_atif=bool(entry["writes_atif"]),
capture=bool(entry["capture"]),
seed_native=bool(entry["seed_native"]),
seed_atif=bool(entry["seed_atif"]),
flaky_hangs=bool(entry.get("flaky_hangs", False)),
enabled=bool(entry.get("enabled", True)),
authoring=bool(entry.get("authoring", False)),
cli=entry.get("cli"),
install=entry.get("install"),
skills_dir=entry.get("skills_dir"),
auth_path=entry.get("auth_path"),
auth_key_env=entry.get("auth_key_env"),
explore_launch=entry.get("explore_launch"),
config_path=entry.get("config_path"),
agent_config=entry.get("agent_config"),
container_config=entry.get("container_config"),
explore_config=entry.get("explore_config"),
extra={k: v for k, v in entry.items() if k not in _KNOWN_FIELDS},
)
_cache: dict[Path, HarnessRegistry] = {}
def load_harness_registry(path: Path | str = REGISTRY_PATH) -> HarnessRegistry:
"""Parse and validate the registry. Raises HarnessRegistryError on a malformed
file, a duplicate id, or an import path claimed by two harnesses (which would
make ``by_import_path`` depend on declaration order)."""
resolved = Path(path).resolve()
if resolved in _cache:
return _cache[resolved]
with open(resolved, "rb") as handle:
doc = tomllib.load(handle)
if "version" not in doc:
raise HarnessRegistryError("harness-registry: missing 'version'")
entries = doc.get("harness") or []
if not entries:
raise HarnessRegistryError("harness-registry: no [[harness]] entries")
harnesses = tuple(_build(entry, i) for i, entry in enumerate(entries))
seen_ids: set[str] = set()
for harness in harnesses:
if harness.id in seen_ids:
raise HarnessRegistryError(
f"harness-registry: duplicate harness id: {harness.id}"
)
seen_ids.add(harness.id)
owners: dict[str, str] = {}
for harness in harnesses:
for import_path in harness.known_import_paths():
owner = owners.get(import_path)
if owner is not None and owner != harness.id:
raise HarnessRegistryError(
f'harness-registry: import path "{import_path}" claimed by both '
f'"{owner}" and "{harness.id}"'
)
owners[import_path] = harness.id
registry = HarnessRegistry(version=int(doc["version"]), harnesses=harnesses)
_cache[resolved] = registry
return registry

View File

@@ -1,351 +0,0 @@
/**
* input-checksums.ts — capture and compare sha256 checksums of the task inputs
* that reference runs and detector reports depend on.
*
* A reference run is only meaningful for the task inputs it actually ran
* against: the prompt (instruction.md), the snapshot session
* (environment/session.jsonl), the workspace patch
* (environment/workspace.patch), and the gitref the workspace is built from
* (task.toml `[metadata].commit`). Detector reports likewise assess a specific
* revision of instruction.md + the task's holistic rubric
* (tests/holistic-rubric.md on current tasks; tests/grader-guidance-consolidated.md
* on tasks created before the rename) — and, for the rubric detectors, the
* task's atomic rubric (tests/atomic-rubric.yaml; tests/rubrics.yaml on tasks
* converted before the rename) and tests/grader-context.md. When any of those
* change after the artifact was produced, the artifact is stale — it describes
* an older revision of the task than the one being packaged.
*
* This module is the single source of truth for WHAT gets checksummed and how
* captures are compared. Capture sites (copy-reference-run.ts,
* record-detector-inputs.ts) write a {@link TaskInputChecksums} record next to
* the artifact; submit-task.ts re-captures at packaging time and diffs.
* Content hashes rather than mtimes: a re-clone / whole-tree touch can fake or
* mask an mtime, but can't change a sha256.
*
* Lives in the worker toolkit's shipped file set (static/scripts/lib/), so in
* a packed toolkit it sits at scripts/lib/ next to both consumers. Repo-side
* callers go through the scripts/lib/input-checksums.ts re-export shim, which
* occupies the same relative path there (mirroring the check-devcontainer
* pattern).
*
* Related but deliberately separate: `computeDeliveryHash` (repo-side
* delivery script — grep the internal repo for it; not shipped with the
* toolkit) hashes an overlapping input set for delivery idempotency. It is
* NOT built on this module because its hash format is load-bearing (a
* changed hash re-delivers every task); if you change WHAT counts as a task
* input here, check whether the delivery hash needs the same change.
*/
import { createHash } from 'node:crypto';
import { existsSync, readFileSync } from 'node:fs';
import { join } from 'node:path';
/** Bump when the record shape changes incompatibly. */
export const INPUT_CHECKSUMS_VERSION = 1;
/** Filename of the record inside a reference-run directory. */
export const INPUT_CHECKSUMS_FILENAME = 'input-checksums.json';
/**
* Where in the artifact lifecycle a capture happened. The moment matters for
* how much a "fresh" verdict can be trusted:
*
* - 'run' — at trial launch (scripts/harbor-run stamps the trial dir).
* The strongest evidence: the record is what the agent ran
* against, whatever got edited afterwards.
* - 'copy' — at copy-reference-run time, the fallback when a trial carries
* no run-time stamp. An input edited between harbor-run and the
* copy is recorded at its post-edit state, so a stale run can
* read fresh.
* - 'stamp' — right after a detector report is written
* (record-detector-inputs.ts in a worker checkout; the
* internal repo's detector save path stamps the same way).
* - 'mirror' — retired: written by the repo-side flow that re-materialized
* canonical detector reports to disk back when reports had a
* remote canonical store. Reports are local-only now, so no
* current code writes it; the member stays so old stamps keep
* their recorded method when read.
* - 'regrade' — a re-grade of an existing run. Present on records already on
* disk; no current code path writes it.
*
* Absent on records written before this field existed.
*/
export type TaskInputCaptureMethod = 'run' | 'copy' | 'stamp' | 'mirror' | 'regrade';
/** Runtime mirror of {@link TaskInputCaptureMethod}, for validating a record read from disk. */
export const TASK_INPUT_CAPTURE_METHODS = Object.freeze([
'run',
'copy',
'stamp',
'mirror',
'regrade',
] as const satisfies readonly TaskInputCaptureMethod[]);
/**
* The checksums of a task's inputs as they stood at capture time. Every hash
* field is a sha256 hex digest, or `null` when the file didn't exist (a null
* that later becomes a hash — or vice versa — is a change like any other).
* `gitref` is the raw `[metadata].commit` string, recorded verbatim rather
* than hashed so a mismatch message can show it.
*/
export interface TaskInputChecksums {
readonly version: number;
/** ISO-8601 timestamp of the capture. */
readonly capturedAt: string;
/** Lifecycle point of the capture ({@link TaskInputCaptureMethod}). */
readonly capturedBy?: TaskInputCaptureMethod;
/**
* The task slug the capture was taken from (harbor-tasks/<slug>). Written
* by launch-time captures so stamping can be scoped to the right task's
* trial dirs when several harbor-runs share a cwd — and so a mis-routed
* stamp is detectable after the fact.
*/
readonly taskSlug?: string;
/**
* Set when the finalize flow re-stamped the prompt/graderGuidance hashes
* after the harbor-path scrub deliberately rewrote those docs
* (scripts/restamp-task-inputs.ts) — the run/report still reflects the
* task; only the doc bytes were normalized.
*/
readonly restampedAt?: string;
readonly inputs: {
readonly prompt: string | null;
readonly graderGuidance: string | null;
readonly sessionJsonl: string | null;
readonly workspacePatch: string | null;
readonly gitref: string | null;
/**
* tests/grader-guidance-consolidated.md — the holistic rubric under its
* pre-rename filename, which every task created before the rename keeps;
* hashed when present, null otherwise. Absent (undefined) on records
* captured before the field existed; comparisons skip a field the record
* predates, so old captures stay fresh until they are re-stamped.
*/
readonly graderGuidanceConsolidated?: string | null;
/**
* tests/holistic-rubric.md — the holistic rubric under its current
* filename (each task carries exactly one of this and the pre-rename
* name above). Hashed when present, null otherwise. Absent (undefined)
* on records captured before the field existed; comparisons skip a
* field the record predates. The task-checksum digest serializes fields
* in this order and appends new fields at the end (see
* scripts/lib/grader-run-checksums.ts).
*/
readonly holisticRubric?: string | null;
/**
* tests/atomic-rubric.yaml — the task's atomic rubric under its current
* filename. Hashed when present, null otherwise. Absent (undefined) on
* records captured before the field existed; comparisons skip a field
* the record predates.
*/
readonly atomicRubric?: string | null;
/**
* tests/rubrics.yaml — the atomic rubric under its pre-rename filename,
* which tasks converted before the rename keep. Hashed when present,
* null otherwise; absent (undefined) on records captured before the
* field existed.
*/
readonly rubricsYaml?: string | null;
/**
* tests/grader-context.md — the context document the rubric grader modes
* read beside the atomic rubric. Hashed when present, null otherwise;
* absent (undefined) on records captured before the field existed. Last
* in field order per the append-at-the-end digest rule above.
*/
readonly graderContext?: string | null;
};
}
export type TaskInputName = keyof TaskInputChecksums['inputs'];
/** Human-readable component names, used verbatim in staleness warnings. Frozen:
* its key set is the runtime source of truth for the task-input axes. */
export const INPUT_LABELS = Object.freeze({
prompt: 'prompt (instruction.md)',
graderGuidance: 'legacy-era grader guidance (tests/grader-guidance.md)',
graderGuidanceConsolidated: 'holistic rubric (tests/grader-guidance-consolidated.md)',
sessionJsonl: 'session snapshot (environment/session.jsonl)',
workspacePatch: 'workspace patch (environment/workspace.patch)',
gitref: 'gitref (task.toml commit)',
holisticRubric: 'holistic rubric (tests/holistic-rubric.md)',
atomicRubric: 'atomic rubric (tests/atomic-rubric.yaml)',
rubricsYaml: 'atomic rubric (tests/rubrics.yaml)',
graderContext: 'grader context (tests/grader-context.md)',
} as const satisfies Record<TaskInputName, string>);
/**
* The inputs that shape what the AGENT saw and did. Changing any of them means
* a captured reference run no longer reflects the task being packaged, and
* only re-running the agent can fix that. The holistic-rubric files are
* deliberately NOT in this set: editing the rubric stales the run's GRADE,
* not the run itself, and `scripts/harbor-regrade` re-derives grades without
* re-running the agent.
*/
export const REFERENCE_RUN_INPUTS = Object.freeze([
'prompt',
'sessionJsonl',
'workspacePatch',
'gitref',
] as const satisfies readonly TaskInputName[]);
/**
* The inputs a detector report assesses — instruction.md plus whichever
* holistic-rubric files the task directory carries (tests/holistic-rubric.md
* on current tasks, tests/grader-guidance-consolidated.md on tasks created
* before the rename, plus the legacy-era plain-named file when a task
* authored on an earlier generation carries one). An absent file hashes to
* null on both sides and never diffs. Compared by content.
*
* The atomic-rubric files are deliberately NOT in this set: fifteen of the
* seventeen detectors never open them, so writing an atomic rubric after
* running the detectors would stale every one of those reports over files
* they never read. The two that do read them use
* {@link RUBRIC_DETECTOR_REPORT_INPUTS}.
*/
export const DETECTOR_REPORT_INPUTS = Object.freeze([
'prompt',
'graderGuidance',
'graderGuidanceConsolidated',
'holisticRubric',
] as const satisfies readonly TaskInputName[]);
/**
* The inputs the two rubric detectors assess: {@link DETECTOR_REPORT_INPUTS}
* plus the atomic-rubric package (tests/atomic-rubric.yaml, the pre-rename
* tests/rubrics.yaml, and tests/grader-context.md), which they compare
* against the holistic rubric.
*/
export const RUBRIC_DETECTOR_REPORT_INPUTS = Object.freeze([
...DETECTOR_REPORT_INPUTS,
'atomicRubric',
'rubricsYaml',
'graderContext',
] as const satisfies readonly TaskInputName[]);
/** Detectors that read the atomic-rubric package, and so are staled by it. */
export const ATOMIC_RUBRIC_DETECTORS: readonly string[] = Object.freeze([
'detector-rubric-coverage',
'detector-rubric-form',
]);
/** The input set a named detector's report is judged against. */
export function detectorReportInputs(detectorName: string): readonly TaskInputName[] {
return ATOMIC_RUBRIC_DETECTORS.includes(detectorName)
? RUBRIC_DETECTOR_REPORT_INPUTS
: DETECTOR_REPORT_INPUTS;
}
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
function sha256File(filePath: string): string | null {
if (!existsSync(filePath)) return null;
return createHash('sha256').update(readFileSync(filePath)).digest('hex');
}
/**
* The gitref (`[metadata].commit`) from a task.toml, or null when the file is
* missing, unreadable, or has no commit line — a read failure downgrades to
* "absent" rather than crashing a capture or a validation sweep.
*
* Deliberately a regex, not a TOML parser: this module ships in the worker
* toolkit, where a new runtime dep would break packaging for every worker
* whose container predates the dep (npm install runs only on container
* create, and containers survive toolkit upgrades). build-workspace.sh reads
* the same key with the same grep-a-`commit`-line approach. The one `commit`
* key in a task.toml is `[metadata].commit`, so anchoring to the first
* `commit = "…"` line is exact in practice.
*/
function readGitref(taskDir: string): string | null {
const tomlPath = join(taskDir, 'task.toml');
if (!existsSync(tomlPath)) return null;
try {
const match = /^\s*commit\s*=\s*(?:"([^"\n]+)"|'([^'\n]+)')\s*(?:#.*)?$/m.exec(
readFileSync(tomlPath, 'utf-8')
);
const commit = match?.[1] ?? match?.[2];
return commit && commit.length > 0 ? commit : null;
} catch {
return null;
}
}
/** Checksum the task inputs as they stand right now under `taskDir`. */
export function captureTaskInputs(
taskDir: string,
capturedBy?: TaskInputCaptureMethod
): TaskInputChecksums {
return Object.freeze({
version: INPUT_CHECKSUMS_VERSION,
capturedAt: new Date().toISOString(),
...(capturedBy ? { capturedBy } : {}),
inputs: Object.freeze({
prompt: sha256File(join(taskDir, 'instruction.md')),
graderGuidance: sha256File(join(taskDir, 'tests', 'grader-guidance.md')),
sessionJsonl: sha256File(join(taskDir, 'environment', 'session.jsonl')),
workspacePatch: sha256File(join(taskDir, 'environment', 'workspace.patch')),
gitref: readGitref(taskDir),
graderGuidanceConsolidated: sha256File(
join(taskDir, 'tests', 'grader-guidance-consolidated.md')
),
holisticRubric: sha256File(join(taskDir, 'tests', 'holistic-rubric.md')),
atomicRubric: sha256File(join(taskDir, 'tests', 'atomic-rubric.yaml')),
rubricsYaml: sha256File(join(taskDir, 'tests', 'rubrics.yaml')),
graderContext: sha256File(join(taskDir, 'tests', 'grader-context.md')),
}),
});
}
/**
* Read a previously captured record. Returns null when the file is missing or
* doesn't look like a capture (pre-tracking artifact, hand-edited JSON, a
* future incompatible version) — callers treat null as "staleness unknowable",
* never as an error. No zod here: this module ships in the worker toolkit,
* whose dependency set stays minimal, so the guard is manual.
*/
export function readTaskInputChecksums(filePath: string): TaskInputChecksums | null {
if (!existsSync(filePath)) return null;
let parsed: unknown;
try {
parsed = JSON.parse(readFileSync(filePath, 'utf-8'));
} catch {
return null;
}
if (typeof parsed !== 'object' || parsed === null) return null;
const record = parsed as TaskInputChecksums;
if (record.version !== INPUT_CHECKSUMS_VERSION) return null;
if (typeof record.inputs !== 'object' || record.inputs === null) return null;
for (const name of Object.keys(INPUT_LABELS) as TaskInputName[]) {
const value = record.inputs[name];
// undefined = the record predates this input field; still a valid capture.
if (value !== undefined && value !== null && typeof value !== 'string') return null;
}
// An unrecognized capture method is dropped, not rejected: the field is
// provenance colour, and rejecting would flip the whole run to "unknowable".
const method: unknown = record.capturedBy;
const isKnown = TASK_INPUT_CAPTURE_METHODS.includes(method as TaskInputCaptureMethod);
if (method !== undefined && !isKnown) {
const { capturedBy: _dropped, ...rest } = record;
return rest;
}
return record;
}
/**
* Which of `names` changed between a recorded capture and the current state?
* Returns the human-readable labels ({@link INPUT_LABELS}) of every component
* whose value differs — including absent→present and present→absent flips. A
* field the recorded capture predates (the key is not in the record at all)
* is skipped: freshness on that axis is unknowable, and flagging every old
* record the moment a new axis ships would drown the real signal. (The
* task-checksum fold folds absence as 'null' instead — it only ever reads a
* fresh capture, which is total, so the two never disagree in practice.)
*/
export function diffTaskInputs(
recorded: TaskInputChecksums,
current: TaskInputChecksums,
names: readonly TaskInputName[]
): string[] {
return names
.filter((name) => name in recorded.inputs)
.filter((name) => recorded.inputs[name] !== current.inputs[name])
.map((name) => INPUT_LABELS[name]);
}

View File

@@ -1,13 +0,0 @@
/**
* Wrap a notice in a banner loud enough to survive a scrollback.
*
* Yellow only when stderr is a terminal, so piped logs stay clean.
*/
export function banner(message: string, headline: string): string {
const RULE = '#'.repeat(78);
const pad = ' '.repeat(Math.max(0, Math.floor((78 - headline.length) / 2)));
const body = [RULE, `${pad}${headline}`, RULE, '', message, RULE].join('\n');
const color = process.stderr.isTTY ? ['\u001b[33m', '\u001b[39m'] : ['', ''];
return `${color[0]}${body}${color[1]}`;
}

View File

@@ -1,96 +0,0 @@
# shellcheck shell=bash
#
# resolve_pin — turn a task's pinned commit into a SHA that exists in the repo,
# translating through a commit map when history has been rewritten under it.
#
# A task pins a commit in task.toml. If that repo's history is later rewritten
# (to strip something that should never have shipped, say), every rewritten
# commit gets a new SHA and the pin stops resolving — including on machines we
# cannot reach, holding tasks we cannot edit. A commit map lets those pins keep
# working: `<old-sha> <new-sha>` per line, at task-shared/commit-maps/<member>.map,
# <member> being the task's `repo` key — a standalone toolkit checks its repo out
# at repo/, so the directory name is not the member name and cannot be the key.
#
# The map is only consulted when the pin does not resolve, so it carries only
# rewritten commits — an unchanged commit resolves on its own and its identity
# row could never be read.
#
# Usage (source, then call):
# . "$(dirname "$0")/lib/resolve-pin.sh"
# sha=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
#
# Writes the resolved SHA to stdout, notes on stderr. Returns non-zero if the
# pin cannot be resolved, having explained why.
RESOLVE_PIN_MAX_HOPS="${RESOLVE_PIN_MAX_HOPS:-25}"
_RESOLVE_PIN_ZERO='0000000000000000000000000000000000000000'
# Look one hop: echo the successor of $1 in map $2, or nothing. Fails if the
# prefix is ambiguous, which would otherwise pick an arbitrary commit.
_resolve_pin_hop() {
local from="$1" map="$2" hits
hits=$(awk -v p="$from" '
/^#/ || NF < 2 { next }
index($1, p) == 1 { print $2 }
' "$map" | sort -u)
[ -z "$hits" ] && return 1
if [ "$(printf '%s\n' "$hits" | wc -l | tr -d ' ')" -gt 1 ]; then
echo " pin $from is ambiguous in $(basename "$map") — use a longer SHA" >&2
return 2
fi
printf '%s\n' "$hits"
}
resolve_pin() {
local repo_dir="$1" commit="$2" map_dir="${3:-}" key="${4:-}" sha map cur hops next rc
# Present in the repo: nothing to translate.
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$commit^{commit}" 2>/dev/null); then
printf '%s\n' "$sha"
return 0
fi
map=""
if [ -n "$map_dir" ]; then
if [ -n "$key" ] && [ -f "$map_dir/$key.map" ]; then
map="$map_dir/$key.map"
elif [ -f "$map_dir/$(basename "$repo_dir").map" ]; then
map="$map_dir/$(basename "$repo_dir").map"
fi
fi
if [ -z "$map" ]; then
echo "Error: pinned commit $commit is not in $repo_dir, and no commit map is available." >&2
echo " The repo may be a shallow or partial copy — try a full clone." >&2
return 1
fi
# Follow the chain: a commit rewritten more than once maps forward a hop per
# rewrite, so keep going until the SHA exists or the trail ends.
cur="$commit"
hops=0
while [ "$hops" -lt "$RESOLVE_PIN_MAX_HOPS" ]; do
next=$(_resolve_pin_hop "$cur" "$map"); rc=$?
[ "$rc" -eq 2 ] && return 1
if [ "$rc" -ne 0 ]; then
echo "Error: pinned commit $commit is not in $repo_dir and is not in $(basename "$map")." >&2
echo " It predates the map, or came from a repo copy this toolkit was not built from." >&2
return 1
fi
if [ "$next" = "$_RESOLVE_PIN_ZERO" ]; then
echo "Error: pinned commit $commit was deleted by a history rewrite, not rewritten." >&2
echo " Re-pin this task to a commit that still exists." >&2
return 1
fi
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$next^{commit}" 2>/dev/null); then
echo " Pin $commit was rewritten; using $sha" >&2
printf '%s\n' "$sha"
return 0
fi
cur="$next"
hops=$((hops + 1))
done
echo "Error: pinned commit $commit did not settle after $RESOLVE_PIN_MAX_HOPS hops." >&2
echo " $(basename "$map") may contain a cycle." >&2
return 1
}

View File

@@ -1,431 +0,0 @@
/**
* task-infra-integrity.ts — detect edits to toolkit-managed task files.
*
* `environment/Dockerfile`, `tests/test.sh`, and
* `tests/grader-system-prompt-consolidated.md` come from `task-shared/` and
* are the same in every task: they decide how the
* trial runs and how the grade is produced. An edit makes a task's reference
* runs incomparable to every other task's, and the scores still look normal,
* so nothing downstream notices.
*
* A task is compared against itself as created. {@link writeManagedStamp} records
* a sha256 of each managed file into `<task>/.toolkit-managed.json` at task
* creation, so a later mismatch is an edit made since. Tasks created before
* stamping have no record and fall back to matching the copies this toolkit
* ships — see {@link IntegrityStatus}.
*
* The toolkit appends to a task's Dockerfile itself (session staging, the
* reference-data corpus). Those blocks are wrapped in
* `# >>> toolkit-managed: <name> >>>` sentinels and stripped before hashing or
* comparing, so they never read as edits.
*/
import { createHash } from 'crypto';
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
import { basename, join } from 'path';
import { banner } from './notice-banner.js';
/**
* Every status is advisory. Nothing here stops a trial or a submission: an author
* who changed one of these files did it because they didn't know we'd rather they
* didn't, and refusing to package their work punishes a misunderstanding. The job
* is to say so clearly, and to record it so a reviewer sees it too.
*
* `ok` — identical to a copy this toolkit ships, or unchanged since the
* task was created.
* `outdated` — unchanged since creation, but the toolkit has shipped a newer
* copy since. Nobody's mistake; it does mean this task's runs
* aren't directly comparable to one built today.
* `modified` — matches neither its baseline nor anything shipped: an edit.
* `unverifiable` — no recorded baseline and matches nothing shipped, so an edit
* and an older release are indistinguishable.
* `missing` — the task doesn't have the file.
* `placeholder` — still the polyglot scaffold placeholder, so no base image has
* been selected yet.
*/
export type IntegrityStatus =
| 'ok'
| 'outdated'
| 'modified'
| 'unverifiable'
| 'missing'
| 'placeholder';
export interface FileVerdict {
/** Task-relative path, e.g. `environment/Dockerfile`. */
taskPath: string;
status: IntegrityStatus;
/** Command that restores the managed version, on `modified` / `unverifiable`. */
restore?: string;
}
export interface IntegrityReport {
/**
* False when this isn't a worker toolkit, or when the task was authored on
* a different toolkit generation (see {@link isTaskFromThisToolkitGeneration})
* — callers should skip silently.
*/
checked: boolean;
files: FileVerdict[];
/** Looks like an edit: matches neither a baseline nor anything shipped. */
modified: FileVerdict[];
/** Can't be told apart from an older release. */
unverifiable: FileVerdict[];
/** Unchanged, but a newer copy has shipped since. */
outdated: FileVerdict[];
}
interface ManagedFile {
taskPath: string;
/** Matches the candidate pristine filenames under `task-shared/`. */
baselinePattern: RegExp;
}
/** The Dockerfile pattern accepts `Dockerfile` and every `Dockerfile.<member>`. */
const MANAGED_FILES: ManagedFile[] = [
{ taskPath: 'environment/Dockerfile', baselinePattern: /^Dockerfile(\.[\w.-]+)?$/ },
{ taskPath: 'tests/test.sh', baselinePattern: /^test\.sh$/ },
{
taskPath: 'tests/grader-system-prompt-consolidated.md',
baselinePattern: /^grader-system-prompt-consolidated\.md$/,
},
];
const SENTINEL_OPEN = /^#\s*>>>\s*toolkit-managed:.*>>>\s*$/;
const SENTINEL_CLOSE = /^#\s*<<<\s*toolkit-managed\s*<<<\s*$/;
/**
* Line shapes from toolkit releases that predate the sentinels. Deliberately
* narrow: each is a literal line the toolkit wrote, not a general "ignore COPY
* lines" rule an edit could hide behind.
*/
const LEGACY_MANAGED_LINES: RegExp[] = [
/^# Stage session files for the snapshot agent adapter to install at runtime\.$/,
/^COPY session\.jsonl \/tmp\/snapshot-session\/session\.jsonl$/,
/^COPY session\/ \/tmp\/snapshot-session\/session\/$/,
/^RUN echo '[0-9a-fA-F-]+' > \/tmp\/snapshot-session\/uuid\.txt$/,
/^# Reference-data corpus at \/data\/zeta-corpus \(staged by build-workspace\)\.$/,
/^COPY corpus\/ \/data\/zeta-corpus\/$/,
];
/** Marker identifying the polyglot scaffold's deliberately-failing placeholder. */
const PLACEHOLDER_MARKER = 'POLYGLOT TOOLKIT';
/** Per-task stamp of the managed files as created. Lives in the task directory. */
export const STAMP_FILENAME = '.toolkit-managed.json';
interface ManagedStamp {
version: number;
stampedAt: string;
/** taskPath → sha256 of the stripped content. */
files: Record<string, string>;
}
/**
* Remove toolkit-appended content so only author-authored differences remain.
* Trailing blank lines go too — an editor adding or trimming a final newline is
* not something to fail a trial over.
*/
export function stripManagedBlocks(content: string): string {
const out: string[] = [];
let inBlock = false;
// Normalize CRLF before anything else: a Windows editor or a checkout with
// core.autocrlf rewrites every line ending, and that must not read as an edit.
for (const line of content.replace(/\r\n/g, '\n').split('\n')) {
if (!inBlock && SENTINEL_OPEN.test(line)) {
inBlock = true;
continue;
}
if (inBlock) {
if (SENTINEL_CLOSE.test(line)) inBlock = false;
continue;
}
if (LEGACY_MANAGED_LINES.some((re) => re.test(line))) continue;
out.push(line);
}
return out.join('\n').replace(/\s+$/, '');
}
export function sha256(content: string): string {
return createHash('sha256').update(content).digest('hex');
}
/** Pristine `task-shared/` filenames matching a managed file's baseline pattern. */
function baselineCandidates(sharedDir: string, pattern: RegExp): string[] {
if (!existsSync(sharedDir)) return [];
return readdirSync(sharedDir)
.filter((f) => pattern.test(f))
.sort();
}
/**
* Record the managed files, so later edits are detectable. Call at task creation
* and after a managed file is first put in place.
*
* A file earns a baseline only by matching a copy this toolkit ships, and an
* entry already recorded is never rewritten. Together those mean a stamp can
* only ever describe a pristine file: re-running this can't turn an author's
* edit into the new baseline, and a file dropped in later (the polyglot
* Dockerfile, which is the scaffold's placeholder at first stamp) still gets a
* baseline once it's in place.
*
* Returns true if anything was recorded.
*/
export function writeManagedStamp(taskDir: string, toolkitRoot: string): boolean {
const sharedDir = join(toolkitRoot, 'task-shared');
const existing = readStamp(taskDir);
const files: Record<string, string> = { ...(existing?.files ?? {}) };
let added = false;
for (const managed of MANAGED_FILES) {
if (files[managed.taskPath]) continue;
const p = join(taskDir, managed.taskPath);
if (!existsSync(p)) continue;
const raw = readFileSync(p, 'utf-8');
// Not a baseline: the author still has to drop in their member's base image.
if (raw.includes(PLACEHOLDER_MARKER)) continue;
const stripped = stripManagedBlocks(raw);
if (!matchesShipped(sharedDir, managed, stripped)) continue;
files[managed.taskPath] = sha256(stripped);
added = true;
}
if (!added) return false;
const stamp: ManagedStamp = {
version: 1,
stampedAt: new Date().toISOString(),
files,
};
writeFileSync(join(taskDir, STAMP_FILENAME), `${JSON.stringify(stamp, null, 2)}\n`);
return true;
}
function readStamp(taskDir: string): ManagedStamp | null {
const stampPath = join(taskDir, STAMP_FILENAME);
if (!existsSync(stampPath)) return null;
try {
const parsed = JSON.parse(readFileSync(stampPath, 'utf-8')) as ManagedStamp;
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
} catch {
// Treat a corrupt stamp as no stamp rather than blocking a trial over it.
return null;
}
}
/**
* How to restore a managed file, or undefined when this toolkit ships no copy to
* restore from. Only the Dockerfile can have several candidates (one per member).
*/
function restoreCommand(taskPath: string, candidates: string[], slug: string): string | undefined {
const dest = `harbor-tasks/${slug}/${taskPath}`;
if (candidates.length === 1) return `cp task-shared/${candidates[0]} ${dest}`;
if (candidates.length > 1) {
return `cp task-shared/Dockerfile.<your-member> ${dest} (list them: ls task-shared/Dockerfile.*)`;
}
// Never guess. Emitting the multi-candidate Dockerfile line here would tell an
// author to copy a Dockerfile over their grader prompt.
return undefined;
}
/** Render a restore line, saying so plainly when there is nothing to restore from. */
function restoreLine(f: FileVerdict): string {
return f.restore
? ` ${f.restore}`
: ` (no copy of ${f.taskPath} ships in task-shared/ — re-extract the toolkit zip)`;
}
/** Does this content match a pristine copy the toolkit ships? */
function matchesShipped(sharedDir: string, managed: ManagedFile, stripped: string): boolean {
return baselineCandidates(sharedDir, managed.baselinePattern).some(
(c) => stripManagedBlocks(readFileSync(join(sharedDir, c), 'utf-8')) === stripped
);
}
/**
* Was this task created by this toolkit generation? Task creation (the packed
* scaffold's task.toml and snapshot-to-task) writes `[metadata].toolkit_version`;
* a task directory without the key was authored on a different toolkit
* generation and grades with the assets frozen in its own tests/ directory, so
* comparing those against this toolkit's copies would report drift that is not
* an edit. Presence-based on purpose: wall-clock stamps cannot separate the
* generations, because tasks from an earlier generation are completed after
* later kits ship.
*
* A regex rather than a TOML parser, for the same shipped-dependency reason as
* input-checksums.ts readGitref: the one `toolkit_version` key in a task.toml
* is `[metadata].toolkit_version`.
*/
export function isTaskFromThisToolkitGeneration(taskDir: string): boolean {
const tomlPath = join(taskDir, 'task.toml');
if (!existsSync(tomlPath)) return false;
try {
return /^\s*toolkit_version\s*=/m.test(readFileSync(tomlPath, 'utf-8'));
} catch {
return false;
}
}
/**
* Compare a task's managed files against its creation-time stamp.
*
* @param taskDir Absolute path to `harbor-tasks/<slug>`.
* @param toolkitRoot Absolute path to the toolkit root (holds `task-shared/`).
*/
export function checkTaskInfraIntegrity(taskDir: string, toolkitRoot: string): IntegrityReport {
const sharedDir = join(toolkitRoot, 'task-shared');
// Without task-shared/ there is nothing to compare against; report "not
// checked" so callers no-op rather than reporting three phantom failures.
if (!existsSync(sharedDir)) {
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
}
// A task authored on a different toolkit generation grades with the assets
// frozen in its own tests/ directory. Comparing those against this toolkit's
// copies would report drift that is not an edit — and the printed remedy
// (restore the current copy) would change how that task grades. Skip it.
if (!isTaskFromThisToolkitGeneration(taskDir)) {
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
}
const slug = basename(taskDir);
const stamp = readStamp(taskDir);
const files: FileVerdict[] = [];
for (const managed of MANAGED_FILES) {
const taskFile = join(taskDir, managed.taskPath);
if (!existsSync(taskFile)) {
files.push({ taskPath: managed.taskPath, status: 'missing' });
continue;
}
const raw = readFileSync(taskFile, 'utf-8');
const candidates = baselineCandidates(sharedDir, managed.baselinePattern);
const restore = restoreCommand(managed.taskPath, candidates, slug);
const stripped = stripManagedBlocks(raw);
// FIRST: is this byte-for-byte something the toolkit ships right now? If so it
// cannot be an author edit, whatever the stamp says — and asking the stamp first
// is what used to make restoring the current copy (which is exactly what we tell
// authors to do) look like an edit, with no way out.
if (matchesShipped(sharedDir, managed, stripped)) {
files.push({ taskPath: managed.taskPath, status: 'ok' });
continue;
}
const expected = stamp?.files[managed.taskPath];
if (expected) {
// Matches its baseline but nothing shipped: untouched by the author, and the
// toolkit has moved on since. Worth saying, nobody's fault.
const status = sha256(stripped) === expected ? 'outdated' : 'modified';
files.push({ taskPath: managed.taskPath, status, restore });
continue;
}
// Checked after the stamp so that adding this marker to a file that HAS a
// baseline can't exempt it from the comparison.
if (raw.includes(PLACEHOLDER_MARKER)) {
files.push({ taskPath: managed.taskPath, status: 'placeholder' });
continue;
}
files.push({ taskPath: managed.taskPath, status: 'unverifiable', restore });
}
return {
checked: true,
files,
modified: files.filter((f) => f.status === 'modified'),
unverifiable: files.filter((f) => f.status === 'unverifiable'),
outdated: files.filter((f) => f.status === 'outdated'),
};
}
/**
* Human-readable report. Returns '' when there is nothing worth saying, so callers
* can `if (msg) print(msg)`.
*
* Deliberately not phrased as a refusal. An author who changed one of these files
* almost always did it to get unstuck, not knowing we'd rather they told us — so
* this explains what it means for their task and what restoring would do, and then
* lets them get on with it.
*/
export function formatIntegrityReport(report: IntegrityReport): string {
const sections: string[] = [];
if (report.modified.length > 0) {
sections.push(
[
'These files look edited since this task was created, and the toolkit manages',
'them — they set up how the trial runs and how the grade is produced, so they',
"have to be identical across every task. Yours aren't, which makes this task's",
"runs hard to compare with everyone else's:",
'',
...report.modified.map((f) => ` ${f.taskPath}`),
'',
'Restoring the shipped version puts that right:',
...report.modified.map(restoreLine),
'',
'If you changed one to work around a problem — a missing package, a grader that',
"wouldn't run — please tell us about the problem instead. It almost certainly",
'affects other authors too, and the fix belongs in the toolkit, not in one task.',
'Nothing here stops you running trials or submitting.',
].join('\n')
);
}
if (report.outdated.length > 0) {
sections.push(
[
'These files are unchanged, but the toolkit has shipped newer copies since this',
'task was created:',
'',
...report.outdated.map((f) => ` ${f.taskPath}`),
'',
"You haven't done anything wrong. It does mean this task was run and graded with",
"older versions than a task built today, so its scores aren't directly",
'comparable. To line them up, restore the current copies and re-run your trials:',
...report.outdated.map(restoreLine),
].join('\n')
);
}
if (report.unverifiable.length > 0) {
sections.push(
[
"These files don't match the copies this toolkit ships, and this task has no",
'record of what they looked like when it was created:',
'',
...report.unverifiable.map((f) => ` ${f.taskPath}`),
'',
'Two things look like this and we cannot tell them apart: a task created on an',
'earlier toolkit release (nothing to fix, though its scores are not directly',
'comparable to a task built today), or a file that was edited. Either way,',
'restoring the current copy and re-running your trials is what makes this task',
"comparable to everyone else's:",
...report.unverifiable.map(restoreLine),
].join('\n')
);
}
return sections.join('\n\n');
}
/**
* Wrap a report in a banner loud enough to survive a scrollback.
*
* Nothing blocks any more, so this notice is the entire mechanism — and an
* unframed paragraph among build output is one a reasonable person scrolls past.
*/
export function bannerize(message: string, report: IntegrityReport): string {
return banner(
message,
report.modified.length > 0
? '!! TOOLKIT-MANAGED FILES LOOK EDITED — PLEASE READ !!'
: '!! TOOLKIT-MANAGED FILES NEED A LOOK — PLEASE READ !!'
);
}

View File

@@ -1,217 +0,0 @@
/**
* toolkit-script-integrity.ts — detect edits to the toolkit's own scripts.
*
* Sibling of task-infra-integrity.ts, which covers a task's managed files. This
* covers `scripts/`. The scripts never ship with a task, so an edit can't reach
* the delivered workspace — but their OUTPUT does: `build-workspace.sh` alone
* stages `tests/test-commands.sh` (the deterministic checks behind the
* correctness score), writes the Dockerfile's toolkit-managed blocks, and
* records the managed stamp and input checksums. Nothing downstream re-derives
* those, and the reference runs can't be re-derived at all.
*
* The baseline is a manifest written at package time ({@link writeScriptManifest}),
* so it ships in the same zip as the scripts it describes. That removes the
* ambiguity a task's managed files have: there is no "created on an older
* release" case to tell apart, so a hash mismatch is an edit. Files absent from
* the manifest are ignored, which keeps a worker's own helper script — or a
* `__pycache__` left by a harbor run — from ever being reported.
*/
import { createHash } from 'crypto';
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
import { join, relative } from 'path';
import { banner } from './notice-banner.js';
/** Manifest of the shipped `scripts/` tree. Lives at the toolkit root. */
export const SCRIPT_MANIFEST_FILENAME = '.toolkit-scripts.json';
const MANIFEST_VERSION = 1;
/** Runtime droppings, never part of the shipped tree. */
const IGNORED_DIRS = new Set(['__pycache__', 'node_modules', '.git']);
const IGNORED_FILES = /\.(pyc|pyo)$/;
/**
* `modified` — content differs from what shipped: an edit.
* `missing` — shipped, but no longer on disk.
* `ok` — unchanged.
*/
export type ScriptStatus = 'ok' | 'modified' | 'missing';
export interface ScriptVerdict {
/** Toolkit-relative path, e.g. `scripts/build-workspace.sh`. */
path: string;
status: ScriptStatus;
}
export interface ScriptIntegrityReport {
/** False when no manifest ships — callers should skip silently. */
checked: boolean;
files: ScriptVerdict[];
modified: ScriptVerdict[];
missing: ScriptVerdict[];
}
interface ScriptManifest {
version: number;
generatedAt: string;
/** Toolkit-relative path → sha256 of the normalized content. */
files: Record<string, string>;
}
/**
* Line endings and trailing whitespace are normalized away: a Windows editor, a
* checkout with core.autocrlf, or a formatter trimming a final newline must not
* read as an edit.
*/
function hashContent(content: string): string {
return createHash('sha256')
.update(content.replace(/\r\n/g, '\n').replace(/\s+$/, ''))
.digest('hex');
}
/** Every shipped file under `scripts/`, as toolkit-relative paths. */
function walkScripts(dir: string, toolkitRoot: string): string[] {
if (!existsSync(dir)) return [];
const out: string[] = [];
for (const entry of readdirSync(dir, { withFileTypes: true }).sort((a, b) =>
a.name.localeCompare(b.name)
)) {
const abs = join(dir, entry.name);
if (entry.isDirectory()) {
if (!IGNORED_DIRS.has(entry.name)) out.push(...walkScripts(abs, toolkitRoot));
continue;
}
if (!entry.isFile() || IGNORED_FILES.test(entry.name)) continue;
out.push(relative(toolkitRoot, abs));
}
return out;
}
/**
* Record the shipped `scripts/` tree. Call at package time, once the tree is
* fully staged — anything written to `scripts/` afterwards reads as an edit.
*
* Returns the number of files recorded.
*/
export function writeScriptManifest(toolkitRoot: string): number {
const files: Record<string, string> = {};
for (const rel of walkScripts(join(toolkitRoot, 'scripts'), toolkitRoot)) {
files[rel] = hashContent(readFileSync(join(toolkitRoot, rel), 'utf-8'));
}
const manifest: ScriptManifest = {
version: MANIFEST_VERSION,
generatedAt: new Date().toISOString(),
files,
};
writeFileSync(
join(toolkitRoot, SCRIPT_MANIFEST_FILENAME),
`${JSON.stringify(manifest, null, 2)}\n`
);
return Object.keys(files).length;
}
function readManifest(toolkitRoot: string): ScriptManifest | null {
const p = join(toolkitRoot, SCRIPT_MANIFEST_FILENAME);
if (!existsSync(p)) return null;
try {
const parsed = JSON.parse(readFileSync(p, 'utf-8')) as ScriptManifest;
if (parsed?.version !== MANIFEST_VERSION) return null;
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
} catch {
// A corrupt manifest is treated as no manifest rather than blocking a trial.
return null;
}
}
/**
* Compare the toolkit's `scripts/` tree against the manifest it shipped with.
*
* @param toolkitRoot Absolute path to the toolkit root (holds `scripts/`).
*/
export function checkToolkitScriptIntegrity(toolkitRoot: string): ScriptIntegrityReport {
const manifest = readManifest(toolkitRoot);
if (!manifest) return { checked: false, files: [], modified: [], missing: [] };
const files: ScriptVerdict[] = Object.entries(manifest.files).map(([path, expected]) => {
const abs = join(toolkitRoot, path);
if (!existsSync(abs)) return { path, status: 'missing' as const };
const status = hashContent(readFileSync(abs, 'utf-8')) === expected ? 'ok' : 'modified';
return { path, status };
});
return {
checked: true,
files,
modified: files.filter((f) => f.status === 'modified'),
missing: files.filter((f) => f.status === 'missing'),
};
}
/**
* Human-readable report. Returns '' when there is nothing worth saying, so callers
* can `if (msg) print(msg)`.
*
* Deliberately not phrased as a refusal, for the same reason the managed-file
* notice isn't: an author who changed one of these did it to get unstuck, and the
* fix they needed almost certainly belongs in the toolkit rather than in their copy.
*/
export function formatScriptIntegrityReport(report: ScriptIntegrityReport): string {
const sections: string[] = [];
if (report.modified.length > 0) {
sections.push(
[
'These toolkit scripts look edited:',
'',
...report.modified.map((f) => ` ${f.path}`),
'',
"They aren't part of any task, so an edit is easy to miss — but what they write",
'is. Building a task stages its deterministic checks, fills in parts of its',
'Dockerfile, and records the checksums a reviewer reads; a script that does any of',
'that differently produces a task that looks normal and behaves differently from',
'every other one.',
'',
'Re-extracting the toolkit zip over your copy restores them. Your tasks, snapshots',
'and reference runs are untouched by that.',
'',
'If you changed one to work around a problem — a build that would not run, a',
'missing dependency — please tell us about the problem instead. It almost',
'certainly affects other authors too, and the fix belongs in the toolkit.',
'Nothing here stops you running trials or submitting.',
].join('\n')
);
}
if (report.missing.length > 0) {
sections.push(
[
'These toolkit scripts shipped with this release but are no longer here:',
'',
...report.missing.map((f) => ` ${f.path}`),
'',
'Something that depends on one will fail partway through rather than up front.',
'Re-extract the toolkit zip over your copy to put them back.',
].join('\n')
);
}
return sections.join('\n\n');
}
/** The full notice, bannered and ready to write to stderr, or '' if all is well. */
export function scriptIntegrityNotice(report: ScriptIntegrityReport): string {
const message = formatScriptIntegrityReport(report);
if (!message) return '';
const headline =
report.modified.length > 0
? '!! TOOLKIT SCRIPTS LOOK EDITED — PLEASE READ !!'
: '!! TOOLKIT SCRIPTS ARE MISSING — PLEASE READ !!';
return banner(message, headline);
}

View File

@@ -1,304 +0,0 @@
/**
* Tests for tree-permissions.ts.
*
* The load-bearing case is the one from the field report: a directory that came
* across without its search bit makes `tar` fail with `Cannot stat` on the files
* *inside* it, so the repair has to fix directory modes, not just ownership.
* These tests run unprivileged, so they exercise the mode axis for real and the
* ownership axis only as far as an unprivileged process can (target resolution +
* graceful EPERM), which is the same shape CI runs in. One case needs root and
* skips otherwise; the rest hold under either uid, which is why the fixtures that
* must look human-owned say so with `ownedByHuman` instead of relying on the
* caller's uid.
*/
import assert from 'node:assert/strict';
import {
chmodSync,
chownSync,
mkdirSync,
rmSync,
statSync,
symlinkSync,
writeFileSync,
} from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { test } from 'node:test';
import {
didRepair,
manualRepairHint,
normalizeTreePermissions,
resolveWorkspaceOwner,
} from './tree-permissions';
function scratch(name: string): string {
const dir = join(tmpdir(), `tree-perms-${name}-${process.pid}`);
rmSync(dir, { recursive: true, force: true });
mkdirSync(dir, { recursive: true });
return dir;
}
const RUNNING_AS_ROOT = process.getuid?.() === 0;
const HUMAN_UID = RUNNING_AS_ROOT ? 1000 : (process.getuid?.() ?? 0);
const HUMAN_GID = RUNNING_AS_ROOT ? 1000 : (process.getgid?.() ?? 0);
/** Give a fixture a non-root owner, so the repair sees a tree it can hand back. */
function ownedByHuman(path: string): string {
chownSync(path, HUMAN_UID, HUMAN_GID);
return path;
}
test('restores the search bit on a directory that lost it', () => {
const root = scratch('searchbit');
const models = join(root, 'agent-output', 'app', 'models');
mkdirSync(models, { recursive: true });
writeFileSync(join(models, 'bill.rb'), 'class Bill; end\n');
// r-- : readdir works, so tar can NAME the file, but stat is refused.
chmodSync(models, 0o400);
const report = normalizeTreePermissions(root);
assert.equal(statSync(models).mode & 0o700, 0o700, 'owner rwx restored on the directory');
assert.ok(report.modeFixed.some((p) => p === models));
assert.ok(didRepair(report));
rmSync(root, { recursive: true, force: true });
});
test('recurses into a directory it had to widen first', () => {
const root = scratch('recurse');
const inner = join(root, 'locked', 'deeper');
mkdirSync(inner, { recursive: true });
const leaf = join(inner, 'leaf.rb');
writeFileSync(leaf, 'x\n');
chmodSync(leaf, 0o000);
chmodSync(inner, 0o400);
chmodSync(join(root, 'locked'), 0o400);
const report = normalizeTreePermissions(root);
// Only reachable if the walk widened each parent before descending.
assert.equal(statSync(leaf).mode & 0o600, 0o600, 'leaf became owner-readable');
assert.ok(report.modeFixed.includes(leaf));
rmSync(root, { recursive: true, force: true });
});
test('leaves already-correct trees untouched', () => {
const root = scratch('noop');
mkdirSync(join(root, 'sub'), { recursive: true });
writeFileSync(join(root, 'sub', 'f.txt'), 'hi\n');
const report = normalizeTreePermissions(root);
assert.deepEqual(report.modeFixed, [], 'no mode changes');
assert.deepEqual(report.ownerFixed, [], 'no owner changes (already ours)');
assert.deepEqual(report.failures, []);
assert.equal(didRepair(report), false);
rmSync(root, { recursive: true, force: true });
});
test('does not widen group/other beyond what was already there', () => {
const root = ownedByHuman(scratch('narrow'));
const f = join(root, 'secret.txt');
writeFileSync(f, 'x\n');
ownedByHuman(f);
chmodSync(f, 0o000);
normalizeTreePermissions(root, { ownerRef: root });
const mode = statSync(f).mode & 0o777;
assert.equal(mode, 0o600, 'owner rw only — group/other stay closed');
rmSync(root, { recursive: true, force: true });
});
test('ignores symlinks rather than following them out of the tree', () => {
const root = scratch('symlink');
const outside = scratch('symlink-outside');
const victim = join(outside, 'victim.txt');
writeFileSync(victim, 'x\n');
chmodSync(victim, 0o000);
symlinkSync(outside, join(root, 'link'));
const report = normalizeTreePermissions(root);
assert.equal(statSync(victim).mode & 0o777, 0o000, 'target outside the tree untouched');
assert.deepEqual(report.failures, []);
rmSync(root, { recursive: true, force: true });
rmSync(outside, { recursive: true, force: true });
});
test('never throws on a missing root, and reports it', () => {
const report = normalizeTreePermissions(join(tmpdir(), 'definitely-not-here-xyz'));
assert.equal(report.failures.length, 1);
assert.equal(report.failures[0].reason, 'ENOENT');
});
test('resolveWorkspaceOwner reads the reference path, not the caller', () => {
const root = scratch('owner');
const owner = resolveWorkspaceOwner(root);
assert.ok(owner, 'resolved');
const st = statSync(root);
assert.equal(owner.uid, st.uid);
assert.equal(owner.gid, st.gid);
assert.equal(resolveWorkspaceOwner(join(tmpdir(), 'nope-xyz')), null);
rmSync(root, { recursive: true, force: true });
});
test('never chowns TO root, even when the owner ref is root-owned', () => {
// The regression this guards: workspace root owned by root (unzipped with
// sudo) while the task files are correctly owned by the human. Chowning to the
// ref's owner would inflict the very lockout this module prevents. `/` is
// root-owned on every platform we run on, so it's a stable stand-in.
const root = scratch('root-ref');
const f = join(root, 'mine.txt');
writeFileSync(f, 'x\n');
const beforeUid = statSync(f).uid;
const report = normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(report.target?.uid, 0, 'resolved a root target');
assert.deepEqual(report.ownerFixed, [], 'declined to chown anything to root');
assert.deepEqual(report.failures, [], 'and did not fail trying');
assert.equal(statSync(f).uid, beforeUid, 'owner untouched');
rmSync(root, { recursive: true, force: true });
});
test('still normalizes modes when the chown target is root', () => {
const root = scratch('root-ref-modes');
const sub = join(root, 'sub');
mkdirSync(sub, { recursive: true });
writeFileSync(join(sub, 'f.txt'), 'x\n');
chmodSync(sub, 0o400);
const report = normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(statSync(sub).mode & 0o700, 0o700, 'mode axis still applied');
assert.ok(report.modeFixed.includes(sub));
rmSync(root, { recursive: true, force: true });
});
test('keeps modes narrow for files that have a real owner, even under a root ref', () => {
// The complement of the case below: we declined to chown, but these entries are
// already the human's, so owner bits reach them and nothing should be widened.
const root = ownedByHuman(scratch('root-ref-narrow'));
const f = join(root, 'mine.txt');
writeFileSync(f, 'x\n');
ownedByHuman(f);
chmodSync(f, 0o600);
normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(statSync(f).mode & 0o077, 0, 'group/other untouched');
rmSync(root, { recursive: true, force: true });
});
test(
'grants read+search to group and other on files stranded root-owned',
{ skip: process.getuid?.() !== 0 ? 'needs root to create root-owned files' : false },
() => {
// The worker authoring container: root process, root-owned workspace. The chown
// is declined, so owner bits land on root and the human — a different uid in
// Explore and on a WSL host — is still locked out of a --w------- capture.
const root = scratch('stranded');
const sub = join(root, 'agent-output');
mkdirSync(sub, { recursive: true });
const f = join(sub, 'answer.md');
writeFileSync(f, 'x\n');
chmodSync(f, 0o200);
chmodSync(sub, 0o300);
normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(
statSync(f).mode & 0o777,
0o644,
'file readable by everyone, writable by none but root'
);
assert.equal(statSync(sub).mode & 0o777, 0o755, 'directory searchable');
}
);
test('walks a tree as deep as the filesystem allows', () => {
const root = scratch('deep');
// PATH_MAX caps how deep a tree can physically get (~300 levels at these name
// lengths — building deeper fails with ENAMETOOLONG), which is well inside any
// call-stack limit. So this isn't a stack test; it just pins that a deep,
// narrow tree walks cleanly end to end.
let path = root;
for (let i = 0; i < 250; i++) {
path = join(path, `d${i}`);
}
mkdirSync(path, { recursive: true });
writeFileSync(join(path, 'leaf.txt'), 'x\n');
chmodSync(join(path, 'leaf.txt'), 0o000);
const report = normalizeTreePermissions(root);
assert.deepEqual(report.failures, [], 'walked the whole depth cleanly');
assert.equal(statSync(join(path, 'leaf.txt')).mode & 0o600, 0o600, 'reached the deepest leaf');
rmSync(root, { recursive: true, force: true });
});
test('a failure in one subtree does not abandon the rest', () => {
const root = scratch('partial');
const good = join(root, 'good');
mkdirSync(good, { recursive: true });
const goodFile = join(good, 'f.txt');
writeFileSync(goodFile, 'x\n');
chmodSync(goodFile, 0o000);
// A dangling symlink and a vanished path both produce per-entry trouble.
symlinkSync(join(root, 'nowhere'), join(root, 'dangling'));
const report = normalizeTreePermissions(root);
assert.equal(statSync(goodFile).mode & 0o600, 0o600, 'the healthy subtree was still repaired');
assert.ok(report.modeFixed.includes(goodFile));
rmSync(root, { recursive: true, force: true });
});
test('reports rather than throws when the root is a file, not a directory', () => {
const root = scratch('file-root');
const f = join(root, 'lonely.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
const report = normalizeTreePermissions(f);
assert.equal(statSync(f).mode & 0o600, 0o600);
assert.deepEqual(report.failures, []);
rmSync(root, { recursive: true, force: true });
});
test('RACCOON_SKIP_PERMISSION_REPAIR=1 makes it a total no-op', () => {
const root = scratch('killswitch');
const sub = join(root, 'sub');
mkdirSync(sub, { recursive: true });
const f = join(sub, 'f.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
chmodSync(sub, 0o400);
const prev = process.env.RACCOON_SKIP_PERMISSION_REPAIR;
process.env.RACCOON_SKIP_PERMISSION_REPAIR = '1';
try {
const report = normalizeTreePermissions(root);
assert.equal(report.skipped, true);
assert.deepEqual(report.modeFixed, []);
assert.deepEqual(report.ownerFixed, []);
assert.deepEqual(report.failures, []);
assert.equal(didRepair(report), false);
assert.equal(statSync(sub).mode & 0o777, 0o400, 'directory left exactly as it was');
} finally {
if (prev === undefined) delete process.env.RACCOON_SKIP_PERMISSION_REPAIR;
else process.env.RACCOON_SKIP_PERMISSION_REPAIR = prev;
}
chmodSync(sub, 0o700);
rmSync(root, { recursive: true, force: true });
});
test('manual hint repairs both axes, ownership first', () => {
const hint = manualRepairHint('harbor-tasks/my-slug');
assert.match(hint, /chown -R/);
assert.match(hint, /chmod -R u\+rwX/);
assert.ok(hint.indexOf('chown') < hint.indexOf('chmod'), 'chown before chmod');
});

View File

@@ -1,126 +0,0 @@
/**
* tree-permissions.ts — make a copied tree readable by whoever owns the workspace.
*
* Files captured from a task run can arrive owned by another user, or with a
* directory missing the permission needed to walk into it. Packaging then fails
* with `Cannot stat: Permission denied`. This repairs both.
*
* Grants owner rwX only, never group or other. Never throws, and never hands
* files to root. Set `RACCOON_SKIP_PERMISSION_REPAIR=1` to turn it off.
*/
import { chmodSync, chownSync, lstatSync, readdirSync, statSync } from 'fs';
import { join } from 'path';
export interface NormalizeReport {
/** Paths whose owner was changed. */
ownerFixed: string[];
/** Paths whose mode gained owner rwX. */
modeFixed: string[];
/** Paths we wanted to change but could not, with the errno. */
failures: { path: string; reason: string }[];
/** Resolved target owner, or null if it couldn't be determined. */
target: { uid: number; gid: number } | null;
/** Set when disabled via RACCOON_SKIP_PERMISSION_REPAIR. */
skipped?: boolean;
}
/** Owner a workspace tree should have: whoever owns `ownerRef`. */
export function resolveWorkspaceOwner(ownerRef: string): { uid: number; gid: number } | null {
try {
const st = statSync(ownerRef);
return { uid: st.uid, gid: st.gid };
} catch {
return null;
}
}
/**
* Owner-rwX mode, preserving every other bit. Dirs also need the search bit.
*
* `stranded` means the file stays root-owned because we have no non-root owner to
* give it to. Owner bits then help nobody — whoever has to read it is a different
* user — so read and search are granted more widely. Never write, never +x on files.
*/
function withOwnerAccess(mode: number, isDir: boolean, stranded: boolean): number {
const owner = isDir ? 0o700 : 0o600;
return mode | owner | (stranded ? (isDir ? 0o055 : 0o044) : 0);
}
/**
* Give every entry under `root` to the workspace owner and make sure that owner
* can read and traverse it. Symlinks are skipped. Repairs what it can and
* reports what it couldn't; it never throws and never blocks its caller.
*/
export function normalizeTreePermissions(
root: string,
options: { ownerRef?: string } = {}
): NormalizeReport {
if (process.env.RACCOON_SKIP_PERMISSION_REPAIR === '1') {
return { ownerFixed: [], modeFixed: [], failures: [], target: null, skipped: true };
}
const target = resolveWorkspaceOwner(options.ownerRef ?? process.cwd());
const report: NormalizeReport = { ownerFixed: [], modeFixed: [], failures: [], target };
// Never hand files to root — that would lock the owner out rather than help.
const chownTarget = target && target.uid !== 0 ? target : null;
try {
const stack: string[] = [root];
while (stack.length > 0) {
const path = stack.pop() as string;
let st;
try {
st = lstatSync(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ELSTAT' });
continue;
}
if (st.isSymbolicLink()) continue;
const isDir = st.isDirectory();
// Mode first: a directory we can't search is one we can't descend into.
const wanted = withOwnerAccess(st.mode, isDir, chownTarget === null && st.uid === 0);
if (wanted !== st.mode) {
try {
chmodSync(path, wanted);
report.modeFixed.push(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHMOD' });
}
}
if (chownTarget && (st.uid !== chownTarget.uid || st.gid !== chownTarget.gid)) {
try {
chownSync(path, chownTarget.uid, chownTarget.gid);
report.ownerFixed.push(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHOWN' });
}
}
if (!isDir) continue;
try {
for (const entry of readdirSync(path)) stack.push(join(path, entry));
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'EREADDIR' });
}
}
} catch (err) {
report.failures.push({ path: root, reason: (err as NodeJS.ErrnoException).code ?? 'EWALK' });
}
return report;
}
/** True when something was actually repaired. */
export function didRepair(report: NormalizeReport): boolean {
return report.ownerFixed.length > 0 || report.modeFixed.length > 0;
}
/** The command to run on your host if we couldn't fix it ourselves. */
export function manualRepairHint(path: string): string {
return `sudo chown -R "$(id -un):$(id -gn)" ${path} && chmod -R u+rwX ${path}`;
}