lots of change - all to start my 3rd redo
This commit is contained in:
@@ -1,34 +0,0 @@
|
||||
# shellcheck shell=bash
|
||||
# call-origin.sh — build the X-Surge-Client-Metadata header value.
|
||||
#
|
||||
# Which surface a proxy call came from (a trial agent, the grader, Explore, a
|
||||
# dev box). Separate from llm-proxy-env.sh, which carries the project id and the
|
||||
# proxy routes: those are platform-internal, this is not, so this file is the
|
||||
# half that ships in the worker toolkit — worker runs go through the same proxy
|
||||
# and are attributed the same way.
|
||||
#
|
||||
# . scripts/lib/call-origin.sh
|
||||
# meta="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
|
||||
#
|
||||
# The proxy rejects the WHOLE CALL over a malformed metadata header (400
|
||||
# invalid_client_metadata), so an origin that is not a plain slug yields an
|
||||
# empty string and the caller sends no header at all: losing attribution beats
|
||||
# failing the call.
|
||||
|
||||
CALL_ORIGIN_HEADER="X-Surge-Client-Metadata"
|
||||
# An unlabelled call is still a real call, so it gets a bucket rather than no
|
||||
# header: a missing origin in the audit log then means an unplumbed surface.
|
||||
DEFAULT_CALL_ORIGIN="local"
|
||||
|
||||
# Compact JSON for the header, or empty when LLM_CALL_ORIGIN is unusable.
|
||||
# Only a slug matching this pattern is ever interpolated, so nothing needs
|
||||
# JSON-escaping and this stays dependency-free (it is sourced in worker
|
||||
# containers, which have no python).
|
||||
call_origin_metadata() {
|
||||
local origin="${LLM_CALL_ORIGIN:-$DEFAULT_CALL_ORIGIN}"
|
||||
case "$origin" in
|
||||
"" | *[!a-z0-9._-]* | [!a-z0-9]*) return 0 ;;
|
||||
esac
|
||||
[ "${#origin}" -le 64 ] || return 0
|
||||
printf '{"origin":"%s"}' "$origin"
|
||||
}
|
||||
@@ -1,19 +0,0 @@
|
||||
import { existsSync } from 'node:fs';
|
||||
|
||||
(function checkDevcontainer() {
|
||||
if (process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE === '1') return;
|
||||
process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE = '1';
|
||||
|
||||
const inContainer = process.env.IN_DEVCONTAINER === '1' || existsSync('/.dockerenv');
|
||||
if (inContainer) return;
|
||||
if (process.env.SUPPRESS_DEVCONTAINER_WARNING === '1') return;
|
||||
if (process.env.CI === 'true' || process.env.CI === '1') return;
|
||||
|
||||
const yellow = '\x1b[33m';
|
||||
const reset = '\x1b[0m';
|
||||
process.stderr.write(
|
||||
`${yellow}Warning: this script is meant to run inside the toolkit devcontainer.${reset}\n` +
|
||||
` Reopen this toolkit folder in its devcontainer and run the command again.\n` +
|
||||
` (suppress with SUPPRESS_DEVCONTAINER_WARNING=1)\n`
|
||||
);
|
||||
})();
|
||||
@@ -1,36 +0,0 @@
|
||||
"""How codex is handed its API key, kept out of codex_agent so it is testable without
|
||||
harbor (whose venv has no pytest, so anything importing it SKIPs in CI).
|
||||
|
||||
codex reads its key from `$CODEX_HOME/auth.json` and its proxy URL from config.toml —
|
||||
`OPENAI_API_KEY` / `OPENAI_BASE_URL` in the environment are both ignored, verified against
|
||||
0.146.0 and 0.152.0 (an env-var-only run sends no `authorization` header at all).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import shlex
|
||||
|
||||
# Characters harbor's own auth.json writer cannot survive: it interpolates the key into a
|
||||
# shell heredoc, so `"` closes the JSON string and `\` starts an escape.
|
||||
_UNESCAPABLE = '"\\\n\r'
|
||||
|
||||
|
||||
AUTH_JSON_ENV_VAR = "RACCOON_CODEX_AUTH_JSON"
|
||||
|
||||
|
||||
def auth_json_setup(key: str, remote_auth_path: str) -> tuple[dict[str, str], str]:
|
||||
"""The one extra env var — returned separately so it reaches ONLY the setup exec — plus
|
||||
shell writing a parseable auth.json. Subshell: the umask must not outlive this write."""
|
||||
env = {AUTH_JSON_ENV_VAR: json.dumps({"OPENAI_API_KEY": key})}
|
||||
command = (
|
||||
f"(umask 077; printf '%s\\n' \"${AUTH_JSON_ENV_VAR}\" "
|
||||
f">{shlex.quote(remote_auth_path)})\n"
|
||||
)
|
||||
return env, command
|
||||
|
||||
|
||||
def unescapable_chars(key: str) -> list[str]:
|
||||
"""Which characters in `key` harbor's stock heredoc writer would corrupt — empty for
|
||||
every ordinary key, so the caller can refuse instead of 401ing three layers down."""
|
||||
return sorted({c for c in _UNESCAPABLE if c in key})
|
||||
@@ -1,43 +0,0 @@
|
||||
/**
|
||||
* Recursive copy for scripts that must not call `cpSync`: it fails EACCES
|
||||
* against a macOS docker bind mount, where the toolkit's job dirs live.
|
||||
*/
|
||||
|
||||
import {
|
||||
chmodSync,
|
||||
copyFileSync,
|
||||
lstatSync,
|
||||
mkdirSync,
|
||||
readdirSync,
|
||||
readlinkSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
symlinkSync,
|
||||
} from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
/** Copy one entry — symlink, directory or file — preserving its mode. */
|
||||
export function copyPath(src: string, dest: string) {
|
||||
const st = lstatSync(src);
|
||||
if (st.isSymbolicLink()) {
|
||||
rmSync(dest, { force: true });
|
||||
symlinkSync(readlinkSync(src), dest);
|
||||
return;
|
||||
}
|
||||
if (st.isDirectory()) {
|
||||
copyTree(src, dest);
|
||||
return;
|
||||
}
|
||||
// Unlink first: copyFileSync onto an existing file keeps that file's mode.
|
||||
rmSync(dest, { force: true });
|
||||
copyFileSync(src, dest);
|
||||
chmodSync(dest, statSync(src).mode & 0o777);
|
||||
}
|
||||
|
||||
/** Copy `src`'s contents into `dest`, creating `dest` if it doesn't exist. */
|
||||
export function copyTree(src: string, dest: string) {
|
||||
mkdirSync(dest, { recursive: true });
|
||||
for (const entry of readdirSync(src, { withFileTypes: true })) {
|
||||
copyPath(join(src, entry.name), join(dest, entry.name));
|
||||
}
|
||||
}
|
||||
@@ -1,137 +0,0 @@
|
||||
#!/bin/sh
|
||||
# Restrict this container's DNS to the hosts in DNSJAIL_ALLOW (space-separated), leaving
|
||||
# every other name unresolvable. Runs as root, inside the container.
|
||||
#
|
||||
# Baked into the task images and invoked by the agent (scripts/dnsjail.py); shipped to the
|
||||
# Explore container by the toolkit packaging. Both surfaces run this same file. Supplied from
|
||||
# outside: DNSJAIL_ALLOW, the hosts the agent will actually dial -- every one must resolve or
|
||||
# no jail happens -- and DNSJAIL_ALLOW_EXTRA, nice-to-haves that only warn if they do not.
|
||||
#
|
||||
# An unreachable model endpoint is a dead trial or a dead session, so nothing here is
|
||||
# applied before it is verified, and any doubt leaves the container's DNS untouched.
|
||||
set -u
|
||||
|
||||
STATE=/tmp/.dnsjail
|
||||
CONTROL=example.com # must NOT resolve through us; proves we reached our own filter
|
||||
|
||||
bounded() { if command -v timeout >/dev/null 2>&1; then timeout 5 "$@"; else "$@"; fi; }
|
||||
# Exact match: docker's own embedded resolver is 127.0.0.11, which a prefix match reads as
|
||||
# already-jailed — and then resolv.orig is never captured, so unjail has nothing to restore.
|
||||
jailed_now() { grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null; }
|
||||
|
||||
# Stop only the dnsmasq we started, so a declined run leaves nothing bound on :53 that a
|
||||
# later run could mistake for its own filter.
|
||||
drop_ours() {
|
||||
if [ -s "$STATE/dnsmasq.pid" ]; then
|
||||
pid=$(cat "$STATE/dnsmasq.pid")
|
||||
# /tmp survives docker stop/start but pids restart at 1, so last boot's pid may now be
|
||||
# some service's child. Confirm it is dnsmasq before signalling it.
|
||||
case "$(cat "/proc/$pid/comm" 2>/dev/null)" in
|
||||
dnsmasq) kill "$pid" 2>/dev/null || true ;;
|
||||
esac
|
||||
rm -f "$STATE/dnsmasq.pid" 2>/dev/null || true
|
||||
fi
|
||||
}
|
||||
|
||||
# Never `exit`: a caller may source this, so bailing out has to fall through rather than
|
||||
# end the caller's shell.
|
||||
dnsjail_apply() {
|
||||
required="${DNSJAIL_ALLOW:-}"
|
||||
extra="${DNSJAIL_ALLOW_EXTRA:-}"
|
||||
allow=$(echo $required $extra) # unquoted: collapses to a single-spaced word list
|
||||
# A blank required list means no model endpoint was found: jailing would strand the agent.
|
||||
set -- $required
|
||||
[ $# -gt 0 ] || return 0
|
||||
|
||||
# Already jailed by us, with our resolver alive and the same allowlist? Do nothing. Tearing
|
||||
# down and rebinding :53 races the kernel releasing the socket, and losing that race ends
|
||||
# in a fail-open restore -- so a second apply (the codex fresh path, rejail, run-app) would
|
||||
# silently UNjail a working container.
|
||||
if jailed_now && [ -s "$STATE/dnsmasq.pid" ] &&
|
||||
[ "$(cat "/proc/$(cat "$STATE/dnsmasq.pid")/comm" 2>/dev/null)" = "dnsmasq" ] &&
|
||||
[ "$(cat "$STATE/allow" 2>/dev/null)" = "$required" ] &&
|
||||
[ "$(cat "$STATE/allow-extra" 2>/dev/null)" = "$extra" ]; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
# The state dir has to work first: it holds what unjail restores, and a failed write here
|
||||
# is what would otherwise truncate /etc/resolv.conf. Sticky world-writable so run-app,
|
||||
# running as the container user in Explore, can drop its own lift markers.
|
||||
mkdir -p "$STATE" 2>/dev/null || return 0
|
||||
chmod 1777 "$STATE" 2>/dev/null || true
|
||||
: > "$STATE/.probe" 2>/dev/null || return 0
|
||||
rm -f "$STATE/.probe" 2>/dev/null || true
|
||||
|
||||
# Never forward to ourselves. Re-applying to an already-jailed container would otherwise
|
||||
# read 127.0.0.1 out of resolv.conf and point dnsmasq at its own socket, blackholing
|
||||
# every name.
|
||||
src=/etc/resolv.conf
|
||||
if jailed_now && [ -s "$STATE/resolv.orig" ]; then src="$STATE/resolv.orig"; fi
|
||||
up=$(awk '/^nameserver[ \t]+[0-9]+\./{print $2; exit}' "$src" 2>/dev/null)
|
||||
[ "$up" = "127.0.0.1" ] && up=""
|
||||
|
||||
if [ -n "$up" ] && command -v dnsmasq >/dev/null 2>&1; then
|
||||
srv=""
|
||||
for h in $allow; do srv="$srv --server=/$h/$up"; done
|
||||
drop_ours
|
||||
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
|
||||
# one would rather than an answer this resolver decided to keep.
|
||||
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
|
||||
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
|
||||
>/dev/null 2>>"$STATE/dnsmasq.err" || true
|
||||
fi
|
||||
|
||||
# Ask the resolver directly: the model endpoint must answer and the control must not --
|
||||
# otherwise we are looking at somebody else's resolver, not our filter. Only the FIRST
|
||||
# host gates the jail: an extra host that CNAMEs outside the allowlist cannot resolve
|
||||
# through the catch-all, and one of those must not silently disable the whole jail.
|
||||
live=1
|
||||
for h in $required; do
|
||||
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 || { live=""; break; }
|
||||
done
|
||||
if [ -n "$live" ] && bounded nslookup "$CONTROL" 127.0.0.1 >/dev/null 2>&1; then live=""; fi
|
||||
# The extras are reported, never fatal: one that CNAMEs outside the allowlist cannot
|
||||
# resolve through the catch-all, and must not take the whole jail down with it.
|
||||
if [ -n "$live" ]; then
|
||||
for h in $extra; do
|
||||
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 ||
|
||||
echo "dns-jail: $h does not resolve through the jail (CNAME outside the allowlist?)" >&2
|
||||
done
|
||||
fi
|
||||
|
||||
if [ -z "$live" ]; then
|
||||
# Say why. A silent decline is indistinguishable from a jail that worked, and the
|
||||
# reason is usually one line from dnsmasq (gVisor sandboxes, for instance, have no
|
||||
# AF_NETLINK, so dnsmasq cannot start there at all).
|
||||
echo "dns-jail: declined, this container keeps normal network access${DNSJAIL_WHY:-}" >&2
|
||||
[ -s "$STATE/dnsmasq.err" ] && sed 's/^/dns-jail: /' "$STATE/dnsmasq.err" >&2
|
||||
drop_ours
|
||||
# Failing open has to mean actually open, including when an earlier run left this
|
||||
# container jailed.
|
||||
if jailed_now && [ -s "$STATE/resolv.orig" ]; then
|
||||
cat "$STATE/resolv.orig" > /etc/resolv.conf 2>/dev/null || true
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Capture what unjail restores — but never overwrite it with an already-jailed file, which
|
||||
# would leave unjail a permanent no-op.
|
||||
if ! jailed_now; then
|
||||
cp /etc/resolv.conf "$STATE/resolv.orig" 2>/dev/null || return 0
|
||||
fi
|
||||
printf '%s\n' "$required" > "$STATE/allow" 2>/dev/null || true
|
||||
printf '%s\n' "$extra" > "$STATE/allow-extra" 2>/dev/null || true
|
||||
# A marker from a run-app that was killed would otherwise keep the jail disarmed forever.
|
||||
rm -rf "$STATE/lifts" 2>/dev/null || true
|
||||
|
||||
# /etc/resolv.conf is a bind mount, so it is truncated in place, never renamed over —
|
||||
# which means the replacement has to be complete BEFORE the write starts. Keep every
|
||||
# non-nameserver directive docker set (options, search).
|
||||
{ printf 'nameserver 127.0.0.1\n'
|
||||
grep -vE '^[[:space:]]*nameserver' /etc/resolv.conf
|
||||
} > "$STATE/resolv.jailed" 2>/dev/null
|
||||
[ -s "$STATE/resolv.jailed" ] || return 0
|
||||
cat "$STATE/resolv.jailed" > /etc/resolv.conf
|
||||
}
|
||||
|
||||
dnsjail_apply || true
|
||||
@@ -1,328 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Read the harness registry and derive per-harness credentials from it.
|
||||
#
|
||||
# Source it — the whole point is exporting into the caller's environment, which a subshell
|
||||
# would lose:
|
||||
#
|
||||
# HARNESS_SCRIPTS_DIR=/workspace/scripts . /workspace/scripts/lib/harness-credentials.sh
|
||||
# harness_setup_credentials
|
||||
#
|
||||
# Three callers: `harbor-run`, which needs only this; `refresh-harness-auth`, which
|
||||
# re-derives and rewrites the auth files before an interactive launch; and
|
||||
# `setup-harnesses.sh`, which sources it and adds installs, config writing and launchers
|
||||
# on top.
|
||||
#
|
||||
# No -e here — this file is SOURCED, and shell options belong to the caller's shell (both
|
||||
# post-creates run with -e). An unguarded failure below therefore aborts container
|
||||
# creation, which is why every failure site is individually guarded rather than relying on
|
||||
# this line.
|
||||
set -uo pipefail
|
||||
|
||||
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
|
||||
|
||||
# The registry is read with tomllib (stdlib from 3.11), and `python3` is not always new
|
||||
# enough — macOS ships 3.9, and a container may symlink an older managed interpreter. Pick
|
||||
# the first one that can actually import it rather than assuming.
|
||||
_raccoon_python() {
|
||||
local p
|
||||
for p in "${RACCOON_PYTHON:-}" python3 python3.13 python3.12 python3.11; do
|
||||
[ -n "$p" ] || continue
|
||||
command -v "$p" >/dev/null 2>&1 || continue
|
||||
if "$p" -c "import tomllib" >/dev/null 2>&1; then
|
||||
printf '%s' "$p"
|
||||
return 0
|
||||
fi
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
_harness_query() {
|
||||
local py
|
||||
py=$(_raccoon_python) || return 1
|
||||
"$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" "$@"
|
||||
}
|
||||
|
||||
# Drop every whitespace character from a value read out of .env. A Windows-saved .env leaves a
|
||||
# \r on each value, which reaches the proxy as a 401; no key or base URL legitimately contains
|
||||
# whitespace anywhere, so deleting rather than trimming needs no cases.
|
||||
_harness_trim() {
|
||||
local out
|
||||
# Fall back to the raw value: a trim that cannot run must never turn a working key into an
|
||||
# empty one, which is what an unavailable `tr` would otherwise do to every caller.
|
||||
out="$(printf '%s' "$1" | tr -d '[:space:]' 2>/dev/null)" || out="$1"
|
||||
printf '%s' "${out:-$1}"
|
||||
}
|
||||
|
||||
# The proxy root: the worker's ANTHROPIC_BASE_URL minus its provider path.
|
||||
_harness_proxy_root() {
|
||||
local base_url
|
||||
base_url="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
|
||||
[ -n "$base_url" ] || return 1
|
||||
base_url="${base_url%"${base_url##*[!/]}"}"
|
||||
# ".../llm_proxy/projects/<id>/anthropic" -> ".../llm_proxy/projects/<id>", so each
|
||||
# harness's proxy_path composes onto the project route. Requires a path to strip: a base
|
||||
# URL that is a bare host with no path — a provider's own API root rather than the
|
||||
# proxy — would yield "https:/", handed to codex as a base URL and failing obscurely.
|
||||
case "${base_url#*://}" in
|
||||
*/*) printf '%s' "${base_url%/*}" ;;
|
||||
*) return 2 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
harness_setup_credentials() {
|
||||
# `|| rc=$?` and not a bare assignment: this is sourced into a `set -e` shell (see the
|
||||
# note at the top), and a bare failing assignment would exit the caller's post-create
|
||||
# outright — silently, since the failure paths below are what do the explaining.
|
||||
local root rc=0
|
||||
root="$(_harness_proxy_root)" || rc=$?
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
if [ "$rc" -eq 2 ]; then
|
||||
echo "harness-setup: ANTHROPIC_BASE_URL (${ANTHROPIC_BASE_URL:-}) has no provider" >&2
|
||||
echo "harness-setup: path, so it is not the proxy URL other harnesses derive their" >&2
|
||||
echo "harness-setup: credentials from. claude will work; codex will not be" >&2
|
||||
echo "harness-setup: authenticated. Use the base URL you were given." >&2
|
||||
else
|
||||
echo "harness-setup: ANTHROPIC_BASE_URL unset — skipping credential derivation" >&2
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
ANTHROPIC_BASE_URL="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
|
||||
export ANTHROPIC_BASE_URL
|
||||
local key
|
||||
key="$(_harness_trim "${ANTHROPIC_API_KEY:-}")"
|
||||
if [ -z "$key" ]; then
|
||||
echo "harness-setup: ANTHROPIC_API_KEY unset — skipping credential derivation" >&2
|
||||
return 0
|
||||
fi
|
||||
# harbor-run sources .env itself and passes ANTHROPIC_* through to the trial sandbox, so
|
||||
# cleaning only the derived per-harness copies would leave a claude trial carrying the CR.
|
||||
export ANTHROPIC_API_KEY="$key"
|
||||
|
||||
local id key_env base_url_env proxy_path
|
||||
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
|
||||
[ -n "$key_env" ] || continue
|
||||
# ${!name} is an indirect expansion. Only set when empty: an explicit key wins.
|
||||
if [ -z "${!key_env:-}" ]; then
|
||||
export "$key_env=$key"
|
||||
fi
|
||||
if [ -n "$base_url_env" ] && [ -n "$proxy_path" ] && [ -z "${!base_url_env:-}" ]; then
|
||||
export "$base_url_env=$root/$proxy_path"
|
||||
fi
|
||||
echo "harness-setup: $id credentials ready ($key_env, ${base_url_env:-no base url})" >&2
|
||||
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# Write the auth file for harnesses that read credentials from disk rather than $ENV.
|
||||
harness_write_auth() {
|
||||
local id auth_path key_env target key py
|
||||
py=$(_raccoon_python) || {
|
||||
echo "harness-setup: no python3.11+ with tomllib — skipping auth files" >&2
|
||||
return 0
|
||||
}
|
||||
while IFS=$'\t' read -r id auth_path key_env; do
|
||||
[ -n "$auth_path" ] && [ -n "$key_env" ] || continue
|
||||
# Last mile: an explicit OPENAI_API_KEY bypasses the derivation above, so trim here
|
||||
# too — this is the value that reaches the file the harness authenticates with.
|
||||
key="$(_harness_trim "${!key_env:-}")"
|
||||
if [ -z "$key" ]; then
|
||||
echo "harness-setup: $key_env unset — skipping $id auth file" >&2
|
||||
continue
|
||||
fi
|
||||
target=$(eval "printf '%s' \"$auth_path\"") || {
|
||||
echo "harness-setup: WARNING $id auth_path could not be expanded — skipping" >&2
|
||||
continue
|
||||
}
|
||||
mkdir -p "$(dirname "$target")" || {
|
||||
echo "harness-setup: WARNING $id auth dir not creatable — skipping $target" >&2
|
||||
continue
|
||||
}
|
||||
# json.dumps, not printf: a key containing a quote or backslash would otherwise
|
||||
# produce a file the CLI cannot parse, and the failure would surface as an auth
|
||||
# error rather than a malformed file.
|
||||
# 0600 tmp + rename, never a redirect onto the target: a redirect truncates the live
|
||||
# file first, so a write dying mid-flight leaves codex an EMPTY auth.json.
|
||||
if ! RACCOON_AUTH_K="$key_env" RACCOON_AUTH_V="$key" RACCOON_AUTH_TARGET="$target" \
|
||||
"$py" -c 'import json, os
|
||||
target = os.environ["RACCOON_AUTH_TARGET"]
|
||||
tmp = target + ".raccoon-tmp." + str(os.getpid())
|
||||
try:
|
||||
with os.fdopen(os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600), "w") as fh:
|
||||
json.dump({os.environ["RACCOON_AUTH_K"]: os.environ["RACCOON_AUTH_V"]}, fh)
|
||||
fh.write("\n")
|
||||
os.replace(tmp, target)
|
||||
except OSError:
|
||||
try:
|
||||
os.unlink(tmp)
|
||||
except OSError:
|
||||
pass
|
||||
raise SystemExit(1)
|
||||
'; then
|
||||
echo "harness-setup: WARNING $id auth file NOT written — $target unwritable." >&2
|
||||
echo "harness-setup: the key already on disk (if any) is left untouched." >&2
|
||||
continue
|
||||
fi
|
||||
echo "harness-setup: $id auth -> $target" >&2
|
||||
done < <(_harness_query --auth-files 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# Re-set just the root keys of a harness's config file (codex's `openai_base_url`),
|
||||
# leaving every other line — the explore surface's [hooks] table included — untouched.
|
||||
harness_refresh_config_keys() {
|
||||
local id config_path blob target py
|
||||
py=$(_raccoon_python) || return 0
|
||||
# The surface only decides what a CREATE writes. An update takes the root keys off the
|
||||
# front of the same blob, so a surface's tables survive byte-for-byte either way.
|
||||
while IFS=$'\t' read -r id config_path blob; do
|
||||
[ -n "$config_path" ] && [ -n "$blob" ] || continue
|
||||
target=$(eval "printf '%s' \"$config_path\"") || continue
|
||||
mkdir -p "$(dirname "$target")" || continue
|
||||
if printf '%s' "$blob" | base64 -d |
|
||||
RACCOON_CONFIG_TARGET="$target" "$py" -c '
|
||||
import os, re, sys, tomllib
|
||||
|
||||
HEADER = "# Generated from harness-registry.toml — edits here are overwritten."
|
||||
|
||||
target = os.environ["RACCOON_CONFIG_TARGET"]
|
||||
text = sys.stdin.read()
|
||||
# Empty counts as unresolved: writing an empty base URL would break a container whose
|
||||
# config is currently right, which is the one thing this must never do.
|
||||
if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1))]:
|
||||
raise SystemExit(1)
|
||||
text = os.path.expandvars(text)
|
||||
|
||||
# Root keys, plus keys inside a [model_providers.*] table: codex reserves its built-in
|
||||
# provider ids, so the proxy URL it must follow lives in a provider table, not at the
|
||||
# root. Every other table, [hooks] on the explore surface included, is left alone.
|
||||
REFRESHABLE_TABLE = re.compile(r"\[model_providers\.[^]]+\]$")
|
||||
wanted = []
|
||||
section = None
|
||||
for line in text.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("["):
|
||||
section = stripped if REFRESHABLE_TABLE.match(stripped) else False
|
||||
continue
|
||||
if section is False:
|
||||
continue
|
||||
m = re.match(r"\s*\"?([A-Za-z0-9_.-]+)\"?\s*=", line)
|
||||
if m:
|
||||
wanted.append((section, m.group(1), line.rstrip()))
|
||||
if not wanted:
|
||||
raise SystemExit(0)
|
||||
|
||||
|
||||
def section_path(header):
|
||||
"""[model_providers.llm-proxy] -> ("model_providers", "llm-proxy")."""
|
||||
return tuple(header.strip("[]").split("."))
|
||||
|
||||
|
||||
def lookup(doc, header, key):
|
||||
"""The value a parsed config holds for a wanted key, or KeyError."""
|
||||
node = doc
|
||||
if header:
|
||||
for part in section_path(header):
|
||||
node = node[part]
|
||||
return node[key]
|
||||
|
||||
mode = None
|
||||
if os.path.exists(target):
|
||||
try:
|
||||
with open(target, encoding="utf-8") as fh:
|
||||
lines = fh.read().splitlines()
|
||||
mode = os.stat(target).st_mode & 0o777
|
||||
except OSError:
|
||||
raise SystemExit(1)
|
||||
def span(header):
|
||||
"""The line range a section owns, or None when the file has no such section.
|
||||
|
||||
Root is everything above the first table header: a key appended below one
|
||||
would be reparented into it, so searches and inserts stay inside the span.
|
||||
"""
|
||||
heads = [i for i, l in enumerate(lines) if l.lstrip().startswith("[")]
|
||||
if header is None:
|
||||
return 0, (heads[0] if heads else len(lines))
|
||||
at = next((i for i in heads if lines[i].strip() == header), None)
|
||||
if at is None:
|
||||
return None
|
||||
after = next((i for i in heads if i > at), len(lines))
|
||||
return at + 1, after
|
||||
|
||||
# Grouped, root first, so a section this file lacks can be written whole.
|
||||
grouped = {}
|
||||
for header, key, line in wanted:
|
||||
grouped.setdefault(header, []).append((key, line))
|
||||
ordered = sorted(grouped, key=lambda h: (h is not None, h or ""))
|
||||
|
||||
changed = False
|
||||
for header in ordered:
|
||||
if span(header) is None:
|
||||
# A config written before this section existed. Write the whole table
|
||||
# rather than leave a root key naming a provider that is not there.
|
||||
if lines and lines[-1].strip():
|
||||
lines.append("")
|
||||
lines.append(header)
|
||||
lines.extend(line for _, line in grouped[header])
|
||||
changed = True
|
||||
continue
|
||||
for key, line in grouped[header]:
|
||||
# Re-read the span: an insert for an earlier key moved it.
|
||||
start, end = span(header)
|
||||
# The quoted spelling is the same key: replace rather than duplicate.
|
||||
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
|
||||
at = next((i for i in range(start, end) if pat.match(lines[i])), None)
|
||||
if at is None:
|
||||
if end < len(lines) and lines[end].strip():
|
||||
lines.insert(end, "")
|
||||
lines.insert(end, line)
|
||||
changed = True
|
||||
elif lines[at] != line:
|
||||
lines[at] = line
|
||||
changed = True
|
||||
if not changed:
|
||||
raise SystemExit(0)
|
||||
out = "\n".join(lines).rstrip("\n") + "\n"
|
||||
else:
|
||||
# No file means container-create could not write one, so write what it would have:
|
||||
# on the explore surface that is the capture hooks too, not just the root keys.
|
||||
out = HEADER + "\n" + text
|
||||
|
||||
try:
|
||||
doc = tomllib.loads(out)
|
||||
except tomllib.TOMLDecodeError:
|
||||
raise SystemExit(1)
|
||||
# Parsing is not enough: a line edit can land inside a multi-line value, which still
|
||||
# parses while leaving the key unset. Require every key to have landed on the value the
|
||||
# blob asks for, in its own section — skipping sections this file does not carry.
|
||||
blob_doc = tomllib.loads(text)
|
||||
for header, key, _ in wanted:
|
||||
try:
|
||||
expected = lookup(blob_doc, header, key)
|
||||
except (KeyError, TypeError):
|
||||
raise SystemExit(1)
|
||||
try:
|
||||
got = lookup(doc, header, key)
|
||||
except (KeyError, TypeError):
|
||||
if header is None:
|
||||
raise SystemExit(1)
|
||||
continue
|
||||
if got != expected:
|
||||
raise SystemExit(1)
|
||||
|
||||
# Pid-suffixed: two launches at once must not write the same scratch path.
|
||||
tmp = target + ".raccoon-tmp." + str(os.getpid())
|
||||
try:
|
||||
with open(tmp, "w", encoding="utf-8") as fh:
|
||||
fh.write(out)
|
||||
if mode is not None:
|
||||
os.chmod(tmp, mode)
|
||||
os.replace(tmp, target)
|
||||
except OSError:
|
||||
try:
|
||||
os.unlink(tmp)
|
||||
except OSError:
|
||||
pass
|
||||
raise SystemExit(1)
|
||||
'; then
|
||||
echo "harness-setup: $id config keys refreshed -> $target" >&2
|
||||
fi
|
||||
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
|
||||
}
|
||||
@@ -1,418 +0,0 @@
|
||||
"""harness_registry.py — Python loader for ``scripts/harness-registry.toml``.
|
||||
|
||||
The ONE loader for the registry: TS callers shell into ``resolve_harness.py`` rather than
|
||||
parse the TOML themselves, which is why the toolkit ships no TOML parser for TS (its
|
||||
package.json has no zod/smol-toml).
|
||||
|
||||
This module supersedes ``benchmark_models_lib``'s ``HARNESS_BY_IMPORT_PATH`` and
|
||||
``LEGACY_BARE_MODEL_AGENTS``; those should read from here rather than keep private
|
||||
copies.
|
||||
|
||||
Harbor-free and dependency-free (stdlib ``tomllib``) so it can be imported from a
|
||||
sandbox agent, a plain unit test, or the devcontainer python alike.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import tomllib
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
REGISTRY_PATH = Path(__file__).resolve().parent.parent / "harness-registry.toml"
|
||||
|
||||
MODEL_ID_SHAPES = frozenset({"bare", "provider/model", "provider:model"})
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Harness:
|
||||
"""One harness, as declared in harness-registry.toml."""
|
||||
|
||||
id: str
|
||||
label: str
|
||||
agent_import_path: str
|
||||
model_id_shape: str
|
||||
writes_atif: bool
|
||||
capture: bool
|
||||
seed_native: bool
|
||||
seed_atif: bool
|
||||
agent_import_path_single_turn: str | None = None
|
||||
# Browser-opt-in variants (`[metadata] browser = true`). A harness that has no variant
|
||||
# keeps its normal class: codex, for instance, gains the browser and its disclosure but
|
||||
# has no `Read` equivalent to switch toolsets for.
|
||||
agent_import_path_browser: str | None = None
|
||||
agent_import_path_single_turn_browser: str | None = None
|
||||
import_path_aliases: tuple[str, ...] = ()
|
||||
legacy_bare_model_rows: bool = False
|
||||
default_model: str | None = None
|
||||
effort_kwarg: str = ""
|
||||
effort_default: str | None = None
|
||||
# Agent kwarg that opts a trial into the harness's fast/priority serving mode
|
||||
# (claude-code: fast mode). Empty means the harness has none and --fast refuses.
|
||||
fast_kwarg: str = ""
|
||||
key_env: str | None = None
|
||||
base_url_env: str | None = None
|
||||
proxy_path: str | None = None
|
||||
flaky_hangs: bool = False
|
||||
enabled: bool = True
|
||||
# Worker-container fields; see the registry header.
|
||||
authoring: bool = False
|
||||
cli: str | None = None
|
||||
install: str | None = None
|
||||
skills_dir: str | None = None
|
||||
auth_path: str | None = None
|
||||
auth_key_env: str | None = None
|
||||
explore_launch: str | None = None
|
||||
config_path: str | None = None
|
||||
# Config the harness needs wherever it runs, trial sandbox included.
|
||||
agent_config: str | None = None
|
||||
# Config for both worker containers (explore and authoring).
|
||||
container_config: str | None = None
|
||||
# Config for the EXPLORE container only — the capture hooks, whose commands ship in
|
||||
# explore/plugins/. Writing them in authoring would register hooks against files that
|
||||
# are not there, firing on every prompt.
|
||||
explore_config: str | None = None
|
||||
# Fields added for a later phase, kept verbatim so this loader doesn't have to
|
||||
# be edited in lockstep with the schema.
|
||||
extra: dict = field(default_factory=dict, compare=False)
|
||||
|
||||
def agent_import_path_for(self, *, multi_turn: bool, browser: bool = False) -> str:
|
||||
"""Agent class to launch. Multi-turn tasks need the resuming class; a
|
||||
single-turn task given it would try to resume a session that isn't there.
|
||||
|
||||
``browser`` selects the opt-in variant, which for claude also carries the ``Read``
|
||||
built-in — a different toolset is a different agent, so it is a different class with
|
||||
its own name rather than a flag on the canonical one. Harnesses without a variant fall
|
||||
through to their normal class."""
|
||||
if browser:
|
||||
variant = (
|
||||
self.agent_import_path_browser
|
||||
if multi_turn
|
||||
else (self.agent_import_path_single_turn_browser or self.agent_import_path_browser)
|
||||
)
|
||||
if variant:
|
||||
return variant
|
||||
if multi_turn:
|
||||
return self.agent_import_path
|
||||
return self.agent_import_path_single_turn or self.agent_import_path
|
||||
|
||||
def row_label(self, model: str) -> str:
|
||||
"""Row identity for one trial: bare model for legacy harnesses (so
|
||||
published manifests keep their labels), else ``<harness>:<model>``."""
|
||||
return model if self.legacy_bare_model_rows else f"{self.id}:{model}"
|
||||
|
||||
def agent_config_overrides(self) -> dict[str, str]:
|
||||
"""``agent_config`` as flat ``dotted.key -> value`` pairs in CLI-override form.
|
||||
|
||||
Values are rendered bare — ``disabled``, not ``"disabled"``. Every consumer
|
||||
interpolates these into a shell command, which would strip the quotes anyway;
|
||||
emitting them would only make the result depend on how many shell layers the
|
||||
string crosses. Bare is what the CLIs document (``-c model="o3"`` reaches the
|
||||
binary as ``model=o3``).
|
||||
|
||||
These settings ride the command line as ``-c dotted.key=value`` everywhere the
|
||||
harness runs, never a config file. A trial sandbox rules the file out: the
|
||||
harness's own runner appends root keys to it, and TOML has no way back to the
|
||||
root scope once a table has opened, so a table we appended would swallow them.
|
||||
Overrides compose in any order and beat the file, so the same rendering serves
|
||||
the explore launcher too — one declaration, one mechanism.
|
||||
"""
|
||||
if not self.agent_config:
|
||||
return {}
|
||||
try:
|
||||
parsed = tomllib.loads(self.agent_config)
|
||||
except tomllib.TOMLDecodeError as exc:
|
||||
raise HarnessRegistryError(
|
||||
f"{self.id}: agent_config is not valid TOML ({exc})"
|
||||
) from exc
|
||||
|
||||
flat: dict[str, str] = {}
|
||||
|
||||
def walk(node: dict, prefix: str) -> None:
|
||||
for key, value in node.items():
|
||||
path = f"{prefix}{key}"
|
||||
if isinstance(value, dict):
|
||||
walk(value, f"{path}.")
|
||||
elif isinstance(value, bool):
|
||||
flat[path] = "true" if value else "false"
|
||||
elif isinstance(value, (int, float)):
|
||||
flat[path] = str(value)
|
||||
elif isinstance(value, str):
|
||||
if value != value.strip() or any(c in value for c in " \"'\\"):
|
||||
raise HarnessRegistryError(
|
||||
f"{self.id}: agent_config key {path!r} has a value needing "
|
||||
"shell quoting, which the -c override form cannot carry"
|
||||
)
|
||||
flat[path] = value
|
||||
else:
|
||||
raise HarnessRegistryError(
|
||||
f"{self.id}: agent_config key {path!r} has type "
|
||||
f"{type(value).__name__}, which has no -c override form"
|
||||
)
|
||||
|
||||
walk(parsed, "")
|
||||
return flat
|
||||
|
||||
def container_config_text(self, *, surface: str) -> str | None:
|
||||
"""Config file body for a worker container. `surface` is "explore" or
|
||||
"authoring"; explore additionally gets `explore_config`. Root keys come from
|
||||
`container_config` first, so appending a table section stays valid TOML."""
|
||||
parts = [self.container_config]
|
||||
if surface == "explore":
|
||||
parts.append(self.explore_config)
|
||||
kept = [part.strip("\n") for part in parts if part and part.strip()]
|
||||
return "\n\n".join(kept) + "\n" if kept else None
|
||||
|
||||
def agent_config_flags(self) -> str:
|
||||
"""``agent_config`` as a ``-c key=value`` command-line string."""
|
||||
return " ".join(
|
||||
f"-c {key}={value}"
|
||||
for key, value in sorted(self.agent_config_overrides().items())
|
||||
)
|
||||
|
||||
def explore_launch_command(self) -> str | None:
|
||||
"""``explore_launch`` with the registry's own values substituted in.
|
||||
|
||||
The worker's Explore session and the trial must run the same agent, so the
|
||||
model, effort and reductions are declared once here and rendered into both.
|
||||
A literal in the launch string would be a second declaration, and the two
|
||||
would drift the first time one of them was updated alone.
|
||||
|
||||
Only these three placeholders are substituted; ``$@`` and
|
||||
``$RACCOON_TOOLSET_NOTE`` stay for the launcher's own shell to expand.
|
||||
"""
|
||||
if not self.explore_launch:
|
||||
return None
|
||||
return (
|
||||
self.explore_launch.replace("$RACCOON_AGENT_FLAGS", self.agent_config_flags())
|
||||
.replace("$RACCOON_MODEL", self.default_model or "")
|
||||
.replace("$RACCOON_EFFORT", self.effort_default or "")
|
||||
)
|
||||
|
||||
def known_import_paths(self) -> tuple[str, ...]:
|
||||
paths = [self.agent_import_path, *self.import_path_aliases]
|
||||
if self.agent_import_path_single_turn:
|
||||
paths.append(self.agent_import_path_single_turn)
|
||||
return tuple(paths)
|
||||
|
||||
|
||||
_KNOWN_FIELDS = frozenset(
|
||||
{
|
||||
"id",
|
||||
"label",
|
||||
"agent_import_path",
|
||||
"agent_import_path_single_turn",
|
||||
"agent_import_path_browser",
|
||||
"agent_import_path_single_turn_browser",
|
||||
"import_path_aliases",
|
||||
"legacy_bare_model_rows",
|
||||
"default_model",
|
||||
"model_id_shape",
|
||||
"effort_kwarg",
|
||||
"effort_default",
|
||||
"fast_kwarg",
|
||||
"key_env",
|
||||
"base_url_env",
|
||||
"proxy_path",
|
||||
"writes_atif",
|
||||
"capture",
|
||||
"seed_native",
|
||||
"seed_atif",
|
||||
"flaky_hangs",
|
||||
"enabled",
|
||||
"authoring",
|
||||
"cli",
|
||||
"install",
|
||||
"skills_dir",
|
||||
"auth_path",
|
||||
"auth_key_env",
|
||||
"explore_launch",
|
||||
"config_path",
|
||||
"agent_config",
|
||||
"container_config",
|
||||
"explore_config",
|
||||
}
|
||||
)
|
||||
|
||||
_REQUIRED_FIELDS = (
|
||||
"id",
|
||||
"label",
|
||||
"agent_import_path",
|
||||
"model_id_shape",
|
||||
"writes_atif",
|
||||
"capture",
|
||||
"seed_native",
|
||||
"seed_atif",
|
||||
)
|
||||
|
||||
|
||||
class HarnessRegistryError(ValueError):
|
||||
"""Malformed registry. Raised rather than tolerated: a broken registry is a
|
||||
broken deployment, and silently defaulting would pick the wrong agent."""
|
||||
|
||||
|
||||
def _references_agent_flags(launch: str) -> bool:
|
||||
return "$RACCOON_AGENT_FLAGS" in launch or "${RACCOON_AGENT_FLAGS}" in launch
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HarnessRegistry:
|
||||
version: int
|
||||
harnesses: tuple[Harness, ...]
|
||||
|
||||
def all(self) -> tuple[Harness, ...]:
|
||||
return self.harnesses
|
||||
|
||||
def enabled(self) -> tuple[Harness, ...]:
|
||||
return tuple(h for h in self.harnesses if h.enabled)
|
||||
|
||||
def authoring(self) -> tuple[Harness, ...]:
|
||||
"""Harnesses a worker can author with — what the worker containers install.
|
||||
Narrower than enabled(): a harness can be runnable in a trial without having
|
||||
an authoring story (no CLI to converse with, or no capture)."""
|
||||
return tuple(h for h in self.harnesses if h.enabled and h.authoring)
|
||||
|
||||
def find(self, harness_id: str) -> Harness | None:
|
||||
return next((h for h in self.harnesses if h.id == harness_id), None)
|
||||
|
||||
def require(self, harness_id: str) -> Harness:
|
||||
harness = self.find(harness_id)
|
||||
if harness is not None:
|
||||
return harness
|
||||
available = ", ".join(sorted(h.id for h in self.enabled()))
|
||||
raise HarnessRegistryError(
|
||||
f'Unknown harness "{harness_id}". Available: {available}'
|
||||
)
|
||||
|
||||
def by_import_path(self, agent: str) -> Harness | None:
|
||||
"""Resolve an agent identity — a ``name()`` or import path from
|
||||
``result.json`` ``config.agent``, or a manifest row — to its harness."""
|
||||
needle = (agent or "").strip()
|
||||
if not needle:
|
||||
return None
|
||||
for harness in self.harnesses:
|
||||
if needle == harness.id or needle in harness.known_import_paths():
|
||||
return harness
|
||||
return None
|
||||
|
||||
|
||||
def _build(entry: dict, index: int) -> Harness:
|
||||
for name in _REQUIRED_FIELDS:
|
||||
if name not in entry:
|
||||
raise HarnessRegistryError(
|
||||
f"harness[{index}]: missing required field '{name}'"
|
||||
)
|
||||
shape = entry["model_id_shape"]
|
||||
if shape not in MODEL_ID_SHAPES:
|
||||
raise HarnessRegistryError(
|
||||
f"harness[{index}] ({entry['id']}): model_id_shape {shape!r} not one of "
|
||||
f"{sorted(MODEL_ID_SHAPES)}"
|
||||
)
|
||||
# These three reach `eval` in setup-harnesses.sh, which is how they support the
|
||||
# `${CODEX_HOME:-$HOME/.codex}` default-value syntax that python's expandvars cannot
|
||||
# express. Under eval a backtick or $( would EXECUTE, so refuse them here — the registry
|
||||
# is ours, but "ours" is not an argument that survives a careless future edit.
|
||||
for shell_field in ("config_path", "auth_path", "skills_dir"):
|
||||
value = entry.get(shell_field)
|
||||
if not isinstance(value, str):
|
||||
continue
|
||||
# A backtick or $( executes outright. A double quote closes the string these are
|
||||
# interpolated into, and a semicolon then starts a new command inside it — same
|
||||
# outcome, one step removed.
|
||||
bad = [t for t in ("`", "$(", '"', ";") if t in value]
|
||||
if bad:
|
||||
raise HarnessRegistryError(
|
||||
f"harness[{index}] ({entry['id']}): {shell_field} contains "
|
||||
f"{', '.join(repr(t) for t in bad)} ({value!r}). This value is shell-"
|
||||
f"expanded, so that would execute; use plain $VAR or ${{VAR:-default}} only."
|
||||
)
|
||||
|
||||
launch = entry.get("explore_launch")
|
||||
if entry.get("agent_config") and launch and not _references_agent_flags(launch):
|
||||
raise HarnessRegistryError(
|
||||
f"harness[{index}] ({entry['id']}): declares agent_config but its "
|
||||
"explore_launch does not pass $RACCOON_AGENT_FLAGS. The worker's session "
|
||||
"would then run with a different toolset than the trial it is authoring "
|
||||
"for, which is the drift agent_config exists to prevent."
|
||||
)
|
||||
return Harness(
|
||||
id=entry["id"],
|
||||
label=entry["label"],
|
||||
agent_import_path=entry["agent_import_path"],
|
||||
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
|
||||
agent_import_path_browser=entry.get("agent_import_path_browser"),
|
||||
agent_import_path_single_turn_browser=entry.get("agent_import_path_single_turn_browser"),
|
||||
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
|
||||
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
|
||||
default_model=entry.get("default_model"),
|
||||
model_id_shape=shape,
|
||||
effort_kwarg=entry.get("effort_kwarg", ""),
|
||||
effort_default=entry.get("effort_default"),
|
||||
fast_kwarg=entry.get("fast_kwarg", ""),
|
||||
key_env=entry.get("key_env"),
|
||||
base_url_env=entry.get("base_url_env"),
|
||||
proxy_path=entry.get("proxy_path"),
|
||||
writes_atif=bool(entry["writes_atif"]),
|
||||
capture=bool(entry["capture"]),
|
||||
seed_native=bool(entry["seed_native"]),
|
||||
seed_atif=bool(entry["seed_atif"]),
|
||||
flaky_hangs=bool(entry.get("flaky_hangs", False)),
|
||||
enabled=bool(entry.get("enabled", True)),
|
||||
authoring=bool(entry.get("authoring", False)),
|
||||
cli=entry.get("cli"),
|
||||
install=entry.get("install"),
|
||||
skills_dir=entry.get("skills_dir"),
|
||||
auth_path=entry.get("auth_path"),
|
||||
auth_key_env=entry.get("auth_key_env"),
|
||||
explore_launch=entry.get("explore_launch"),
|
||||
config_path=entry.get("config_path"),
|
||||
agent_config=entry.get("agent_config"),
|
||||
container_config=entry.get("container_config"),
|
||||
explore_config=entry.get("explore_config"),
|
||||
extra={k: v for k, v in entry.items() if k not in _KNOWN_FIELDS},
|
||||
)
|
||||
|
||||
|
||||
_cache: dict[Path, HarnessRegistry] = {}
|
||||
|
||||
|
||||
def load_harness_registry(path: Path | str = REGISTRY_PATH) -> HarnessRegistry:
|
||||
"""Parse and validate the registry. Raises HarnessRegistryError on a malformed
|
||||
file, a duplicate id, or an import path claimed by two harnesses (which would
|
||||
make ``by_import_path`` depend on declaration order)."""
|
||||
resolved = Path(path).resolve()
|
||||
if resolved in _cache:
|
||||
return _cache[resolved]
|
||||
|
||||
with open(resolved, "rb") as handle:
|
||||
doc = tomllib.load(handle)
|
||||
|
||||
if "version" not in doc:
|
||||
raise HarnessRegistryError("harness-registry: missing 'version'")
|
||||
entries = doc.get("harness") or []
|
||||
if not entries:
|
||||
raise HarnessRegistryError("harness-registry: no [[harness]] entries")
|
||||
|
||||
harnesses = tuple(_build(entry, i) for i, entry in enumerate(entries))
|
||||
|
||||
seen_ids: set[str] = set()
|
||||
for harness in harnesses:
|
||||
if harness.id in seen_ids:
|
||||
raise HarnessRegistryError(
|
||||
f"harness-registry: duplicate harness id: {harness.id}"
|
||||
)
|
||||
seen_ids.add(harness.id)
|
||||
|
||||
owners: dict[str, str] = {}
|
||||
for harness in harnesses:
|
||||
for import_path in harness.known_import_paths():
|
||||
owner = owners.get(import_path)
|
||||
if owner is not None and owner != harness.id:
|
||||
raise HarnessRegistryError(
|
||||
f'harness-registry: import path "{import_path}" claimed by both '
|
||||
f'"{owner}" and "{harness.id}"'
|
||||
)
|
||||
owners[import_path] = harness.id
|
||||
|
||||
registry = HarnessRegistry(version=int(doc["version"]), harnesses=harnesses)
|
||||
_cache[resolved] = registry
|
||||
return registry
|
||||
@@ -1,351 +0,0 @@
|
||||
/**
|
||||
* input-checksums.ts — capture and compare sha256 checksums of the task inputs
|
||||
* that reference runs and detector reports depend on.
|
||||
*
|
||||
* A reference run is only meaningful for the task inputs it actually ran
|
||||
* against: the prompt (instruction.md), the snapshot session
|
||||
* (environment/session.jsonl), the workspace patch
|
||||
* (environment/workspace.patch), and the gitref the workspace is built from
|
||||
* (task.toml `[metadata].commit`). Detector reports likewise assess a specific
|
||||
* revision of instruction.md + the task's holistic rubric
|
||||
* (tests/holistic-rubric.md on current tasks; tests/grader-guidance-consolidated.md
|
||||
* on tasks created before the rename) — and, for the rubric detectors, the
|
||||
* task's atomic rubric (tests/atomic-rubric.yaml; tests/rubrics.yaml on tasks
|
||||
* converted before the rename) and tests/grader-context.md. When any of those
|
||||
* change after the artifact was produced, the artifact is stale — it describes
|
||||
* an older revision of the task than the one being packaged.
|
||||
*
|
||||
* This module is the single source of truth for WHAT gets checksummed and how
|
||||
* captures are compared. Capture sites (copy-reference-run.ts,
|
||||
* record-detector-inputs.ts) write a {@link TaskInputChecksums} record next to
|
||||
* the artifact; submit-task.ts re-captures at packaging time and diffs.
|
||||
* Content hashes rather than mtimes: a re-clone / whole-tree touch can fake or
|
||||
* mask an mtime, but can't change a sha256.
|
||||
*
|
||||
* Lives in the worker toolkit's shipped file set (static/scripts/lib/), so in
|
||||
* a packed toolkit it sits at scripts/lib/ next to both consumers. Repo-side
|
||||
* callers go through the scripts/lib/input-checksums.ts re-export shim, which
|
||||
* occupies the same relative path there (mirroring the check-devcontainer
|
||||
* pattern).
|
||||
*
|
||||
* Related but deliberately separate: `computeDeliveryHash` (repo-side
|
||||
* delivery script — grep the internal repo for it; not shipped with the
|
||||
* toolkit) hashes an overlapping input set for delivery idempotency. It is
|
||||
* NOT built on this module because its hash format is load-bearing (a
|
||||
* changed hash re-delivers every task); if you change WHAT counts as a task
|
||||
* input here, check whether the delivery hash needs the same change.
|
||||
*/
|
||||
|
||||
import { createHash } from 'node:crypto';
|
||||
import { existsSync, readFileSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
|
||||
/** Bump when the record shape changes incompatibly. */
|
||||
export const INPUT_CHECKSUMS_VERSION = 1;
|
||||
|
||||
/** Filename of the record inside a reference-run directory. */
|
||||
export const INPUT_CHECKSUMS_FILENAME = 'input-checksums.json';
|
||||
|
||||
/**
|
||||
* Where in the artifact lifecycle a capture happened. The moment matters for
|
||||
* how much a "fresh" verdict can be trusted:
|
||||
*
|
||||
* - 'run' — at trial launch (scripts/harbor-run stamps the trial dir).
|
||||
* The strongest evidence: the record is what the agent ran
|
||||
* against, whatever got edited afterwards.
|
||||
* - 'copy' — at copy-reference-run time, the fallback when a trial carries
|
||||
* no run-time stamp. An input edited between harbor-run and the
|
||||
* copy is recorded at its post-edit state, so a stale run can
|
||||
* read fresh.
|
||||
* - 'stamp' — right after a detector report is written
|
||||
* (record-detector-inputs.ts in a worker checkout; the
|
||||
* internal repo's detector save path stamps the same way).
|
||||
* - 'mirror' — retired: written by the repo-side flow that re-materialized
|
||||
* canonical detector reports to disk back when reports had a
|
||||
* remote canonical store. Reports are local-only now, so no
|
||||
* current code writes it; the member stays so old stamps keep
|
||||
* their recorded method when read.
|
||||
* - 'regrade' — a re-grade of an existing run. Present on records already on
|
||||
* disk; no current code path writes it.
|
||||
*
|
||||
* Absent on records written before this field existed.
|
||||
*/
|
||||
export type TaskInputCaptureMethod = 'run' | 'copy' | 'stamp' | 'mirror' | 'regrade';
|
||||
|
||||
/** Runtime mirror of {@link TaskInputCaptureMethod}, for validating a record read from disk. */
|
||||
export const TASK_INPUT_CAPTURE_METHODS = Object.freeze([
|
||||
'run',
|
||||
'copy',
|
||||
'stamp',
|
||||
'mirror',
|
||||
'regrade',
|
||||
] as const satisfies readonly TaskInputCaptureMethod[]);
|
||||
|
||||
/**
|
||||
* The checksums of a task's inputs as they stood at capture time. Every hash
|
||||
* field is a sha256 hex digest, or `null` when the file didn't exist (a null
|
||||
* that later becomes a hash — or vice versa — is a change like any other).
|
||||
* `gitref` is the raw `[metadata].commit` string, recorded verbatim rather
|
||||
* than hashed so a mismatch message can show it.
|
||||
*/
|
||||
export interface TaskInputChecksums {
|
||||
readonly version: number;
|
||||
/** ISO-8601 timestamp of the capture. */
|
||||
readonly capturedAt: string;
|
||||
/** Lifecycle point of the capture ({@link TaskInputCaptureMethod}). */
|
||||
readonly capturedBy?: TaskInputCaptureMethod;
|
||||
/**
|
||||
* The task slug the capture was taken from (harbor-tasks/<slug>). Written
|
||||
* by launch-time captures so stamping can be scoped to the right task's
|
||||
* trial dirs when several harbor-runs share a cwd — and so a mis-routed
|
||||
* stamp is detectable after the fact.
|
||||
*/
|
||||
readonly taskSlug?: string;
|
||||
/**
|
||||
* Set when the finalize flow re-stamped the prompt/graderGuidance hashes
|
||||
* after the harbor-path scrub deliberately rewrote those docs
|
||||
* (scripts/restamp-task-inputs.ts) — the run/report still reflects the
|
||||
* task; only the doc bytes were normalized.
|
||||
*/
|
||||
readonly restampedAt?: string;
|
||||
readonly inputs: {
|
||||
readonly prompt: string | null;
|
||||
readonly graderGuidance: string | null;
|
||||
readonly sessionJsonl: string | null;
|
||||
readonly workspacePatch: string | null;
|
||||
readonly gitref: string | null;
|
||||
/**
|
||||
* tests/grader-guidance-consolidated.md — the holistic rubric under its
|
||||
* pre-rename filename, which every task created before the rename keeps;
|
||||
* hashed when present, null otherwise. Absent (undefined) on records
|
||||
* captured before the field existed; comparisons skip a field the record
|
||||
* predates, so old captures stay fresh until they are re-stamped.
|
||||
*/
|
||||
readonly graderGuidanceConsolidated?: string | null;
|
||||
/**
|
||||
* tests/holistic-rubric.md — the holistic rubric under its current
|
||||
* filename (each task carries exactly one of this and the pre-rename
|
||||
* name above). Hashed when present, null otherwise. Absent (undefined)
|
||||
* on records captured before the field existed; comparisons skip a
|
||||
* field the record predates. The task-checksum digest serializes fields
|
||||
* in this order and appends new fields at the end (see
|
||||
* scripts/lib/grader-run-checksums.ts).
|
||||
*/
|
||||
readonly holisticRubric?: string | null;
|
||||
/**
|
||||
* tests/atomic-rubric.yaml — the task's atomic rubric under its current
|
||||
* filename. Hashed when present, null otherwise. Absent (undefined) on
|
||||
* records captured before the field existed; comparisons skip a field
|
||||
* the record predates.
|
||||
*/
|
||||
readonly atomicRubric?: string | null;
|
||||
/**
|
||||
* tests/rubrics.yaml — the atomic rubric under its pre-rename filename,
|
||||
* which tasks converted before the rename keep. Hashed when present,
|
||||
* null otherwise; absent (undefined) on records captured before the
|
||||
* field existed.
|
||||
*/
|
||||
readonly rubricsYaml?: string | null;
|
||||
/**
|
||||
* tests/grader-context.md — the context document the rubric grader modes
|
||||
* read beside the atomic rubric. Hashed when present, null otherwise;
|
||||
* absent (undefined) on records captured before the field existed. Last
|
||||
* in field order per the append-at-the-end digest rule above.
|
||||
*/
|
||||
readonly graderContext?: string | null;
|
||||
};
|
||||
}
|
||||
|
||||
export type TaskInputName = keyof TaskInputChecksums['inputs'];
|
||||
|
||||
/** Human-readable component names, used verbatim in staleness warnings. Frozen:
|
||||
* its key set is the runtime source of truth for the task-input axes. */
|
||||
export const INPUT_LABELS = Object.freeze({
|
||||
prompt: 'prompt (instruction.md)',
|
||||
graderGuidance: 'legacy-era grader guidance (tests/grader-guidance.md)',
|
||||
graderGuidanceConsolidated: 'holistic rubric (tests/grader-guidance-consolidated.md)',
|
||||
sessionJsonl: 'session snapshot (environment/session.jsonl)',
|
||||
workspacePatch: 'workspace patch (environment/workspace.patch)',
|
||||
gitref: 'gitref (task.toml commit)',
|
||||
holisticRubric: 'holistic rubric (tests/holistic-rubric.md)',
|
||||
atomicRubric: 'atomic rubric (tests/atomic-rubric.yaml)',
|
||||
rubricsYaml: 'atomic rubric (tests/rubrics.yaml)',
|
||||
graderContext: 'grader context (tests/grader-context.md)',
|
||||
} as const satisfies Record<TaskInputName, string>);
|
||||
|
||||
/**
|
||||
* The inputs that shape what the AGENT saw and did. Changing any of them means
|
||||
* a captured reference run no longer reflects the task being packaged, and
|
||||
* only re-running the agent can fix that. The holistic-rubric files are
|
||||
* deliberately NOT in this set: editing the rubric stales the run's GRADE,
|
||||
* not the run itself, and `scripts/harbor-regrade` re-derives grades without
|
||||
* re-running the agent.
|
||||
*/
|
||||
export const REFERENCE_RUN_INPUTS = Object.freeze([
|
||||
'prompt',
|
||||
'sessionJsonl',
|
||||
'workspacePatch',
|
||||
'gitref',
|
||||
] as const satisfies readonly TaskInputName[]);
|
||||
|
||||
/**
|
||||
* The inputs a detector report assesses — instruction.md plus whichever
|
||||
* holistic-rubric files the task directory carries (tests/holistic-rubric.md
|
||||
* on current tasks, tests/grader-guidance-consolidated.md on tasks created
|
||||
* before the rename, plus the legacy-era plain-named file when a task
|
||||
* authored on an earlier generation carries one). An absent file hashes to
|
||||
* null on both sides and never diffs. Compared by content.
|
||||
*
|
||||
* The atomic-rubric files are deliberately NOT in this set: fifteen of the
|
||||
* seventeen detectors never open them, so writing an atomic rubric after
|
||||
* running the detectors would stale every one of those reports over files
|
||||
* they never read. The two that do read them use
|
||||
* {@link RUBRIC_DETECTOR_REPORT_INPUTS}.
|
||||
*/
|
||||
export const DETECTOR_REPORT_INPUTS = Object.freeze([
|
||||
'prompt',
|
||||
'graderGuidance',
|
||||
'graderGuidanceConsolidated',
|
||||
'holisticRubric',
|
||||
] as const satisfies readonly TaskInputName[]);
|
||||
|
||||
/**
|
||||
* The inputs the two rubric detectors assess: {@link DETECTOR_REPORT_INPUTS}
|
||||
* plus the atomic-rubric package (tests/atomic-rubric.yaml, the pre-rename
|
||||
* tests/rubrics.yaml, and tests/grader-context.md), which they compare
|
||||
* against the holistic rubric.
|
||||
*/
|
||||
export const RUBRIC_DETECTOR_REPORT_INPUTS = Object.freeze([
|
||||
...DETECTOR_REPORT_INPUTS,
|
||||
'atomicRubric',
|
||||
'rubricsYaml',
|
||||
'graderContext',
|
||||
] as const satisfies readonly TaskInputName[]);
|
||||
|
||||
/** Detectors that read the atomic-rubric package, and so are staled by it. */
|
||||
export const ATOMIC_RUBRIC_DETECTORS: readonly string[] = Object.freeze([
|
||||
'detector-rubric-coverage',
|
||||
'detector-rubric-form',
|
||||
]);
|
||||
|
||||
/** The input set a named detector's report is judged against. */
|
||||
export function detectorReportInputs(detectorName: string): readonly TaskInputName[] {
|
||||
return ATOMIC_RUBRIC_DETECTORS.includes(detectorName)
|
||||
? RUBRIC_DETECTOR_REPORT_INPUTS
|
||||
: DETECTOR_REPORT_INPUTS;
|
||||
}
|
||||
|
||||
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
|
||||
function sha256File(filePath: string): string | null {
|
||||
if (!existsSync(filePath)) return null;
|
||||
return createHash('sha256').update(readFileSync(filePath)).digest('hex');
|
||||
}
|
||||
|
||||
/**
|
||||
* The gitref (`[metadata].commit`) from a task.toml, or null when the file is
|
||||
* missing, unreadable, or has no commit line — a read failure downgrades to
|
||||
* "absent" rather than crashing a capture or a validation sweep.
|
||||
*
|
||||
* Deliberately a regex, not a TOML parser: this module ships in the worker
|
||||
* toolkit, where a new runtime dep would break packaging for every worker
|
||||
* whose container predates the dep (npm install runs only on container
|
||||
* create, and containers survive toolkit upgrades). build-workspace.sh reads
|
||||
* the same key with the same grep-a-`commit`-line approach. The one `commit`
|
||||
* key in a task.toml is `[metadata].commit`, so anchoring to the first
|
||||
* `commit = "…"` line is exact in practice.
|
||||
*/
|
||||
function readGitref(taskDir: string): string | null {
|
||||
const tomlPath = join(taskDir, 'task.toml');
|
||||
if (!existsSync(tomlPath)) return null;
|
||||
try {
|
||||
const match = /^\s*commit\s*=\s*(?:"([^"\n]+)"|'([^'\n]+)')\s*(?:#.*)?$/m.exec(
|
||||
readFileSync(tomlPath, 'utf-8')
|
||||
);
|
||||
const commit = match?.[1] ?? match?.[2];
|
||||
return commit && commit.length > 0 ? commit : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Checksum the task inputs as they stand right now under `taskDir`. */
|
||||
export function captureTaskInputs(
|
||||
taskDir: string,
|
||||
capturedBy?: TaskInputCaptureMethod
|
||||
): TaskInputChecksums {
|
||||
return Object.freeze({
|
||||
version: INPUT_CHECKSUMS_VERSION,
|
||||
capturedAt: new Date().toISOString(),
|
||||
...(capturedBy ? { capturedBy } : {}),
|
||||
inputs: Object.freeze({
|
||||
prompt: sha256File(join(taskDir, 'instruction.md')),
|
||||
graderGuidance: sha256File(join(taskDir, 'tests', 'grader-guidance.md')),
|
||||
sessionJsonl: sha256File(join(taskDir, 'environment', 'session.jsonl')),
|
||||
workspacePatch: sha256File(join(taskDir, 'environment', 'workspace.patch')),
|
||||
gitref: readGitref(taskDir),
|
||||
graderGuidanceConsolidated: sha256File(
|
||||
join(taskDir, 'tests', 'grader-guidance-consolidated.md')
|
||||
),
|
||||
holisticRubric: sha256File(join(taskDir, 'tests', 'holistic-rubric.md')),
|
||||
atomicRubric: sha256File(join(taskDir, 'tests', 'atomic-rubric.yaml')),
|
||||
rubricsYaml: sha256File(join(taskDir, 'tests', 'rubrics.yaml')),
|
||||
graderContext: sha256File(join(taskDir, 'tests', 'grader-context.md')),
|
||||
}),
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Read a previously captured record. Returns null when the file is missing or
|
||||
* doesn't look like a capture (pre-tracking artifact, hand-edited JSON, a
|
||||
* future incompatible version) — callers treat null as "staleness unknowable",
|
||||
* never as an error. No zod here: this module ships in the worker toolkit,
|
||||
* whose dependency set stays minimal, so the guard is manual.
|
||||
*/
|
||||
export function readTaskInputChecksums(filePath: string): TaskInputChecksums | null {
|
||||
if (!existsSync(filePath)) return null;
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(readFileSync(filePath, 'utf-8'));
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
if (typeof parsed !== 'object' || parsed === null) return null;
|
||||
const record = parsed as TaskInputChecksums;
|
||||
if (record.version !== INPUT_CHECKSUMS_VERSION) return null;
|
||||
if (typeof record.inputs !== 'object' || record.inputs === null) return null;
|
||||
for (const name of Object.keys(INPUT_LABELS) as TaskInputName[]) {
|
||||
const value = record.inputs[name];
|
||||
// undefined = the record predates this input field; still a valid capture.
|
||||
if (value !== undefined && value !== null && typeof value !== 'string') return null;
|
||||
}
|
||||
// An unrecognized capture method is dropped, not rejected: the field is
|
||||
// provenance colour, and rejecting would flip the whole run to "unknowable".
|
||||
const method: unknown = record.capturedBy;
|
||||
const isKnown = TASK_INPUT_CAPTURE_METHODS.includes(method as TaskInputCaptureMethod);
|
||||
if (method !== undefined && !isKnown) {
|
||||
const { capturedBy: _dropped, ...rest } = record;
|
||||
return rest;
|
||||
}
|
||||
return record;
|
||||
}
|
||||
|
||||
/**
|
||||
* Which of `names` changed between a recorded capture and the current state?
|
||||
* Returns the human-readable labels ({@link INPUT_LABELS}) of every component
|
||||
* whose value differs — including absent→present and present→absent flips. A
|
||||
* field the recorded capture predates (the key is not in the record at all)
|
||||
* is skipped: freshness on that axis is unknowable, and flagging every old
|
||||
* record the moment a new axis ships would drown the real signal. (The
|
||||
* task-checksum fold folds absence as 'null' instead — it only ever reads a
|
||||
* fresh capture, which is total, so the two never disagree in practice.)
|
||||
*/
|
||||
export function diffTaskInputs(
|
||||
recorded: TaskInputChecksums,
|
||||
current: TaskInputChecksums,
|
||||
names: readonly TaskInputName[]
|
||||
): string[] {
|
||||
return names
|
||||
.filter((name) => name in recorded.inputs)
|
||||
.filter((name) => recorded.inputs[name] !== current.inputs[name])
|
||||
.map((name) => INPUT_LABELS[name]);
|
||||
}
|
||||
@@ -1,13 +0,0 @@
|
||||
/**
|
||||
* Wrap a notice in a banner loud enough to survive a scrollback.
|
||||
*
|
||||
* Yellow only when stderr is a terminal, so piped logs stay clean.
|
||||
*/
|
||||
export function banner(message: string, headline: string): string {
|
||||
const RULE = '#'.repeat(78);
|
||||
const pad = ' '.repeat(Math.max(0, Math.floor((78 - headline.length) / 2)));
|
||||
|
||||
const body = [RULE, `${pad}${headline}`, RULE, '', message, RULE].join('\n');
|
||||
const color = process.stderr.isTTY ? ['\u001b[33m', '\u001b[39m'] : ['', ''];
|
||||
return `${color[0]}${body}${color[1]}`;
|
||||
}
|
||||
@@ -1,96 +0,0 @@
|
||||
# shellcheck shell=bash
|
||||
#
|
||||
# resolve_pin — turn a task's pinned commit into a SHA that exists in the repo,
|
||||
# translating through a commit map when history has been rewritten under it.
|
||||
#
|
||||
# A task pins a commit in task.toml. If that repo's history is later rewritten
|
||||
# (to strip something that should never have shipped, say), every rewritten
|
||||
# commit gets a new SHA and the pin stops resolving — including on machines we
|
||||
# cannot reach, holding tasks we cannot edit. A commit map lets those pins keep
|
||||
# working: `<old-sha> <new-sha>` per line, at task-shared/commit-maps/<member>.map,
|
||||
# <member> being the task's `repo` key — a standalone toolkit checks its repo out
|
||||
# at repo/, so the directory name is not the member name and cannot be the key.
|
||||
#
|
||||
# The map is only consulted when the pin does not resolve, so it carries only
|
||||
# rewritten commits — an unchanged commit resolves on its own and its identity
|
||||
# row could never be read.
|
||||
#
|
||||
# Usage (source, then call):
|
||||
# . "$(dirname "$0")/lib/resolve-pin.sh"
|
||||
# sha=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
|
||||
#
|
||||
# Writes the resolved SHA to stdout, notes on stderr. Returns non-zero if the
|
||||
# pin cannot be resolved, having explained why.
|
||||
|
||||
RESOLVE_PIN_MAX_HOPS="${RESOLVE_PIN_MAX_HOPS:-25}"
|
||||
_RESOLVE_PIN_ZERO='0000000000000000000000000000000000000000'
|
||||
|
||||
# Look one hop: echo the successor of $1 in map $2, or nothing. Fails if the
|
||||
# prefix is ambiguous, which would otherwise pick an arbitrary commit.
|
||||
_resolve_pin_hop() {
|
||||
local from="$1" map="$2" hits
|
||||
hits=$(awk -v p="$from" '
|
||||
/^#/ || NF < 2 { next }
|
||||
index($1, p) == 1 { print $2 }
|
||||
' "$map" | sort -u)
|
||||
[ -z "$hits" ] && return 1
|
||||
if [ "$(printf '%s\n' "$hits" | wc -l | tr -d ' ')" -gt 1 ]; then
|
||||
echo " pin $from is ambiguous in $(basename "$map") — use a longer SHA" >&2
|
||||
return 2
|
||||
fi
|
||||
printf '%s\n' "$hits"
|
||||
}
|
||||
|
||||
resolve_pin() {
|
||||
local repo_dir="$1" commit="$2" map_dir="${3:-}" key="${4:-}" sha map cur hops next rc
|
||||
|
||||
# Present in the repo: nothing to translate.
|
||||
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$commit^{commit}" 2>/dev/null); then
|
||||
printf '%s\n' "$sha"
|
||||
return 0
|
||||
fi
|
||||
|
||||
map=""
|
||||
if [ -n "$map_dir" ]; then
|
||||
if [ -n "$key" ] && [ -f "$map_dir/$key.map" ]; then
|
||||
map="$map_dir/$key.map"
|
||||
elif [ -f "$map_dir/$(basename "$repo_dir").map" ]; then
|
||||
map="$map_dir/$(basename "$repo_dir").map"
|
||||
fi
|
||||
fi
|
||||
if [ -z "$map" ]; then
|
||||
echo "Error: pinned commit $commit is not in $repo_dir, and no commit map is available." >&2
|
||||
echo " The repo may be a shallow or partial copy — try a full clone." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Follow the chain: a commit rewritten more than once maps forward a hop per
|
||||
# rewrite, so keep going until the SHA exists or the trail ends.
|
||||
cur="$commit"
|
||||
hops=0
|
||||
while [ "$hops" -lt "$RESOLVE_PIN_MAX_HOPS" ]; do
|
||||
next=$(_resolve_pin_hop "$cur" "$map"); rc=$?
|
||||
[ "$rc" -eq 2 ] && return 1
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo "Error: pinned commit $commit is not in $repo_dir and is not in $(basename "$map")." >&2
|
||||
echo " It predates the map, or came from a repo copy this toolkit was not built from." >&2
|
||||
return 1
|
||||
fi
|
||||
if [ "$next" = "$_RESOLVE_PIN_ZERO" ]; then
|
||||
echo "Error: pinned commit $commit was deleted by a history rewrite, not rewritten." >&2
|
||||
echo " Re-pin this task to a commit that still exists." >&2
|
||||
return 1
|
||||
fi
|
||||
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$next^{commit}" 2>/dev/null); then
|
||||
echo " Pin $commit was rewritten; using $sha" >&2
|
||||
printf '%s\n' "$sha"
|
||||
return 0
|
||||
fi
|
||||
cur="$next"
|
||||
hops=$((hops + 1))
|
||||
done
|
||||
|
||||
echo "Error: pinned commit $commit did not settle after $RESOLVE_PIN_MAX_HOPS hops." >&2
|
||||
echo " $(basename "$map") may contain a cycle." >&2
|
||||
return 1
|
||||
}
|
||||
@@ -1,431 +0,0 @@
|
||||
/**
|
||||
* task-infra-integrity.ts — detect edits to toolkit-managed task files.
|
||||
*
|
||||
* `environment/Dockerfile`, `tests/test.sh`, and
|
||||
* `tests/grader-system-prompt-consolidated.md` come from `task-shared/` and
|
||||
* are the same in every task: they decide how the
|
||||
* trial runs and how the grade is produced. An edit makes a task's reference
|
||||
* runs incomparable to every other task's, and the scores still look normal,
|
||||
* so nothing downstream notices.
|
||||
*
|
||||
* A task is compared against itself as created. {@link writeManagedStamp} records
|
||||
* a sha256 of each managed file into `<task>/.toolkit-managed.json` at task
|
||||
* creation, so a later mismatch is an edit made since. Tasks created before
|
||||
* stamping have no record and fall back to matching the copies this toolkit
|
||||
* ships — see {@link IntegrityStatus}.
|
||||
*
|
||||
* The toolkit appends to a task's Dockerfile itself (session staging, the
|
||||
* reference-data corpus). Those blocks are wrapped in
|
||||
* `# >>> toolkit-managed: <name> >>>` sentinels and stripped before hashing or
|
||||
* comparing, so they never read as edits.
|
||||
*/
|
||||
|
||||
import { createHash } from 'crypto';
|
||||
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
|
||||
import { basename, join } from 'path';
|
||||
|
||||
import { banner } from './notice-banner.js';
|
||||
|
||||
/**
|
||||
* Every status is advisory. Nothing here stops a trial or a submission: an author
|
||||
* who changed one of these files did it because they didn't know we'd rather they
|
||||
* didn't, and refusing to package their work punishes a misunderstanding. The job
|
||||
* is to say so clearly, and to record it so a reviewer sees it too.
|
||||
*
|
||||
* `ok` — identical to a copy this toolkit ships, or unchanged since the
|
||||
* task was created.
|
||||
* `outdated` — unchanged since creation, but the toolkit has shipped a newer
|
||||
* copy since. Nobody's mistake; it does mean this task's runs
|
||||
* aren't directly comparable to one built today.
|
||||
* `modified` — matches neither its baseline nor anything shipped: an edit.
|
||||
* `unverifiable` — no recorded baseline and matches nothing shipped, so an edit
|
||||
* and an older release are indistinguishable.
|
||||
* `missing` — the task doesn't have the file.
|
||||
* `placeholder` — still the polyglot scaffold placeholder, so no base image has
|
||||
* been selected yet.
|
||||
*/
|
||||
export type IntegrityStatus =
|
||||
| 'ok'
|
||||
| 'outdated'
|
||||
| 'modified'
|
||||
| 'unverifiable'
|
||||
| 'missing'
|
||||
| 'placeholder';
|
||||
|
||||
export interface FileVerdict {
|
||||
/** Task-relative path, e.g. `environment/Dockerfile`. */
|
||||
taskPath: string;
|
||||
status: IntegrityStatus;
|
||||
/** Command that restores the managed version, on `modified` / `unverifiable`. */
|
||||
restore?: string;
|
||||
}
|
||||
|
||||
export interface IntegrityReport {
|
||||
/**
|
||||
* False when this isn't a worker toolkit, or when the task was authored on
|
||||
* a different toolkit generation (see {@link isTaskFromThisToolkitGeneration})
|
||||
* — callers should skip silently.
|
||||
*/
|
||||
checked: boolean;
|
||||
files: FileVerdict[];
|
||||
/** Looks like an edit: matches neither a baseline nor anything shipped. */
|
||||
modified: FileVerdict[];
|
||||
/** Can't be told apart from an older release. */
|
||||
unverifiable: FileVerdict[];
|
||||
/** Unchanged, but a newer copy has shipped since. */
|
||||
outdated: FileVerdict[];
|
||||
}
|
||||
|
||||
interface ManagedFile {
|
||||
taskPath: string;
|
||||
/** Matches the candidate pristine filenames under `task-shared/`. */
|
||||
baselinePattern: RegExp;
|
||||
}
|
||||
|
||||
/** The Dockerfile pattern accepts `Dockerfile` and every `Dockerfile.<member>`. */
|
||||
const MANAGED_FILES: ManagedFile[] = [
|
||||
{ taskPath: 'environment/Dockerfile', baselinePattern: /^Dockerfile(\.[\w.-]+)?$/ },
|
||||
{ taskPath: 'tests/test.sh', baselinePattern: /^test\.sh$/ },
|
||||
{
|
||||
taskPath: 'tests/grader-system-prompt-consolidated.md',
|
||||
baselinePattern: /^grader-system-prompt-consolidated\.md$/,
|
||||
},
|
||||
];
|
||||
|
||||
const SENTINEL_OPEN = /^#\s*>>>\s*toolkit-managed:.*>>>\s*$/;
|
||||
const SENTINEL_CLOSE = /^#\s*<<<\s*toolkit-managed\s*<<<\s*$/;
|
||||
|
||||
/**
|
||||
* Line shapes from toolkit releases that predate the sentinels. Deliberately
|
||||
* narrow: each is a literal line the toolkit wrote, not a general "ignore COPY
|
||||
* lines" rule an edit could hide behind.
|
||||
*/
|
||||
const LEGACY_MANAGED_LINES: RegExp[] = [
|
||||
/^# Stage session files for the snapshot agent adapter to install at runtime\.$/,
|
||||
/^COPY session\.jsonl \/tmp\/snapshot-session\/session\.jsonl$/,
|
||||
/^COPY session\/ \/tmp\/snapshot-session\/session\/$/,
|
||||
/^RUN echo '[0-9a-fA-F-]+' > \/tmp\/snapshot-session\/uuid\.txt$/,
|
||||
/^# Reference-data corpus at \/data\/zeta-corpus \(staged by build-workspace\)\.$/,
|
||||
/^COPY corpus\/ \/data\/zeta-corpus\/$/,
|
||||
];
|
||||
|
||||
/** Marker identifying the polyglot scaffold's deliberately-failing placeholder. */
|
||||
const PLACEHOLDER_MARKER = 'POLYGLOT TOOLKIT';
|
||||
|
||||
/** Per-task stamp of the managed files as created. Lives in the task directory. */
|
||||
export const STAMP_FILENAME = '.toolkit-managed.json';
|
||||
|
||||
interface ManagedStamp {
|
||||
version: number;
|
||||
stampedAt: string;
|
||||
/** taskPath → sha256 of the stripped content. */
|
||||
files: Record<string, string>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove toolkit-appended content so only author-authored differences remain.
|
||||
* Trailing blank lines go too — an editor adding or trimming a final newline is
|
||||
* not something to fail a trial over.
|
||||
*/
|
||||
export function stripManagedBlocks(content: string): string {
|
||||
const out: string[] = [];
|
||||
let inBlock = false;
|
||||
|
||||
// Normalize CRLF before anything else: a Windows editor or a checkout with
|
||||
// core.autocrlf rewrites every line ending, and that must not read as an edit.
|
||||
for (const line of content.replace(/\r\n/g, '\n').split('\n')) {
|
||||
if (!inBlock && SENTINEL_OPEN.test(line)) {
|
||||
inBlock = true;
|
||||
continue;
|
||||
}
|
||||
if (inBlock) {
|
||||
if (SENTINEL_CLOSE.test(line)) inBlock = false;
|
||||
continue;
|
||||
}
|
||||
if (LEGACY_MANAGED_LINES.some((re) => re.test(line))) continue;
|
||||
out.push(line);
|
||||
}
|
||||
|
||||
return out.join('\n').replace(/\s+$/, '');
|
||||
}
|
||||
|
||||
export function sha256(content: string): string {
|
||||
return createHash('sha256').update(content).digest('hex');
|
||||
}
|
||||
|
||||
/** Pristine `task-shared/` filenames matching a managed file's baseline pattern. */
|
||||
function baselineCandidates(sharedDir: string, pattern: RegExp): string[] {
|
||||
if (!existsSync(sharedDir)) return [];
|
||||
return readdirSync(sharedDir)
|
||||
.filter((f) => pattern.test(f))
|
||||
.sort();
|
||||
}
|
||||
|
||||
/**
|
||||
* Record the managed files, so later edits are detectable. Call at task creation
|
||||
* and after a managed file is first put in place.
|
||||
*
|
||||
* A file earns a baseline only by matching a copy this toolkit ships, and an
|
||||
* entry already recorded is never rewritten. Together those mean a stamp can
|
||||
* only ever describe a pristine file: re-running this can't turn an author's
|
||||
* edit into the new baseline, and a file dropped in later (the polyglot
|
||||
* Dockerfile, which is the scaffold's placeholder at first stamp) still gets a
|
||||
* baseline once it's in place.
|
||||
*
|
||||
* Returns true if anything was recorded.
|
||||
*/
|
||||
export function writeManagedStamp(taskDir: string, toolkitRoot: string): boolean {
|
||||
const sharedDir = join(toolkitRoot, 'task-shared');
|
||||
const existing = readStamp(taskDir);
|
||||
const files: Record<string, string> = { ...(existing?.files ?? {}) };
|
||||
let added = false;
|
||||
|
||||
for (const managed of MANAGED_FILES) {
|
||||
if (files[managed.taskPath]) continue;
|
||||
const p = join(taskDir, managed.taskPath);
|
||||
if (!existsSync(p)) continue;
|
||||
const raw = readFileSync(p, 'utf-8');
|
||||
// Not a baseline: the author still has to drop in their member's base image.
|
||||
if (raw.includes(PLACEHOLDER_MARKER)) continue;
|
||||
const stripped = stripManagedBlocks(raw);
|
||||
if (!matchesShipped(sharedDir, managed, stripped)) continue;
|
||||
files[managed.taskPath] = sha256(stripped);
|
||||
added = true;
|
||||
}
|
||||
|
||||
if (!added) return false;
|
||||
|
||||
const stamp: ManagedStamp = {
|
||||
version: 1,
|
||||
stampedAt: new Date().toISOString(),
|
||||
files,
|
||||
};
|
||||
writeFileSync(join(taskDir, STAMP_FILENAME), `${JSON.stringify(stamp, null, 2)}\n`);
|
||||
return true;
|
||||
}
|
||||
|
||||
function readStamp(taskDir: string): ManagedStamp | null {
|
||||
const stampPath = join(taskDir, STAMP_FILENAME);
|
||||
if (!existsSync(stampPath)) return null;
|
||||
try {
|
||||
const parsed = JSON.parse(readFileSync(stampPath, 'utf-8')) as ManagedStamp;
|
||||
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
|
||||
} catch {
|
||||
// Treat a corrupt stamp as no stamp rather than blocking a trial over it.
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* How to restore a managed file, or undefined when this toolkit ships no copy to
|
||||
* restore from. Only the Dockerfile can have several candidates (one per member).
|
||||
*/
|
||||
function restoreCommand(taskPath: string, candidates: string[], slug: string): string | undefined {
|
||||
const dest = `harbor-tasks/${slug}/${taskPath}`;
|
||||
if (candidates.length === 1) return `cp task-shared/${candidates[0]} ${dest}`;
|
||||
if (candidates.length > 1) {
|
||||
return `cp task-shared/Dockerfile.<your-member> ${dest} (list them: ls task-shared/Dockerfile.*)`;
|
||||
}
|
||||
// Never guess. Emitting the multi-candidate Dockerfile line here would tell an
|
||||
// author to copy a Dockerfile over their grader prompt.
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/** Render a restore line, saying so plainly when there is nothing to restore from. */
|
||||
function restoreLine(f: FileVerdict): string {
|
||||
return f.restore
|
||||
? ` ${f.restore}`
|
||||
: ` (no copy of ${f.taskPath} ships in task-shared/ — re-extract the toolkit zip)`;
|
||||
}
|
||||
|
||||
/** Does this content match a pristine copy the toolkit ships? */
|
||||
function matchesShipped(sharedDir: string, managed: ManagedFile, stripped: string): boolean {
|
||||
return baselineCandidates(sharedDir, managed.baselinePattern).some(
|
||||
(c) => stripManagedBlocks(readFileSync(join(sharedDir, c), 'utf-8')) === stripped
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Was this task created by this toolkit generation? Task creation (the packed
|
||||
* scaffold's task.toml and snapshot-to-task) writes `[metadata].toolkit_version`;
|
||||
* a task directory without the key was authored on a different toolkit
|
||||
* generation and grades with the assets frozen in its own tests/ directory, so
|
||||
* comparing those against this toolkit's copies would report drift that is not
|
||||
* an edit. Presence-based on purpose: wall-clock stamps cannot separate the
|
||||
* generations, because tasks from an earlier generation are completed after
|
||||
* later kits ship.
|
||||
*
|
||||
* A regex rather than a TOML parser, for the same shipped-dependency reason as
|
||||
* input-checksums.ts readGitref: the one `toolkit_version` key in a task.toml
|
||||
* is `[metadata].toolkit_version`.
|
||||
*/
|
||||
export function isTaskFromThisToolkitGeneration(taskDir: string): boolean {
|
||||
const tomlPath = join(taskDir, 'task.toml');
|
||||
if (!existsSync(tomlPath)) return false;
|
||||
try {
|
||||
return /^\s*toolkit_version\s*=/m.test(readFileSync(tomlPath, 'utf-8'));
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare a task's managed files against its creation-time stamp.
|
||||
*
|
||||
* @param taskDir Absolute path to `harbor-tasks/<slug>`.
|
||||
* @param toolkitRoot Absolute path to the toolkit root (holds `task-shared/`).
|
||||
*/
|
||||
export function checkTaskInfraIntegrity(taskDir: string, toolkitRoot: string): IntegrityReport {
|
||||
const sharedDir = join(toolkitRoot, 'task-shared');
|
||||
|
||||
// Without task-shared/ there is nothing to compare against; report "not
|
||||
// checked" so callers no-op rather than reporting three phantom failures.
|
||||
if (!existsSync(sharedDir)) {
|
||||
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
|
||||
}
|
||||
|
||||
// A task authored on a different toolkit generation grades with the assets
|
||||
// frozen in its own tests/ directory. Comparing those against this toolkit's
|
||||
// copies would report drift that is not an edit — and the printed remedy
|
||||
// (restore the current copy) would change how that task grades. Skip it.
|
||||
if (!isTaskFromThisToolkitGeneration(taskDir)) {
|
||||
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
|
||||
}
|
||||
|
||||
const slug = basename(taskDir);
|
||||
const stamp = readStamp(taskDir);
|
||||
const files: FileVerdict[] = [];
|
||||
|
||||
for (const managed of MANAGED_FILES) {
|
||||
const taskFile = join(taskDir, managed.taskPath);
|
||||
if (!existsSync(taskFile)) {
|
||||
files.push({ taskPath: managed.taskPath, status: 'missing' });
|
||||
continue;
|
||||
}
|
||||
|
||||
const raw = readFileSync(taskFile, 'utf-8');
|
||||
const candidates = baselineCandidates(sharedDir, managed.baselinePattern);
|
||||
const restore = restoreCommand(managed.taskPath, candidates, slug);
|
||||
const stripped = stripManagedBlocks(raw);
|
||||
|
||||
// FIRST: is this byte-for-byte something the toolkit ships right now? If so it
|
||||
// cannot be an author edit, whatever the stamp says — and asking the stamp first
|
||||
// is what used to make restoring the current copy (which is exactly what we tell
|
||||
// authors to do) look like an edit, with no way out.
|
||||
if (matchesShipped(sharedDir, managed, stripped)) {
|
||||
files.push({ taskPath: managed.taskPath, status: 'ok' });
|
||||
continue;
|
||||
}
|
||||
|
||||
const expected = stamp?.files[managed.taskPath];
|
||||
if (expected) {
|
||||
// Matches its baseline but nothing shipped: untouched by the author, and the
|
||||
// toolkit has moved on since. Worth saying, nobody's fault.
|
||||
const status = sha256(stripped) === expected ? 'outdated' : 'modified';
|
||||
files.push({ taskPath: managed.taskPath, status, restore });
|
||||
continue;
|
||||
}
|
||||
|
||||
// Checked after the stamp so that adding this marker to a file that HAS a
|
||||
// baseline can't exempt it from the comparison.
|
||||
if (raw.includes(PLACEHOLDER_MARKER)) {
|
||||
files.push({ taskPath: managed.taskPath, status: 'placeholder' });
|
||||
continue;
|
||||
}
|
||||
|
||||
files.push({ taskPath: managed.taskPath, status: 'unverifiable', restore });
|
||||
}
|
||||
|
||||
return {
|
||||
checked: true,
|
||||
files,
|
||||
modified: files.filter((f) => f.status === 'modified'),
|
||||
unverifiable: files.filter((f) => f.status === 'unverifiable'),
|
||||
outdated: files.filter((f) => f.status === 'outdated'),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Human-readable report. Returns '' when there is nothing worth saying, so callers
|
||||
* can `if (msg) print(msg)`.
|
||||
*
|
||||
* Deliberately not phrased as a refusal. An author who changed one of these files
|
||||
* almost always did it to get unstuck, not knowing we'd rather they told us — so
|
||||
* this explains what it means for their task and what restoring would do, and then
|
||||
* lets them get on with it.
|
||||
*/
|
||||
export function formatIntegrityReport(report: IntegrityReport): string {
|
||||
const sections: string[] = [];
|
||||
|
||||
if (report.modified.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
'These files look edited since this task was created, and the toolkit manages',
|
||||
'them — they set up how the trial runs and how the grade is produced, so they',
|
||||
"have to be identical across every task. Yours aren't, which makes this task's",
|
||||
"runs hard to compare with everyone else's:",
|
||||
'',
|
||||
...report.modified.map((f) => ` ${f.taskPath}`),
|
||||
'',
|
||||
'Restoring the shipped version puts that right:',
|
||||
...report.modified.map(restoreLine),
|
||||
'',
|
||||
'If you changed one to work around a problem — a missing package, a grader that',
|
||||
"wouldn't run — please tell us about the problem instead. It almost certainly",
|
||||
'affects other authors too, and the fix belongs in the toolkit, not in one task.',
|
||||
'Nothing here stops you running trials or submitting.',
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
if (report.outdated.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
'These files are unchanged, but the toolkit has shipped newer copies since this',
|
||||
'task was created:',
|
||||
'',
|
||||
...report.outdated.map((f) => ` ${f.taskPath}`),
|
||||
'',
|
||||
"You haven't done anything wrong. It does mean this task was run and graded with",
|
||||
"older versions than a task built today, so its scores aren't directly",
|
||||
'comparable. To line them up, restore the current copies and re-run your trials:',
|
||||
...report.outdated.map(restoreLine),
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
if (report.unverifiable.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
"These files don't match the copies this toolkit ships, and this task has no",
|
||||
'record of what they looked like when it was created:',
|
||||
'',
|
||||
...report.unverifiable.map((f) => ` ${f.taskPath}`),
|
||||
'',
|
||||
'Two things look like this and we cannot tell them apart: a task created on an',
|
||||
'earlier toolkit release (nothing to fix, though its scores are not directly',
|
||||
'comparable to a task built today), or a file that was edited. Either way,',
|
||||
'restoring the current copy and re-running your trials is what makes this task',
|
||||
"comparable to everyone else's:",
|
||||
...report.unverifiable.map(restoreLine),
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
return sections.join('\n\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Wrap a report in a banner loud enough to survive a scrollback.
|
||||
*
|
||||
* Nothing blocks any more, so this notice is the entire mechanism — and an
|
||||
* unframed paragraph among build output is one a reasonable person scrolls past.
|
||||
*/
|
||||
export function bannerize(message: string, report: IntegrityReport): string {
|
||||
return banner(
|
||||
message,
|
||||
report.modified.length > 0
|
||||
? '!! TOOLKIT-MANAGED FILES LOOK EDITED — PLEASE READ !!'
|
||||
: '!! TOOLKIT-MANAGED FILES NEED A LOOK — PLEASE READ !!'
|
||||
);
|
||||
}
|
||||
@@ -1,217 +0,0 @@
|
||||
/**
|
||||
* toolkit-script-integrity.ts — detect edits to the toolkit's own scripts.
|
||||
*
|
||||
* Sibling of task-infra-integrity.ts, which covers a task's managed files. This
|
||||
* covers `scripts/`. The scripts never ship with a task, so an edit can't reach
|
||||
* the delivered workspace — but their OUTPUT does: `build-workspace.sh` alone
|
||||
* stages `tests/test-commands.sh` (the deterministic checks behind the
|
||||
* correctness score), writes the Dockerfile's toolkit-managed blocks, and
|
||||
* records the managed stamp and input checksums. Nothing downstream re-derives
|
||||
* those, and the reference runs can't be re-derived at all.
|
||||
*
|
||||
* The baseline is a manifest written at package time ({@link writeScriptManifest}),
|
||||
* so it ships in the same zip as the scripts it describes. That removes the
|
||||
* ambiguity a task's managed files have: there is no "created on an older
|
||||
* release" case to tell apart, so a hash mismatch is an edit. Files absent from
|
||||
* the manifest are ignored, which keeps a worker's own helper script — or a
|
||||
* `__pycache__` left by a harbor run — from ever being reported.
|
||||
*/
|
||||
|
||||
import { createHash } from 'crypto';
|
||||
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
|
||||
import { join, relative } from 'path';
|
||||
|
||||
import { banner } from './notice-banner.js';
|
||||
|
||||
/** Manifest of the shipped `scripts/` tree. Lives at the toolkit root. */
|
||||
export const SCRIPT_MANIFEST_FILENAME = '.toolkit-scripts.json';
|
||||
|
||||
const MANIFEST_VERSION = 1;
|
||||
|
||||
/** Runtime droppings, never part of the shipped tree. */
|
||||
const IGNORED_DIRS = new Set(['__pycache__', 'node_modules', '.git']);
|
||||
const IGNORED_FILES = /\.(pyc|pyo)$/;
|
||||
|
||||
/**
|
||||
* `modified` — content differs from what shipped: an edit.
|
||||
* `missing` — shipped, but no longer on disk.
|
||||
* `ok` — unchanged.
|
||||
*/
|
||||
export type ScriptStatus = 'ok' | 'modified' | 'missing';
|
||||
|
||||
export interface ScriptVerdict {
|
||||
/** Toolkit-relative path, e.g. `scripts/build-workspace.sh`. */
|
||||
path: string;
|
||||
status: ScriptStatus;
|
||||
}
|
||||
|
||||
export interface ScriptIntegrityReport {
|
||||
/** False when no manifest ships — callers should skip silently. */
|
||||
checked: boolean;
|
||||
files: ScriptVerdict[];
|
||||
modified: ScriptVerdict[];
|
||||
missing: ScriptVerdict[];
|
||||
}
|
||||
|
||||
interface ScriptManifest {
|
||||
version: number;
|
||||
generatedAt: string;
|
||||
/** Toolkit-relative path → sha256 of the normalized content. */
|
||||
files: Record<string, string>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Line endings and trailing whitespace are normalized away: a Windows editor, a
|
||||
* checkout with core.autocrlf, or a formatter trimming a final newline must not
|
||||
* read as an edit.
|
||||
*/
|
||||
function hashContent(content: string): string {
|
||||
return createHash('sha256')
|
||||
.update(content.replace(/\r\n/g, '\n').replace(/\s+$/, ''))
|
||||
.digest('hex');
|
||||
}
|
||||
|
||||
/** Every shipped file under `scripts/`, as toolkit-relative paths. */
|
||||
function walkScripts(dir: string, toolkitRoot: string): string[] {
|
||||
if (!existsSync(dir)) return [];
|
||||
const out: string[] = [];
|
||||
|
||||
for (const entry of readdirSync(dir, { withFileTypes: true }).sort((a, b) =>
|
||||
a.name.localeCompare(b.name)
|
||||
)) {
|
||||
const abs = join(dir, entry.name);
|
||||
if (entry.isDirectory()) {
|
||||
if (!IGNORED_DIRS.has(entry.name)) out.push(...walkScripts(abs, toolkitRoot));
|
||||
continue;
|
||||
}
|
||||
if (!entry.isFile() || IGNORED_FILES.test(entry.name)) continue;
|
||||
out.push(relative(toolkitRoot, abs));
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Record the shipped `scripts/` tree. Call at package time, once the tree is
|
||||
* fully staged — anything written to `scripts/` afterwards reads as an edit.
|
||||
*
|
||||
* Returns the number of files recorded.
|
||||
*/
|
||||
export function writeScriptManifest(toolkitRoot: string): number {
|
||||
const files: Record<string, string> = {};
|
||||
|
||||
for (const rel of walkScripts(join(toolkitRoot, 'scripts'), toolkitRoot)) {
|
||||
files[rel] = hashContent(readFileSync(join(toolkitRoot, rel), 'utf-8'));
|
||||
}
|
||||
|
||||
const manifest: ScriptManifest = {
|
||||
version: MANIFEST_VERSION,
|
||||
generatedAt: new Date().toISOString(),
|
||||
files,
|
||||
};
|
||||
writeFileSync(
|
||||
join(toolkitRoot, SCRIPT_MANIFEST_FILENAME),
|
||||
`${JSON.stringify(manifest, null, 2)}\n`
|
||||
);
|
||||
return Object.keys(files).length;
|
||||
}
|
||||
|
||||
function readManifest(toolkitRoot: string): ScriptManifest | null {
|
||||
const p = join(toolkitRoot, SCRIPT_MANIFEST_FILENAME);
|
||||
if (!existsSync(p)) return null;
|
||||
try {
|
||||
const parsed = JSON.parse(readFileSync(p, 'utf-8')) as ScriptManifest;
|
||||
if (parsed?.version !== MANIFEST_VERSION) return null;
|
||||
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
|
||||
} catch {
|
||||
// A corrupt manifest is treated as no manifest rather than blocking a trial.
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare the toolkit's `scripts/` tree against the manifest it shipped with.
|
||||
*
|
||||
* @param toolkitRoot Absolute path to the toolkit root (holds `scripts/`).
|
||||
*/
|
||||
export function checkToolkitScriptIntegrity(toolkitRoot: string): ScriptIntegrityReport {
|
||||
const manifest = readManifest(toolkitRoot);
|
||||
if (!manifest) return { checked: false, files: [], modified: [], missing: [] };
|
||||
|
||||
const files: ScriptVerdict[] = Object.entries(manifest.files).map(([path, expected]) => {
|
||||
const abs = join(toolkitRoot, path);
|
||||
if (!existsSync(abs)) return { path, status: 'missing' as const };
|
||||
const status = hashContent(readFileSync(abs, 'utf-8')) === expected ? 'ok' : 'modified';
|
||||
return { path, status };
|
||||
});
|
||||
|
||||
return {
|
||||
checked: true,
|
||||
files,
|
||||
modified: files.filter((f) => f.status === 'modified'),
|
||||
missing: files.filter((f) => f.status === 'missing'),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Human-readable report. Returns '' when there is nothing worth saying, so callers
|
||||
* can `if (msg) print(msg)`.
|
||||
*
|
||||
* Deliberately not phrased as a refusal, for the same reason the managed-file
|
||||
* notice isn't: an author who changed one of these did it to get unstuck, and the
|
||||
* fix they needed almost certainly belongs in the toolkit rather than in their copy.
|
||||
*/
|
||||
export function formatScriptIntegrityReport(report: ScriptIntegrityReport): string {
|
||||
const sections: string[] = [];
|
||||
|
||||
if (report.modified.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
'These toolkit scripts look edited:',
|
||||
'',
|
||||
...report.modified.map((f) => ` ${f.path}`),
|
||||
'',
|
||||
"They aren't part of any task, so an edit is easy to miss — but what they write",
|
||||
'is. Building a task stages its deterministic checks, fills in parts of its',
|
||||
'Dockerfile, and records the checksums a reviewer reads; a script that does any of',
|
||||
'that differently produces a task that looks normal and behaves differently from',
|
||||
'every other one.',
|
||||
'',
|
||||
'Re-extracting the toolkit zip over your copy restores them. Your tasks, snapshots',
|
||||
'and reference runs are untouched by that.',
|
||||
'',
|
||||
'If you changed one to work around a problem — a build that would not run, a',
|
||||
'missing dependency — please tell us about the problem instead. It almost',
|
||||
'certainly affects other authors too, and the fix belongs in the toolkit.',
|
||||
'Nothing here stops you running trials or submitting.',
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
if (report.missing.length > 0) {
|
||||
sections.push(
|
||||
[
|
||||
'These toolkit scripts shipped with this release but are no longer here:',
|
||||
'',
|
||||
...report.missing.map((f) => ` ${f.path}`),
|
||||
'',
|
||||
'Something that depends on one will fail partway through rather than up front.',
|
||||
'Re-extract the toolkit zip over your copy to put them back.',
|
||||
].join('\n')
|
||||
);
|
||||
}
|
||||
|
||||
return sections.join('\n\n');
|
||||
}
|
||||
|
||||
/** The full notice, bannered and ready to write to stderr, or '' if all is well. */
|
||||
export function scriptIntegrityNotice(report: ScriptIntegrityReport): string {
|
||||
const message = formatScriptIntegrityReport(report);
|
||||
if (!message) return '';
|
||||
|
||||
const headline =
|
||||
report.modified.length > 0
|
||||
? '!! TOOLKIT SCRIPTS LOOK EDITED — PLEASE READ !!'
|
||||
: '!! TOOLKIT SCRIPTS ARE MISSING — PLEASE READ !!';
|
||||
return banner(message, headline);
|
||||
}
|
||||
@@ -1,304 +0,0 @@
|
||||
/**
|
||||
* Tests for tree-permissions.ts.
|
||||
*
|
||||
* The load-bearing case is the one from the field report: a directory that came
|
||||
* across without its search bit makes `tar` fail with `Cannot stat` on the files
|
||||
* *inside* it, so the repair has to fix directory modes, not just ownership.
|
||||
* These tests run unprivileged, so they exercise the mode axis for real and the
|
||||
* ownership axis only as far as an unprivileged process can (target resolution +
|
||||
* graceful EPERM), which is the same shape CI runs in. One case needs root and
|
||||
* skips otherwise; the rest hold under either uid, which is why the fixtures that
|
||||
* must look human-owned say so with `ownedByHuman` instead of relying on the
|
||||
* caller's uid.
|
||||
*/
|
||||
import assert from 'node:assert/strict';
|
||||
import {
|
||||
chmodSync,
|
||||
chownSync,
|
||||
mkdirSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
symlinkSync,
|
||||
writeFileSync,
|
||||
} from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { test } from 'node:test';
|
||||
|
||||
import {
|
||||
didRepair,
|
||||
manualRepairHint,
|
||||
normalizeTreePermissions,
|
||||
resolveWorkspaceOwner,
|
||||
} from './tree-permissions';
|
||||
|
||||
function scratch(name: string): string {
|
||||
const dir = join(tmpdir(), `tree-perms-${name}-${process.pid}`);
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
mkdirSync(dir, { recursive: true });
|
||||
return dir;
|
||||
}
|
||||
|
||||
const RUNNING_AS_ROOT = process.getuid?.() === 0;
|
||||
const HUMAN_UID = RUNNING_AS_ROOT ? 1000 : (process.getuid?.() ?? 0);
|
||||
const HUMAN_GID = RUNNING_AS_ROOT ? 1000 : (process.getgid?.() ?? 0);
|
||||
|
||||
/** Give a fixture a non-root owner, so the repair sees a tree it can hand back. */
|
||||
function ownedByHuman(path: string): string {
|
||||
chownSync(path, HUMAN_UID, HUMAN_GID);
|
||||
return path;
|
||||
}
|
||||
|
||||
test('restores the search bit on a directory that lost it', () => {
|
||||
const root = scratch('searchbit');
|
||||
const models = join(root, 'agent-output', 'app', 'models');
|
||||
mkdirSync(models, { recursive: true });
|
||||
writeFileSync(join(models, 'bill.rb'), 'class Bill; end\n');
|
||||
// r-- : readdir works, so tar can NAME the file, but stat is refused.
|
||||
chmodSync(models, 0o400);
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.equal(statSync(models).mode & 0o700, 0o700, 'owner rwx restored on the directory');
|
||||
assert.ok(report.modeFixed.some((p) => p === models));
|
||||
assert.ok(didRepair(report));
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('recurses into a directory it had to widen first', () => {
|
||||
const root = scratch('recurse');
|
||||
const inner = join(root, 'locked', 'deeper');
|
||||
mkdirSync(inner, { recursive: true });
|
||||
const leaf = join(inner, 'leaf.rb');
|
||||
writeFileSync(leaf, 'x\n');
|
||||
chmodSync(leaf, 0o000);
|
||||
chmodSync(inner, 0o400);
|
||||
chmodSync(join(root, 'locked'), 0o400);
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
// Only reachable if the walk widened each parent before descending.
|
||||
assert.equal(statSync(leaf).mode & 0o600, 0o600, 'leaf became owner-readable');
|
||||
assert.ok(report.modeFixed.includes(leaf));
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('leaves already-correct trees untouched', () => {
|
||||
const root = scratch('noop');
|
||||
mkdirSync(join(root, 'sub'), { recursive: true });
|
||||
writeFileSync(join(root, 'sub', 'f.txt'), 'hi\n');
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.deepEqual(report.modeFixed, [], 'no mode changes');
|
||||
assert.deepEqual(report.ownerFixed, [], 'no owner changes (already ours)');
|
||||
assert.deepEqual(report.failures, []);
|
||||
assert.equal(didRepair(report), false);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('does not widen group/other beyond what was already there', () => {
|
||||
const root = ownedByHuman(scratch('narrow'));
|
||||
const f = join(root, 'secret.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
ownedByHuman(f);
|
||||
chmodSync(f, 0o000);
|
||||
|
||||
normalizeTreePermissions(root, { ownerRef: root });
|
||||
|
||||
const mode = statSync(f).mode & 0o777;
|
||||
assert.equal(mode, 0o600, 'owner rw only — group/other stay closed');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('ignores symlinks rather than following them out of the tree', () => {
|
||||
const root = scratch('symlink');
|
||||
const outside = scratch('symlink-outside');
|
||||
const victim = join(outside, 'victim.txt');
|
||||
writeFileSync(victim, 'x\n');
|
||||
chmodSync(victim, 0o000);
|
||||
symlinkSync(outside, join(root, 'link'));
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.equal(statSync(victim).mode & 0o777, 0o000, 'target outside the tree untouched');
|
||||
assert.deepEqual(report.failures, []);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
rmSync(outside, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('never throws on a missing root, and reports it', () => {
|
||||
const report = normalizeTreePermissions(join(tmpdir(), 'definitely-not-here-xyz'));
|
||||
assert.equal(report.failures.length, 1);
|
||||
assert.equal(report.failures[0].reason, 'ENOENT');
|
||||
});
|
||||
|
||||
test('resolveWorkspaceOwner reads the reference path, not the caller', () => {
|
||||
const root = scratch('owner');
|
||||
const owner = resolveWorkspaceOwner(root);
|
||||
assert.ok(owner, 'resolved');
|
||||
const st = statSync(root);
|
||||
assert.equal(owner.uid, st.uid);
|
||||
assert.equal(owner.gid, st.gid);
|
||||
assert.equal(resolveWorkspaceOwner(join(tmpdir(), 'nope-xyz')), null);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('never chowns TO root, even when the owner ref is root-owned', () => {
|
||||
// The regression this guards: workspace root owned by root (unzipped with
|
||||
// sudo) while the task files are correctly owned by the human. Chowning to the
|
||||
// ref's owner would inflict the very lockout this module prevents. `/` is
|
||||
// root-owned on every platform we run on, so it's a stable stand-in.
|
||||
const root = scratch('root-ref');
|
||||
const f = join(root, 'mine.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
const beforeUid = statSync(f).uid;
|
||||
|
||||
const report = normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(report.target?.uid, 0, 'resolved a root target');
|
||||
assert.deepEqual(report.ownerFixed, [], 'declined to chown anything to root');
|
||||
assert.deepEqual(report.failures, [], 'and did not fail trying');
|
||||
assert.equal(statSync(f).uid, beforeUid, 'owner untouched');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('still normalizes modes when the chown target is root', () => {
|
||||
const root = scratch('root-ref-modes');
|
||||
const sub = join(root, 'sub');
|
||||
mkdirSync(sub, { recursive: true });
|
||||
writeFileSync(join(sub, 'f.txt'), 'x\n');
|
||||
chmodSync(sub, 0o400);
|
||||
|
||||
const report = normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(statSync(sub).mode & 0o700, 0o700, 'mode axis still applied');
|
||||
assert.ok(report.modeFixed.includes(sub));
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('keeps modes narrow for files that have a real owner, even under a root ref', () => {
|
||||
// The complement of the case below: we declined to chown, but these entries are
|
||||
// already the human's, so owner bits reach them and nothing should be widened.
|
||||
const root = ownedByHuman(scratch('root-ref-narrow'));
|
||||
const f = join(root, 'mine.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
ownedByHuman(f);
|
||||
chmodSync(f, 0o600);
|
||||
|
||||
normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(statSync(f).mode & 0o077, 0, 'group/other untouched');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test(
|
||||
'grants read+search to group and other on files stranded root-owned',
|
||||
{ skip: process.getuid?.() !== 0 ? 'needs root to create root-owned files' : false },
|
||||
() => {
|
||||
// The worker authoring container: root process, root-owned workspace. The chown
|
||||
// is declined, so owner bits land on root and the human — a different uid in
|
||||
// Explore and on a WSL host — is still locked out of a --w------- capture.
|
||||
const root = scratch('stranded');
|
||||
const sub = join(root, 'agent-output');
|
||||
mkdirSync(sub, { recursive: true });
|
||||
const f = join(sub, 'answer.md');
|
||||
writeFileSync(f, 'x\n');
|
||||
chmodSync(f, 0o200);
|
||||
chmodSync(sub, 0o300);
|
||||
|
||||
normalizeTreePermissions(root, { ownerRef: '/' });
|
||||
|
||||
assert.equal(
|
||||
statSync(f).mode & 0o777,
|
||||
0o644,
|
||||
'file readable by everyone, writable by none but root'
|
||||
);
|
||||
assert.equal(statSync(sub).mode & 0o777, 0o755, 'directory searchable');
|
||||
}
|
||||
);
|
||||
|
||||
test('walks a tree as deep as the filesystem allows', () => {
|
||||
const root = scratch('deep');
|
||||
// PATH_MAX caps how deep a tree can physically get (~300 levels at these name
|
||||
// lengths — building deeper fails with ENAMETOOLONG), which is well inside any
|
||||
// call-stack limit. So this isn't a stack test; it just pins that a deep,
|
||||
// narrow tree walks cleanly end to end.
|
||||
let path = root;
|
||||
for (let i = 0; i < 250; i++) {
|
||||
path = join(path, `d${i}`);
|
||||
}
|
||||
mkdirSync(path, { recursive: true });
|
||||
writeFileSync(join(path, 'leaf.txt'), 'x\n');
|
||||
chmodSync(join(path, 'leaf.txt'), 0o000);
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.deepEqual(report.failures, [], 'walked the whole depth cleanly');
|
||||
assert.equal(statSync(join(path, 'leaf.txt')).mode & 0o600, 0o600, 'reached the deepest leaf');
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('a failure in one subtree does not abandon the rest', () => {
|
||||
const root = scratch('partial');
|
||||
const good = join(root, 'good');
|
||||
mkdirSync(good, { recursive: true });
|
||||
const goodFile = join(good, 'f.txt');
|
||||
writeFileSync(goodFile, 'x\n');
|
||||
chmodSync(goodFile, 0o000);
|
||||
// A dangling symlink and a vanished path both produce per-entry trouble.
|
||||
symlinkSync(join(root, 'nowhere'), join(root, 'dangling'));
|
||||
|
||||
const report = normalizeTreePermissions(root);
|
||||
|
||||
assert.equal(statSync(goodFile).mode & 0o600, 0o600, 'the healthy subtree was still repaired');
|
||||
assert.ok(report.modeFixed.includes(goodFile));
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('reports rather than throws when the root is a file, not a directory', () => {
|
||||
const root = scratch('file-root');
|
||||
const f = join(root, 'lonely.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
chmodSync(f, 0o000);
|
||||
|
||||
const report = normalizeTreePermissions(f);
|
||||
|
||||
assert.equal(statSync(f).mode & 0o600, 0o600);
|
||||
assert.deepEqual(report.failures, []);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('RACCOON_SKIP_PERMISSION_REPAIR=1 makes it a total no-op', () => {
|
||||
const root = scratch('killswitch');
|
||||
const sub = join(root, 'sub');
|
||||
mkdirSync(sub, { recursive: true });
|
||||
const f = join(sub, 'f.txt');
|
||||
writeFileSync(f, 'x\n');
|
||||
chmodSync(f, 0o000);
|
||||
chmodSync(sub, 0o400);
|
||||
|
||||
const prev = process.env.RACCOON_SKIP_PERMISSION_REPAIR;
|
||||
process.env.RACCOON_SKIP_PERMISSION_REPAIR = '1';
|
||||
try {
|
||||
const report = normalizeTreePermissions(root);
|
||||
assert.equal(report.skipped, true);
|
||||
assert.deepEqual(report.modeFixed, []);
|
||||
assert.deepEqual(report.ownerFixed, []);
|
||||
assert.deepEqual(report.failures, []);
|
||||
assert.equal(didRepair(report), false);
|
||||
assert.equal(statSync(sub).mode & 0o777, 0o400, 'directory left exactly as it was');
|
||||
} finally {
|
||||
if (prev === undefined) delete process.env.RACCOON_SKIP_PERMISSION_REPAIR;
|
||||
else process.env.RACCOON_SKIP_PERMISSION_REPAIR = prev;
|
||||
}
|
||||
chmodSync(sub, 0o700);
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test('manual hint repairs both axes, ownership first', () => {
|
||||
const hint = manualRepairHint('harbor-tasks/my-slug');
|
||||
assert.match(hint, /chown -R/);
|
||||
assert.match(hint, /chmod -R u\+rwX/);
|
||||
assert.ok(hint.indexOf('chown') < hint.indexOf('chmod'), 'chown before chmod');
|
||||
});
|
||||
@@ -1,126 +0,0 @@
|
||||
/**
|
||||
* tree-permissions.ts — make a copied tree readable by whoever owns the workspace.
|
||||
*
|
||||
* Files captured from a task run can arrive owned by another user, or with a
|
||||
* directory missing the permission needed to walk into it. Packaging then fails
|
||||
* with `Cannot stat: Permission denied`. This repairs both.
|
||||
*
|
||||
* Grants owner rwX only, never group or other. Never throws, and never hands
|
||||
* files to root. Set `RACCOON_SKIP_PERMISSION_REPAIR=1` to turn it off.
|
||||
*/
|
||||
import { chmodSync, chownSync, lstatSync, readdirSync, statSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
export interface NormalizeReport {
|
||||
/** Paths whose owner was changed. */
|
||||
ownerFixed: string[];
|
||||
/** Paths whose mode gained owner rwX. */
|
||||
modeFixed: string[];
|
||||
/** Paths we wanted to change but could not, with the errno. */
|
||||
failures: { path: string; reason: string }[];
|
||||
/** Resolved target owner, or null if it couldn't be determined. */
|
||||
target: { uid: number; gid: number } | null;
|
||||
/** Set when disabled via RACCOON_SKIP_PERMISSION_REPAIR. */
|
||||
skipped?: boolean;
|
||||
}
|
||||
|
||||
/** Owner a workspace tree should have: whoever owns `ownerRef`. */
|
||||
export function resolveWorkspaceOwner(ownerRef: string): { uid: number; gid: number } | null {
|
||||
try {
|
||||
const st = statSync(ownerRef);
|
||||
return { uid: st.uid, gid: st.gid };
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Owner-rwX mode, preserving every other bit. Dirs also need the search bit.
|
||||
*
|
||||
* `stranded` means the file stays root-owned because we have no non-root owner to
|
||||
* give it to. Owner bits then help nobody — whoever has to read it is a different
|
||||
* user — so read and search are granted more widely. Never write, never +x on files.
|
||||
*/
|
||||
function withOwnerAccess(mode: number, isDir: boolean, stranded: boolean): number {
|
||||
const owner = isDir ? 0o700 : 0o600;
|
||||
return mode | owner | (stranded ? (isDir ? 0o055 : 0o044) : 0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Give every entry under `root` to the workspace owner and make sure that owner
|
||||
* can read and traverse it. Symlinks are skipped. Repairs what it can and
|
||||
* reports what it couldn't; it never throws and never blocks its caller.
|
||||
*/
|
||||
export function normalizeTreePermissions(
|
||||
root: string,
|
||||
options: { ownerRef?: string } = {}
|
||||
): NormalizeReport {
|
||||
if (process.env.RACCOON_SKIP_PERMISSION_REPAIR === '1') {
|
||||
return { ownerFixed: [], modeFixed: [], failures: [], target: null, skipped: true };
|
||||
}
|
||||
|
||||
const target = resolveWorkspaceOwner(options.ownerRef ?? process.cwd());
|
||||
const report: NormalizeReport = { ownerFixed: [], modeFixed: [], failures: [], target };
|
||||
|
||||
// Never hand files to root — that would lock the owner out rather than help.
|
||||
const chownTarget = target && target.uid !== 0 ? target : null;
|
||||
|
||||
try {
|
||||
const stack: string[] = [root];
|
||||
while (stack.length > 0) {
|
||||
const path = stack.pop() as string;
|
||||
|
||||
let st;
|
||||
try {
|
||||
st = lstatSync(path);
|
||||
} catch (err) {
|
||||
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ELSTAT' });
|
||||
continue;
|
||||
}
|
||||
if (st.isSymbolicLink()) continue;
|
||||
|
||||
const isDir = st.isDirectory();
|
||||
|
||||
// Mode first: a directory we can't search is one we can't descend into.
|
||||
const wanted = withOwnerAccess(st.mode, isDir, chownTarget === null && st.uid === 0);
|
||||
if (wanted !== st.mode) {
|
||||
try {
|
||||
chmodSync(path, wanted);
|
||||
report.modeFixed.push(path);
|
||||
} catch (err) {
|
||||
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHMOD' });
|
||||
}
|
||||
}
|
||||
|
||||
if (chownTarget && (st.uid !== chownTarget.uid || st.gid !== chownTarget.gid)) {
|
||||
try {
|
||||
chownSync(path, chownTarget.uid, chownTarget.gid);
|
||||
report.ownerFixed.push(path);
|
||||
} catch (err) {
|
||||
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHOWN' });
|
||||
}
|
||||
}
|
||||
|
||||
if (!isDir) continue;
|
||||
try {
|
||||
for (const entry of readdirSync(path)) stack.push(join(path, entry));
|
||||
} catch (err) {
|
||||
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'EREADDIR' });
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
report.failures.push({ path: root, reason: (err as NodeJS.ErrnoException).code ?? 'EWALK' });
|
||||
}
|
||||
|
||||
return report;
|
||||
}
|
||||
|
||||
/** True when something was actually repaired. */
|
||||
export function didRepair(report: NormalizeReport): boolean {
|
||||
return report.ownerFixed.length > 0 || report.modeFixed.length > 0;
|
||||
}
|
||||
|
||||
/** The command to run on your host if we couldn't fix it ourselves. */
|
||||
export function manualRepairHint(path: string): string {
|
||||
return `sudo chown -R "$(id -un):$(id -gn)" ${path} && chmod -R u+rwX ${path}`;
|
||||
}
|
||||
Reference in New Issue
Block a user