lots of change - all to start my 3rd redo

This commit is contained in:
2026-09-26 14:31:52 -04:00
parent 7f4d388e19
commit bceb52e8ee
1046 changed files with 4476 additions and 0 deletions

View File

@@ -1,436 +0,0 @@
"""Convert seed conversations between harness-native session formats, via ATIF.
Parsers turn a native session into ATIF; renderers turn ATIF back into a native
session. Adding a harness is one parser plus one renderer.
Renderers flatten tool calls to narration (`[ran Bash: {...}]` / `[result: ...]`)
rather than rebuilding native tool-call records. Every renderer must flatten
identically — see flatten_steps.
"""
from __future__ import annotations
import json
from typing import Any
ATIF_SCHEMA_VERSION = "ATIF-v1.7"
# Codex rollout record types, used to tell the formats apart.
_CODEX_ROLLOUT_TYPES = frozenset(
{"session_meta", "response_item", "event_msg", "turn_context", "compacted"}
)
# ---------------------------------------------------------------------------
# Format detection
# ---------------------------------------------------------------------------
def detect_format(text: str) -> str | None:
"""Return 'atif', 'claude', 'codex', or None for an unrecognised/empty blob."""
stripped = text.strip()
if not stripped:
return None
# ATIF is a single JSON object, not JSONL.
if stripped.startswith("{") and '"steps"' in stripped:
try:
doc = json.loads(stripped)
except (json.JSONDecodeError, ValueError):
doc = None
if isinstance(doc, dict) and isinstance(doc.get("steps"), list):
return "atif"
for raw in stripped.splitlines():
raw = raw.strip()
if not raw:
continue
try:
rec = json.loads(raw)
except (json.JSONDecodeError, ValueError):
continue
if not isinstance(rec, dict):
continue
if rec.get("type") in _CODEX_ROLLOUT_TYPES and "message" not in rec:
return "codex"
if rec.get("type") in ("user", "assistant") or "message" in rec:
return "claude"
return None
# ---------------------------------------------------------------------------
# ATIF construction helpers
# ---------------------------------------------------------------------------
def _trajectory(steps: list[dict], *, session_id: str | None = None) -> dict:
return {
"schema_version": ATIF_SCHEMA_VERSION,
"session_id": session_id,
"agent": {"name": "unknown"},
"steps": steps,
}
def _step(
step_id: int,
source: str,
*,
message: str = "",
reasoning: str | None = None,
tool_calls: list[dict] | None = None,
observations: list[dict] | None = None,
timestamp: str | None = None,
) -> dict:
step: dict[str, Any] = {
"step_id": step_id,
"source": source,
"message": message,
"is_copied_context": True,
}
if timestamp:
step["timestamp"] = timestamp
if reasoning:
step["reasoning_content"] = reasoning
if tool_calls:
step["tool_calls"] = tool_calls
if observations:
step["observation"] = {"results": observations}
return step
def _content_text(content: Any) -> str:
"""Text of an ATIF message or a ContentPart list."""
if isinstance(content, str):
return content
if isinstance(content, list):
return "".join(
part.get("text") or "" for part in content if isinstance(part, dict)
)
return ""
# ---------------------------------------------------------------------------
# Parsers: native -> ATIF
# ---------------------------------------------------------------------------
def claude_session_to_atif(jsonl_text: str) -> dict:
"""Parse a Claude Code session.jsonl into ATIF steps.
Claude records tool results on `user` records; they become observations.
"""
steps: list[dict] = []
session_id: str | None = None
for raw in jsonl_text.splitlines():
raw = raw.strip()
if not raw:
continue
try:
rec = json.loads(raw)
except (json.JSONDecodeError, ValueError):
continue
if not isinstance(rec, dict):
continue
session_id = session_id or rec.get("sessionId")
msg = rec.get("message") or {}
role = msg.get("role") or rec.get("type")
if role not in ("user", "assistant"):
continue
content = msg.get("content")
if content is None:
continue
source = "user" if role == "user" else "agent"
if isinstance(content, str):
steps.append(
_step(
len(steps) + 1, source, message=content, timestamp=rec.get("timestamp")
)
)
continue
text_parts: list[str] = []
reasoning_parts: list[str] = []
tool_calls: list[dict] = []
observations: list[dict] = []
for block in content:
if not isinstance(block, dict):
text_parts.append(str(block))
continue
btype = block.get("type")
if btype == "text":
text_parts.append(block.get("text") or "")
elif btype == "thinking":
reasoning_parts.append(block.get("thinking") or "")
elif btype == "tool_use":
tool_calls.append(
{
"tool_call_id": block.get("id") or f"call_{len(tool_calls) + 1}",
"function_name": block.get("name") or "tool",
"arguments": block.get("input") or {},
}
)
elif btype == "tool_result":
observations.append(
{
"source_call_id": block.get("tool_use_id"),
"content": _content_text(block.get("content")),
}
)
steps.append(
_step(
len(steps) + 1,
source,
message="".join(text_parts),
reasoning="".join(reasoning_parts) or None,
tool_calls=tool_calls or None,
observations=observations or None,
timestamp=rec.get("timestamp"),
)
)
return _trajectory(steps, session_id=session_id)
def codex_rollout_to_atif(jsonl_text: str) -> dict:
"""Parse a codex rollout JSONL into ATIF steps."""
steps: list[dict] = []
session_id: str | None = None
for raw in jsonl_text.splitlines():
raw = raw.strip()
if not raw:
continue
try:
rec = json.loads(raw)
except (json.JSONDecodeError, ValueError):
continue
if not isinstance(rec, dict):
continue
rtype = rec.get("type")
payload = rec.get("payload") or {}
if rtype == "session_meta":
session_id = session_id or payload.get("id")
continue
if rtype != "response_item":
continue
ptype = payload.get("type")
timestamp = rec.get("timestamp")
if ptype == "message":
role = payload.get("role")
if role not in ("user", "assistant"):
continue
steps.append(
_step(
len(steps) + 1,
"user" if role == "user" else "agent",
message=_content_text(payload.get("content")),
timestamp=timestamp,
)
)
elif ptype in ("function_call", "local_shell_call", "custom_tool_call"):
arguments = payload.get("arguments")
if isinstance(arguments, str):
try:
arguments = json.loads(arguments)
except (json.JSONDecodeError, ValueError):
arguments = {"raw": arguments}
steps.append(
_step(
len(steps) + 1,
"agent",
tool_calls=[
{
"tool_call_id": payload.get("call_id") or f"call_{len(steps)}",
"function_name": payload.get("name") or "tool",
"arguments": arguments or {},
}
],
timestamp=timestamp,
)
)
elif ptype in ("function_call_output", "custom_tool_call_output"):
output = payload.get("output")
if isinstance(output, dict):
output = output.get("content") or json.dumps(output, ensure_ascii=False)
steps.append(
_step(
len(steps) + 1,
"agent",
observations=[
{
"source_call_id": payload.get("call_id"),
"content": output if isinstance(output, str) else "",
}
],
timestamp=timestamp,
)
)
elif ptype == "reasoning":
summary = payload.get("summary")
text = ""
if isinstance(summary, list):
text = "".join(
s.get("text") or "" for s in summary if isinstance(s, dict)
)
if text:
steps.append(_step(len(steps) + 1, "agent", reasoning=text, timestamp=timestamp))
return _trajectory(steps, session_id=session_id)
def to_atif(text: str) -> dict:
"""Parse whichever native format `text` is into ATIF."""
fmt = detect_format(text)
if fmt == "atif":
return json.loads(text)
if fmt == "claude":
return claude_session_to_atif(text)
if fmt == "codex":
return codex_rollout_to_atif(text)
raise ValueError("unrecognised session format (not ATIF, Claude JSONL, or codex rollout)")
# ---------------------------------------------------------------------------
# Flattening — shared by every renderer so the loss stays symmetric
# ---------------------------------------------------------------------------
def flatten_steps(atif: dict) -> list[tuple[str, str]]:
"""ATIF steps -> ordered (role, text) pairs, where role is 'user' or 'agent'.
Tool calls and observations become agent narration; `reasoning_content` is dropped.
"""
out: list[tuple[str, str]] = []
for step in atif.get("steps") or []:
if not isinstance(step, dict):
continue
role = "user" if step.get("source") == "user" else "agent"
text = _content_text(step.get("message"))
if text:
out.append((role, text))
for call in step.get("tool_calls") or []:
if not isinstance(call, dict):
continue
args = json.dumps(call.get("arguments") or {}, ensure_ascii=False)
out.append(("agent", f"[ran {call.get('function_name') or 'tool'}: {args}]"))
observation = step.get("observation") or {}
for result in observation.get("results") or []:
if not isinstance(result, dict):
continue
out.append(("agent", f"[result: {_content_text(result.get('content'))}]"))
return out
# ---------------------------------------------------------------------------
# Renderers: ATIF -> native
# ---------------------------------------------------------------------------
def atif_to_codex_rollout(
atif: dict,
iso_ts: str,
*,
session_meta: dict,
max_total: int | None = None,
) -> list[str]:
"""Render ATIF as codex rollout JSONL that `codex exec resume` can continue.
`session_meta` is supplied by the caller so this module reads no files.
"""
lines = [json.dumps(session_meta)]
budget = float("inf") if max_total is None else max_total
for role, text in flatten_steps(atif):
text = (text or "").strip()
if not text or budget <= 0:
continue
text = text[: int(min(budget, len(text)))]
ctype = "input_text" if role == "user" else "output_text"
lines.append(
json.dumps(
{
"timestamp": iso_ts,
"type": "response_item",
"payload": {
"type": "message",
"role": "user" if role == "user" else "assistant",
"content": [{"type": ctype, "text": text}],
},
}
)
)
budget -= len(text)
return lines
def atif_to_claude_session(
atif: dict,
*,
session_id: str,
cwd: str = "/workspace",
git_branch: str = "main",
version: str = "2.1.87",
iso_ts: str,
max_total: int | None = None,
) -> list[str]:
"""Render ATIF as Claude Code session.jsonl that `claude --resume` can continue.
Records are chained by parentUuid: Claude resumes by walking that chain, not by
file order.
"""
lines: list[str] = []
parent_uuid: str | None = None
budget = float("inf") if max_total is None else max_total
for index, (role, text) in enumerate(flatten_steps(atif), start=1):
text = (text or "").strip()
if not text or budget <= 0:
continue
text = text[: int(min(budget, len(text)))]
uuid = _deterministic_uuid(session_id, index)
claude_role = "user" if role == "user" else "assistant"
record: dict[str, Any] = {
"parentUuid": parent_uuid,
"isSidechain": False,
"userType": "external",
"cwd": cwd,
"sessionId": session_id,
"version": version,
"gitBranch": git_branch,
"type": claude_role,
"uuid": uuid,
"timestamp": iso_ts,
}
if claude_role == "user":
record["message"] = {"role": "user", "content": text}
else:
record["message"] = {
"role": "assistant",
"content": [{"type": "text", "text": text}],
"stop_reason": "end_turn",
}
lines.append(json.dumps(record))
parent_uuid = uuid
budget -= len(text)
return lines
def _deterministic_uuid(session_id: str, index: int) -> str:
"""A stable uuid5 per (session, position), so re-rendering is byte-identical."""
import uuid as _uuid
return str(_uuid.uuid5(_uuid.NAMESPACE_URL, f"raccoon-seed/{session_id}/{index}"))

View File

@@ -1,53 +0,0 @@
"""Shared browser-capability disclosure for the agent harnesses.
Only images for browser-facing repos ship Playwright, so the note is conditional on probing
the sandbox for the `pw` wrapper rather than on anything about the task. Probing keeps the
claim true by construction: telling an agent it has a browser it does not have sends it after
a missing binary. To check an image yourself: `command -v pw`.
Both harnesses disclose the same text through their own mechanism:
- Claude Code: appended to --append-system-prompt (scripts/snapshot_agent.py)
- codex: -c developer_instructions=... (scripts/codex_agent.py), which prepends a
developer message and LEAVES codex's base instructions intact. Verified with
`codex debug prompt-input`. Do not switch to model_instructions_file — that
REPLACES the base instructions.
This module exists so the probe and the text live in one place; a copy in each adapter would
drift and the drift would be invisible (both would still run, just disclosing differently).
"""
from __future__ import annotations
import logging
from pathlib import Path
_log = logging.getLogger(__name__)
_NOTE_FILE = Path(__file__).resolve().parent / "toolset_note_browser.md"
_PROBE = "command -v pw >/dev/null 2>&1 && echo yes || echo no"
def browser_note() -> str:
"""The disclosure text, or "" if the note file is missing (never fatal)."""
try:
return _NOTE_FILE.read_text(encoding="utf-8").strip()
except OSError:
_log.warning("%s missing; browser note omitted", _NOTE_FILE.name)
return ""
async def probe_browser(environment) -> bool:
"""True when this image ships the `pw` wrapper. Best-effort: a failed probe means no
note, never a failed run."""
try:
result = await environment.exec(command=_PROBE, timeout_sec=30)
except Exception as exc:
_log.warning("browser probe failed (%s); omitting the browser note", exc)
return False
# Exact tail match, not a substring: several harbor environments exec through a LOGIN
# shell, whose profile scripts can print to stdout. A banner containing "yes" would
# otherwise claim a browser that isn't there — the precise failure this module exists
# to prevent.
found = (getattr(result, "stdout", "") or "").strip().endswith("yes")
_log.info("browser probe: pw %s", "present" if found else "absent")
return found

View File

@@ -1,328 +0,0 @@
#!/bin/bash
# Build a task's workspace from the local repo.
#
# Usage: scripts/build-workspace.sh <task-slug> [commit]
# Example: scripts/build-workspace.sh my-cool-task 3af4366a6
#
# If commit is omitted, reads it from the task's task.toml.
set -euo pipefail
TOOLKIT_ROOT="$(cd "$(dirname "$0")/.." && pwd)"
REPO_DIR="$TOOLKIT_ROOT/repo"
TASK_SLUG="$1"
TASK_DIR="$TOOLKIT_ROOT/harbor-tasks/$TASK_SLUG"
if [ ! -d "$TASK_DIR" ]; then
echo "Error: task directory not found at $TASK_DIR" >&2
exit 1
fi
# The member this task targets, per task.toml ([metadata].repo). Used to resolve both
# the source repo (polyglot) and the member's deterministic checks (below).
# `|| true` is load-bearing: a task.toml with no `repo =` line is perfectly valid
# (single-repo tasks don't need one), but under `set -o pipefail` grep's exit 1
# propagates out of the pipeline and `set -e` would kill the script here.
MEMBER=""
if [ -f "$TASK_DIR/task.toml" ]; then
MEMBER=$(grep -E '^repo[[:space:]]*=' "$TASK_DIR/task.toml" | head -1 | sed -E 's/.*=[[:space:]]*"?([^"]+)"?.*/\1/' || true)
fi
# Single-repo toolkits keep the repo at $ROOT/repo; a polyglot toolkit keeps each member
# at $ROOT/repos/<member>. If the single-repo path is absent, use the member from task.toml
# so a graded task builds against the right member repo.
if [ ! -d "$REPO_DIR/.git" ] && [ -n "$MEMBER" ]; then
if [ -d "$TOOLKIT_ROOT/repos/$MEMBER/.git" ]; then
REPO_DIR="$TOOLKIT_ROOT/repos/$MEMBER"
fi
fi
if [ ! -d "$REPO_DIR/.git" ]; then
echo "Error: repo not found at $REPO_DIR" >&2
exit 1
fi
# Get commit from arg or task.toml
if [ -n "${2:-}" ]; then
COMMIT="$2"
else
# `|| true` for the same reason as MEMBER above: without it, pipefail turns a
# task.toml with no `commit` line into a bare `set -e` abort, and the explicit
# error below never gets a chance to print.
COMMIT=$(grep 'commit' "$TASK_DIR/task.toml" | head -1 | sed 's/.*"\(.*\)".*/\1/' || true)
if [ -z "$COMMIT" ]; then
echo "Error: no commit specified and could not read from task.toml" >&2
exit 1
fi
fi
WORKSPACE="$TASK_DIR/environment/workspace"
echo "Building workspace for $TASK_SLUG"
echo " Commit: $COMMIT"
# `browser = true` in task.toml gives the trial Playwright + Chromium. The build has no way
# to read task.toml — a Dockerfile can only see its build context — so the answer is written
# here as a file the Dockerfile COPYs.
#
# ALWAYS write it, including the "0" case: the COPY is unconditional, and a missing source
# fails the build. Accepts `true` and `"true"`, since the quoted form is a plausible hand-edit
# and rejecting it would silently give a task no browser after its author asked for one.
BROWSER_OPTIN=0
if [ -f "$TASK_DIR/task.toml" ] &&
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
BROWSER_OPTIN=1
fi
mkdir -p "$TASK_DIR/environment"
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin"
# --- DNS jail script staging -------------------------------------------------
# The Dockerfile COPYs this directory, so it must always exist (same rule as the marker
# above: a missing COPY source fails the build). The script itself is optional -- without it
# the image installs no resolver and trials simply run with normal network access.
mkdir -p "$TASK_DIR/environment/dns-jail"
if [ -f "$TOOLKIT_ROOT/task-shared/dns-jail-container.sh" ]; then
cp "$TOOLKIT_ROOT/task-shared/dns-jail-container.sh" \
"$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
fi
# Not every member's image ships a browser, and the Explore container has one either way — so
# a task can ask for a browser it will not get. Say so here rather than let it pass silently.
if [ "$BROWSER_OPTIN" = "1" ]; then
if grep -q "COPY browser-optin" "$TASK_DIR/environment/Dockerfile" 2>/dev/null; then
echo " Browser: Playwright + Chromium (browser = true)"
else
echo " WARNING: browser = true, but this task's Dockerfile has no browser. The agent" >&2
echo " will get the Read tool and no Chromium. Either drop the flag, or use a" >&2
echo " member whose image ships one:" >&2
echo " grep -l 'COPY browser-optin' task-shared/Dockerfile.*" >&2
fi
fi
# Resolve commit. Goes through resolve_pin so a task pinned before a history
# rewrite still builds: the pin is translated via task-shared/commit-maps/.
# shellcheck source=lib/resolve-pin.sh
. "$(dirname "$0")/lib/resolve-pin.sh"
RESOLVED_SHA=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
echo " Resolved SHA: $RESOLVED_SHA"
# --- Member-specific setup ---------------------------------------------------
#
# A polyglot _task-scaffold can't know which member a task targets, so anything
# member-specific is resolved here instead of left to the author to remember. This is
# the one step every task runs on both the manual and snapshot paths, and task.toml
# already tells us the member. Both actions below are idempotent and never clobber
# authored content, so re-running is always safe.
SHARED_DIR="$TOOLKIT_ROOT/task-shared"
MEMBER_LC=$(echo "${MEMBER:-}" | tr '[:upper:]' '[:lower:]')
# 1. Base image. Replace the placeholder Dockerfile with the member's real base. Guarded
# on the placeholder marker so an authored Dockerfile is never touched — snapshot tasks
# append session staging to theirs, and any task may be customized by hand. The marker
# must match POLYGLOT_SCAFFOLD_DOCKERFILE in package-worker-toolkit.ts; a packaging test
# asserts the two agree so this can't silently stop matching.
TASK_DOCKERFILE="$TASK_DIR/environment/Dockerfile"
if [ -f "$TASK_DOCKERFILE" ] && grep -q 'POLYGLOT TOOLKIT' "$TASK_DOCKERFILE"; then
if [ -n "$MEMBER_LC" ] && [ -f "$SHARED_DIR/Dockerfile.$MEMBER_LC" ]; then
cp "$SHARED_DIR/Dockerfile.$MEMBER_LC" "$TASK_DOCKERFILE"
echo " Set base image: environment/Dockerfile (from Dockerfile.$MEMBER_LC)"
else
echo " WARN: environment/Dockerfile is still the scaffold placeholder and no" >&2
echo " task-shared/Dockerfile.${MEMBER_LC:-<member>} exists to replace it with." >&2
echo " Set [metadata].repo in task.toml to your member, then re-run this script." >&2
echo " Members: $(cd "$SHARED_DIR" 2>/dev/null && ls Dockerfile.* 2>/dev/null | sed 's/Dockerfile\.//' | tr '\n' ' ')" >&2
fi
fi
# 2. Deterministic checks (tests/typecheck/lint). tests/test.sh sources this file and
# hands its output to the grader as evidence for the CORRECTNESS score, so a task without
# it gets a correctness score judged from the code alone — no test signal behind it. The
# absent-only guard leaves an existing file untouched (a single-repo scaffold ships one).
if [ ! -f "$TASK_DIR/tests/test-commands.sh" ]; then
CHECKS_SRC=""
if [ -n "$MEMBER_LC" ] && [ -f "$SHARED_DIR/test-commands.$MEMBER_LC.sh" ]; then
CHECKS_SRC="$SHARED_DIR/test-commands.$MEMBER_LC.sh"
elif [ -f "$SHARED_DIR/test-commands.sh" ]; then
CHECKS_SRC="$SHARED_DIR/test-commands.sh"
fi
if [ -n "$CHECKS_SRC" ]; then
mkdir -p "$TASK_DIR/tests"
cp "$CHECKS_SRC" "$TASK_DIR/tests/test-commands.sh"
chmod +x "$TASK_DIR/tests/test-commands.sh"
echo " Staged deterministic checks: tests/test-commands.sh (from $(basename "$CHECKS_SRC"))"
else
# Say it out loud. Absence is legitimate for members with no runnable checks, but
# silence is indistinguishable from a mistake — and it changes how the correctness
# score is arrived at, so the author should know either way.
echo " NOTE: no deterministic checks available for ${MEMBER:-this repo} — the grader will"
echo " score correctness from the code alone, with no test/typecheck/lint signal."
fi
fi
# Clean and recreate
rm -rf "$WORKSPACE"
mkdir -p "$WORKSPACE"
# Export repo at target commit (no git history).
# --no-same-owner: `git archive` stamps every entry as uid/gid 0, so GNU tar
# running as (container) root tries to chown files back to 0/0. On nested /
# rootless / Sysbox runtimes the container "root" is a userns-mapped uid with no
# CAP_CHOWN, so that chown fails with EPERM. --no-same-owner skips the restore
# (files are owned by the extracting user) — a no-op for real root and for
# non-root extraction, and the fix for the mapped-root case.
git -C "$REPO_DIR" archive "$RESOLVED_SHA" | tar -x --no-same-owner -C "$WORKSPACE"
# Apply workspace patch if one exists
PATCH_FILE="$TASK_DIR/environment/workspace.patch"
if [ -f "$PATCH_FILE" ]; then
echo " Applying workspace.patch..."
cd "$WORKSPACE"
# gc.auto=0 / maintenance.auto=false / gc.autoDetach=false prevent git
# from launching background processes (gc, commit-graph, fsmonitor) that
# can write into .git/objects after the foreground command returns. If
# such a write races with the `rm -rf .git` below, rmdir trips on
# "Directory not empty" and the build fails non-deterministically.
GIT_FLAGS=(-c gc.auto=0 -c gc.autoDetach=false -c maintenance.auto=false)
git "${GIT_FLAGS[@]}" init --quiet
git "${GIT_FLAGS[@]}" add -A
# Inject identity inline so this works on containers without a global
# git config (e.g., native Linux Docker, fresh container images).
# The .git directory is deleted on the next line, so these values are
# throwaway and never reach the patch, the workspace, or the agent.
git "${GIT_FLAGS[@]}" -c user.email=toolkit@local -c user.name=Toolkit commit -m "base" --quiet
git "${GIT_FLAGS[@]}" apply "$PATCH_FILE"
# Belt-and-suspenders: retry rm a few times in case anything still races.
for _ in 1 2 3; do
if rm -rf .git 2>/dev/null; then
break
fi
sleep 0.5
done
# Final attempt without swallowing errors, so a genuine failure surfaces.
if [ -d .git ]; then
rm -rf .git
fi
cd "$TOOLKIT_ROOT"
echo " Patch applied."
fi
# Bundle transitive poetry sibling deps. Some polyglot Python members poetry-depend on
# sibling repos via `ssh://git@github.com/AskZeta/<name>`, which can't resolve in a single-member
# harbor image (no SSH key / network). Archive the transitive closure into workspace/.zeta-siblings/<name>/
# from the toolkit's repos/zeta-<name>/ (members are packaged under their display name zeta-<name>);
# the generated Dockerfile rewrites those git deps to
# these local paths before `poetry install`. No-op for members without such deps.
if [ -f "$WORKSPACE/pyproject.toml" ]; then
SIB_DIR="$WORKSPACE/.zeta-siblings"
queue=("$WORKSPACE/pyproject.toml")
seen=" "
while [ "${#queue[@]}" -gt 0 ]; do
pp="${queue[0]}"; queue=("${queue[@]:1}")
[ -f "$pp" ] || continue
for name in $(grep -oE 'ssh://git@github\.com/AskZeta/[A-Za-z0-9._-]+' "$pp" 2>/dev/null | sed -E 's#.*/AskZeta/##; s#\.git$##' | sort -u); do
case "$seen" in *" $name "*) continue ;; esac
seen="$seen$name "
sib="$TOOLKIT_ROOT/repos/zeta-$name"
[ -e "$sib/.git" ] || { echo " WARN: sibling repo not found: $name" >&2; continue; }
mkdir -p "$SIB_DIR/$name"
git -C "$sib" archive HEAD | tar -x --no-same-owner -C "$SIB_DIR/$name"
queue+=("$SIB_DIR/$name/pyproject.toml")
done
done
[ -d "$SIB_DIR" ] && echo " Bundled siblings:$(printf '%s' "${seen# }" | sed 's/ $//' | sed 's/^/ /')"
fi
# Bundle Maven sibling libs. The swingbell-polyglot Java services depend on sibling shared
# artifacts (com.swingbell*: common-repository, common-aws-service, jasper-report from
# `reports`, jwt-encryption-decryption) at 0.0.1-SNAPSHOT — resolvable only from a local
# reactor install, never a registry. Walk the dependency closure (a bundled provider's own
# pom can name further siblings — `reports` needs common-repository) into
# workspace/.sbl-siblings/<name>/ with an ORDER file in install order (commons before
# consumers); the generated java Dockerfile `mvn install`s them into ~/.m2 before building
# the member. No-op without a pom or refs.
#
# Twin forks: the book-my-minutes-* repos publish the SAME coordinates as the swingbell
# commons (com.swingbell.common:common-repository:0.0.1-SNAPSHOT etc. — the twin naming is
# repo-level only, invisible to Maven), so an artifactId resolves to the provider from the
# member's own family.
if [ -f "$WORKSPACE/pom.xml" ] && grep -q 'com\.swingbell' "$WORKSPACE/pom.xml" 2>/dev/null; then
SBL_DIR="$WORKSPACE/.sbl-siblings"
case "$MEMBER" in book-my-minutes-*) SBL_TWIN=book-my-minutes- ;; *) SBL_TWIN= ;; esac
queue=("$WORKSPACE/pom.xml")
seen=" "
while [ "${#queue[@]}" -gt 0 ]; do
pom="${queue[0]}"; queue=("${queue[@]:1}")
[ -f "$pom" ] || continue
for artifact in $(grep -oE '<artifactId>(common-repository|common-aws-service|jasper-report|jwt-encryption-decryption)</artifactId>' "$pom" 2>/dev/null | sed -E 's#</?artifactId>##g' | sort -u); do
case "$artifact" in
common-repository|common-aws-service) provider="$SBL_TWIN$artifact" ;;
jasper-report) provider=reports ;;
jwt-encryption-decryption) provider=jwt-encryption-decryption ;;
esac
case "$seen" in *" $provider "*) continue ;; esac
# never bundle the member into itself: the pom's OWN <artifactId> declaration
# matches the grep above just like a dependency would ($MEMBER is the task repo)
[ "$provider" = "$MEMBER" ] && continue
seen="$seen$provider "
sib="$TOOLKIT_ROOT/repos/$provider"
[ -e "$sib/.git" ] || { echo " WARN: maven sibling repo not found: $provider" >&2; continue; }
mkdir -p "$SBL_DIR/$provider"
git -C "$sib" archive HEAD | tar -x --no-same-owner -C "$SBL_DIR/$provider"
queue+=("$SBL_DIR/$provider/pom.xml")
done
done
# ORDER = canonical install order (providers before their consumers), filtered to the
# closure just bundled — discovery order is consumer-first, which is backwards for install.
for provider in "${SBL_TWIN}common-repository" "${SBL_TWIN}common-aws-service" jwt-encryption-decryption reports; do
case "$seen" in *" $provider "*) echo "$provider" >> "$SBL_DIR/ORDER" ;; esac
done
[ -f "$SBL_DIR/ORDER" ] && echo " Bundled maven siblings: $(tr '\n' ' ' < "$SBL_DIR/ORDER")"
fi
# --- Reference-data corpus: mounted at /data/zeta-corpus in the trial -----------------------
# If this toolkit ships the supplementary data corpus, it's included in every trial — staged into
# the build context + a COPY added to the Dockerfile, so what you see while authoring (bind-mounted
# at /data/zeta-corpus) is exactly what the trial sees. Toolkits without a corpus never include it.
CORPUS_SRC=""
for cand in "${ZETA_CORPUS_DIR:-}" "$TOOLKIT_ROOT/data/zeta-corpus" "/data/zeta-corpus"; do
[ -n "$cand" ] && [ -d "$cand" ] && { CORPUS_SRC="$cand"; break; }
done
if [ -n "$CORPUS_SRC" ]; then
CORPUS_STAGE="$TASK_DIR/environment/corpus"
rm -rf "$CORPUS_STAGE"
# hardlink-stage (cp -al ~free, same filesystem as the toolkit); full copy fallback.
cp -al "$CORPUS_SRC/." "$CORPUS_STAGE" 2>/dev/null || cp -a "$CORPUS_SRC/." "$CORPUS_STAGE"
DF="$TASK_DIR/environment/Dockerfile"
# Wrapped in toolkit-managed sentinels so scripts/check-task-infra.ts can tell
# this append apart from an author's edit — see scripts/lib/task-infra-integrity.ts.
if [ -f "$DF" ] && ! grep -qF 'COPY corpus/ /data/zeta-corpus' "$DF"; then
{ echo ""; echo "# >>> toolkit-managed: corpus >>>"; \
echo "# Reference-data corpus at /data/zeta-corpus (staged by build-workspace)."; \
echo "COPY corpus/ /data/zeta-corpus/"; \
echo "# <<< toolkit-managed <<<"; } >> "$DF"
fi
if [ -f "$TASK_DIR/task.toml" ]; then
CUR=$(grep -oE '^[[:space:]]*storage_mb[[:space:]]*=[[:space:]]*[0-9]+' "$TASK_DIR/task.toml" | grep -oE '[0-9]+' | head -1 || echo 0)
# 10240 = the sandbox disk ceiling (a higher request is rejected downstream).
if [ "${CUR:-0}" -lt 10240 ] && grep -qE '^[[:space:]]*storage_mb[[:space:]]*=' "$TASK_DIR/task.toml"; then
sed -i.bak -E 's/^([[:space:]]*storage_mb[[:space:]]*=[[:space:]]*)[0-9]+/\110240/' "$TASK_DIR/task.toml"
rm -f "$TASK_DIR/task.toml.bak"
fi
fi
echo " Corpus: staged from $CORPUS_SRC -> environment/corpus + Dockerfile COPY (storage_mb>=10240)"
fi
FILE_COUNT=$(find "$WORKSPACE" -type f | wc -l | tr -d ' ')
echo " Workspace: $WORKSPACE ($FILE_COUNT files)"
# Toolkit-managed files. Stamp them if they aren't already (tasks copied from
# _task-scaffold arrive stamped; this covers the ones built by snapshot-to-task), then
# report. Advisory only — this script writes to the Dockerfile itself, so it never
# blocks; harbor-run and submit-task do.
CHECK_INFRA="$TOOLKIT_ROOT/scripts/check-task-infra.ts"
if [ -f "$CHECK_INFRA" ]; then
(cd "$TOOLKIT_ROOT" && npx tsx "$CHECK_INFRA" --stamp "$TASK_SLUG") || true
(cd "$TOOLKIT_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_SLUG") || true
fi
echo "Done."

View File

@@ -1,138 +0,0 @@
/**
* check-task-infra.ts — report edits to toolkit-managed files.
*
* Called by `scripts/harbor-run` before a trial and by `scripts/submit-task.ts`
* before packaging, so an accidental edit to the trial Dockerfile, the grader
* orchestration, or the grader system prompt surfaces at the moment it matters
* rather than after a submission is reviewed.
*
* Covers two sets: the task's own managed files (environment/Dockerfile,
* tests/test.sh, the grader system prompts) and the toolkit's `scripts/` tree,
* which is checked once per invocation regardless of which task was named.
*
* Always exits 0. Both checks are advisory — see the notes on IntegrityStatus
* and formatIntegrityReport.
*
* Usage:
* npx tsx scripts/check-task-infra.ts <task-slug-or-dir>
* npx tsx scripts/check-task-infra.ts my-task --json
*/
import { existsSync } from 'fs';
import { basename, isAbsolute, join, resolve } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import {
bannerize,
checkTaskInfraIntegrity,
formatIntegrityReport,
writeManagedStamp,
} from './lib/task-infra-integrity.js';
import {
checkToolkitScriptIntegrity,
scriptIntegrityNotice,
} from './lib/toolkit-script-integrity.js';
const argv = yargs(hideBin(process.argv))
.usage('Usage: $0 <task> [options]')
.positional('task', { type: 'string', describe: 'Task slug, or a path to harbor-tasks/<slug>' })
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.option('stamp', {
type: 'boolean',
default: false,
describe:
'Record the managed files as created, so later edits are detectable. No-op if already stamped.',
})
.demandCommand(1, 'Provide a task slug or directory')
.help()
.parseSync();
const log = pino(
{ name: 'check-task-infra', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
const arg = String(argv._[0]);
const toolkitRoot = process.cwd();
// Accept both a bare slug and a path, since harbor-run is invoked with a path
// (`scripts/harbor-run harbor-tasks/<slug>`) and submit-task with a slug.
const taskDir = isAbsolute(arg)
? arg
: existsSync(resolve(toolkitRoot, arg))
? resolve(toolkitRoot, arg)
: join(toolkitRoot, 'harbor-tasks', arg);
if (!existsSync(taskDir)) {
log.fatal({ taskDir }, 'Task directory not found');
process.exit(1);
}
const slug = basename(taskDir);
// --stamp records a task's baseline. Runs at task creation; never overwrites.
if (argv.stamp) {
if (!existsSync(join(toolkitRoot, 'task-shared'))) {
log.debug('Not a worker toolkit (no task-shared/); nothing to stamp');
process.exit(0);
}
const wrote = writeManagedStamp(taskDir, toolkitRoot);
log.debug({ slug, wrote }, wrote ? 'Stamped toolkit-managed files' : 'Already stamped');
process.exit(0);
}
// Toolkit scripts, not this task's files: checked here because this is already the
// preflight both `harbor-run` and `submit-task.ts` reach. An edited script ships no
// trace of itself, only its output — see toolkit-script-integrity.ts.
const scripts = checkToolkitScriptIntegrity(toolkitRoot);
if (scripts.checked) {
const notice = scriptIntegrityNotice(scripts);
if (notice) {
log.warn(
{
edited: scripts.modified.map((f) => f.path),
missing: scripts.missing.map((f) => f.path),
},
'Toolkit scripts need a look'
);
process.stderr.write(`\n${notice}\n\n`);
} else {
log.info({ files: scripts.files.length }, 'Toolkit scripts are unmodified');
}
}
const report = checkTaskInfraIntegrity(taskDir, toolkitRoot);
if (!report.checked) {
log.debug(
'Managed-file check skipped: no task-shared/ here, or the task was authored on a different toolkit generation (its tests/ assets are its own)'
);
process.exit(0);
}
const message = formatIntegrityReport(report);
if (!message) {
log.info({ files: report.files.length }, 'Toolkit-managed files are unmodified');
process.exit(0);
}
// Advisory, always. Exiting non-zero here is what used to let a false positive stop
// an author's trial with no way out; the report is the whole product.
log.warn(
{
edited: report.modified.map((f) => f.taskPath),
outdated: report.outdated.map((f) => f.taskPath),
unverifiable: report.unverifiable.map((f) => f.taskPath),
},
'Toolkit-managed files need a look'
);
process.stderr.write(`\n${bannerize(message, report)}\n\n`);
process.exit(0);

View File

@@ -1,295 +0,0 @@
#!/bin/bash
# Check that a task's live environment/workspace matches what a rebuild from
# the pinned commit + environment/workspace.patch would produce — i.e. the
# workspace every downstream consumer of the task actually sees. Files edited
# (or added/deleted) directly in the built workspace are visible to your local
# trials but do NOT survive packaging: your own tarball may carry them, but
# the finalized task keeps only the rebuild inputs (the workspace/ dir itself
# is gitignored), and everywhere downstream the workspace is rebuilt from the
# gitref in task.toml plus workspace.patch (see build-workspace.sh) — anything
# not captured there is silently dropped.
#
# Usage:
# bash scripts/check-workspace-sync.sh <task-dir> # check (advisory)
# bash scripts/check-workspace-sync.sh --update-patch <task-dir> # fold live edits into workspace.patch
#
# Check mode is run automatically at the start of every `scripts/harbor-run`.
# It warns loudly when the workspace has uncaptured changes, and always exits
# 0 — it never blocks a run. It also exits 0 (silently) when it can't resolve
# the source repo or the pinned commit, since it can't tell anything useful
# then.
#
# --update-patch regenerates environment/workspace.patch as the full diff from
# the pinned commit to the live workspace (the previous patch's changes are
# preserved — they're part of that diff). After updating the patch, re-run
# your trials: reference runs should be captured against the workspace every
# downstream rebuild produces.
#
# Mechanics: the pinned commit's tree is read into a THROWAWAY git index (with
# a throwaway object directory layered over the repo's, so the source repo is
# never written to), workspace.patch is applied to that index, and the live
# workspace directory is compared against it. Files matched by the repo's
# .gitignore are not considered — they can't be captured in workspace.patch
# either, so they never ship either way. File-mode-only changes are ignored
# (core.fileMode=false), matching how patches are generated here.
set -euo pipefail
MODE="check"
if [ "${1:-}" = "--update-patch" ]; then
MODE="update"
shift
fi
if [ -z "${1:-}" ]; then
echo "Usage: $0 [--update-patch] <task-dir>" >&2
exit 1
fi
# Normalize the task dir (tolerates relative paths and trailing slashes).
TASK_DIR="$(cd "$1" 2>/dev/null && pwd)" || {
echo "Error: task directory not found: $1" >&2
exit 1
}
SLUG="$(basename "$TASK_DIR")"
# Tasks live at <root>/harbor-tasks/<slug> in every layout this script ships to.
ROOT="$(cd "$TASK_DIR/../.." && pwd)"
WORKSPACE="$TASK_DIR/environment/workspace"
PATCH_FILE="$TASK_DIR/environment/workspace.patch"
TASK_TOML="$TASK_DIR/task.toml"
# How to spell this script in the recommendations we print. In a packed
# toolkit it lives at <root>/scripts/ (the worker's usual cwd is <root>), so
# the short form works; anywhere else (e.g. invoked from the internal repo
# layout via harbor-run) fall back to the invoked path.
SELF_DISPLAY="bash scripts/check-workspace-sync.sh"
if [ ! -f "$ROOT/scripts/check-workspace-sync.sh" ]; then
SELF_DISPLAY="bash $0"
fi
# In check mode every "can't verify" path exits 0 quietly: this is an advisory
# preflight and a task we can't reason about must never break a run. In
# --update-patch mode the same conditions are hard errors — the user asked for
# a patch and we can't produce one.
skip() {
if [ "$MODE" = "update" ]; then
echo "Error: $1" >&2
exit 1
fi
exit 0
}
[ -d "$WORKSPACE" ] || skip "workspace not built at $WORKSPACE (run build-workspace.sh first)"
[ -f "$TASK_TOML" ] || skip "no task.toml at $TASK_TOML"
# Pinned commit: the `commit = "..."` line in task.toml. Anchored to the line
# start so prose mentions (e.g. a `source = "... commit abc"` note) don't
# match. No commit line is legitimate for some internally-built tasks — then
# there's nothing to compare against.
COMMIT="$(grep -E '^[[:space:]]*commit[[:space:]]*=' "$TASK_TOML" | head -1 | sed 's/.*"\(.*\)".*/\1/' || true)"
[ -n "$COMMIT" ] || skip "no commit pinned in task.toml"
# The member this task targets, per task.toml ([metadata].repo) — used to
# resolve the source repo in polyglot layouts. Same extraction as
# build-workspace.sh.
MEMBER="$(grep -E '^repo[[:space:]]*=' "$TASK_TOML" | head -1 | sed -E 's/.*=[[:space:]]*"?([^"]+)"?.*/\1/' || true)"
# Source repo resolution, in order:
# <root>/repo — single-repo toolkit
# <root>/repos/<member> — polyglot toolkit
# <root>/repos/<member>/repo — internal submodule layout
REPO_DIR=""
for cand in "$ROOT/repo" ${MEMBER:+"$ROOT/repos/$MEMBER" "$ROOT/repos/$MEMBER/repo"}; do
if [ -e "$cand/.git" ]; then
REPO_DIR="$cand"
break
fi
done
[ -n "$REPO_DIR" ] || skip "source repo not found under $ROOT"
# Resolve the repo's real git dir (handles submodules, whose .git is a file).
GITDIR="$(git -C "$REPO_DIR" rev-parse --absolute-git-dir 2>/dev/null)" || skip "not a git repo: $REPO_DIR"
# Via resolve_pin, so a task pinned before a history rewrite keeps being checked
# rather than silently skipping every run once its SHA stops resolving.
# shellcheck source=lib/resolve-pin.sh
. "$(dirname "$0")/lib/resolve-pin.sh"
RESOLVED_SHA="$(resolve_pin "$REPO_DIR" "$COMMIT" "$ROOT/task-shared/commit-maps" "$MEMBER" 2>/dev/null)" || \
skip "pinned commit $COMMIT not found in $REPO_DIR"
# --- Throwaway git state ------------------------------------------------------
# A temp index + temp object dir (with the real object dir as a read-only
# alternate) lets us build "commit + patch" as an index and diff the live
# workspace against it without ever writing to the source repo or creating a
# .git inside the workspace.
TMP="$(mktemp -d)"
trap 'rm -rf "$TMP"' EXIT
export GIT_INDEX_FILE="$TMP/index"
export GIT_OBJECT_DIRECTORY="$TMP/objects"
export GIT_ALTERNATE_OBJECT_DIRECTORIES="$GITDIR/objects"
mkdir -p "$GIT_OBJECT_DIRECTORY"
# Suppress mode-bit and line-ending munging so the comparison is about content,
# and keep non-ASCII paths readable instead of C-quoted ("\360\237...").
GIT_FLAGS=(-c core.fileMode=false -c core.autocrlf=false -c core.quotePath=false)
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" read-tree "$RESOLVED_SHA"
# `.zeta-siblings/` is staged INTO the workspace by build-workspace.sh on some
# toolkits (bundled sibling deps) — a build artifact, never part of the patch.
# `.raccoon-setup-done` is run-app's per-repo first-use setup marker (polyglot
# toolkits) — authoring-machine state, never task content. run-app git-ignores
# it via the repo's .git/info/exclude, but this script diffs through a
# throwaway --git-dir that never reads that file, so exclude it here too.
# The leading `.` positive pathspec is load-bearing: several git commands
# reject a pathspec made of nothing but exclusions.
EXCLUDES=("." ":(exclude).zeta-siblings" ":(exclude).raccoon-setup-done")
cd "$WORKSPACE"
export GIT_WORK_TREE="$WORKSPACE"
if [ "$MODE" = "update" ]; then
# Stage the live workspace on top of the pinned tree, then emit the full
# tree -> index diff as the new workspace.patch. --binary --full-index so
# binary additions survive a later `git apply`.
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" add -A -- "${EXCLUDES[@]}"
NEW_PATCH="$TMP/workspace.patch"
git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --cached --binary --full-index "$RESOLVED_SHA" -- "${EXCLUDES[@]}" > "$NEW_PATCH"
if [ ! -s "$NEW_PATCH" ]; then
if [ -f "$PATCH_FILE" ]; then
rm -f "$PATCH_FILE"
echo "Workspace matches commit $COMMIT exactly — removed the now-empty environment/workspace.patch."
else
echo "Workspace matches commit $COMMIT exactly — no workspace.patch needed."
fi
exit 0
fi
# Verify the regenerated patch applies to the pristine tree before
# installing it, so we never leave behind a patch build-workspace.sh
# would choke on.
GIT_INDEX_FILE="$TMP/verify-index" git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" read-tree "$RESOLVED_SHA"
GIT_INDEX_FILE="$TMP/verify-index" git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" apply --cached --check "$NEW_PATCH" || {
echo "Error: regenerated patch does not apply cleanly to $COMMIT — workspace.patch left unchanged." >&2
exit 1
}
cp "$NEW_PATCH" "$PATCH_FILE"
FILE_COUNT="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --cached --name-only "$RESOLVED_SHA" -- "${EXCLUDES[@]}" | wc -l | tr -d ' ')"
echo "Wrote environment/workspace.patch: $FILE_COUNT file(s) differ from commit $COMMIT."
if [ -f "$TASK_DIR/environment/session.jsonl" ]; then
echo ""
echo "NOTE: this task was built from a session snapshot, so workspace.patch is"
echo "meant to mirror the workspace state the captured session describes. Make"
echo "sure these folded-in changes don't contradict the session transcript —"
echo "if they belong to the session's story, re-capturing the snapshot"
echo "(/create-snapshot:snapshot, then scripts/snapshot-to-task.ts) is the"
echo "cleaner fix."
fi
echo ""
echo "Re-run your trials so your reference runs match what now ships:"
echo " scripts/harbor-run harbor-tasks/$SLUG -k 4"
exit 0
fi
# --- Check mode ---------------------------------------------------------------
# Apply workspace.patch to the throwaway index — the index then holds exactly
# the tree build-workspace.sh would produce. A patch that no longer applies is
# its own (serious) problem: the shipped inputs can't even rebuild.
if [ -s "$PATCH_FILE" ]; then
if ! git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" apply --cached "$PATCH_FILE" 2>/dev/null; then
echo "" >&2
echo "==============================================================================" >&2
echo "!! WARNING: environment/workspace.patch does not apply to commit $COMMIT." >&2
echo "!! A rebuild of this task from its shipped inputs (build-workspace.sh)" >&2
echo "!! would FAIL, and your live workspace can't be checked against them." >&2
echo "!! Did the gitref or the patch change after the workspace was built?" >&2
echo "==============================================================================" >&2
echo "" >&2
exit 0
fi
fi
# Tracked files that differ between the index (commit + patch) and the live
# workspace, plus files that exist only in the live workspace. Both respect
# the repo's .gitignore.
DIFF_RAW="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" diff --name-status -- "${EXCLUDES[@]}")"
UNTRACKED="$(git --git-dir="$GITDIR" "${GIT_FLAGS[@]}" ls-files --others --exclude-standard -- "${EXCLUDES[@]}")"
# Deletions of symlinks are skipped: the internal workspace build prunes
# dangling symlinks after applying the patch, so their absence is expected,
# not a worker edit.
CHANGES=""
while IFS=$'\t' read -r st path; do
[ -n "$st" ] || continue
if [ "$st" = "D" ]; then
entry_mode="$(git --git-dir="$GITDIR" ls-files -s -- "$path" | awk '{print $1}')"
[ "$entry_mode" = "120000" ] && continue
fi
CHANGES="${CHANGES} ${st} ${path}
"
done <<< "$DIFF_RAW"
while IFS= read -r path; do
[ -n "$path" ] || continue
CHANGES="${CHANGES} ?? ${path}
"
done <<< "$UNTRACKED"
[ -n "$CHANGES" ] || exit 0
TOTAL="$(printf '%s' "$CHANGES" | wc -l | tr -d ' ')"
LISTED="$(printf '%s' "$CHANGES" | head -25)"
{
echo ""
echo "=============================================================================="
echo "!! WARNING: environment/workspace has changes that will NOT survive"
echo "!! packaging."
echo "=============================================================================="
echo ""
echo "The workspace/ directory itself is never kept: everywhere downstream the"
echo "task is rebuilt from the commit pinned in task.toml ($COMMIT) plus"
echo "environment/workspace.patch — exactly what scripts/build-workspace.sh"
echo "produces. These $TOTAL file(s) differ from that rebuild, so your local trials"
echo "see them, but they will not survive packaging:"
echo ""
echo "$LISTED"
if [ "$TOTAL" -gt 25 ]; then
echo " ... and $((TOTAL - 25)) more"
fi
echo ""
echo " (M = modified, D = deleted, ?? = only in the live workspace. Files matched"
echo " by the repo's .gitignore are not checked — they never ship either way.)"
echo ""
if [ -f "$TASK_DIR/environment/session.jsonl" ]; then
echo "This task was built from a session snapshot, and workspace.patch mirrors"
echo "the workspace state captured with that session. If these changes belong in"
echo "the task, the cleanest fix is to make them in the Explore session and"
echo "re-capture (/create-snapshot:snapshot, then scripts/snapshot-to-task.ts),"
echo "so the session transcript and the workspace stay consistent."
echo ""
echo "To fold them into workspace.patch anyway — only if they don't contradict"
echo "what the captured session says about the workspace:"
echo ""
echo " $SELF_DISPLAY --update-patch harbor-tasks/$SLUG"
else
echo "To fold these changes into workspace.patch so they ship with the task:"
echo ""
echo " $SELF_DISPLAY --update-patch harbor-tasks/$SLUG"
if [ -f "$ROOT/scripts/build-workspace.sh" ]; then
echo ""
echo "To discard them instead (rebuild the workspace from commit + patch):"
echo ""
echo " bash scripts/build-workspace.sh $SLUG"
fi
fi
echo ""
echo "Either way, re-run your trials afterwards so your reference runs match the"
echo "workspace every downstream rebuild produces."
echo "=============================================================================="
echo ""
} >&2
exit 0

File diff suppressed because one or more lines are too long

View File

@@ -1,615 +0,0 @@
"""Custom Codex agents for our devcontainer-based task images.
Harbor's stock Codex agent (`harbor.agents.installed.codex.Codex`) installs Node
via nvm into `$HOME/.nvm` during `install()`, and `setup()` ALWAYS calls
`install()` (the version probe only runs afterward). That install fails on our
task images: they are `FROM mcr.microsoft.com/devcontainers/typescript-node:20`,
which provides Node through the devcontainer nvm at `/usr/local/share/nvm`, and
the tasks run as root, where that nvm isn't auto-loaded — so harbor's
`$HOME/.nvm/nvm.sh` doesn't exist and the agent dies with "NVM failed to load".
`SystemNodeCodex` overrides `install()` to load the image's existing Node and
install only the codex CLI (no second Node via nvm). Everything else — the
trajectory parsing, the codex exec, reasoning_effort kwargs — is inherited
unchanged from the stock agent.
Use via: `--agent-import-path codex_agent:SystemNodeCodex` (PYTHONPATH=scripts).
"""
from __future__ import annotations
import json
import os
import shlex
import sys
import tempfile
import tomllib
import uuid
from pathlib import Path
import atif_session
import browser_note
try:
from dnsjail import apply_dns_jail
except ImportError: # no helper shipped -> no jail, rather than no trials
async def apply_dns_jail(agent, environment) -> None: # type: ignore[misc]
return None
from harbor.agents.installed.codex import Codex
from harbor.models.trial.paths import EnvironmentPaths
sys.path.insert(0, str(Path(__file__).resolve().parent / "lib"))
import codex_auth # noqa: E402
from harness_registry import load_harness_registry # noqa: E402
# Installs the codex CLI into /usr/local/bin so harbor's plain-sh execs find it.
#
# Primary path is the official standalone installer, which fetches a prebuilt
# native binary and needs only curl + tar — no Node in the image. That matters
# because most task images (Ruby/Python) ship no Node at all, and the npm route
# below can only run on the Node-bearing minority.
#
# The npm route is kept as a fallback for images where the installer can't run
# (e.g. a native binary the image's glibc rejects) but a usable npm exists.
_INSTALL_CMD = (
"set -x; "
"if command -v apt-get >/dev/null 2>&1; then "
" apt-get update -qq >/dev/null 2>&1 && apt-get install -y -qq curl ripgrep >/dev/null 2>&1 || true; "
"fi; "
'if ! command -v codex >/dev/null 2>&1; then '
' CODEX_INSTALL_DIR=/usr/local/bin CODEX_NON_INTERACTIVE=true '
' sh -c "curl -fsSL https://chatgpt.com/codex/install.sh | sh" >&2 || true; '
"fi; "
'if ! command -v codex >/dev/null 2>&1 && [ -x "$HOME/.local/bin/codex" ]; then '
' ln -sf "$HOME/.local/bin/codex" /usr/local/bin/codex; '
"fi; "
# npm fallback: load the devcontainer nvm, else find npm anywhere plausible.
'if ! command -v codex >/dev/null 2>&1; then '
' export NVM_DIR="${NVM_DIR:-/usr/local/share/nvm}"; '
' [ -s "$NVM_DIR/nvm.sh" ] && . "$NVM_DIR/nvm.sh" >/dev/null 2>&1 || true; '
' if ! command -v npm >/dev/null 2>&1; then '
' npm_path="$(find /usr/local/share/nvm /usr/local /usr/lib /opt -name npm -type f 2>/dev/null | head -1)"; '
' [ -n "$npm_path" ] && export PATH="$PATH:$(dirname "$npm_path")"; '
" fi; "
' command -v npm >/dev/null 2>&1 && npm install -g @openai/codex@latest; '
"fi; "
'for bin in node codex; do '
' p="$(command -v "$bin" 2>/dev/null || true)"; '
' [ -n "$p" ] && [ "$p" != "/usr/local/bin/$bin" ] && ln -sf "$p" "/usr/local/bin/$bin" || true; '
"done; "
'command -v codex >/dev/null 2>&1 '
' || { echo "FATAL: codex CLI unavailable (standalone installer and npm both failed)" >&2; exit 1; }; '
"codex --version"
)
# codex reserves its built-in provider ids, so the call-origin header needs a provider of
# our own. A trial's origin is a constant, so it is a static http_headers literal here
# rather than the env-var indirection the containers need — nothing to plumb into a
# sandbox, and no way for a missing var to lose the attribution.
PROXY_PROVIDER_TOML = """\
model_provider = "llm-proxy"
[model_providers.llm-proxy]
name = "LLM proxy"
base_url = "${OPENAI_BASE_URL}"
env_key = "OPENAI_API_KEY"
wire_api = "responses"
http_headers = { "X-Surge-Client-Metadata" = '{"origin":"harbor-trial"}' }
"""
# A custom provider reads its key from env_key and never from auth.json, so the proxy
# path must carry this even when an auth file was uploaded.
PROXY_KEY_VAR = "OPENAI_API_KEY"
def proxy_provider_config(openai_base_url: str) -> dict:
"""PROXY_PROVIDER_TOML as harbor's config dict, pointed at this trial's URL."""
return tomllib.loads(PROXY_PROVIDER_TOML.replace("${OPENAI_BASE_URL}", openai_base_url))
class SystemNodeCodex(Codex):
# Set by install()'s probe, read by build_cli_flags(). Mirrors the claude adapter.
_has_browser = False
# Non-snapshot codex tasks run this class directly; the native-snapshot resume path
# overrides run() and applies the jail itself.
async def run(self, instruction, environment, context): # type: ignore[override]
self._refuse_shell_hostile_key()
await apply_dns_jail(self, environment)
await super().run(instruction, environment, context)
def _auth_json_setup(self, remote_auth_path: str) -> tuple[dict[str, str], str]:
return codex_auth.auth_json_setup(
self._get_env("OPENAI_API_KEY") or "", remote_auth_path
)
def _proxy_provider_flags(self) -> str:
"""`-c` overrides putting the trial on our own provider — the only place codex
can be told to send the call-origin header.
On the command line rather than in config.toml because harbor writes that file
itself, differently per version (0.20 hardcodes the block inline), while these
flags are ours in every version.
"""
base_url = self._get_env("OPENAI_BASE_URL") or ""
if "/llm_proxy/" not in base_url:
return ""
# A custom provider ignores auth.json, so leave that flow on the built-in
# provider: losing attribution beats breaking the run's auth.
if self._resolve_auth_json_path():
return ""
config = proxy_provider_config(base_url)
provider_id = config["model_provider"]
parts = [f"-c model_provider={provider_id}"]
for key, value in config["model_providers"][provider_id].items():
for path, leaf in (
[(f"{key}.{k}", v) for k, v in value.items()]
if isinstance(value, dict)
else [(key, value)]
):
# A TOML literal string, since the header value is JSON and carries its
# own double quotes.
quoted = f"'{leaf}'" if '"' in leaf else f'"{leaf}"'
parts.append(
"-c "
+ shlex.quote(f"model_providers.{provider_id}.{path}={quoted}")
)
return " ".join(parts)
def _refuse_shell_hostile_key(self) -> None:
"""Harbor's own Codex.run interpolates the key into a heredoc, so a key it cannot
escape would 401 with no stated cause. Refuse up front instead."""
if self._resolve_auth_json_path():
return
bad = codex_auth.unescapable_chars(self._get_env("OPENAI_API_KEY") or "")
if bad:
raise ValueError(
"OPENAI_API_KEY contains "
+ ", ".join(repr(c) for c in bad)
+ ", which harbor's stock auth.json writer cannot escape. Point "
"CODEX_AUTH_JSON_PATH at a pre-written auth.json instead."
)
async def install(self, environment) -> None: # type: ignore[override]
await self.exec_as_root(environment, command=_INSTALL_CMD)
self._has_browser = await browser_note.probe_browser(environment)
def build_cli_flags(self) -> str: # type: ignore[override]
"""Harbor's flags plus the registry's `agent_config`, so a trial's toolset
matches the explore launcher's — which passes the same rendering as
$RACCOON_AGENT_FLAGS. Both run paths go through here."""
flags = super().build_cli_flags()
reductions = load_harness_registry().require("codex").agent_config_flags()
if reductions:
flags = f"{flags} {reductions}".strip()
# Both run paths go through here, so this is where the trial picks up the
# provider that carries the call-origin header.
provider = self._proxy_provider_flags()
if provider:
flags = f"{flags} {provider}".strip()
return f"{flags} {self._browser_flag()}".strip() if self._browser_flag() else flags
def _browser_flag(self) -> str:
"""Disclose the browser to codex the way codex takes extra instructions.
`developer_instructions` PREPENDS a developer message and leaves codex's own base
instructions in place — verified with `codex debug prompt-input`. That makes it the
equivalent of claude's --append-system-prompt. `model_instructions_file`, the other
instruction-shaped key, REPLACES the base instructions; do not use it here.
"""
if not self._has_browser:
return ""
note = browser_note.browser_note()
if not note:
return ""
return f"-c developer_instructions={shlex.quote(note)}"
# Where harbor's run-prep stages the prior Claude Code session for snapshot tasks
# (the same file our snapshot_agent reads). Agent-agnostic, so codex sees it too.
_STAGED_SESSION = "/tmp/snapshot-session/session.jsonl"
def render_claude_session(jsonl_text: str, max_block: int = 4000) -> str:
"""Render a Claude Code session JSONL transcript into readable plain text so a
non-Claude agent (codex) can be handed the prior conversation as context.
Each line is a Claude record: {"type": "user"|"assistant", "message": {"role",
"content"}}. `content` is either a string or a list of blocks
(text / tool_use / tool_result / thinking). We flatten to labeled turns and
truncate oversized tool payloads so the context stays bounded."""
out: list[str] = []
def clip(s: str) -> str:
s = s.rstrip()
return s if len(s) <= max_block else s[:max_block] + "\n…[truncated]"
for line in jsonl_text.splitlines():
line = line.strip()
if not line:
continue
try:
rec = json.loads(line)
except (json.JSONDecodeError, ValueError):
continue
rtype = rec.get("type")
msg = rec.get("message") or {}
role = msg.get("role") or rtype
content = msg.get("content")
if content is None:
# non-message records (summaries, etc.) — skip unless they carry text
txt = rec.get("summary") or rec.get("content")
if isinstance(txt, str) and txt.strip():
out.append(f"[{rtype}] {clip(txt)}")
continue
if isinstance(content, str):
out.append(f"{role.upper()}: {clip(content)}")
continue
# content is a list of blocks
for block in content:
if not isinstance(block, dict):
out.append(f"{role.upper()}: {clip(str(block))}")
continue
btype = block.get("type")
if btype == "text":
out.append(f"{role.upper()}: {clip(block.get('text', ''))}")
elif btype == "thinking":
out.append(f"{role.upper()} (thinking): {clip(block.get('thinking', ''))}")
elif btype == "tool_use":
name = block.get("name", "?")
inp = json.dumps(block.get("input", {}), ensure_ascii=False)
out.append(f"{role.upper()} [tool_use {name}]: {clip(inp)}")
elif btype == "tool_result":
res = block.get("content")
if isinstance(res, list):
res = "".join(
b.get("text", "") for b in res if isinstance(b, dict)
)
out.append(f"[tool_result]: {clip(str(res))}")
return "\n".join(out)
_INLINE_PREAMBLE = (
"You are continuing an in-progress pair-programming session. Below is the FULL "
"prior conversation between the user and the previous assistant (you), including "
"the tool calls that assistant made and their results. Treat it as your own prior "
"context — the workspace already reflects any edits made in it. Then respond to the "
"user's newest message at the end.\n\n"
"================ PRIOR CONVERSATION ================\n"
)
class InlineSnapshotCodex(SystemNodeCodex):
"""Bridge A: run codex on snapshot tasks by INLINING the prior Claude session as
plain-text context ahead of the user's next-turn instruction. Works for any
provider — codex just sees a long prompt: [rendered prior conversation] + [the
user's newest message]. For non-snapshot tasks (no staged session) it behaves
exactly like the stock codex agent."""
async def run(self, instruction, environment, context): # type: ignore[override]
session_text = ""
try:
result = await environment.exec(
command=f"cat {_STAGED_SESSION} 2>/dev/null || true"
)
session_text = (getattr(result, "stdout", "") or "").strip()
except Exception as exc: # best-effort; fall back to bare instruction
self.logger.warning("InlineSnapshotCodex: could not read session: %s", exc)
if session_text:
rendered = render_claude_session(session_text)
if rendered.strip():
instruction = (
_INLINE_PREAMBLE
+ rendered
+ "\n\n================ USER'S NEWEST MESSAGE ================\n"
+ instruction
)
self.logger.info(
"InlineSnapshotCodex: injected %d chars of rendered prior session",
len(rendered),
)
else:
self.logger.warning("InlineSnapshotCodex: session rendered empty")
else:
self.logger.info(
"InlineSnapshotCodex: no staged session (non-snapshot task or empty); "
"running bare instruction"
)
await super().run(instruction, environment, context)
# ---------------------------------------------------------------------------
# Bridge B: native codex resume.
#
# Instead of inlining the whole prior Claude session into one giant prompt
# (Bridge A, which makes codex stall on a ~50k-token blob), we translate the
# staged session into codex's OWN rollout JSONL format, drop it into
# $CODEX_HOME/sessions/<date>/rollout-<ts>-<uuid>.jsonl, and invoke
# `codex exec resume <uuid> -- <instruction>`. codex then treats the prior turns
# as its own conversation history — prompt-cached and incremental — and only has
# to reason about the user's newest message.
#
# We resume by EXPLICIT session id (not --last): --last is cwd-filtered (help:
# "--all ... disables cwd filtering"), and we can't guarantee the rollout's
# recorded cwd matches the sandbox cwd at runtime; an explicit UUID is a direct
# lookup that sidesteps that entirely.
# ---------------------------------------------------------------------------
# A real recorded codex session_meta line (with codex's base_instructions) is the
# most reliable seed for `resume`. The fixture lives in-repo (mounted into the
# devcontainer where this agent code runs); a captured host copy is a secondary
# source, and a synthesized minimal record is the final fallback.
_ROLLOUT_TEMPLATE_CANDIDATES = (
os.path.join(os.path.dirname(os.path.abspath(__file__)), "codex-rollout-template.jsonl"),
"/Users/nickheiner/.claude/jobs/e8fade29/tmp/codex-rollout-template.jsonl",
)
def _now_iso() -> str:
from datetime import datetime, timezone
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.000Z")
def _session_meta(new_id: str, iso_ts: str) -> dict:
"""Return a codex `session_meta` rollout record, reusing the captured real
template (best fidelity for resume) when readable, else a minimal synthesized
one. The id/timestamp are always overwritten with our fresh values."""
for template_path in _ROLLOUT_TEMPLATE_CANDIDATES:
try:
with open(template_path, encoding="utf-8") as fh:
for line in fh:
line = line.strip()
if not line:
continue
rec = json.loads(line)
if rec.get("type") == "session_meta":
rec["timestamp"] = iso_ts
rec.setdefault("payload", {})
rec["payload"]["id"] = new_id
rec["payload"]["timestamp"] = iso_ts
return rec
except (OSError, ValueError):
continue
return {
"timestamp": iso_ts,
"type": "session_meta",
"payload": {
"id": new_id,
"timestamp": iso_ts,
"cwd": "/workspace",
"originator": "codex_exec",
"cli_version": "0.135.0",
"source": "exec",
"thread_source": "user",
"model_provider": "openai",
},
}
def is_codex_rollout(session_jsonl_text: str) -> bool:
"""True when the staged session is already a codex rollout rather than a Claude Code
transcript. Delegates to atif_session, which owns format detection — a second copy of
the record-type set here is how the two would eventually disagree."""
return atif_session.detect_format(session_jsonl_text) == "codex"
def reid_codex_rollout(session_jsonl_text: str, new_id: str, iso_ts: str) -> list[str]:
"""Re-key an already-native codex rollout onto `new_id` so `codex exec resume
<new_id>` finds it. Conversation records pass through byte-identical — a
codex-authored snapshot resumed by codex needs no translation, which is the
whole fidelity argument for native seeding."""
lines = [json.dumps(_session_meta(new_id, iso_ts))]
for raw in session_jsonl_text.splitlines():
raw = raw.strip()
if not raw:
continue
try:
rec = json.loads(raw)
except (json.JSONDecodeError, ValueError):
continue
if not isinstance(rec, dict) or rec.get("type") == "session_meta":
continue
lines.append(raw)
return lines
def stage_to_codex_rollout(session_jsonl_text: str, new_id: str, iso_ts: str) -> list[str]:
"""Build a resumable codex rollout from whichever format the snapshot staged."""
if is_codex_rollout(session_jsonl_text):
return reid_codex_rollout(session_jsonl_text, new_id, iso_ts)
return claude_to_codex_rollout(session_jsonl_text, new_id, iso_ts)
def claude_to_codex_rollout(
session_jsonl_text: str, new_id: str, iso_ts: str, max_total: int | None = None
) -> list[str]:
"""Translate a staged Claude Code session into codex rollout JSONL lines so
`codex exec resume` can continue it natively.
Parsing and rendering live in atif_session, which routes every harness pair
through ATIF; this stays as the codex-side entry point. `max_total` is an optional
char cap used by tests; the default is uncapped — our seeded sessions (~20-95k
tokens) fit every supported model's context."""
return atif_session.atif_to_codex_rollout(
atif_session.claude_session_to_atif(session_jsonl_text),
iso_ts,
session_meta=_session_meta(new_id, iso_ts),
max_total=max_total,
)
class NativeSnapshotCodex(SystemNodeCodex):
"""Bridge B: continue the staged Claude session via NATIVE codex resume.
For snapshot tasks we translate `/tmp/snapshot-session/session.jsonl` into a
codex rollout, write it under `$CODEX_HOME/sessions/`, and run
`codex exec resume <uuid> -- <instruction>`. For non-snapshot tasks (no staged
session) we defer to the stock fresh `codex exec` via the base agent."""
async def _read_staged_session(self, environment) -> str:
try:
result = await environment.exec(
command=f"cat {_STAGED_SESSION} 2>/dev/null || true"
)
return (getattr(result, "stdout", "") or "").strip()
except Exception as exc: # best-effort
self.logger.warning("NativeSnapshotCodex: could not read session: %s", exc)
return ""
async def run(self, instruction, environment, context): # type: ignore[override]
# NOTE: codex unconditionally declares its `tool_search` (MCP apps tool-
# discovery) tool, which the OpenAI API REJECTS for nano models with HTTP
# 400 "Tool 'tool_search' is not supported". None of codex's knobs
# (--disable tool_search / features.tool_search / enable_mcp_apps=false /
# disabled_tools) suppress it as of codex 0.135, so nano models are NOT
# runnable under this harness. Use a mini (e.g. gpt-5.4-mini) for the small
# end instead. Non-nano models are unaffected.
await apply_dns_jail(self, environment)
session_text = await self._read_staged_session(environment)
if not session_text:
self.logger.info(
"NativeSnapshotCodex: no staged session (non-snapshot task or empty); "
"running stock fresh codex exec"
)
await SystemNodeCodex.run(self, instruction, environment, context)
return
if not self.model_name:
raise ValueError("Model name is required")
model = self.model_name.split("/")[-1]
new_id = str(uuid.uuid4())
iso_ts = _now_iso()
rollout_lines = stage_to_codex_rollout(session_text, new_id, iso_ts)
self.logger.info(
"NativeSnapshotCodex: %s rollout of %d records (~%d chars) for resume %s",
"re-keyed native" if is_codex_rollout(session_text) else "translated Claude",
len(rollout_lines),
sum(len(line) for line in rollout_lines),
new_id,
)
# --- auth/setup: faithful to harbor's Codex.run (OPENAI_API_KEY → auth.json) ---
escaped_instruction = shlex.quote(instruction)
cli_flags = self.build_cli_flags()
cli_flags_arg = (cli_flags + " ") if cli_flags else ""
auth_json_path = self._resolve_auth_json_path()
remote_codex_home = self._REMOTE_CODEX_HOME.as_posix()
remote_secrets_dir = self._REMOTE_CODEX_SECRETS_DIR.as_posix()
remote_auth_path = (self._REMOTE_CODEX_SECRETS_DIR / "auth.json").as_posix()
env: dict[str, str] = {"CODEX_HOME": remote_codex_home}
setup_env: dict[str, str] = {}
await self.exec_as_agent(
environment,
command=(
f'mkdir -p "$CODEX_HOME" {shlex.quote(remote_secrets_dir)} '
f"{shlex.quote(EnvironmentPaths.agent_dir.as_posix())}"
),
env=env,
)
if auth_json_path:
await environment.upload_file(auth_json_path, remote_auth_path)
if environment.default_user is not None:
await self.exec_as_root(
environment,
command=f"chown {environment.default_user} {remote_auth_path}",
)
setup_command = f'ln -sf {shlex.quote(remote_auth_path)} "$CODEX_HOME/auth.json"\n'
else:
env["OPENAI_API_KEY"] = self._get_env("OPENAI_API_KEY") or ""
setup_env, auth_command = self._auth_json_setup(remote_auth_path)
setup_command = (
auth_command
+ f"ln -sf {shlex.quote(remote_auth_path)} \"$CODEX_HOME/auth.json\"\n"
)
if openai_base_url := self._get_env("OPENAI_BASE_URL"):
env["OPENAI_BASE_URL"] = openai_base_url
# The provider that carries the origin header rides in on build_cli_flags,
# so this stays harbor's plain root key.
setup_command += (
'\ncat >>"$CODEX_HOME/config.toml" <<TOML\n'
'openai_base_url = "${OPENAI_BASE_URL}"\n'
"TOML"
)
# env_key names this, and the provider cannot fall back to auth.json.
if proxy_key := self._get_env(PROXY_KEY_VAR):
env[PROXY_KEY_VAR] = proxy_key
skills_command = self._build_register_skills_command()
if skills_command:
setup_command += f"\n{skills_command}"
mcp_command = self._build_register_mcp_servers_command()
if mcp_command:
setup_command += f"\n{mcp_command}"
if setup_command.strip():
await self.exec_as_agent(
environment, command=setup_command, env={**env, **setup_env}
)
# --- write the converted rollout into $CODEX_HOME/sessions/<date>/ ---
date_parts = iso_ts[:10].split("-") # YYYY, MM, DD
sessions_dir = f"{remote_codex_home}/sessions/{date_parts[0]}/{date_parts[1]}/{date_parts[2]}"
rollout_name = f"rollout-{iso_ts.replace(':', '-')}-{new_id}.jsonl"
remote_rollout = f"{sessions_dir}/{rollout_name}"
await self.exec_as_agent(
environment, command=f"mkdir -p {shlex.quote(sessions_dir)}", env=env
)
with tempfile.NamedTemporaryFile(
"w", suffix=".jsonl", delete=False, encoding="utf-8"
) as tmp:
tmp.write("\n".join(rollout_lines) + "\n")
host_rollout = tmp.name
try:
await environment.upload_file(host_rollout, remote_rollout)
if environment.default_user is not None:
await self.exec_as_root(
environment,
command=f"chown {environment.default_user} {shlex.quote(remote_rollout)}",
)
finally:
try:
os.unlink(host_rollout)
except OSError:
pass
# --- resume by explicit session id ---
try:
await self.exec_as_agent(
environment,
command=(
"if [ -s ~/.nvm/nvm.sh ]; then . ~/.nvm/nvm.sh; fi; "
f"codex exec resume {new_id} "
"--dangerously-bypass-approvals-and-sandbox "
"--skip-git-repo-check "
f"--model {model} "
"--json "
"--enable unified_exec "
f"{cli_flags_arg}"
"-- "
f"{escaped_instruction} "
f"2>&1 </dev/null | tee {EnvironmentPaths.agent_dir / self._OUTPUT_FILENAME}"
),
env=env,
)
finally:
try:
await self.exec_as_agent(
environment,
command=(
f"mkdir -p {EnvironmentPaths.agent_dir.as_posix()}\n"
'if [ -d "$CODEX_HOME/sessions" ]; then\n'
f" rm -rf {(EnvironmentPaths.agent_dir / 'sessions').as_posix()}\n"
f' cp -R "$CODEX_HOME/sessions" {(EnvironmentPaths.agent_dir / "sessions").as_posix()}\n'
"fi"
),
env=env,
)
except Exception:
pass

View File

@@ -1,561 +0,0 @@
/**
* copy-reference-run.ts - Copy Harbor job trials into a task's reference-runs directory.
*
* Usage:
* npx tsx scripts/copy-reference-run.ts <trial-path> [trial-path...]
* npx tsx scripts/copy-reference-run.ts --rubric-regrade <trial-path>
* npx tsx scripts/copy-reference-run.ts --supersede <old-run-dir> <trial-path>
*
* Examples:
* # Copy a single trial
* npx tsx scripts/copy-reference-run.ts harbor-jobs/2026-04-02__11-43-55/my-task-slug__3Df3Bjr
*
* # Copy all trials from a job
* npx tsx scripts/copy-reference-run.ts harbor-jobs/2026-04-02__11-43-55/my-task-slug__*
*
* What gets copied:
* - verifier/agent-output/ (answer.md, etc.)
* - verifier/reward.txt (the run's reward) + reward-correctness.txt (reads
* N/A by design — correctness lives inside the graded criteria)
* - verifier/reward.json (the machine-readable reward record)
* - verifier/signals-status.txt (whether every deterministic check actually ran,
* i.e. whether the signals the grader was fed are complete)
* - verifier/grade.md + every grade-<N>.md grader sample
* - verifier/grader-result(-<N>).json, grader-stderr(-<N>).log, grader-samples.txt
* - verifier/grader-regime.json (the grading regime this grade actually ran
* under — unrecoverable after the fact, so it must travel with the grade)
* - agent/claude-code.txt or agent/codex.txt (the harness's own log), agent/trajectory.json
* - session.jsonl — the resumable session log, hoisted to the top of the run dir so
* `view-harbor-session.ts <run-dir>/session.jsonl` can load it without further
* indirection. Its location is per harness: Claude Code writes
* `agent/sessions/projects/-workspace/<id>.jsonl` with the id printed in
* `agent/claude-code.txt`; codex writes `agent/sessions/<Y>/<M>/<D>/rollout-*.jsonl`.
* - verifier/test-stdout.txt (the verifier's console output, where the reward is printed)
* - verifier/deterministic-signals.txt, verifier/rubric-grade(-<N>).json
* - config.json, result.json, trial.log
* - input-checksums.json — sha256 checksums of the task inputs the run was
* generated against (prompt, session snapshot, workspace patch, gitref),
* so submit-task.ts can warn when the run goes stale. Copied from the
* trial dir when harbor-run stamped one at launch time (capturedBy:
* 'run' — immune to edits made between the run and this copy);
* otherwise captured here at copy time as a fallback (capturedBy:
* 'copy').
*
* --rubric-regrade files an atomic-rubric regrade into rubric-regrades/<run>/
* instead: a second grade of a run that keeps its own, under the name the trial
* itself records for the run it graded. The trial is copied verbatim — verifier/
* and all — so a stored grade has the same shape whoever stored it.
*
* --supersede is the other direction: a holistic regrade replaces a run's grade,
* and the new copy lands under a name minted from the NEW reward and trial id, so
* adopting it means removing the directory it supersedes. Deleting is deliberate
* over merging into the old directory: a regrade grades a different number of
* samples than the run often did, so a merge would leave grade-2.md/grade-3.md
* from the previous grade beside the new grade-1.md with nothing marking the
* generation. A swapped directory holds exactly one grade by construction. The
* cost is the old run's session.jsonl, which a replay cannot reproduce; the
* trajectory it would be needed to rebuild travels with the replay itself.
*
* What is NOT copied:
* - agent/sessions/, agent/setup/ (workspace data — the resumable JSONL
* is hoisted out as `session.jsonl` above; everything else here is
* untyped workspace state)
* - artifacts/
*/
import './lib/check-devcontainer';
import {
existsSync,
mkdirSync,
readdirSync,
readFileSync,
renameSync,
rmSync,
statSync,
writeFileSync,
} from 'fs';
import { basename, join, resolve } from 'path';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { copyPath, copyTree } from './lib/copy-tree';
import {
captureTaskInputs,
INPUT_CHECKSUMS_FILENAME,
readTaskInputChecksums,
} from './lib/input-checksums';
import { manualRepairHint, normalizeTreePermissions } from './lib/tree-permissions';
import { readSessionId } from './session-id';
const HARBOR_TASKS_DIR = 'harbor-tasks';
function findTaskDir(trialPrefix: string): string | null {
if (!existsSync(HARBOR_TASKS_DIR)) return null;
const entries = readdirSync(HARBOR_TASKS_DIR, { withFileTypes: true });
for (const entry of entries) {
if (entry.isDirectory() && entry.name.startsWith(trialPrefix)) {
return join(HARBOR_TASKS_DIR, entry.name);
}
}
return null;
}
/** Newest `rollout-*.jsonl` anywhere under a codex `sessions/` tree, or null. */
function newestRollout(sessionsDir: string): string | null {
if (!existsSync(sessionsDir)) return null;
const found: Array<{ path: string; mtime: number }> = [];
const walk = (dir: string) => {
for (const entry of readdirSync(dir, { withFileTypes: true })) {
const full = join(dir, entry.name);
if (entry.isDirectory()) walk(full);
else if (entry.name.startsWith('rollout-') && entry.name.endsWith('.jsonl')) {
found.push({ path: full, mtime: statSync(full).mtimeMs });
}
}
};
walk(sessionsDir);
if (found.length === 0) return null;
found.sort((a, b) => b.mtime - a.mtime);
return found[0].path;
}
/** What a replay trial records about itself: the run it graded, and whether the
* grade came from a rubric grader mode rather than the holistic one. */
function readReplayProvenance(trialPath: string): {
sourceRun: string | null;
rubricMode: boolean;
} {
let config: Record<string, unknown> | undefined;
try {
const parsed = JSON.parse(readFileSync(join(trialPath, 'result.json'), 'utf-8')) as {
config?: Record<string, unknown>;
};
config = parsed.config;
} catch {
config = undefined;
}
const agent = (config?.agent ?? {}) as { kwargs?: { reference_run_dir?: unknown } };
const src = agent.kwargs?.reference_run_dir;
const verifier = (config?.verifier ?? {}) as { env?: Record<string, unknown> };
const graderMode = verifier.env?.GRADER_MODE;
// Either marker is enough: the env records the mode harbor was handed, the
// rubric-grade file records what the grader actually produced.
const rubricMode =
(typeof graderMode === 'string' && graderMode.startsWith('rubric-')) ||
existsSync(join(trialPath, 'verifier', 'rubric-grade.json'));
return {
sourceRun: typeof src === 'string' && src.length > 0 ? basename(src.replace(/\/+$/, '')) : null,
rubricMode,
};
}
/**
* Resolve the task dir from the trial's result.json `task_name` — the FULL,
* unambiguous slug. Harbor TRUNCATES long task names in the trial DIRNAME, so two
* distinct tasks sharing a truncated prefix (e.g. `foo--hash` and `foo--hash-2`,
* both truncating to `foo--ha`) collide: a dirname-prefix scan returns whichever
* sorts first and misroutes the other's trials (observed in the wild as base +
* `-2` reference-runs sharing trial IDs). result.json is written per-trial with
* the real task_name, so it disambiguates exactly. Returns null when result.json
* is absent/unparseable or names a task dir that doesn't exist (caller then falls
* back to the prefix scan).
*/
function findTaskDirByResultJson(trialPath: string): string | null {
const resultPath = join(trialPath, 'result.json');
if (!existsSync(resultPath)) return null;
let taskName: unknown;
try {
taskName = (JSON.parse(readFileSync(resultPath, 'utf-8')) as { task_name?: unknown }).task_name;
} catch {
return null;
}
if (typeof taskName !== 'string' || taskName.length === 0) return null;
// Hub-published task_names are org-prefixed (`<org>/<slug>`); the dir is bare.
// Inlined (not the shared bareSlug helper) because this script ships in the
// worker toolkit and must not import outside its shipped file set.
const dir = join(HARBOR_TASKS_DIR, taskName.replace(/^[^/]+\//, ''));
return existsSync(dir) ? dir : null;
}
function copyTrial(
trialPath: string,
opts: { destName?: string; rubricRegrade?: boolean; supersede?: string } = {}
) {
const { destName, rubricRegrade = false, supersede } = opts;
trialPath = trialPath.replace(/\/$/, '');
if (!existsSync(trialPath)) {
console.error(`Error: ${trialPath} does not exist`);
process.exit(1);
}
// Repair the SOURCE before reading a byte of it. A trial can leave files
// write-only, which locks out their own owner: everything below — reading
// reward.txt, copying agent-output — fails on them, and any that do get
// through land in the task dir, where harbor hashes every file on every
// later trial and one unreadable path aborts the run.
let sourcePerms = null;
try {
sourcePerms = normalizeTreePermissions(trialPath);
} catch (err) {
console.warn(`Warning: could not normalize permissions on ${trialPath}: ${String(err)}`);
console.warn(` If the copy below fails on permissions:`);
console.warn(` ${manualRepairHint(trialPath)}`);
}
if (sourcePerms && sourcePerms.failures.length > 0) {
console.warn(
`Warning: could not normalize permissions on ${sourcePerms.failures.length} path(s) under ${trialPath}.`
);
console.warn(` If the copy below fails on permissions, run:`);
console.warn(` ${manualRepairHint(trialPath)}`);
}
const rewardPath = join(trialPath, 'verifier', 'reward.txt');
if (!existsSync(rewardPath)) {
console.error(`Error: No reward.txt found in ${trialPath}/verifier/`);
process.exit(1);
}
const reward = readFileSync(rewardPath, 'utf-8').trim();
const trialDir = basename(trialPath);
// Trial dir format: <task-slug-truncated>__<trialId>
const separatorIndex = trialDir.lastIndexOf('__');
if (separatorIndex === -1) {
console.error(
`Error: Trial directory '${trialDir}' does not match expected format <slug>__<trialId>`
);
process.exit(1);
}
const trialPrefix = trialDir.substring(0, separatorIndex);
const trialId = trialDir.substring(separatorIndex + 2);
// Prefer the exact task_name from result.json (handles truncated-prefix
// collisions like `foo--hash` vs `foo--hash-2`); fall back to the dirname
// prefix scan only when result.json can't resolve it.
const taskDir = findTaskDirByResultJson(trialPath) ?? findTaskDir(trialPrefix);
if (!taskDir) {
console.error(
`Error: Could not find task directory matching prefix '${trialPrefix}' in ${HARBOR_TASKS_DIR}/`
);
console.error('Available tasks:');
readdirSync(HARBOR_TASKS_DIR).forEach((d) => console.error(` ${d}`));
process.exit(1);
}
// destName (--dest-name) makes a RECORDED rollout id authoritative: the run
// is copied to exactly that name instead of the minted reward-<r>-<id> —
// used by the SxS pipeline so task.toml preference_rollouts, this dir, and
// the publish-manifest run_id stay byte-identical by construction.
let dest = join(taskDir, 'reference-runs', destName ?? `reward-${reward}-${trialId}`);
const prov = readReplayProvenance(trialPath);
// An atomic grade in reference-runs/ would overwrite the holistic grade it exists
// to be compared against, and the two are not interchangeable.
if (!rubricRegrade && prov.rubricMode) {
console.error(
`Error: ${trialPath} is an atomic-rubric regrade, which does not belong in ` +
`reference-runs/. File it beside the run it graded:\n` +
` npx tsx scripts/copy-reference-run.ts --rubric-regrade ${trialPath}`
);
process.exit(1);
}
if (rubricRegrade) {
if (!prov.rubricMode) {
console.error(
`Error: ${trialPath} is not an atomic-rubric regrade (no rubric grader mode ` +
`in result.json, no verifier/rubric-grade.json). A holistic regrade replaces ` +
`the run's own grade — copy it without --rubric-regrade.`
);
process.exit(1);
}
if (!prov.sourceRun) {
console.error(
`Error: ${trialPath}/result.json records no reference_run_dir, so there is no ` +
`run to file this regrade under. Copy it by hand into ` +
`${join(taskDir, 'rubric-regrades')}/<run-id>/.`
);
process.exit(1);
}
dest = join(taskDir, 'rubric-regrades', prov.sourceRun);
}
// Build in a sibling directory and swap at the end. A grade costs 15-30 minutes and
// cannot be reproduced exactly, so an interrupted copy must leave the stored one intact.
const staged = `${dest}.staging-${process.pid}`;
rmSync(staged, { recursive: true, force: true });
mkdirSync(staged, { recursive: true });
// A stored atomic grade is the trial verbatim. Matching that shape exactly matters
// more than trimming it: grades stored by hand have it, and a reviewer opening one
// should not have to work out which way it was written.
if (rubricRegrade) {
copyTree(trialPath, staged);
} else {
// Copy verifier outputs. The grader writes one reward (reward.txt) with its
// companion reward-correctness.txt (N/A by design) and the machine-readable
// reward.json, plus signals-status.txt, its aggregate reasoning (grade.md)
// PLUS one grade-<N>.md per grader sample, the structured grade(-<N>).json
// (the source of truth grade.md/reward.txt are rendered from), the
// machine-readable grader-result(-<N>).json, the grader-stderr(-<N>).log,
// render-stderr(-<N>).log, and grader-samples.txt.
// Copy the whole set — glob so it stays agnostic to the sample count. (Earlier
// this took only reward.txt/grade.md/grader-stderr.log and silently dropped
// every per-sample record.)
//
// signals-status.txt qualifies the deterministic signals the grader was fed:
// it records whether every deterministic check actually produced a verdict
// ("ok") or one or more was killed before finishing ("degraded" — the grade
// is then NOT fully signal-backed). Without it a copied run is
// indistinguishable from a run whose checks all passed, so it must travel
// with the reward files.
const verifierDir = join(trialPath, 'verifier');
if (existsSync(verifierDir)) {
for (const entry of readdirSync(verifierDir, { withFileTypes: true })) {
if (!entry.isFile()) continue;
const f = entry.name;
if (
f === 'reward.json' ||
f === 'signals-status.txt' ||
f === 'grader-samples.txt' ||
f === 'grader-regime.json' ||
f === 'test-stdout.txt' ||
f === 'deterministic-signals.txt' ||
/^reward(-\d+)?\.txt$/.test(f) ||
/^reward-correctness(-\d+)?\.txt$/.test(f) ||
/^grade(-\d+)?\.md$/.test(f) ||
/^grade(-\d+)?\.json$/.test(f) ||
/^rubric-grade(-\d+)?\.json$/.test(f) ||
/^grader-result(-\d+)?\.json$/.test(f) ||
/^grader-stderr(-\d+)?\.log$/.test(f) ||
// The renderer retries, so the real name is render-stderr-<sample>-attempt<n>.log.
/^render-stderr(-\d+)?(-attempt\d+)?\.log$/.test(f)
) {
copyPath(join(verifierDir, f), join(staged, f));
}
}
}
// A rubric regrade re-grades a run that already ships its own agent-output,
// trajectory and session; a second copy would only double the tarball.
const agentOutputDir = join(verifierDir, 'agent-output');
if (!rubricRegrade && existsSync(agentOutputDir)) {
copyTree(agentOutputDir, join(staged, 'agent-output'));
}
// Copy agent session log and trajectory (not workspace)
const agentDir = join(trialPath, 'agent');
if (!rubricRegrade && existsSync(agentDir)) {
mkdirSync(join(staged, 'agent'), { recursive: true });
// The agent's own log is named per harness (claude-code.txt / codex.txt); copying only
// the Claude one left codex reference runs with nothing but the trajectory.
for (const file of ['claude-code.txt', 'codex.txt', 'trajectory.json']) {
const src = join(agentDir, file);
if (existsSync(src)) copyPath(src, join(staged, 'agent', file));
}
// The grader (and downstream worldbench export / replay) reads
// agent/trajectory.json. If it's missing, scream so we don't silently
// ship a reference run that's only half-useful.
if (!existsSync(join(agentDir, 'trajectory.json'))) {
console.warn(
`WARNING: ${trialPath}/agent/trajectory.json is missing. ` +
`Harbor's adapter for this harness failed to write it (typically because ` +
`the converter choked on the session log). Downstream consumers ` +
`(grader replay, worldbench export) need this file — investigate ` +
`before relying on this reference run.`
);
}
// Hoist the resumable session JSONL to <run-dir>/session.jsonl. The id
// gets embedded in claude-code.txt's first non-empty line; we use it to
// locate the sibling JSONL Claude Code wrote in the same trial.
// Layout is per harness, so each needs a case here — the same reason
// harness-session.mjs has one finder per CLI. A harness with no case gets no hoisted
// session, which is what codex got before this: nothing at all.
// Silent when there is nothing to hoist: no shipped tool reads this file and nothing
// validates it, so its absence is not worth a line of output.
const claudeCodeTxt = join(agentDir, 'claude-code.txt');
if (existsSync(claudeCodeTxt)) {
// Claude Code: the id is in claude-code.txt, the JSONL is its sibling.
const sessionId = readSessionId(claudeCodeTxt);
if (sessionId) {
const jsonl = join(agentDir, 'sessions', 'projects', '-workspace', `${sessionId}.jsonl`);
if (existsSync(jsonl)) copyPath(jsonl, join(staged, 'session.jsonl'));
}
} else {
// codex: sessions/<Y>/<M>/<D>/rollout-<ts>-<id>.jsonl, newest wins.
const rollout = newestRollout(join(agentDir, 'sessions'));
if (rollout) copyPath(rollout, join(staged, 'session.jsonl'));
}
}
// Copy top-level metadata. result.json is what makes a regrade self-describing
// (which run it graded, under which grader mode), so it travels either way.
for (const file of rubricRegrade
? ['result.json']
: ['config.json', 'result.json', 'trial.log']) {
const src = join(trialPath, file);
if (existsSync(src)) copyPath(src, join(staged, file));
}
// Not for a rubric regrade: this input set is what stales a RUN (prompt, snapshot,
// patch, gitref), and the run it re-graded already carries its own stamp.
if (!rubricRegrade) {
// Record the checksums of the task inputs this run was generated against
// (prompt, session snapshot, workspace patch, gitref, holistic rubric).
// submit-task.ts re-captures at packaging time and warns when any of them
// changed — the run then describes an older revision of the task than the
// one being shipped. Preferred source: the launch-time stamp harbor-run
// wrote into the trial dir (scripts/stamp-trial-inputs.ts, capturedBy:
// 'run') — it records the inputs the agent actually ran against, so an
// input edited BETWEEN harbor-run and this copy is still caught. Fallback
// (trials from an older harbor-run, or a failed stamp): capture here at
// copy time, marked capturedBy: 'copy' so staleness.json is honest about
// the weaker evidence.
const trialStampPath = join(trialPath, INPUT_CHECKSUMS_FILENAME);
const trialStamp = readTaskInputChecksums(trialStampPath);
// A stamp that names a DIFFERENT task got mis-routed (e.g. concurrent
// harbor-runs sharing a cwd) — its hashes describe some other task's
// inputs, so treat it as absent rather than importing false evidence.
const stampSlug = trialStamp?.taskSlug;
const stampMisrouted = typeof stampSlug === 'string' && stampSlug !== basename(taskDir);
if (trialStamp && !stampMisrouted) {
copyPath(trialStampPath, join(staged, INPUT_CHECKSUMS_FILENAME));
} else {
if (stampMisrouted) {
console.warn(
`Warning: ${trialStampPath} was captured for task '${stampSlug}', not ` +
`'${basename(taskDir)}' — ignoring it and capturing at copy time instead`
);
}
writeFileSync(
join(staged, INPUT_CHECKSUMS_FILENAME),
JSON.stringify(captureTaskInputs(taskDir, 'copy'), null, 2) + '\n'
);
}
}
}
// Files captured from a run can land unreadable to you, which makes packaging
// fail later. Fix that now. Guarded so it can never fail a copy that worked.
let perms = null;
try {
perms = normalizeTreePermissions(staged);
} catch (err) {
console.warn(`Warning: could not normalize permissions on ${dest}: ${String(err)}`);
console.warn(` The run copied fine. If packaging later fails on permissions:`);
console.warn(` ${manualRepairHint(dest)}`);
}
if (perms && perms.failures.length > 0) {
console.warn(
`Warning: could not normalize permissions on ${perms.failures.length} path(s) under ${dest}.`
);
console.warn(
` If packaging later fails with 'Cannot stat: Permission denied', run:\n` +
` ${manualRepairHint(dest)}`
);
}
if (existsSync(dest)) {
console.warn(`Warning: ${dest} already exists, overwriting`);
rmSync(dest, { recursive: true });
}
renameSync(staged, dest);
// Remove what this copy supersedes, now that the copy is on disk. Skipped when the
// minted name landed on the superseded directory itself — that is the copy, not a
// leftover.
if (supersede) {
const old = resolve(supersede);
if (!existsSync(old)) {
console.warn(`Warning: nothing to supersede at ${supersede} — it is already gone`);
} else if (old === resolve(dest)) {
console.log(`Superseded ${supersede} in place (same name)`);
} else {
// A stored atomic grade is keyed on the run's folder name, and superseding
// changes that name. Move it with the run — it grades the same behaviour — or
// it is left pointing at a run that no longer exists.
const storedGrade = join(taskDir, 'rubric-regrades', basename(old));
const movedGrade = join(taskDir, 'rubric-regrades', basename(dest));
if (existsSync(storedGrade) && existsSync(movedGrade)) {
console.warn(
`Warning: ${storedGrade} names the superseded run, but ${movedGrade} already ` +
`exists. Leaving both — remove whichever is obsolete.`
);
} else if (existsSync(storedGrade)) {
renameSync(storedGrade, movedGrade);
console.log(`Moved the stored atomic grade to ${movedGrade}`);
console.log(' It grades the same run. Grade it again under the atomic rubric');
console.log(' if the rubric changed since it was stored.');
}
rmSync(old, { recursive: true });
console.log(`Superseded ${supersede} (removed)`);
}
}
console.log(`Copied to ${dest}`);
console.log(` reward: ${reward}`);
console.log(` task: ${taskDir}`);
console.log(` trial: ${trialId}`);
const ownerFixed = (sourcePerms?.ownerFixed.length ?? 0) + (perms?.ownerFixed.length ?? 0);
const modeFixed = (sourcePerms?.modeFixed.length ?? 0) + (perms?.modeFixed.length ?? 0);
if (ownerFixed > 0 || modeFixed > 0) {
console.log(` perms: normalized ${ownerFixed} owner / ${modeFixed} mode`);
}
}
// Main
const rawArgs = process.argv.slice(2);
let destName: string | undefined;
let rubricRegrade = false;
let supersede: string | undefined;
const args: string[] = [];
for (let i = 0; i < rawArgs.length; i++) {
if (rawArgs[i] === '--rubric-regrade') {
rubricRegrade = true;
} else if (rawArgs[i] === '--supersede') {
supersede = rawArgs[++i];
if (!supersede) {
console.error('Error: --supersede requires the run directory being replaced');
process.exit(1);
}
} else if (rawArgs[i] === '--dest-name') {
destName = rawArgs[++i];
if (!destName) {
console.error('Error: --dest-name requires a value');
process.exit(1);
}
} else {
args.push(rawArgs[i]);
}
}
if (args.length === 0) {
console.error(
'Usage: npx tsx scripts/copy-reference-run.ts [--dest-name <run-id>] <trial-path> [trial-path...]\n' +
' npx tsx scripts/copy-reference-run.ts --rubric-regrade <trial-path>\n' +
' npx tsx scripts/copy-reference-run.ts --supersede <old-run-dir> <trial-path>'
);
process.exit(1);
}
if (destName && args.length !== 1) {
console.error('Error: --dest-name applies to exactly one trial path');
process.exit(1);
}
if (supersede && args.length !== 1) {
console.error('Error: --supersede applies to exactly one trial path');
process.exit(1);
}
if (supersede && rubricRegrade) {
console.error('Error: --rubric-regrade adds a grade beside a run; there is nothing to supersede');
process.exit(1);
}
if (destName && rubricRegrade) {
console.error(
'Error: --rubric-regrade takes its name from the trial, so --dest-name cannot apply'
);
process.exit(1);
}
for (const trialPath of args) {
copyTrial(trialPath, { destName, rubricRegrade, supersede });
}

View File

@@ -1,98 +0,0 @@
"""Apply the DNS jail to a trial container: the model endpoint resolves, nothing else does.
Opt-in with RACCOON_DNS_JAIL=1. Runs from the agent's own turn rather than from a compose
overlay — the allowlist comes from the proxy URL this process already holds (plus any hosts
RACCOON_DNS_JAIL_ALLOW adds), so nothing has to be injected into the container, and the jail works on every harbor backend. Deliberately
after agent-setup: a harness that downloads its CLI there still reaches the network to do it.
"""
import logging
import os
import shlex
from typing import Any
JAIL = "/usr/local/bin/raccoon-dns-jail"
_NO_SCRIPT = "raccoon-dns-jail: not in this image"
_URL_VARS = (
"ANTHROPIC_BASE_URL", "OPENAI_BASE_URL", "GOOGLE_GEMINI_BASE_URL",
"HTTPS_PROXY", "https_proxy", "HTTP_PROXY", "http_proxy", "ALL_PROXY", "all_proxy",
)
_EXTRA_VAR = "RACCOON_DNS_JAIL_ALLOW"
_log = logging.getLogger(__name__)
def _host(url: str) -> str:
"""Hostname out of a URL, or "" when it is not a plain hostname we can allow."""
h = url.split("://", 1)[-1].split("/", 1)[0].rsplit("@", 1)[-1].split(":", 1)[0]
if not h or h.startswith((".", "-")) or h.endswith(".") or not all(
c.isascii() and (c.isalnum() or c in ".-") for c in h
):
return ""
# An IP-literal endpoint (a loopback proxy shim, say) needs no DNS at all, and a
# --server rule for it would only be checked by a PTR query the catch-all answers.
if all(part.isdigit() for part in h.split(".")):
return ""
return h
def dns_jail_allowlist() -> tuple[list[str], list[str]]:
"""(required, advisory).
Required = the hosts this process's own env says the agent will dial; every one must
resolve through the jail or no jail is applied, because a host the agent needs and
cannot resolve is a dead trial. Advisory = whatever RACCOON_DNS_JAIL_ALLOW adds, which
only warns: an added host that CNAMEs outside the allowlist cannot resolve through the
catch-all, and must not take the whole jail down with it.
"""
required: list[str] = []
for var in _URL_VARS:
h = _host(os.environ.get(var) or "")
if h and h not in required:
required.append(h)
advisory: list[str] = []
for entry in (os.environ.get(_EXTRA_VAR) or "").replace(",", " ").split():
# Bare hostnames only: a URL silently truncated to its first path segment would
# allow a name nobody asked for and block the one they meant.
h = "" if ("/" in entry or ":" in entry) else _host(entry)
if not h:
_log.warning("DNS jail: ignoring unusable %s entry %r", _EXTRA_VAR, entry)
elif h not in required and h not in advisory:
advisory.append(h)
return required, advisory
def dns_jail_enabled() -> bool:
return os.environ.get("RACCOON_DNS_JAIL") == "1"
async def apply_dns_jail(agent: Any, environment: Any) -> None:
"""No-op unless enabled; leaves the container's DNS untouched on any doubt."""
if not dns_jail_enabled():
return
required, advisory = dns_jail_allowlist()
allow = " ".join(required)
# A blank allowlist means no model endpoint was found: jailing would strand the agent.
if not allow:
_log.warning("DNS jail: no usable model endpoint — the trial keeps normal network access")
return
try:
result = await agent.exec_as_root(
environment,
command=(
f"if [ -x {JAIL} ]; then DNSJAIL_ALLOW={shlex.quote(allow)} "
f"DNSJAIL_ALLOW_EXTRA={shlex.quote(' '.join(advisory))} {JAIL}; "
f'else echo "{_NO_SCRIPT}"; fi'
),
)
except Exception as exc: # a jail that cannot be applied must not fail the trial
_log.warning("DNS jail: could not apply (%s) — the trial keeps normal network access", exc)
return
# An image frozen before this feature has nothing to invoke. Say so: a launcher that
# believes the network is restricted when it is not is worse than no jail at all.
if _NO_SCRIPT in (getattr(result, "stdout", "") or ""):
_log.warning(
"DNS jail: this task's image ships no resolver — the trial keeps normal network access"
)

View File

@@ -1,48 +0,0 @@
#!/bin/bash
# guidance-target.sh — print the holistic-rubric file the grader reads for a task.
#
# The grader reads the task's holistic rubric under the Grading Standard.
# Detector skills call this resolver so they always assess the file the grader
# will actually read, and so the resolution rule lives in one place.
#
# Resolution order (renames are forward-only, so every generation stays readable):
# tests/holistic-rubric.md the current name; new tasks use it
# tests/grader-guidance-consolidated.md tasks created before the rename
# tests/grader-guidance.md legacy-generation tasks
# When none exists yet, the current name is printed — that is the file a new
# task's rubric will be written to.
#
# Usage:
# bash scripts/guidance-target.sh <slug-or-task-dir>
#
# Output (one line): the path to the rubric file.
set -eu
arg="${1:?usage: bash scripts/guidance-target.sh <slug-or-task-dir>}"
dir="$arg"
[ -d "$dir" ] || dir="harbor-tasks/$arg"
tests="$dir/tests"
[ -d "$tests" ] || { echo "ERROR: no tests/ directory at $dir" >&2; exit 1; }
new="$tests/holistic-rubric.md"
old="$tests/grader-guidance-consolidated.md"
legacy="$tests/grader-guidance.md"
if [ -f "$new" ] && [ -f "$old" ]; then
# Both names present: the grader's pick depends on the harness generation,
# so an assessment of either could be an assessment of the wrong file.
# Byte-identical copies are safe; anything else is a hard stop.
if ! cmp -s "$new" "$old"; then
echo "ERROR: $tests carries both holistic-rubric.md and grader-guidance-consolidated.md with different content — keep exactly one (tests/holistic-rubric.md is the current name)" >&2
exit 1
fi
echo "$new"
elif [ -f "$new" ]; then
echo "$new"
elif [ -f "$old" ]; then
echo "$old"
elif [ -f "$legacy" ]; then
echo "$legacy"
else
echo "$new"
fi

View File

@@ -1,600 +0,0 @@
#!/bin/bash
# Re-grade an existing reference run without re-invoking the agent.
#
# Spins up a normal harbor trial, but plugs in scripts/replay_agent.py
# instead of a real agent. The replay agent overlays the captured
# agent-output into /workspace, applies any captured deletions, drops
# the captured trajectory at /logs/agent/trajectory.json so the grader
# reads the same transcript it would for the original run, then exits.
# The verifier (real test.sh, real LLM grader if present) runs as it
# would for any other trial.
#
# Usage:
# scripts/harbor-regrade <task-dir> <reference-run-dir> [extra harbor args]
#
# Examples:
# # Single regrade
# scripts/harbor-regrade \
# harbor-tasks/<slug> \
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg
#
# # Every captured run for a task, 2 at a time (after a rubric edit)
# scripts/harbor-regrade harbor-tasks/<slug> --all
#
# # Ten regrades of the same reference run (independent grader trials)
# scripts/harbor-regrade \
# harbor-tasks/<slug> \
# harbor-tasks/<slug>/reference-runs/reward-0.62-h4KNEAg \
# -k 10
#
# --fast runs the GRADER in claude's fast serving mode (faster output at a higher
# token rate). The replay agent runs no model, so the grader is the only model in
# this path. Serving speed and cost change; the grade itself is not steered.
#
# A finished regrade files itself when nothing is filed for that run yet — which is
# only ever the atomic case, since a run always carries a holistic grade already.
# Otherwise it grades and leaves the result in its job dir, so the tune-and-diff loop
# is untouched; --replace adopts it, superseding what was there.
#
# See scripts/replay_agent.py for what the agent actually does, and the
# `verifier: capture tracked-file deletions in agent-output` PR for the
# capture half of this flow (_HARBOR_DELETIONS.txt).
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
# A sample count from the caller wins over .env, so a prefix can override a standing
# setting for one regrade. Captured as one value so a caller's spelling beats .env's,
# whichever each used (test.sh reads GRADER_SAMPLES; this script has long taken both).
_CALLER_SAMPLES="${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}"
# Source API key + any verifier env from the repo's .env
if [ -f "$REPO_ROOT/.env" ]; then
set -a
source "$REPO_ROOT/.env"
set +a
fi
# The grader authenticates with ANTHROPIC_API_KEY straight out of the .env sourced above, so
# a .env saved on Windows would hand it a value with a carriage return still attached.
if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then
# shellcheck disable=SC1091
HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true
harness_setup_credentials >/dev/null 2>&1 || true
fi
if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
. "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
fi
# When running inside a devcontainer, harbor needs HOST paths for docker
# bind mounts (the docker daemon is on the host).
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
cd "$HOST_WORKSPACE"
fi
usage() {
cat >&2 <<EOF
Usage: scripts/harbor-regrade <task-dir> <reference-run-dir>... [extra harbor args]
scripts/harbor-regrade <task-dir> --all [extra harbor args]
Required arguments:
<task-dir> harbor-tasks/<slug> — same dir you'd pass to scripts/harbor-run.
<reference-run-dir> harbor-tasks/<slug>/reference-runs/<run-id> — must contain
agent-output/ (and ideally agent/trajectory.json). Name several
to re-grade them all; each gets its own harbor job.
Optional arguments:
--all re-grade every run under <task-dir>/reference-runs/ — the usual
thing to do after editing the rubric.
--jobs N how many runs to re-grade at once (default 2). Each one is a
container, so raise it only as far as your machine allows.
--fast grade in claude's fast serving mode (higher token rate,
faster output). Anything else is passed through to harbor.
--replace adopt the result, replacing the grade already filed for this
run. Without it, a regrade that would overwrite an existing
grade is left in its job dir for you to compare first.
Note: -k re-grades the SAME run N times (N independent grades of one trajectory).
To re-grade DIFFERENT runs, name them all, or use --all.
EOF
exit 1
}
# `--fast` is consumed here (harbor has no such flag); everything else stays an opaque
# passthrough. Stripped first so it may appear anywhere, not just after the positionals.
FAST_REQUESTED=""
ALL_RUNS=""
REPLACE=""
JOBS=2
REGRADE_ARGS=()
while [ $# -gt 0 ]; do
case "$1" in
--fast) FAST_REQUESTED=1; shift ;;
--replace) REPLACE=1; shift ;;
--all) ALL_RUNS=1; shift ;;
--jobs)
[ $# -ge 2 ] || { echo "Error: --jobs needs a number." >&2; exit 1; }
JOBS="$2"; shift 2 ;;
--jobs=*) JOBS="${1#--jobs=}"; shift ;;
*) REGRADE_ARGS+=("$1"); shift ;;
esac
done
set -- "${REGRADE_ARGS[@]+"${REGRADE_ARGS[@]}"}"
# Normalised to base 10 before any (( )) sees it: "09" passes a -ge test but is
# invalid octal in arithmetic, which spun the dispatch loop forever.
case "$JOBS" in
'' | *[!0-9]*) JOBS_OK="" ;;
*) JOBS=$((10#$JOBS)); [ "$JOBS" -ge 1 ] && JOBS_OK=1 || JOBS_OK="" ;;
esac
if [ -z "$JOBS_OK" ]; then
echo "Error: --jobs must be a positive integer (got '$JOBS')." >&2
exit 1
fi
[ $# -lt 1 ] && usage
TASK_DIR="$1"
shift
# Leading non-flag positionals are reference runs; collection stops at the first
# harbor flag so `<task> <ref> -k 4` keeps working and `4` is never read as a run.
REF_DIRS=()
while [ $# -gt 0 ]; do
case "$1" in
-*) break ;;
*) REF_DIRS+=("$1"); shift ;;
esac
done
# Resolve to absolute paths — harbor cd's around internally; the replay
# agent receives the path as an --agent-kwarg and won't know our cwd.
TASK_DIR_ABS=$(cd "$TASK_DIR" 2>/dev/null && pwd) || {
echo "Error: task-dir does not exist: $TASK_DIR" >&2
exit 1
}
# Unfiltered on purpose: the child, not a name test here, decides what can be replayed.
if [ -n "$ALL_RUNS" ]; then
[ ${#REF_DIRS[@]} -gt 0 ] && {
echo "Error: pass --all or explicit reference-run dirs, not both." >&2
exit 1
}
REF_PARENT="$TASK_DIR_ABS/reference-runs"
[ -d "$REF_PARENT" ] || {
echo "Error: --all needs $REF_PARENT, which does not exist." >&2
exit 1
}
for _cand in "$REF_PARENT"/*/; do
[ -d "$_cand" ] && REF_DIRS+=("${_cand%/}")
done
[ ${#REF_DIRS[@]} -eq 0 ] && {
echo "Error: $REF_PARENT holds no reference runs." >&2
exit 1
}
fi
[ ${#REF_DIRS[@]} -eq 0 ] && usage
# Several runs: re-run ITSELF once each, so every child does the full preflight in an
# output dir of its own — harbor names job dirs by the second, and sharing one corrupts.
if [ ${#REF_DIRS[@]} -gt 1 ]; then
OUT_BASE="${HARBOR_REGRADE_OUT:-harbor-jobs}"
mkdir -p "$OUT_BASE"
CHILD_FLAGS=()
[ -n "$FAST_REQUESTED" ] && CHILD_FLAGS+=(--fast)
[ -n "$REPLACE" ] && CHILD_FLAGS+=(--replace)
echo "Re-grading ${#REF_DIRS[@]} reference runs, $JOBS at a time."
FAILED=()
FAILED_LOGS=()
BATCH_PIDS=()
# Without this a killed driver leaves its children running, and on a cloud backend
# each one is a sandbox that bills until something else reaps it.
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 143' TERM
trap '[ ${#BATCH_PIDS[@]} -gt 0 ] && kill -TERM "${BATCH_PIDS[@]}" 2>/dev/null; exit 130' INT
IDX=0
TOTAL=${#REF_DIRS[@]}
while [ "$IDX" -lt "$TOTAL" ]; do
BATCH_PIDS=()
BATCH_REFS=()
BATCH_LOGS=()
for ((_j = 0; _j < JOBS && IDX < TOTAL; _j++)); do
REF="${REF_DIRS[$IDX]}"
RUN_ID=$(basename "$REF")
# --job-name, not a nested output dir: children keep harbor's own
# harbor-jobs/<job>/<trial> shape. Indexed so duplicate args cannot collide.
JOB_NAME="regrade-$((IDX + 1))-$RUN_ID"
LOG="$OUT_BASE/$JOB_NAME.log"
echo " starting $RUN_ID (log: $LOG)"
"$SCRIPT_DIR/harbor-regrade" "$TASK_DIR_ABS" "$REF" \
${CHILD_FLAGS[@]+"${CHILD_FLAGS[@]}"} "$@" \
--job-name "$JOB_NAME" > "$LOG" 2>&1 &
BATCH_PIDS+=($!)
BATCH_REFS+=("$RUN_ID")
BATCH_LOGS+=("$LOG")
IDX=$((IDX + 1))
done
for ((_i = 0; _i < ${#BATCH_PIDS[@]}; _i++)); do
if wait "${BATCH_PIDS[$_i]}"; then
echo " ok ${BATCH_REFS[$_i]}"
else
echo " FAILED ${BATCH_REFS[$_i]}"
FAILED+=("${BATCH_REFS[$_i]}")
FAILED_LOGS+=("${BATCH_LOGS[$_i]}")
fi
done
done
if [ ${#FAILED[@]} -gt 0 ]; then
echo "${#FAILED[@]} of $TOTAL failed. Their output, in full — re-running is safe:" >&2
for _f in "${FAILED_LOGS[@]}"; do echo " $_f" >&2; done
exit 1
fi
echo "All $TOTAL re-graded."
# Children file their own grades into per-run logs we do not echo, so count the
# results rather than claiming them. Only the atomic destination is countable:
# a superseded run is gone, so there is no before/after to compare against.
case "${HARBOR_GRADER_MODE:-}" in
rubric-*)
FILED=0
for _r in "${REF_DIRS[@]}"; do
[ -d "$TASK_DIR_ABS/rubric-regrades/$(basename "$_r")" ] && FILED=$((FILED + 1))
done
echo "Filed $FILED of $TOTAL into $TASK_DIR_ABS/rubric-regrades/."
if [ "$FILED" -lt "$TOTAL" ]; then
echo "The rest are still in their job dirs; their logs above say why." >&2
fi
;;
esac
exit 0
fi
REF_RUN_DIR="${REF_DIRS[0]}"
REF_RUN_DIR_ABS=$(cd "$REF_RUN_DIR" 2>/dev/null && pwd) || {
echo "Error: reference-run-dir does not exist: $REF_RUN_DIR" >&2
exit 1
}
# Newer-generation Dockerfiles COPY environment/dns-jail/, a derived directory
# that harbor-run pre-stages but a bare regrade context may lack — the sandbox
# build then fails before the verifier ever starts. Recreate it the same way.
if [ -d "$TASK_DIR_ABS/environment" ]; then
mkdir -p "$TASK_DIR_ABS/environment/dns-jail"
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
if [ -f "$DNSJAIL_SRC" ]; then
# Rename into place: children share this task dir, and a half-written
# script is one a sibling's image build can pick up.
_dnsjail_dst="$TASK_DIR_ABS/environment/dns-jail/dns-jail-container.sh"
cp "$DNSJAIL_SRC" "$_dnsjail_dst.tmp.$$" && mv -f "$_dnsjail_dst.tmp.$$" "$_dnsjail_dst"
break
fi
done
fi
# Rubric grader modes (HARBOR_GRADER_MODE=rubric-*) read tests/render-rubric-grade.py,
# a shared asset like the dns-jail script above. A task created before the rubric
# renderer shipped has no copy, and a stale copy aggregates with outdated weights;
# either way the regrade must run against the current shared copy. Stage it
# host-side — the container only ever sees the task directory. The source lives at
# harbor-tasks/raccoon-shared/ in the internal repo and task-shared/ in a worker
# toolkit checkout; first one present wins.
case "${HARBOR_GRADER_MODE:-}" in
rubric-*)
RUBRIC_RENDER_DEST="$TASK_DIR_ABS/tests/render-rubric-grade.py"
for RUBRIC_RENDER_SRC in "$REPO_ROOT/harbor-tasks/raccoon-shared/render-rubric-grade.py" \
"$REPO_ROOT/task-shared/render-rubric-grade.py"; do
[ -f "$RUBRIC_RENDER_SRC" ] || continue
if [ ! -f "$RUBRIC_RENDER_DEST" ] || ! cmp -s "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST"; then
mkdir -p "$TASK_DIR_ABS/tests"
cp "$RUBRIC_RENDER_SRC" "$RUBRIC_RENDER_DEST.tmp.$$" &&
mv -f "$RUBRIC_RENDER_DEST.tmp.$$" "$RUBRIC_RENDER_DEST"
echo "Note: staged the current shared render-rubric-grade.py into $TASK_DIR_ABS/tests (rubric grader mode)." >&2
fi
break
done
;;
esac
# NOTE: agent-output/ is intentionally NOT required here. Advisory tasks (the
# agent only reads + answers in chat) make no workspace edits, so a faithful
# capture has an empty/absent agent-output/ — the deliverable lives in the
# captured transcript (agent/trajectory.json) that the grader reads. ReplayAgent
# overlays agent-output/ when present and otherwise grades base-workspace +
# transcript, but FAILS LOUDLY if the transcript shows file-mutating tool calls
# with no agent-output/ (genuine lost edits). So we let it make that call.
if [ ! -d "$REF_RUN_DIR_ABS/agent-output" ]; then
echo "Note: $REF_RUN_DIR_ABS has no agent-output/ — replaying as an" >&2
echo " advisory run (base workspace + captured transcript). See" >&2
echo " scripts/replay_agent.py for the lost-edits safety guard." >&2
fi
# Make scripts/ importable so harbor can find replay_agent:ReplayAgent.
# ${PYTHONPATH:+...} so an unset PYTHONPATH doesn't leave a trailing colon —
# python treats the resulting empty entry as the CWD, silently putting
# whatever directory the user ran this from on harbor's sys.path.
export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
# Environment backend. Explicit HARBOR_ENV wins; otherwise default to docker in
# a worker-toolkit checkout (detected by toolkit.json at the repo root) and
# daytona in the internal repo. See scripts/harbor-run for the full rationale
# (why the toolkit needs docker, why the marker is a workspace file not an image
# env, and why daytona must NOT pass --no-delete — billed sandbox).
if [ -n "${HARBOR_ENV:-}" ]; then
ENV_TYPE="$HARBOR_ENV"
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
ENV_TYPE="docker"
else
ENV_TYPE="daytona"
fi
DELETE_FLAGS="--no-delete"
case "$ENV_TYPE" in daytona|modal) DELETE_FLAGS="" ;; esac
# Modal bills a sandbox's REQUESTED cpu/memory for its whole life, and harbor's default
# `auto` requests the task's full cpus/memory_mb; `limit` requests Modal's floor and caps
# at the declared values. Override: HARBOR_CPUS_POLICY / HARBOR_MEMORY_POLICY.
RESOURCE_FLAGS=""
[ "$ENV_TYPE" = "modal" ] && RESOURCE_FLAGS="--cpus ${HARBOR_CPUS_POLICY:-limit} --memory ${HARBOR_MEMORY_POLICY:-limit}"
if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
exit 1
fi
# Output dir. harbor names the job subdir by second-granularity timestamp, so
# many regrades launched in the same second under one -o collide
# ("Job directory ... already exists and cannot be resumed"). Set
# HARBOR_REGRADE_OUT to a per-run unique dir when running a parallel sweep.
OUT_DIR="${HARBOR_REGRADE_OUT:-harbor-jobs}"
# Optional grader mode: HARBOR_GRADER_MODE=one-shot flips the task's test.sh into
# the no-tools one-shot grader (vs the default agentic grader) via verifier env —
# lets us A/B the agenticity gap without forking the task. See raccoon-shared/test.sh.
GRADER_MODE_FLAG=()
[ -n "${HARBOR_GRADER_MODE:-}" ] && GRADER_MODE_FLAG=(--verifier-env "GRADER_MODE=$HARBOR_GRADER_MODE")
# Optional grader model: HARBOR_GRADER_MODEL=claude-fable-5-1 overrides the grader's
# model (default: the `opus` alias) via verifier env — lets us A/B the grader model
# inside the unchanged agentic harness. See raccoon-shared/test.sh GRADER_MODEL.
[ -n "${HARBOR_GRADER_MODEL:-}" ] && GRADER_MODE_FLAG+=(--verifier-env "GRADER_MODEL=$HARBOR_GRADER_MODEL")
# Optional sample count, under either name. Unset, the task's own frozen tests/test.sh
# decides: tasks created before Sep 2026 average 3 samples, newer ones grade once.
_SAMPLES="${_CALLER_SAMPLES:-${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}}"
if [ -n "$_SAMPLES" ]; then
if ! [ "$_SAMPLES" -ge 1 ] 2>/dev/null; then
echo "harbor-regrade: ERROR — GRADER_SAMPLES must be a positive integer (got '$_SAMPLES')." >&2
exit 1
fi
GRADER_MODE_FLAG+=(--verifier-env "GRADER_SAMPLES=$_SAMPLES")
fi
# Optional API timeout: HARBOR_API_TIMEOUT_MS=<ms> raises the per-request API
# timeout of the grader's claude calls via verifier env (API_TIMEOUT_MS is the
# env var Claude Code itself reads, so test.sh needs no knowledge of it). For
# proxies whose responses outlast the CLI default.
[ -n "${HARBOR_API_TIMEOUT_MS:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "API_TIMEOUT_MS=$HARBOR_API_TIMEOUT_MS")
# Fast serving mode for the grader's own claude calls (--fast). Per-task tests/ assets
# are frozen at creation, so say when this task's copy cannot act on the flag. The note
# names that copy, not a shared path — this script ships to workers under another layout.
if [ -n "$FAST_REQUESTED" ]; then
GRADER_MODE_FLAG+=(--verifier-env "GRADER_FAST_MODE=true")
if [ -f "$TASK_DIR_ABS/tests/test.sh" ] &&
! grep -q 'GRADER_FAST_MODE' "$TASK_DIR_ABS/tests/test.sh"; then
echo "Note: --fast passed, but this task's tests/test.sh does not read" >&2
echo " GRADER_FAST_MODE, so the grader will run at normal speed —" >&2
echo " its copy predates the flag. Refresh the task's tests/test.sh" >&2
echo " from the current shared grader assets to enable it." >&2
fi
fi
# Carry the SOURCE run's agent identity + model into this replay's own record.
#
# A replay reports `replay_agent:ReplayAgent` with model_name null, because no model
# ran — the behaviour being graded came from the source run. Recording only
# reference_run_dir makes that a pointer, and pointers dangle: a regrade is normally
# copied back over the run it regraded, so the source usually no longer exists (501 of
# 643 on-disk replays already point at a missing dir, none of them in the published
# manifest either). Stamping the values here makes the replay self-describing, so the
# originating harness and model survive the source's deletion.
#
# Regrading a REGRADE means the source is itself a replay, so copying its own identity
# forward would overwrite the real provenance with a self-reference: inherit what it
# inherited instead.
#
# Read with python3 rather than jq — jq is not guaranteed on a worker's box, and a
# missing source result.json must degrade to "unknown", never abort the regrade.
SOURCE_PROV_FLAGS=()
if [ -f "$REF_RUN_DIR_ABS/result.json" ]; then
SOURCE_PROV=$(python3 -c '
import json, sys
try:
a = (json.load(open(sys.argv[1])).get("config") or {}).get("agent") or {}
except Exception:
sys.exit(0)
kw = a.get("kwargs") or {}
REPLAY = "replay_agent:ReplayAgent"
if (a.get("import_path") or a.get("name")) == REPLAY:
agent, model = kw.get("source_agent_import_path"), kw.get("source_model_name")
else:
agent, model = a.get("import_path") or a.get("name"), a.get("model_name")
# An already-damaged chain cannot be recovered; leave it honestly unstamped rather
# than propagating a source that names the replay agent itself.
print("" if agent == REPLAY else agent or "")
print(model or "")
' "$REF_RUN_DIR_ABS/result.json" 2>/dev/null || true)
SOURCE_AGENT=$(printf '%s\n' "$SOURCE_PROV" | sed -n 1p)
SOURCE_MODEL=$(printf '%s\n' "$SOURCE_PROV" | sed -n 2p)
[ -n "$SOURCE_AGENT" ] && SOURCE_PROV_FLAGS+=(--ak "source_agent_import_path=$SOURCE_AGENT")
[ -n "$SOURCE_MODEL" ] && SOURCE_PROV_FLAGS+=(--ak "source_model_name=$SOURCE_MODEL")
fi
# HARBOR_GRADING_STANDARD, when set, is passed through to the task's own
# tests/test.sh as the GRADING_STANDARD verifier env var. What (if anything)
# it does is decided by the scripts inside that tests/ directory; a test.sh
# that reads no such variable ignores it.
[ -n "${HARBOR_GRADING_STANDARD:-}" ] &&
GRADER_MODE_FLAG+=(--verifier-env "GRADING_STANDARD=$HARBOR_GRADING_STANDARD")
# A regrade is all verifier, so every call it makes is the grader's.
if [ -f "$REPO_ROOT/scripts/lib/call-origin.sh" ]; then
. "$REPO_ROOT/scripts/lib/call-origin.sh"
_GRADER_METADATA="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
# `if`, not `[ ... ] && ...`: an AND-list is this block's last statement, so under
# `set -e` an empty value would abort the regrade rather than just skip the header.
if [ -n "$_GRADER_METADATA" ]; then
GRADER_MODE_FLAG+=(
--verifier-env
"ANTHROPIC_CUSTOM_HEADERS=$CALL_ORIGIN_HEADER: $_GRADER_METADATA"
)
fi
fi
# Where this grade would be filed, and whether filing it destroys anything. An atomic
# regrade lands beside the run in rubric-regrades/<run>; a holistic one replaces the
# run's own grade, so its destination is always occupied and never files unasked.
rel() { case "$1" in "$PWD"/*) printf '%s' "${1#"$PWD"/}" ;; *) printf '%s' "$1" ;; esac; }
RUN_ID=$(basename "$REF_RUN_DIR_ABS")
RUBRIC_MODE=""
case "${HARBOR_GRADER_MODE:-}" in rubric-*) RUBRIC_MODE=1 ;; esac
if [ -n "$RUBRIC_MODE" ]; then
FILE_DEST="$TASK_DIR_ABS/rubric-regrades/$RUN_ID"
else
FILE_DEST="$REF_RUN_DIR_ABS"
fi
AUTOFILE=""
if [ ! -e "$FILE_DEST" ] || [ -n "$REPLACE" ]; then
AUTOFILE=1
fi
if [ -z "$AUTOFILE" ]; then
echo "Note: $(rel "$FILE_DEST")" >&2
echo " already holds a grade, so this regrade will not be filed." >&2
echo " Where it landed is printed when it finishes." >&2
fi
# Where the trial will land. Harbor's own job name is a second-granularity timestamp,
# so naming it here is what makes the trial findable afterwards (the fan-out passes one).
CALLER_JOB_NAME=""
CALLER_OUT=""
_prev=""
for _a in "$@"; do
case "$_prev" in
--job-name) CALLER_JOB_NAME="$_a" ;;
-o | --output-dir) CALLER_OUT="$_a" ;;
esac
case "$_a" in
--job-name=*) CALLER_JOB_NAME="${_a#--job-name=}" ;;
--output-dir=*) CALLER_OUT="${_a#--output-dir=}" ;;
esac
_prev="$_a"
done
JOB_NAME="$CALLER_JOB_NAME"
JOB_NAME_FLAG=()
if [ -n "$AUTOFILE" ] && [ -n "$CALLER_OUT" ]; then
echo "Note: -o/--output-dir passed, so this regrade will not be filed into" >&2
echo " $(rel "$FILE_DEST") — copy it yourself when it finishes." >&2
AUTOFILE=""
fi
if [ -z "$JOB_NAME" ] && [ -z "$CALLER_OUT" ]; then
# $$ as well as the epoch: harbor refuses an existing job dir outright, and two
# regrades of one run can start in the same second.
JOB_NAME="regrade-$RUN_ID-$(date +%s)-$$"
JOB_NAME_FLAG=(--job-name "$JOB_NAME")
fi
# RACCOON_DNS_JAIL is inert here on purpose: replay_agent runs no agent turn, so there is
# nothing to restrict — only the verifier runs, and it needs the proxy.
#
# harbor runs as a child, not under `exec`, so the filing below runs after it exits.
# TERM/INT are forwarded so a kill here never orphans a job (on cloud, a billing sandbox).
HARBOR_EXIT=0
HARBOR_SIGNALLED=""
harbor run \
-p "$TASK_DIR_ABS" \
--agent-import-path replay_agent:ReplayAgent \
--ak "reference_run_dir=$REF_RUN_DIR_ABS" \
${SOURCE_PROV_FLAGS[@]+"${SOURCE_PROV_FLAGS[@]}"} \
-e "$ENV_TYPE" \
$DELETE_FLAGS \
$RESOURCE_FLAGS \
--yes \
-o "$OUT_DIR" \
${GRADER_MODE_FLAG[@]+"${GRADER_MODE_FLAG[@]}"} \
${JOB_NAME_FLAG[@]+"${JOB_NAME_FLAG[@]}"} \
"$@" &
HARBOR_PID=$!
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT
wait "$HARBOR_PID" || HARBOR_EXIT=$?
if [ -n "$HARBOR_SIGNALLED" ]; then
# The first wait was interrupted by the trap; wait again for harbor's real status.
wait "$HARBOR_PID" || HARBOR_EXIT=$?
fi
trap - TERM INT
# A failed grade has nothing to report on, and a caller-directed -o put the trial
# somewhere this script cannot name.
if [ "$HARBOR_EXIT" -ne 0 ] || [ -n "$CALLER_OUT" ]; then
exit "$HARBOR_EXIT"
fi
JOB_DIR="$OUT_DIR/$JOB_NAME"
TRIALS=()
for _t in "$JOB_DIR"/*__*/; do
[ -d "$_t" ] && TRIALS+=("${_t%/}")
done
# Atomic grades sit beside the run; a holistic one replaces it, which means removing
# the directory it supersedes (see copy-reference-run.ts for why swap, not merge).
COPY_FLAGS=(--rubric-regrade)
[ -z "$RUBRIC_MODE" ] && COPY_FLAGS=(--supersede "$(rel "$REF_RUN_DIR_ABS")")
FILE_CMD="npx tsx $(rel "$SCRIPT_DIR")/copy-reference-run.ts ${COPY_FLAGS[*]}"
if [ ${#TRIALS[@]} -eq 0 ]; then
echo "Note: no trial directory under $JOB_DIR, so there is no grade." >&2
exit "$HARBOR_EXIT"
fi
if [ ${#TRIALS[@]} -gt 1 ]; then
echo "Note: ${#TRIALS[@]} trials under $JOB_DIR (-k grades one run repeatedly)." >&2
echo " Pick the one to keep and file it: $FILE_CMD <trial-dir>" >&2
exit "$HARBOR_EXIT"
fi
# Not filing: say where the grade is and how the two compare, since comparing them is
# the whole reason it was left alone. Adopting is a copy — re-running with --replace
# would spend the 15-30 minutes again for a grade already sitting on disk.
if [ -z "$AUTOFILE" ]; then
NEW_REWARD=$(cat "${TRIALS[0]}/verifier/reward.txt" 2>/dev/null || echo "?")
# A stored atomic grade is a whole trial, so its reward sits under verifier/;
# a reference run keeps its own at the top.
OLD_REWARD=$(cat "$FILE_DEST/verifier/reward.txt" 2>/dev/null ||
cat "$FILE_DEST/reward.txt" 2>/dev/null || echo "?")
echo "" >&2
echo "Graded, not filed — that run already holds a grade." >&2
echo " new $NEW_REWARD $(rel "${TRIALS[0]}")" >&2
echo " filed $OLD_REWARD $(rel "$FILE_DEST")" >&2
echo "" >&2
echo "Adopt this grade:" >&2
echo " npx tsx $(rel "$SCRIPT_DIR")/copy-reference-run.ts \\" >&2
echo " ${COPY_FLAGS[*]} \\" >&2
echo " $(rel "${TRIALS[0]}")" >&2
echo "" >&2
echo "Or pass --replace next time to file it without this step." >&2
exit "$HARBOR_EXIT"
fi
# Best-effort from here: the grade already cost 15-30 minutes, so a filing problem
# prints the command to finish by hand rather than failing the regrade.
if ! npx tsx "$SCRIPT_DIR/copy-reference-run.ts" "${COPY_FLAGS[@]}" "${TRIALS[0]}" >&2; then
echo "Warning: the grade is in ${TRIALS[0]} but could not be filed. Retry with:" >&2
echo " $FILE_CMD ${TRIALS[0]}" >&2
fi
exit "$HARBOR_EXIT"

View File

@@ -1,510 +0,0 @@
#!/bin/bash
# Run raccoon tasks via Harbor with standard defaults.
#
# Automatically detects snapshot-based tasks (those with environment/session.jsonl)
# and uses the snapshot agent adapter for session resume.
#
# Usage: scripts/harbor-run <task-dir> [extra harbor args...]
# Example: scripts/harbor-run harbor-tasks/my-task-slug
# Example: scripts/harbor-run harbor-tasks/my-task-slug -k 4 --force-build
#
# To change the model, use --model (consumed here). Passing harbor's own -m does NOT
# override: harbor's -m is repeatable and builds one agent per value, so `-m X` runs the
# registry default AND X — two trials.
#
# --fast runs both models of the trial in fast serving mode (faster output at a higher
# token rate): the trial agent (claude-code only) and the grader the verifier launches.
# Serving speed and cost change; the grade itself is not steered.
#
# GRADER_SAMPLES=N (a prefix, or a line in .env) sets how many times the grader scores the
# run and averages. A task's own tests/test.sh supplies the default when it is unset.
#
# Needs python 3.11+ on PATH (tomllib, to read the harness registry). Containers have it;
# a macOS host running HARBOR_ENV=docker may not — set RACCOON_PYTHON if so.
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
# An explicit RACCOON_DNS_JAIL from the caller wins over .env, so a worker who opted in
# there can still turn the jail off for one run.
_RJ_SET="${RACCOON_DNS_JAIL+set}"; _RJ_VAL="${RACCOON_DNS_JAIL:-}"
_RJA_SET="${RACCOON_DNS_JAIL_ALLOW+set}"; _RJA_VAL="${RACCOON_DNS_JAIL_ALLOW:-}"
# Same for the grader sample count, under either name (HARBOR_GRADER_SAMPLES is what
# harbor-regrade calls it; GRADER_SAMPLES is what the task's tests/test.sh reads).
# Captured as one value so a caller's spelling beats .env's, whichever each used.
_CALLER_SAMPLES="${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}"
# Source API key
if [ -f "$REPO_ROOT/.env" ]; then
set -a
source "$REPO_ROOT/.env"
set +a
fi
if [ -n "$_RJ_SET" ]; then RACCOON_DNS_JAIL="$_RJ_VAL"; fi
if [ -n "$_RJA_SET" ]; then RACCOON_DNS_JAIL_ALLOW="$_RJA_VAL"; fi
if [ -f "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" ]; then
. "$REPO_ROOT/scripts/lib/llm-proxy-env.sh" && apply_llm_proxy_env
fi
# Per-harness credentials (OPENAI_API_KEY and friends) are DERIVED from the proxy root in
# ANTHROPIC_BASE_URL — they are not in .env. The container-create derivation exported them
# into a process that has long since exited, and only the auth FILES it wrote survive, so a
# fresh shell has the key on disk but not in its environment. resolve_harness checks the
# environment, and the trial passes it through to the sandbox, so re-derive here.
# Quiet on purpose: if it does not work, resolve_harness refuses by name a second later.
if [ -f "$SCRIPT_DIR/lib/harness-credentials.sh" ]; then
# shellcheck disable=SC1091
HARNESS_SCRIPTS_DIR="$SCRIPT_DIR" . "$SCRIPT_DIR/lib/harness-credentials.sh" >/dev/null 2>&1 || true
harness_setup_credentials >/dev/null 2>&1 || true
fi
export ANTHROPIC_BASE_URL="${ANTHROPIC_BASE_URL:-}"
# When running inside a devcontainer, harbor computes absolute paths for
# Docker bind mounts. These paths must be HOST paths because docker compose
# talks to the host daemon via the shared socket. Switching CWD to the
# host-equivalent workspace path makes harbor resolve paths correctly.
if [ -n "${HOST_WORKSPACE:-}" ] && [ -d "$HOST_WORKSPACE" ]; then
cd "$HOST_WORKSPACE"
fi
TASK_DIR="$1"
shift
# Harness selection. `--harness` / `--model` / `--fast` / `--check-model` are consumed
# here; everything else passes through to harbor untouched, so existing invocations keep
# working. Parsed with a loop rather than getopts because the remaining args are an
# opaque harbor passthrough that getopts would try to interpret.
HARNESS_ARGS=()
PASSTHROUGH=()
FAST_REQUESTED=""
while [ $# -gt 0 ]; do
case "$1" in
--harness) HARNESS_ARGS+=(--harness "$2"); shift 2 ;;
--harness=*) HARNESS_ARGS+=(--harness "${1#*=}"); shift ;;
--model) HARNESS_ARGS+=(--model "$2"); shift 2 ;;
--model=*) HARNESS_ARGS+=(--model "${1#*=}"); shift ;;
--fast) HARNESS_ARGS+=(--fast); FAST_REQUESTED=1; shift ;;
--check-model) HARNESS_ARGS+=(--check-model); shift ;;
*) PASSTHROUGH+=("$1"); shift ;;
esac
done
set -- "${PASSTHROUGH[@]+"${PASSTHROUGH[@]}"}"
# Preflight: workspace must be populated before harbor tries to docker-build it.
# Without this, the Dockerfile's `COPY workspace/ .` fails with an opaque
# "failed to calculate checksum of ref ...: \"/workspace\": not found" buried
# several frames deep in harbor's asyncio + docker-compose traceback. Surface
# the real fix here instead.
WORKSPACE_DIR="$TASK_DIR/environment/workspace"
if [ ! -d "$WORKSPACE_DIR" ] || [ -z "$(ls -A "$WORKSPACE_DIR" 2>/dev/null)" ]; then
SLUG="$(basename "$TASK_DIR")"
echo "Error: $WORKSPACE_DIR is missing or empty." >&2
echo "Build it first: bash scripts/build-workspace.sh $SLUG" >&2
echo "(reads commit from $TASK_DIR/task.toml; applies environment/workspace.patch if present.)" >&2
exit 1
fi
# Preflight: refuse a task dir harbor would not accept as a task.
#
# `harbor run -p` falls back to reading a rejected dir as a DATASET of tasks, finds none,
# and dies with "Either datasets or tasks must be provided." — naming neither the path nor
# the missing file. Name it here instead. Harbor's own interpreter (the uv-tool venv, per
# the shim's shebang) is the only one that can import harbor; skip the check when it or the
# helper is absent, so a stripped environment never blocks a runnable task.
VALIDATE_TASK_DIR="$SCRIPT_DIR/validate_task_dir.py"
# Harbor's own interpreter is the only one that can import harbor. RACCOON_HARBOR_PYTHON
# overrides it for tests, which have no harbor to read a shebang from.
HARBOR_PY="${RACCOON_HARBOR_PYTHON:-}"
if [ -z "$HARBOR_PY" ] && HARBOR_BIN="$(command -v harbor 2>/dev/null)"; then
HARBOR_PY="$(sed -n '1s|^#!||p' "$HARBOR_BIN" 2>/dev/null || true)"
fi
if [ -f "$VALIDATE_TASK_DIR" ] && [ -n "$HARBOR_PY" ] && [ -x "${HARBOR_PY%% *}" ]; then
# --install-only implies --disable-verification in harbor, for task validation too.
VALIDATE_ARGS=()
case " $* " in
*" --disable-verification "* | *" --install-only "*) VALIDATE_ARGS+=(--disable-verification) ;;
esac
VALIDATE_ERR="$(mktemp)"
# Gate on the printed verdict, never the exit code: an interpreter that cannot run
# the helper at all (a stub harbor with a bash shebang) also exits non-zero, and a
# preflight that can refuse a run must never refuse one that would have worked.
VALIDATE_OUT="$("$HARBOR_PY" "$VALIDATE_TASK_DIR" "$TASK_DIR" \
${VALIDATE_ARGS[@]+"${VALIDATE_ARGS[@]}"} 2>"$VALIDATE_ERR")" || true
if [ "$VALIDATE_OUT" = "verdict=invalid" ]; then
echo "Error: harbor will not accept $TASK_DIR as a task." >&2
sed 's|^| |' "$VALIDATE_ERR" >&2
echo "Fix the file named above, then re-run; harbor's own error names no file." >&2
if [ -d "$REPO_ROOT/harbor-tasks/_task-scaffold" ]; then
echo "A missing toolkit-managed file (tests/test.sh, tests/render-grade-consolidated.py)" >&2
echo "can be copied from harbor-tasks/_task-scaffold/ at the same relative path. Never" >&2
echo "overwrite a task.toml or instruction.md you have already written — repair it." >&2
fi
rm -f "$VALIDATE_ERR"
exit 1
fi
rm -f "$VALIDATE_ERR"
fi
# Preflight: recompute the browser marker from task.toml.
#
# `browser = true` decides whether the image installs Playwright, and a Dockerfile can only
# learn it from its build context. build-workspace.sh writes the marker — but a task.toml
# edited afterwards leaves it stale, and flipping the flag off would otherwise still build a
# browser into a `browser = false` task. The file is derived, so there is nothing to preserve
# by leaving it alone.
BROWSER_OPTIN=0
if [ -f "$TASK_DIR/task.toml" ] &&
grep -qE '^[[:space:]]*browser[[:space:]]*=[[:space:]]*"?true"?[[:space:]]*$' "$TASK_DIR/task.toml"; then
BROWSER_OPTIN=1
fi
if [ -d "$TASK_DIR/environment" ]; then
# Rename into place: a concurrent run against this task dir must never read the
# instant between truncate and write.
printf '%s\n' "$BROWSER_OPTIN" > "$TASK_DIR/environment/browser-optin.tmp.$$" &&
mv -f "$TASK_DIR/environment/browser-optin.tmp.$$" "$TASK_DIR/environment/browser-optin"
fi
# Preflight: restage the DNS jail script, for the same reason as the marker above.
#
# The Dockerfile COPYs environment/dns-jail/, and a missing COPY source fails the BUILD --
# which would kill every trial on the task, the one outcome the jail must never cause. A
# context staged before this existed passes the workspace guard above and would then die at
# build, so recreate the directory here and refill it when the source is around. Derived,
# so there is nothing to preserve by leaving it alone.
if [ -d "$TASK_DIR/environment" ]; then
mkdir -p "$TASK_DIR/environment/dns-jail"
for DNSJAIL_SRC in "$REPO_ROOT/scripts/lib/dns-jail-container.sh" \
"$REPO_ROOT/task-shared/dns-jail-container.sh"; do
if [ -f "$DNSJAIL_SRC" ]; then
_dnsjail_dst="$TASK_DIR/environment/dns-jail/dns-jail-container.sh"
cp "$DNSJAIL_SRC" "$_dnsjail_dst.tmp.$$" && mv -f "$_dnsjail_dst.tmp.$$" "$_dnsjail_dst"
break
fi
done
fi
# Preflight: report — never block — on edits to toolkit-managed files.
#
# environment/Dockerfile, tests/test.sh and tests/grader-system-prompt-consolidated.md ship from
# task-shared/ and decide how the trial runs and how the grade is produced, so an edit
# makes a task's runs hard to compare with the rest. Surface that here, before a trial
# burns agent time. It is advisory on purpose: an author who edited one did it to get
# unstuck, and refusing to run their trial punishes a misunderstanding. `|| true` also
# means a checker that can't run (a fresh unzip with no node_modules) never reads as an
# edit. The checker only exists in the worker toolkit; here the file is absent.
CHECK_INFRA="$REPO_ROOT/scripts/check-task-infra.ts"
if [ -f "$CHECK_INFRA" ]; then
(cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true
fi
# Preflight: warn — loudly, but never block — when the live workspace has
# changes that a rebuild from the pinned commit + workspace.patch would lose.
# Trials run against the live workspace, but the finalized task keeps only the
# rebuild inputs (the workspace/ dir is gitignored) and every downstream
# consumer rebuilds from them, so anything uncaptured silently vanishes after
# packaging.
# The helper sits next to this script in a packed toolkit and under the
# toolkit's static scripts in the internal repo layout.
for WS_SYNC in "$SCRIPT_DIR/check-workspace-sync.sh" \
"$REPO_ROOT/raccoon-worker-toolkit/static/scripts/check-workspace-sync.sh"; do
if [ -f "$WS_SYNC" ]; then
bash "$WS_SYNC" "$TASK_DIR" || true
break
fi
done
# Agent + model selection, from scripts/harness-registry.toml via resolve_harness.
# The agent classes come from scripts/{snapshot,codex,gemini}_agent.py or
# harness_agents.py (hence the PYTHONPATH). The Claude variants reuse the claude
# binary baked into the task image instead of re-downloading it at agent-setup —
# stock claude-code's runtime download (~240 MB) races the 360s agent-setup timeout
# and loses on slow-egress hosts (AgentSetupTimeoutError). Tasks that ship a
# non-empty environment/session.jsonl additionally resume the staged session;
# single-turn tasks get the non-resuming class.
#
# resolve_harness exits non-zero (and prints why) when the selection could not
# produce a usable grade — an unknown/disabled harness, one that writes no ATIF
# trajectory, a missing credential, or a task needing resume on a harness that
# can't. Failing here costs a second; failing later costs the whole trial, and the
# resume case wouldn't fail at all, it would silently grade the wrong thing.
# _raccoon_python comes from lib/harness-credentials.sh, sourced above. Define a fallback
# only if that file was missing, so the error below is about the interpreter rather than an
# unbound function.
command -v _raccoon_python >/dev/null 2>&1 || _raccoon_python() { return 1; }
export PYTHONPATH="$SCRIPT_DIR${PYTHONPATH:+:$PYTHONPATH}"
RACCOON_PY=$(_raccoon_python) || {
echo "harbor-run: ERROR — no python3.11+ with tomllib on PATH, so the harness registry" >&2
echo "harbor-run: cannot be read and the agent class cannot be resolved. Set" >&2
echo "harbor-run: RACCOON_PYTHON to an interpreter that has tomllib (3.11+)." >&2
exit 1
}
RESOLVED="$("$RACCOON_PY" "$SCRIPT_DIR/resolve_harness.py" \
--task-dir "$TASK_DIR" \
${HARNESS_ARGS[@]+"${HARNESS_ARGS[@]}"})" || exit 1
eval "$RESOLVED"
# --agent-import-path, not --agent: harbor 0.20 deprecates it but still accepts it, and it
# is the flag whose recorded shape (`agents[0].import_path`) every run on disk and every
# reader expects. --agent leaves import_path null and puts the class in `name`, which
# silently empties the agent field in published benchmark rows. Revisit if the pin moves.
AGENT_FLAGS="--agent-import-path $AGENT_IMPORT_PATH"
# Empty EFFORT_KWARG means "run the harness's native default config" — pass no
# effort kwarg at all rather than an empty one, which harbor would reject.
EFFORT_FLAGS=""
[ -n "$EFFORT_KWARG" ] && EFFORT_FLAGS="--ak $EFFORT_KWARG=$EFFORT_VALUE"
# FAST_KWARG is non-empty only when --fast was passed AND the harness declares one
# (resolve_harness refuses the flag otherwise).
FAST_FLAGS=""
[ -n "${FAST_KWARG:-}" ] && FAST_FLAGS="--ak $FAST_KWARG=true"
# The kwarg above reaches the trial agent only; the verifier launches its own grader
# claude, which the task's tests/test.sh puts in fast mode from GRADER_FAST_MODE.
GRADER_FAST_FLAGS=""
[ -n "$FAST_REQUESTED" ] && GRADER_FAST_FLAGS="--verifier-env GRADER_FAST_MODE=true"
# A task's tests/test.sh is frozen at creation and defaults its own sample count, so
# forwarding the var is the only way to change one that already exists.
GRADER_SAMPLES_FLAGS=""
_SAMPLES="${_CALLER_SAMPLES:-${GRADER_SAMPLES:-${HARBOR_GRADER_SAMPLES:-}}}"
if [ -n "$_SAMPLES" ]; then
if ! [ "$_SAMPLES" -ge 1 ] 2>/dev/null; then
echo "harbor-run: ERROR — GRADER_SAMPLES must be a positive integer (got '$_SAMPLES')." >&2
exit 1
fi
GRADER_SAMPLES_FLAGS="--verifier-env GRADER_SAMPLES=$_SAMPLES"
fi
# The verifier launches its own grader claude, so it names its own call origin
# rather than inheriting the surface that launched harbor. An array because the
# header value contains a space.
GRADER_ORIGIN_FLAGS=()
if [ -f "$REPO_ROOT/scripts/lib/call-origin.sh" ]; then
. "$REPO_ROOT/scripts/lib/call-origin.sh"
_GRADER_METADATA="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
# `if`, not `[ ... ] && ...`: an AND-list is this block's last statement, so under
# `set -e` an empty value would abort the run rather than just skip the header.
if [ -n "$_GRADER_METADATA" ]; then
GRADER_ORIGIN_FLAGS=(
--verifier-env
"ANTHROPIC_CUSTOM_HEADERS=$CALL_ORIGIN_HEADER: $_GRADER_METADATA"
)
fi
fi
# Environment backend. An explicit HARBOR_ENV always wins (either direction).
# Otherwise the default is context-dependent:
# - daytona for the internal repo: runs the trial in a cloud sandbox over
# HTTP, so it needs no local docker daemon and works *inside* the primary
# devcontainer. (HARBOR_ENV=docker uses the host docker daemon instead —
# free and offline, but host-only; the devcontainer has no docker.sock.)
# - docker for the worker toolkit: it's provisioned only for the local docker
# backend (docker CLI + bind-mounted docker.sock, no DAYTONA_API_KEY), so a
# daytona default would just error out. We detect a toolkit checkout by
# toolkit.json at the repo root — a file the packaging step writes that the
# internal repo never has. It lives in the bind-mounted workspace, not a
# baked image layer, so this holds even when the container's HARBOR_ENV pin
# is missing (e.g. a stale, pre-pin image).
if [ -n "${HARBOR_ENV:-}" ]; then
ENV_TYPE="$HARBOR_ENV"
elif [ -f "$REPO_ROOT/toolkit.json" ]; then
ENV_TYPE="docker"
else
ENV_TYPE="daytona"
fi
# --no-delete keeps the environment around after the trial for inspection.
# That's free for a local docker container, but a Daytona or Modal sandbox is
# *billed* while it exists — keeping it would leak a paid sandbox on every run.
# Harbor downloads the trial logs into harbor-jobs before teardown either way, so
# for the cloud backends we let it delete the sandbox; for docker we keep the container.
DELETE_FLAGS="--no-delete"
case "$ENV_TYPE" in daytona|modal) DELETE_FLAGS="" ;; esac
# Orphan resilience (daytona): harbor tears sandboxes down per-trial + via an
# atexit that only closes the client — neither runs on SIGTERM/SIGKILL/crash, so
# a killed run leaks STARTED sandboxes that hog the shared pool until (if ever)
# an account default reaps them. Tell Daytona to auto-stop an IDLE sandbox after
# 20 min (auto-delete on stop), so orphans self-clean however the process dies.
# Safe for live trials: a running agent/grader keeps the sandbox active.
AUTOSTOP_FLAGS=""
[ "$ENV_TYPE" = "daytona" ] && AUTOSTOP_FLAGS="--ek auto_stop_interval_mins=20"
# Modal bills a sandbox's REQUESTED cpu/memory for its whole life, and harbor's default
# `auto` requests the task's full cpus/memory_mb; `limit` requests Modal's floor and caps
# at the declared values. Override: HARBOR_CPUS_POLICY / HARBOR_MEMORY_POLICY.
RESOURCE_FLAGS=""
[ "$ENV_TYPE" = "modal" ] && RESOURCE_FLAGS="--cpus ${HARBOR_CPUS_POLICY:-limit} --memory ${HARBOR_MEMORY_POLICY:-limit}"
if [ "$ENV_TYPE" = "daytona" ] && [ -z "${DAYTONA_API_KEY:-}" ]; then
echo "Error: harbor backend resolved to daytona but DAYTONA_API_KEY is unset." >&2
echo "Add DAYTONA_API_KEY to $REPO_ROOT/.env, or run with HARBOR_ENV=docker." >&2
exit 1
fi
# Task-image Claude Code floor. A task image installs Claude Code when it is first built and
# Docker reuses that layer on every later build, --force-build included, so an image built
# before the grader model's minimum CLI shipped fails every grade with "does not support this
# model" and the trial ends in RewardFileNotFoundError. Before a local docker trial, check the
# hb__ task images on this daemon and remove any that are too old, together with the build
# cache, so harbor's build below installs a current CLI. Advisory: no docker, no daemon, no
# images, or RACCOON_SKIP_IMAGE_PREFLIGHT=1 means nothing happens. The floor follows
# GRADER_CLI_MIN in the shared test.sh; RACCOON_CLAUDE_CODE_MIN overrides it.
CLAUDE_CODE_MIN="${RACCOON_CLAUDE_CODE_MIN:-}"
if [ -z "$CLAUDE_CODE_MIN" ]; then
for _ts in "$REPO_ROOT/task-shared/test.sh" "$REPO_ROOT/harbor-tasks/raccoon-shared/test.sh"; do
[ -f "$_ts" ] || continue
CLAUDE_CODE_MIN="$(sed -n 's/^GRADER_CLI_MIN=.*:-\([0-9][0-9.]*\)}.*/\1/p' "$_ts" | head -1)"
[ -n "$CLAUDE_CODE_MIN" ] && break
done
fi
CLAUDE_CODE_MIN="${CLAUDE_CODE_MIN:-2.1.251}"
if [ "$ENV_TYPE" = "docker" ] && [ "${RACCOON_SKIP_IMAGE_PREFLIGHT:-0}" != "1" ] \
&& command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1; then
STALE_IMAGES=""
for img in $(docker images --format '{{.Repository}}:{{.Tag}}' 2>/dev/null | grep '^hb__' || true); do
v="$(docker run --rm --entrypoint claude "$img" --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
[ -n "$v" ] || continue
if [ "$(printf '%s\n%s\n' "$CLAUDE_CODE_MIN" "$v" | sort -V | head -1)" != "$CLAUDE_CODE_MIN" ]; then
STALE_IMAGES="$STALE_IMAGES $img"
echo "Task image $img carries Claude Code $v; the grader needs $CLAUDE_CODE_MIN or newer. Removing it so the next build installs a current one." >&2
fi
done
if [ -n "$STALE_IMAGES" ]; then
for img in $STALE_IMAGES; do
# Trial containers harbor kept for inspection pin the image; drop them first.
for c in $(docker ps -aq --filter "ancestor=$img" 2>/dev/null); do docker rm -f "$c" >/dev/null 2>&1 || true; done
docker image rm -f "$img" >/dev/null 2>&1 || true
done
# The stale install layer also lives in the build cache, where a rebuild would find it.
docker builder prune -af >/dev/null 2>&1 || true
echo "The task image rebuilds once with a current Claude Code; later runs reuse it." >&2
fi
fi
# Launch-time input-checksum capture. Snapshot the task inputs BEFORE harbor
# starts, so the recorded hashes are what the agent actually ran against —
# an input edited between this run and copy-reference-run no longer records
# post-edit state and masks staleness. The capture is stamped into the trial
# dirs after the run (below); copy-reference-run prefers it over its weaker
# capture-at-copy fallback. Advisory end to end: a failure here never blocks
# the run.
#
# The post-run stamping watches the default `harbor-jobs` output dir. If the
# caller overrides the output dir via extra args, we can't know where the
# trials will land — skip stamping and say so, rather than silently stamping
# nothing (runs then fall back to copy-time capture in copy-reference-run).
STAMP_FILE=""
for arg in "$@"; do
case "$arg" in
-o|--output*)
echo "Note: custom harbor output dir passed ($arg) — skipping launch-time input-checksum stamping; reference runs will fall back to copy-time capture." >&2
STAMP_FILE="skip"
break
;;
esac
done
if [ "$STAMP_FILE" != "skip" ]; then
STAMP_FILE="$(mktemp)"
if ! npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" capture "$TASK_DIR" --out "$STAMP_FILE"; then
echo "Warning: could not capture launch-time input checksums (staleness will be judged from copy-time capture instead)" >&2
rm -f "$STAMP_FILE"
STAMP_FILE=""
fi
else
STAMP_FILE=""
fi
JOBS_BEFORE="$(ls -1 harbor-jobs 2>/dev/null || true)"
# DNS jail: the trial resolves the LLM proxy and nothing else. Opt-in with
# RACCOON_DNS_JAIL=1, which .env can set; the agent applies it inside the container from
# the proxy URL it already holds, so nothing is passed on the harbor command line.
#
# Claude and codex only (see scripts/dnsjail.py): both are baked into the task images,
# while every other harness downloads its CLI at agent-setup. Say so out loud — a silently
# unjailed trial is worse than no jail, because the caller believes otherwise.
if [ "${RACCOON_DNS_JAIL:-0}" = "1" ]; then
case "$AGENT_IMPORT_PATH" in
snapshot_agent:* | codex_agent:*)
# The script is baked into the image, so a task frozen before this feature has
# nothing to invoke. Most hand-authored per-task Dockerfiles are in that group.
if [ -f "$TASK_DIR/environment/Dockerfile" ] &&
! grep -q 'raccoon-dns-jail' "$TASK_DIR/environment/Dockerfile"; then
echo "Note: DNS jail skipped — this task's image predates it and ships no" \
"resolver. This trial has full network access." >&2
fi
;;
*) echo "Note: DNS jail skipped — it is applied by the claude and codex agents" \
"only. This trial has full network access." >&2 ;;
esac
fi
# The model comes from the registry (already resolved above) and is always a
# CONCRETE id, never a shorthand alias. On the manual (non-snapshot) path the agent
# passes the model via ANTHROPIC_MODEL, where a shorthand is NOT alias-resolved, so
# the configured base-URL endpoint rejects it (400 "Invalid model: <shorthand>") and
# trials die on turn 1. Bump `default_model` in scripts/harness-registry.toml when a
# newer model ships.
#
# harbor runs as a child (this script used to `exec` it, but the post-run
# stamping needs to run after harbor exits), so forward TERM/INT: a `kill`
# aimed at this wrapper's PID must take harbor down with it, not orphan a
# running job (this repo has been bitten by zombie harbor coordinators
# before).
HARBOR_EXIT=0
HARBOR_SIGNALLED=""
harbor run \
-p "$TASK_DIR" \
$AGENT_FLAGS \
-m "$MODEL" \
-e "$ENV_TYPE" \
$DELETE_FLAGS \
$AUTOSTOP_FLAGS \
$RESOURCE_FLAGS \
--yes \
-o harbor-jobs \
$EFFORT_FLAGS \
$FAST_FLAGS \
$GRADER_FAST_FLAGS \
$GRADER_SAMPLES_FLAGS \
${GRADER_ORIGIN_FLAGS[@]+"${GRADER_ORIGIN_FLAGS[@]}"} \
"$@" &
HARBOR_PID=$!
trap 'HARBOR_SIGNALLED=1; kill -TERM "$HARBOR_PID" 2>/dev/null || true' TERM
trap 'HARBOR_SIGNALLED=1; kill -INT "$HARBOR_PID" 2>/dev/null || true' INT
wait "$HARBOR_PID" || HARBOR_EXIT=$?
if [ -n "$HARBOR_SIGNALLED" ]; then
# The first wait was interrupted by the trap; wait again so harbor's real
# exit status (not the shell's signal status) is what we propagate.
wait "$HARBOR_PID" || HARBOR_EXIT=$?
fi
trap - TERM INT
# Stamp the launch-time capture into this task's trial dirs in the job dir(s)
# this run created (the harbor-jobs entries that didn't exist before the
# run). Harbor names job dirs with a timestamp, so new entries are this run's
# output — plus, when several harbor-runs share a cwd, possibly a concurrent
# run's; `apply` is slug-scoped so another task's trials are never stamped
# with this task's inputs.
if [ -n "$STAMP_FILE" ]; then
NEW_JOBS="$(comm -13 <(printf '%s\n' "$JOBS_BEFORE" | sort) <(ls -1 harbor-jobs 2>/dev/null | sort) | sed 's|^|harbor-jobs/|')"
if [ -n "$NEW_JOBS" ]; then
# shellcheck disable=SC2086 # job-dir names are timestamps, never spaced
npx tsx "$SCRIPT_DIR/stamp-trial-inputs.ts" apply "$STAMP_FILE" $NEW_JOBS || \
echo "Warning: could not stamp trial dirs with launch-time input checksums" >&2
fi
rm -f "$STAMP_FILE"
fi
# Repeat the toolkit-managed-file notice AFTER the trial. The preflight copy is
# minutes of harbor output up the scrollback by now, which for a notice nothing
# enforces means nobody reads it. This one lands where the author is looking.
if [ -f "$CHECK_INFRA" ]; then
(cd "$REPO_ROOT" && npx tsx "$CHECK_INFRA" "$TASK_DIR") || true
fi
exit "$HARBOR_EXIT"

View File

@@ -1,286 +0,0 @@
version = 1
[[harness]]
id = "claude-code"
label = "Claude Code"
agent_import_path = "snapshot_agent:SnapshotClaudeCode"
# `[metadata] browser = true` swaps in these: same reduced toolset plus `Read`, so an agent
# given a browser can look at the screenshot it just took. Distinct classes with distinct
# names, because a different toolset is a different agent.
agent_import_path_browser = "snapshot_agent:BrowserSnapshotClaudeCode"
agent_import_path_single_turn_browser = "snapshot_agent:BrowserPreinstalledClaudeCode"
agent_import_path_single_turn = "snapshot_agent:PreinstalledClaudeCode"
import_path_aliases = [
"snapshot_agent:FullToolsetSnapshotClaudeCode",
"snapshot_agent:FullToolsetPreinstalledClaudeCode",
"harbor.agents.installed.claude_code:ClaudeCode",
]
legacy_bare_model_rows = true
default_model = "claude-opus-5[1m]"
model_id_shape = "bare"
effort_kwarg = "reasoning_effort"
effort_default = "max"
fast_kwarg = "fast_mode"
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = true
seed_native = true
seed_atif = true
authoring = true
cli = "claude"
install = "for i in 1 2 3; do curl -fsSL https://claude.ai/install.sh | bash && break; echo \"claude install attempt $i failed; retrying in 10s\" >&2; sleep 10; done"
# No agent_config: claude reduces its toolset with `--tools`, not `-c key=value`, so the
# reduction is a launch flag here and `--tools Bash` in snapshot_agent.py for the trial.
# Two expressions of one intent, which the $RACCOON_AGENT_FLAGS guard cannot police —
# unlike model and effort, which are interpolated from this row.
explore_launch = """exec claude --model '$RACCOON_MODEL' --effort $RACCOON_EFFORT --tools "$RACCOON_TOOLS" --append-system-prompt "$RACCOON_TOOLSET_NOTE" --plugin-dir /workspace/plugins/create-snapshot --dangerously-skip-permissions "$@""""
[[harness]]
id = "codex"
label = "OpenAI Codex CLI"
agent_import_path = "codex_agent:NativeSnapshotCodex"
agent_import_path_single_turn = "codex_agent:SystemNodeCodex"
import_path_aliases = [
"codex_agent:InlineSnapshotCodex",
"harbor.agents.installed.codex:Codex",
]
legacy_bare_model_rows = true
default_model = "gpt-5.6-sol"
model_id_shape = "bare"
effort_kwarg = "reasoning_effort"
effort_default = "max"
key_env = "OPENAI_API_KEY"
base_url_env = "OPENAI_BASE_URL"
proxy_path = "openai/v1"
writes_atif = true
capture = true
seed_native = true
seed_atif = true
authoring = true
cli = "codex"
install = "for i in 1 2 3; do curl -fsSL https://chatgpt.com/codex/install.sh | CODEX_NON_INTERACTIVE=1 sh && break; echo \"codex install attempt $i failed; retrying in 10s\" >&2; sleep 10; done"
skills_dir = "$HOME/.agents/skills"
config_path = "${CODEX_HOME:-$HOME/.codex}/config.toml"
auth_path = "${CODEX_HOME:-$HOME/.codex}/auth.json"
auth_key_env = "OPENAI_API_KEY"
agent_config = """
web_search = "disabled"
[agents]
enabled = false
[tools]
update_plan = { enabled = false }
experimental_request_user_input = { enabled = false }
[features]
goals = false
multi_agent = false
multi_agent_v2 = false
memories = false
external_agent_memory_import = false
"""
# A provider of our own, not the built-in `openai`: codex reserves built-in provider ids,
# and env_http_headers — the only place codex can be told to send the call-origin header —
# is a per-provider setting. base_url has to live in the table with it (a provider without
# one silently falls back to api.openai.com), so harness_refresh_config_keys refreshes
# [model_providers.*] keys as well as root ones.
container_config = """
model_provider = "llm-proxy"
[model_providers.llm-proxy]
name = "LLM proxy"
base_url = "${OPENAI_BASE_URL}"
# A custom provider reads its key from this env var and never from auth.json, so it
# names the one key .env actually carries. That makes the key live per launch rather
# than baked at container create — better than the auth-file path it replaces.
env_key = "ANTHROPIC_API_KEY"
wire_api = "responses"
env_http_headers = { "X-Surge-Client-Metadata" = "LLM_CALL_METADATA" }
"""
explore_config = """
[hooks]
SessionStart = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/save-session-info.mjs" } ] } ]
UserPromptSubmit = [ { hooks = [ { type = "command", command = "/workspace/plugins/create-snapshot/bin/checkpoint-workspace.mjs" } ] } ]
"""
explore_launch = """exec codex $RACCOON_AGENT_FLAGS --model $RACCOON_MODEL -c model_reasoning_effort=$RACCOON_EFFORT ${RACCOON_BROWSER_FLAGS[@]+"${RACCOON_BROWSER_FLAGS[@]}"} --dangerously-bypass-approvals-and-sandbox --dangerously-bypass-hook-trust "$@""""
[[harness]]
id = "gemini-cli"
label = "Gemini CLI"
agent_import_path = "gemini_agent:NativeSnapshotGeminiCli"
agent_import_path_single_turn = "gemini_agent:SystemNodeGeminiCli"
import_path_aliases = ["harbor.agents.installed.gemini_cli:GeminiCli"]
legacy_bare_model_rows = true
default_model = "gemini-3.5-flash"
model_id_shape = "provider/model"
effort_kwarg = "reasoning_effort"
effort_default = "high"
key_env = "GEMINI_API_KEY"
base_url_env = "GEMINI_API_BASE"
proxy_path = "gemini"
writes_atif = true
capture = false
seed_native = true
seed_atif = false
[[harness]]
id = "antigravity-cli"
label = "Antigravity CLI"
agent_import_path = "harness_agents:BenchAntigravity"
import_path_aliases = ["harbor.agents.installed.antigravity_cli:AntigravityCli"]
legacy_bare_model_rows = false
# The prefix is load-bearing: harbor's adapter raises without a "/" in the id.
# agy carries its own model catalogue and DROPS entries between point releases
# (1.1.25 removed gemini-3.5-flash, breaking every run). If trials start failing
# with "not recognized as a known model", run `agy --model bogus --prompt=x` to
# print the current catalogue and update this.
default_model = "google/gemini-3.8-flash"
model_id_shape = "provider/model"
# Not optional: agy refuses a Gemini 3 model with no --effort ("requires --effort
# (available: low, medium, high)"). low/high are safe on pro and flash alike.
effort_kwarg = "reasoning_effort"
effort_default = "high"
key_env = "GEMINI_API_KEY"
base_url_env = "GOOGLE_GEMINI_BASE_URL"
proxy_path = "gemini"
writes_atif = true
capture = false
# agy cannot be handed externally-produced history, so multi-turn tasks must
# hard-fail rather than silently run cold. See work-logs/antigravity-harness.md.
seed_native = false
seed_atif = false
[[harness]]
id = "opencode"
label = "OpenCode"
agent_import_path = "harness_agents:BenchOpenCode"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
flaky_hangs = true
[[harness]]
id = "goose"
label = "Goose"
agent_import_path = "harness_agents:BenchGoose"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "mini-swe-agent"
label = "mini-swe-agent"
agent_import_path = "harness_agents:BenchMiniSweAgent"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "cline-cli"
label = "Cline CLI"
agent_import_path = "harness_agents:BenchCline"
legacy_bare_model_rows = false
model_id_shape = "provider:model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
[[harness]]
id = "crush"
label = "Crush"
agent_import_path = "harness_agents:Crush"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
key_env = "ANTHROPIC_API_KEY"
base_url_env = "ANTHROPIC_BASE_URL"
proxy_path = "anthropic"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
flaky_hangs = true
[[harness]]
id = "amp"
label = "Amp"
agent_import_path = "harness_agents:Amp"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "AMP_API_KEY"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "cursor-cli"
label = "Cursor CLI"
agent_import_path = "harness_agents:BenchCursorCli"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "CURSOR_API_KEY"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "copilot-cli"
label = "GitHub Copilot CLI"
agent_import_path = "harness_agents:BenchCopilotCli"
legacy_bare_model_rows = false
model_id_shape = "bare"
effort_kwarg = ""
key_env = "GITHUB_TOKEN"
writes_atif = true
capture = false
seed_native = false
seed_atif = false
enabled = false
[[harness]]
id = "aider"
label = "Aider"
agent_import_path = "harness_agents:BenchAider"
legacy_bare_model_rows = false
model_id_shape = "provider/model"
effort_kwarg = ""
writes_atif = false
capture = false
seed_native = false
seed_atif = false
enabled = false

View File

@@ -1,29 +0,0 @@
// Types for harness-session.mjs, so TS consumers (its test, snapshot-to-task) see a
// real shape instead of `any`.
export interface Turn {
/** Line index in the native session file. */
index: number;
role: 'user' | 'assistant';
text: string;
/** A slash-command turn, not real conversation. */
isCommand: boolean;
/** This record concluded its turn — the truncation boundary. */
endsTurn: boolean;
}
export interface Session {
harness: string;
rawPath: string;
/** The harness own id for this conversation. */
sessionId: string | null;
lines: string[];
turns: Turn[];
}
export function supportedHarnesses(): string[];
export function readSession(harness: string, recordedPath?: string): Session | null;
export function truncationIndex(turns: Turn[]): number;
export function turnsFromLines(harness: string, lines: string[]): Turn[];
export function linearSnapshotLines(session: Session, startLine?: number): string[];
export function stripAuthoringScaffolding(harness: string, lines: string[]): string[];

View File

@@ -1,325 +0,0 @@
// Locate and read a harness's native conversation, so capture-snapshot can work
// against any harness. Everything else in capture (snapshot.patch, restore.sh,
// annotation, metadata) is harness-agnostic.
//
// The returned session stays in the harness's OWN native format: the seeding design
// hands a native blob back to the same harness, and codex_agent reads the same staged
// /tmp/snapshot-session/session.jsonl path that snapshot_agent does.
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
/**
* @typedef {object} Turn
* @property {number} index line index in the native session file
* @property {'user'|'assistant'} role
* @property {string} text
* @property {boolean} isCommand a slash-command turn, not real conversation
* @property {boolean} endsTurn this record concluded its turn
*/
/**
* @typedef {object} Session
* @property {string} harness
* @property {string} rawPath
* @property {string|null} sessionId the harness's own id for this conversation
* @property {string[]} lines
* @property {Turn[]} turns
*/
// Newest matching file beneath `root`, or null. Ties on mtime break on path so the
// answer is stable — two sessions written in the same millisecond are common.
function newestUnder(root, matches) {
if (!fs.existsSync(root)) return null;
const found = [];
const walk = (dir) => {
let entries;
try {
entries = fs.readdirSync(dir, { withFileTypes: true });
} catch {
return;
}
for (const entry of entries) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) walk(full);
else if (matches(entry.name)) found.push({ full, mtimeMs: fs.statSync(full).mtimeMs });
}
};
walk(root);
if (found.length === 0) return null;
found.sort((a, b) => b.mtimeMs - a.mtimeMs || b.full.localeCompare(a.full));
return found[0].full;
}
// User-role records codex writes that the human did not type: its own environment
// preamble, a `$name` skill invocation, and the SKILL.md body injected in response.
// Matched only at the START of the text, so a turn that merely quotes one is still real
// conversation.
function isCodexCommandText(text) {
const trimmed = (text || '').trimStart();
if (trimmed.startsWith('<skill>') || trimmed.startsWith('<environment_context>')) return true;
return /^\$[\w:.-]+\s*$/.test(trimmed);
}
const HARNESSES = {
'claude-code': {
/** Claude Code records one JSONL per session under ~/.claude/projects/<encoded-cwd>/. */
findSession() {
return newestUnder(path.join(os.homedir(), '.claude', 'projects'), (n) =>
n.endsWith('.jsonl')
);
},
/** Claude names the transcript for its session id. */
sessionId(rawPath) {
return path.basename(rawPath, '.jsonl');
},
/**
* One turn per conversational record. `endsTurn` marks an assistant record that
* concluded its turn — the truncation boundary. Bookkeeping records (attachments,
* file-history, permission-mode) carry no role and are skipped.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let entry;
try {
entry = JSON.parse(line);
} catch {
continue;
}
const role =
entry.type === 'user' ? 'user' : entry.type === 'assistant' ? 'assistant' : null;
if (!role) continue;
const content = entry.message?.content;
const text =
typeof content === 'string'
? content
: Array.isArray(content)
? content
.filter((b) => b && b.type === 'text')
.map((b) => b.text ?? '')
.join('')
: '';
turns.push({
index,
role,
text,
isCommand:
role === 'user' &&
typeof content === 'string' &&
/<command-name>|<command-message>|<local-command-caveat>/.test(content),
endsTurn: role === 'assistant' && entry.message?.stop_reason === 'end_turn',
});
}
return turns;
},
},
codex: {
/** codex writes rollout JSONL under $CODEX_HOME/sessions/<date>/. */
findSession() {
const home = process.env.CODEX_HOME || path.join(os.homedir(), '.codex');
return newestUnder(
path.join(home, 'sessions'),
(n) => n.startsWith('rollout-') && n.endsWith('.jsonl')
);
},
/** `codex resume <id>` resolves the id recorded in session_meta, not the filename. */
sessionId(rawPath, lines) {
for (const line of lines) {
try {
const rec = JSON.parse(line);
if (rec.type === 'session_meta' && rec.payload?.id) return rec.payload.id;
} catch {
continue;
}
}
return null;
},
/**
* codex rollouts carry `response_item` records whose payload is a message with a
* role. An assistant message with no following tool activity ends the turn; codex
* records no stop_reason, so a turn ends where the next user message begins —
* resolved after the fact below.
*/
/** @param {string[]} lines @returns {Turn[]} */
readTurns(lines) {
/** @type {Turn[]} */
const turns = [];
for (const [index, line] of lines.entries()) {
let record;
try {
record = JSON.parse(line);
} catch {
continue;
}
if (record.type !== 'response_item') continue;
const payload = record.payload ?? {};
if (payload.type !== 'message') continue;
const role =
payload.role === 'user' ? 'user' : payload.role === 'assistant' ? 'assistant' : null;
if (!role) continue;
const text = Array.isArray(payload.content)
? payload.content.map((b) => b?.text ?? '').join('')
: typeof payload.content === 'string'
? payload.content
: '';
turns.push({
index,
role,
text,
isCommand: role === 'user' && isCodexCommandText(text),
endsTurn: false,
});
}
// An assistant turn ends where the next user turn starts, or at the end.
for (let i = 0; i < turns.length; i += 1) {
if (turns[i].role !== 'assistant') continue;
const next = turns[i + 1];
turns[i].endsTurn = !next || next.role === 'user';
}
return turns;
},
},
};
/** @returns {string[]} */
export function supportedHarnesses() {
return Object.keys(HARNESSES);
}
/**
* Read the current session for `harness`. Returns null when nothing is found, so the
* caller can report which harness had no conversation to capture.
*/
/**
* @param {string} harness
* @param {string} [recordedPath] transcript recorded by the SessionStart hook; preferred
* over the newest-file scan, which can pick a different session in a busy container.
* @returns {Session | null}
*/
export function readSession(harness, recordedPath) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
const rawPath = recordedPath && fs.existsSync(recordedPath) ? recordedPath : reader.findSession();
if (!rawPath) return null;
const lines = fs.readFileSync(rawPath, 'utf8').trimEnd().split('\n');
return {
harness,
rawPath,
lines,
turns: reader.readTurns(lines),
sessionId: reader.sessionId(rawPath, lines),
};
}
/**
* Index of the last record to keep: the last turn-ending assistant record before the
* final real user turn. Drops the prompt that elicited the failure and the failure
* response, so the test agent inherits context but not the answer.
*
* Returns -1 when there is no such boundary (a one-shot conversation), which callers
* treat as "seed nothing and run cold".
*/
/**
* @param {Turn[]} turns
* @returns {number}
*/
export function truncationIndex(turns) {
let lastUser = -1;
for (const turn of turns) {
if (turn.role === 'user' && !turn.isCommand && turn.text.trim()) lastUser = turn.index;
}
if (lastUser < 0) return -1;
let cut = -1;
for (const turn of turns) {
if (turn.index >= lastUser) break;
if (turn.role === 'assistant' && turn.endsTurn) cut = turn.index;
}
return cut;
}
/**
* Parse already-read lines with a harness's reader, for callers that have the text
* rather than a path.
*
* @param {string} harness
* @param {string[]} lines
* @returns {Turn[]}
*/
export function turnsFromLines(harness, lines) {
const reader = HARNESSES[harness];
if (!reader) {
throw new Error(
`capture: no session reader for harness "${harness}" (have: ${supportedHarnesses().join(', ')})`
);
}
return reader.readTurns(lines);
}
/**
* Lines to stage as the captured `session.jsonl` for a linear-transcript harness:
* everything up to the snapshot invocation, matching what Claude Code stages when it
* cuts at its slash-command line. Dropping the failure-eliciting turn happens later,
* in snapshot-to-task — capture keeps the full conversation.
*
* `startLine` is the rollout length recorded when the snapshot was invoked; without it
* the whole session is kept, which would include the snapshot's own Q&A.
*
* @param {Session} session
* @param {number} [startLine]
* @returns {string[]}
*/
export function linearSnapshotLines(session, startLine) {
if (typeof startLine === 'number' && startLine >= 0) {
return session.lines.slice(0, startLine);
}
return session.lines;
}
/**
* Drop records that describe the AUTHORING container rather than the conversation.
*
* codex records both its skill catalogue (a `developer` turn) and the machine it ran on (a
* `user` turn of `<environment_context>`). Native resume replays records byte-identically,
* so without this the test agent inherits a list of skills it does not have — one described
* as "capture the current conversation and repo state as a snapshot" — and a working
* directory that does not exist in the trial. codex re-injects both for the trial, and base
* instructions travel in `session_meta`, so removing them loses nothing. Claude's fork
* already re-records with the trial's own cwd; this brings codex to the same place.
*
* @param {string} harness
* @param {string[]} lines
* @returns {string[]}
*/
export function stripAuthoringScaffolding(harness, lines) {
if (harness === 'claude-code') return lines;
return lines.filter((raw) => {
let rec;
try {
rec = JSON.parse(raw);
} catch {
return true;
}
const payload = rec?.payload;
if (rec?.type !== 'response_item' || payload?.type !== 'message') return true;
const text = (payload.content ?? [])
.map((block) => (typeof block?.text === 'string' ? block.text : ''))
.join('')
.trim();
// Match the machine-generated shape only — a turn that STARTS with the tag — so a
// worker who quotes one of these strings mid-conversation keeps their turn.
if (payload.role === 'developer') return !text.startsWith('<skills_instructions>');
if (payload.role === 'user') return !text.startsWith('<environment_context>');
return true;
});
}

View File

@@ -1,88 +0,0 @@
/** The task's holistic-rubric template. Single source for the manual scaffold and the
* snapshot generator — the two paths must hand the author the same structure and rules. */
export const HOLISTIC_RUBRIC_SCAFFOLD = `# Holistic Rubric — <task-slug>
The shared grading standard (\`task-shared/grading-standard.md\`, embedded in
\`tests/grader-system-prompt-consolidated.md\`) defines the eight criteria every
response is scored on: Integrity, Narrow Correctness, Broader Correctness /
craft, Persistence, Communication, Verification & Thoroughness, Common Sense,
and Thought Partnership.
This file is the task's holistic rubric. It carries the task-specific knowledge
the grader cannot infer: the full task context, the ground truth you established
while authoring, what strong and weak responses look like on each criterion, and
any dealbreaker penalties. This document must stand alone. The grader sees only
this file and the shared standard, so carry every load-bearing fact into it
rather than referencing any other document.
Replace each bracketed section. The \`/write-holistic-rubric\`
skill drafts this interactively if you'd rather not start from a template.
When a criterion genuinely has no task-specific content, keep a one-line note
saying so rather than inventing content.
## Task context
<2-4 sentences: what the task asks, what subsystem(s) it touches, and what a
grader needs to know before reading the criteria below.>
## Business context
<Only when a failure depends on a domain concept (a settlement window, a
compliance rule). Delete this section otherwise.>
## Ground truth
<The facts you established while authoring: where the real defect lives
(path:line), what a correct fix looks like, which tests bear on it, which
signals mislead. The grader trusts this section over its own reading.>
## Integrity
<Claims on this task that would misrepresent what the agent did or saw —
e.g. asserting a file says X after reading it say Y. Charge only on an
observable basis.>
## Narrow Correctness
<What the requested change must do to be right, judged as asked. Anchors a
working result must satisfy, checkable by path:line.>
## Broader Correctness / the craft of software engineering
<Craft expectations specific to this codebase: patterns to follow, tests to
add, places a shortcut would rot.>
## Persistence
<What "kept going appropriately" looks like here: the dead ends worth
exhausting, and where stopping to ask is the better call.>
## Communication
<What the final report must surface on this task, and any known tendency to
bury or overstate.>
## Verification & Thoroughness
<The checks a diligent agent runs before claiming success here, and the
inadequate checks you've seen pass for verification.>
## Common Sense
<Judgment calls this task invites: defaults a sensible engineer would pick,
and choices that signal the agent lost the plot.>
## Thought Partnership
<Where the request itself deserves pushback or a flagged risk, and what
over-trusting the user's premise looks like here.>
## Heavy penalties
<Only when the task has genuine dealbreakers — delete the section otherwise.
Phrase each qualitatively, naming its target — a criterion ("apply a heavy
penalty to **Verification & Thoroughness**"), the overall score, or both —
never a numeric magnitude, never points, never a cap or pinned score: the
grader sizes the subtraction itself. Always state the behavior that does NOT trip the penalty.
Never describe how criteria combine into an overall score.>
`;

View File

@@ -1,34 +0,0 @@
# shellcheck shell=bash
# call-origin.sh — build the X-Surge-Client-Metadata header value.
#
# Which surface a proxy call came from (a trial agent, the grader, Explore, a
# dev box). Separate from llm-proxy-env.sh, which carries the project id and the
# proxy routes: those are platform-internal, this is not, so this file is the
# half that ships in the worker toolkit — worker runs go through the same proxy
# and are attributed the same way.
#
# . scripts/lib/call-origin.sh
# meta="$(LLM_CALL_ORIGIN=harbor-grading call_origin_metadata)"
#
# The proxy rejects the WHOLE CALL over a malformed metadata header (400
# invalid_client_metadata), so an origin that is not a plain slug yields an
# empty string and the caller sends no header at all: losing attribution beats
# failing the call.
CALL_ORIGIN_HEADER="X-Surge-Client-Metadata"
# An unlabelled call is still a real call, so it gets a bucket rather than no
# header: a missing origin in the audit log then means an unplumbed surface.
DEFAULT_CALL_ORIGIN="local"
# Compact JSON for the header, or empty when LLM_CALL_ORIGIN is unusable.
# Only a slug matching this pattern is ever interpolated, so nothing needs
# JSON-escaping and this stays dependency-free (it is sourced in worker
# containers, which have no python).
call_origin_metadata() {
local origin="${LLM_CALL_ORIGIN:-$DEFAULT_CALL_ORIGIN}"
case "$origin" in
"" | *[!a-z0-9._-]* | [!a-z0-9]*) return 0 ;;
esac
[ "${#origin}" -le 64 ] || return 0
printf '{"origin":"%s"}' "$origin"
}

View File

@@ -1,19 +0,0 @@
import { existsSync } from 'node:fs';
(function checkDevcontainer() {
if (process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE === '1') return;
process.env._RACCOON_TOOLKIT_DEVCONTAINER_CHECK_DONE = '1';
const inContainer = process.env.IN_DEVCONTAINER === '1' || existsSync('/.dockerenv');
if (inContainer) return;
if (process.env.SUPPRESS_DEVCONTAINER_WARNING === '1') return;
if (process.env.CI === 'true' || process.env.CI === '1') return;
const yellow = '\x1b[33m';
const reset = '\x1b[0m';
process.stderr.write(
`${yellow}Warning: this script is meant to run inside the toolkit devcontainer.${reset}\n` +
` Reopen this toolkit folder in its devcontainer and run the command again.\n` +
` (suppress with SUPPRESS_DEVCONTAINER_WARNING=1)\n`
);
})();

View File

@@ -1,36 +0,0 @@
"""How codex is handed its API key, kept out of codex_agent so it is testable without
harbor (whose venv has no pytest, so anything importing it SKIPs in CI).
codex reads its key from `$CODEX_HOME/auth.json` and its proxy URL from config.toml —
`OPENAI_API_KEY` / `OPENAI_BASE_URL` in the environment are both ignored, verified against
0.146.0 and 0.152.0 (an env-var-only run sends no `authorization` header at all).
"""
from __future__ import annotations
import json
import shlex
# Characters harbor's own auth.json writer cannot survive: it interpolates the key into a
# shell heredoc, so `"` closes the JSON string and `\` starts an escape.
_UNESCAPABLE = '"\\\n\r'
AUTH_JSON_ENV_VAR = "RACCOON_CODEX_AUTH_JSON"
def auth_json_setup(key: str, remote_auth_path: str) -> tuple[dict[str, str], str]:
"""The one extra env var — returned separately so it reaches ONLY the setup exec — plus
shell writing a parseable auth.json. Subshell: the umask must not outlive this write."""
env = {AUTH_JSON_ENV_VAR: json.dumps({"OPENAI_API_KEY": key})}
command = (
f"(umask 077; printf '%s\\n' \"${AUTH_JSON_ENV_VAR}\" "
f">{shlex.quote(remote_auth_path)})\n"
)
return env, command
def unescapable_chars(key: str) -> list[str]:
"""Which characters in `key` harbor's stock heredoc writer would corrupt — empty for
every ordinary key, so the caller can refuse instead of 401ing three layers down."""
return sorted({c for c in _UNESCAPABLE if c in key})

View File

@@ -1,43 +0,0 @@
/**
* Recursive copy for scripts that must not call `cpSync`: it fails EACCES
* against a macOS docker bind mount, where the toolkit's job dirs live.
*/
import {
chmodSync,
copyFileSync,
lstatSync,
mkdirSync,
readdirSync,
readlinkSync,
rmSync,
statSync,
symlinkSync,
} from 'fs';
import { join } from 'path';
/** Copy one entry — symlink, directory or file — preserving its mode. */
export function copyPath(src: string, dest: string) {
const st = lstatSync(src);
if (st.isSymbolicLink()) {
rmSync(dest, { force: true });
symlinkSync(readlinkSync(src), dest);
return;
}
if (st.isDirectory()) {
copyTree(src, dest);
return;
}
// Unlink first: copyFileSync onto an existing file keeps that file's mode.
rmSync(dest, { force: true });
copyFileSync(src, dest);
chmodSync(dest, statSync(src).mode & 0o777);
}
/** Copy `src`'s contents into `dest`, creating `dest` if it doesn't exist. */
export function copyTree(src: string, dest: string) {
mkdirSync(dest, { recursive: true });
for (const entry of readdirSync(src, { withFileTypes: true })) {
copyPath(join(src, entry.name), join(dest, entry.name));
}
}

View File

@@ -1,137 +0,0 @@
#!/bin/sh
# Restrict this container's DNS to the hosts in DNSJAIL_ALLOW (space-separated), leaving
# every other name unresolvable. Runs as root, inside the container.
#
# Baked into the task images and invoked by the agent (scripts/dnsjail.py); shipped to the
# Explore container by the toolkit packaging. Both surfaces run this same file. Supplied from
# outside: DNSJAIL_ALLOW, the hosts the agent will actually dial -- every one must resolve or
# no jail happens -- and DNSJAIL_ALLOW_EXTRA, nice-to-haves that only warn if they do not.
#
# An unreachable model endpoint is a dead trial or a dead session, so nothing here is
# applied before it is verified, and any doubt leaves the container's DNS untouched.
set -u
STATE=/tmp/.dnsjail
CONTROL=example.com # must NOT resolve through us; proves we reached our own filter
bounded() { if command -v timeout >/dev/null 2>&1; then timeout 5 "$@"; else "$@"; fi; }
# Exact match: docker's own embedded resolver is 127.0.0.11, which a prefix match reads as
# already-jailed — and then resolv.orig is never captured, so unjail has nothing to restore.
jailed_now() { grep -qE '^nameserver[[:space:]]+127\.0\.0\.1[[:space:]]*$' /etc/resolv.conf 2>/dev/null; }
# Stop only the dnsmasq we started, so a declined run leaves nothing bound on :53 that a
# later run could mistake for its own filter.
drop_ours() {
if [ -s "$STATE/dnsmasq.pid" ]; then
pid=$(cat "$STATE/dnsmasq.pid")
# /tmp survives docker stop/start but pids restart at 1, so last boot's pid may now be
# some service's child. Confirm it is dnsmasq before signalling it.
case "$(cat "/proc/$pid/comm" 2>/dev/null)" in
dnsmasq) kill "$pid" 2>/dev/null || true ;;
esac
rm -f "$STATE/dnsmasq.pid" 2>/dev/null || true
fi
}
# Never `exit`: a caller may source this, so bailing out has to fall through rather than
# end the caller's shell.
dnsjail_apply() {
required="${DNSJAIL_ALLOW:-}"
extra="${DNSJAIL_ALLOW_EXTRA:-}"
allow=$(echo $required $extra) # unquoted: collapses to a single-spaced word list
# A blank required list means no model endpoint was found: jailing would strand the agent.
set -- $required
[ $# -gt 0 ] || return 0
# Already jailed by us, with our resolver alive and the same allowlist? Do nothing. Tearing
# down and rebinding :53 races the kernel releasing the socket, and losing that race ends
# in a fail-open restore -- so a second apply (the codex fresh path, rejail, run-app) would
# silently UNjail a working container.
if jailed_now && [ -s "$STATE/dnsmasq.pid" ] &&
[ "$(cat "/proc/$(cat "$STATE/dnsmasq.pid")/comm" 2>/dev/null)" = "dnsmasq" ] &&
[ "$(cat "$STATE/allow" 2>/dev/null)" = "$required" ] &&
[ "$(cat "$STATE/allow-extra" 2>/dev/null)" = "$extra" ]; then
return 0
fi
# The state dir has to work first: it holds what unjail restores, and a failed write here
# is what would otherwise truncate /etc/resolv.conf. Sticky world-writable so run-app,
# running as the container user in Explore, can drop its own lift markers.
mkdir -p "$STATE" 2>/dev/null || return 0
chmod 1777 "$STATE" 2>/dev/null || true
: > "$STATE/.probe" 2>/dev/null || return 0
rm -f "$STATE/.probe" 2>/dev/null || true
# Never forward to ourselves. Re-applying to an already-jailed container would otherwise
# read 127.0.0.1 out of resolv.conf and point dnsmasq at its own socket, blackholing
# every name.
src=/etc/resolv.conf
if jailed_now && [ -s "$STATE/resolv.orig" ]; then src="$STATE/resolv.orig"; fi
up=$(awk '/^nameserver[ \t]+[0-9]+\./{print $2; exit}' "$src" 2>/dev/null)
[ "$up" = "127.0.0.1" ] && up=""
if [ -n "$up" ] && command -v dnsmasq >/dev/null 2>&1; then
srv=""
for h in $allow; do srv="$srv --server=/$h/$up"; done
drop_ours
# cache-size=0: every lookup goes upstream, so a jailed container sees what an unjailed
# one would rather than an answer this resolver decided to keep.
dnsmasq --no-resolv --no-hosts --listen-address=127.0.0.1 --bind-interfaces \
--cache-size=0 --pid-file="$STATE/dnsmasq.pid" --address=/#/ $srv \
>/dev/null 2>>"$STATE/dnsmasq.err" || true
fi
# Ask the resolver directly: the model endpoint must answer and the control must not --
# otherwise we are looking at somebody else's resolver, not our filter. Only the FIRST
# host gates the jail: an extra host that CNAMEs outside the allowlist cannot resolve
# through the catch-all, and one of those must not silently disable the whole jail.
live=1
for h in $required; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 || { live=""; break; }
done
if [ -n "$live" ] && bounded nslookup "$CONTROL" 127.0.0.1 >/dev/null 2>&1; then live=""; fi
# The extras are reported, never fatal: one that CNAMEs outside the allowlist cannot
# resolve through the catch-all, and must not take the whole jail down with it.
if [ -n "$live" ]; then
for h in $extra; do
bounded nslookup "$h" 127.0.0.1 >/dev/null 2>&1 ||
echo "dns-jail: $h does not resolve through the jail (CNAME outside the allowlist?)" >&2
done
fi
if [ -z "$live" ]; then
# Say why. A silent decline is indistinguishable from a jail that worked, and the
# reason is usually one line from dnsmasq (gVisor sandboxes, for instance, have no
# AF_NETLINK, so dnsmasq cannot start there at all).
echo "dns-jail: declined, this container keeps normal network access${DNSJAIL_WHY:-}" >&2
[ -s "$STATE/dnsmasq.err" ] && sed 's/^/dns-jail: /' "$STATE/dnsmasq.err" >&2
drop_ours
# Failing open has to mean actually open, including when an earlier run left this
# container jailed.
if jailed_now && [ -s "$STATE/resolv.orig" ]; then
cat "$STATE/resolv.orig" > /etc/resolv.conf 2>/dev/null || true
fi
return 0
fi
# Capture what unjail restores — but never overwrite it with an already-jailed file, which
# would leave unjail a permanent no-op.
if ! jailed_now; then
cp /etc/resolv.conf "$STATE/resolv.orig" 2>/dev/null || return 0
fi
printf '%s\n' "$required" > "$STATE/allow" 2>/dev/null || true
printf '%s\n' "$extra" > "$STATE/allow-extra" 2>/dev/null || true
# A marker from a run-app that was killed would otherwise keep the jail disarmed forever.
rm -rf "$STATE/lifts" 2>/dev/null || true
# /etc/resolv.conf is a bind mount, so it is truncated in place, never renamed over —
# which means the replacement has to be complete BEFORE the write starts. Keep every
# non-nameserver directive docker set (options, search).
{ printf 'nameserver 127.0.0.1\n'
grep -vE '^[[:space:]]*nameserver' /etc/resolv.conf
} > "$STATE/resolv.jailed" 2>/dev/null
[ -s "$STATE/resolv.jailed" ] || return 0
cat "$STATE/resolv.jailed" > /etc/resolv.conf
}
dnsjail_apply || true

View File

@@ -1,328 +0,0 @@
#!/bin/bash
# Read the harness registry and derive per-harness credentials from it.
#
# Source it — the whole point is exporting into the caller's environment, which a subshell
# would lose:
#
# HARNESS_SCRIPTS_DIR=/workspace/scripts . /workspace/scripts/lib/harness-credentials.sh
# harness_setup_credentials
#
# Three callers: `harbor-run`, which needs only this; `refresh-harness-auth`, which
# re-derives and rewrites the auth files before an interactive launch; and
# `setup-harnesses.sh`, which sources it and adds installs, config writing and launchers
# on top.
#
# No -e here — this file is SOURCED, and shell options belong to the caller's shell (both
# post-creates run with -e). An unguarded failure below therefore aborts container
# creation, which is why every failure site is individually guarded rather than relying on
# this line.
set -uo pipefail
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# The registry is read with tomllib (stdlib from 3.11), and `python3` is not always new
# enough — macOS ships 3.9, and a container may symlink an older managed interpreter. Pick
# the first one that can actually import it rather than assuming.
_raccoon_python() {
local p
for p in "${RACCOON_PYTHON:-}" python3 python3.13 python3.12 python3.11; do
[ -n "$p" ] || continue
command -v "$p" >/dev/null 2>&1 || continue
if "$p" -c "import tomllib" >/dev/null 2>&1; then
printf '%s' "$p"
return 0
fi
done
return 1
}
_harness_query() {
local py
py=$(_raccoon_python) || return 1
"$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" "$@"
}
# Drop every whitespace character from a value read out of .env. A Windows-saved .env leaves a
# \r on each value, which reaches the proxy as a 401; no key or base URL legitimately contains
# whitespace anywhere, so deleting rather than trimming needs no cases.
_harness_trim() {
local out
# Fall back to the raw value: a trim that cannot run must never turn a working key into an
# empty one, which is what an unavailable `tr` would otherwise do to every caller.
out="$(printf '%s' "$1" | tr -d '[:space:]' 2>/dev/null)" || out="$1"
printf '%s' "${out:-$1}"
}
# The proxy root: the worker's ANTHROPIC_BASE_URL minus its provider path.
_harness_proxy_root() {
local base_url
base_url="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
[ -n "$base_url" ] || return 1
base_url="${base_url%"${base_url##*[!/]}"}"
# ".../llm_proxy/projects/<id>/anthropic" -> ".../llm_proxy/projects/<id>", so each
# harness's proxy_path composes onto the project route. Requires a path to strip: a base
# URL that is a bare host with no path — a provider's own API root rather than the
# proxy — would yield "https:/", handed to codex as a base URL and failing obscurely.
case "${base_url#*://}" in
*/*) printf '%s' "${base_url%/*}" ;;
*) return 2 ;;
esac
}
harness_setup_credentials() {
# `|| rc=$?` and not a bare assignment: this is sourced into a `set -e` shell (see the
# note at the top), and a bare failing assignment would exit the caller's post-create
# outright — silently, since the failure paths below are what do the explaining.
local root rc=0
root="$(_harness_proxy_root)" || rc=$?
if [ "$rc" -ne 0 ]; then
if [ "$rc" -eq 2 ]; then
echo "harness-setup: ANTHROPIC_BASE_URL (${ANTHROPIC_BASE_URL:-}) has no provider" >&2
echo "harness-setup: path, so it is not the proxy URL other harnesses derive their" >&2
echo "harness-setup: credentials from. claude will work; codex will not be" >&2
echo "harness-setup: authenticated. Use the base URL you were given." >&2
else
echo "harness-setup: ANTHROPIC_BASE_URL unset — skipping credential derivation" >&2
fi
return 0
fi
ANTHROPIC_BASE_URL="$(_harness_trim "${ANTHROPIC_BASE_URL:-}")"
export ANTHROPIC_BASE_URL
local key
key="$(_harness_trim "${ANTHROPIC_API_KEY:-}")"
if [ -z "$key" ]; then
echo "harness-setup: ANTHROPIC_API_KEY unset — skipping credential derivation" >&2
return 0
fi
# harbor-run sources .env itself and passes ANTHROPIC_* through to the trial sandbox, so
# cleaning only the derived per-harness copies would leave a claude trial carrying the CR.
export ANTHROPIC_API_KEY="$key"
local id key_env base_url_env proxy_path
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
[ -n "$key_env" ] || continue
# ${!name} is an indirect expansion. Only set when empty: an explicit key wins.
if [ -z "${!key_env:-}" ]; then
export "$key_env=$key"
fi
if [ -n "$base_url_env" ] && [ -n "$proxy_path" ] && [ -z "${!base_url_env:-}" ]; then
export "$base_url_env=$root/$proxy_path"
fi
echo "harness-setup: $id credentials ready ($key_env, ${base_url_env:-no base url})" >&2
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
}
# Write the auth file for harnesses that read credentials from disk rather than $ENV.
harness_write_auth() {
local id auth_path key_env target key py
py=$(_raccoon_python) || {
echo "harness-setup: no python3.11+ with tomllib — skipping auth files" >&2
return 0
}
while IFS=$'\t' read -r id auth_path key_env; do
[ -n "$auth_path" ] && [ -n "$key_env" ] || continue
# Last mile: an explicit OPENAI_API_KEY bypasses the derivation above, so trim here
# too — this is the value that reaches the file the harness authenticates with.
key="$(_harness_trim "${!key_env:-}")"
if [ -z "$key" ]; then
echo "harness-setup: $key_env unset — skipping $id auth file" >&2
continue
fi
target=$(eval "printf '%s' \"$auth_path\"") || {
echo "harness-setup: WARNING $id auth_path could not be expanded — skipping" >&2
continue
}
mkdir -p "$(dirname "$target")" || {
echo "harness-setup: WARNING $id auth dir not creatable — skipping $target" >&2
continue
}
# json.dumps, not printf: a key containing a quote or backslash would otherwise
# produce a file the CLI cannot parse, and the failure would surface as an auth
# error rather than a malformed file.
# 0600 tmp + rename, never a redirect onto the target: a redirect truncates the live
# file first, so a write dying mid-flight leaves codex an EMPTY auth.json.
if ! RACCOON_AUTH_K="$key_env" RACCOON_AUTH_V="$key" RACCOON_AUTH_TARGET="$target" \
"$py" -c 'import json, os
target = os.environ["RACCOON_AUTH_TARGET"]
tmp = target + ".raccoon-tmp." + str(os.getpid())
try:
with os.fdopen(os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600), "w") as fh:
json.dump({os.environ["RACCOON_AUTH_K"]: os.environ["RACCOON_AUTH_V"]}, fh)
fh.write("\n")
os.replace(tmp, target)
except OSError:
try:
os.unlink(tmp)
except OSError:
pass
raise SystemExit(1)
'; then
echo "harness-setup: WARNING $id auth file NOT written — $target unwritable." >&2
echo "harness-setup: the key already on disk (if any) is left untouched." >&2
continue
fi
echo "harness-setup: $id auth -> $target" >&2
done < <(_harness_query --auth-files 2>/dev/null || true)
}
# Re-set just the root keys of a harness's config file (codex's `openai_base_url`),
# leaving every other line — the explore surface's [hooks] table included — untouched.
harness_refresh_config_keys() {
local id config_path blob target py
py=$(_raccoon_python) || return 0
# The surface only decides what a CREATE writes. An update takes the root keys off the
# front of the same blob, so a surface's tables survive byte-for-byte either way.
while IFS=$'\t' read -r id config_path blob; do
[ -n "$config_path" ] && [ -n "$blob" ] || continue
target=$(eval "printf '%s' \"$config_path\"") || continue
mkdir -p "$(dirname "$target")" || continue
if printf '%s' "$blob" | base64 -d |
RACCOON_CONFIG_TARGET="$target" "$py" -c '
import os, re, sys, tomllib
HEADER = "# Generated from harness-registry.toml — edits here are overwritten."
target = os.environ["RACCOON_CONFIG_TARGET"]
text = sys.stdin.read()
# Empty counts as unresolved: writing an empty base URL would break a container whose
# config is currently right, which is the one thing this must never do.
if [m for m in re.finditer(r"\$\{(\w+)\}", text) if not os.environ.get(m.group(1))]:
raise SystemExit(1)
text = os.path.expandvars(text)
# Root keys, plus keys inside a [model_providers.*] table: codex reserves its built-in
# provider ids, so the proxy URL it must follow lives in a provider table, not at the
# root. Every other table, [hooks] on the explore surface included, is left alone.
REFRESHABLE_TABLE = re.compile(r"\[model_providers\.[^]]+\]$")
wanted = []
section = None
for line in text.splitlines():
stripped = line.strip()
if stripped.startswith("["):
section = stripped if REFRESHABLE_TABLE.match(stripped) else False
continue
if section is False:
continue
m = re.match(r"\s*\"?([A-Za-z0-9_.-]+)\"?\s*=", line)
if m:
wanted.append((section, m.group(1), line.rstrip()))
if not wanted:
raise SystemExit(0)
def section_path(header):
"""[model_providers.llm-proxy] -> ("model_providers", "llm-proxy")."""
return tuple(header.strip("[]").split("."))
def lookup(doc, header, key):
"""The value a parsed config holds for a wanted key, or KeyError."""
node = doc
if header:
for part in section_path(header):
node = node[part]
return node[key]
mode = None
if os.path.exists(target):
try:
with open(target, encoding="utf-8") as fh:
lines = fh.read().splitlines()
mode = os.stat(target).st_mode & 0o777
except OSError:
raise SystemExit(1)
def span(header):
"""The line range a section owns, or None when the file has no such section.
Root is everything above the first table header: a key appended below one
would be reparented into it, so searches and inserts stay inside the span.
"""
heads = [i for i, l in enumerate(lines) if l.lstrip().startswith("[")]
if header is None:
return 0, (heads[0] if heads else len(lines))
at = next((i for i in heads if lines[i].strip() == header), None)
if at is None:
return None
after = next((i for i in heads if i > at), len(lines))
return at + 1, after
# Grouped, root first, so a section this file lacks can be written whole.
grouped = {}
for header, key, line in wanted:
grouped.setdefault(header, []).append((key, line))
ordered = sorted(grouped, key=lambda h: (h is not None, h or ""))
changed = False
for header in ordered:
if span(header) is None:
# A config written before this section existed. Write the whole table
# rather than leave a root key naming a provider that is not there.
if lines and lines[-1].strip():
lines.append("")
lines.append(header)
lines.extend(line for _, line in grouped[header])
changed = True
continue
for key, line in grouped[header]:
# Re-read the span: an insert for an earlier key moved it.
start, end = span(header)
# The quoted spelling is the same key: replace rather than duplicate.
pat = re.compile(r"\s*\"?" + re.escape(key) + r"\"?\s*=")
at = next((i for i in range(start, end) if pat.match(lines[i])), None)
if at is None:
if end < len(lines) and lines[end].strip():
lines.insert(end, "")
lines.insert(end, line)
changed = True
elif lines[at] != line:
lines[at] = line
changed = True
if not changed:
raise SystemExit(0)
out = "\n".join(lines).rstrip("\n") + "\n"
else:
# No file means container-create could not write one, so write what it would have:
# on the explore surface that is the capture hooks too, not just the root keys.
out = HEADER + "\n" + text
try:
doc = tomllib.loads(out)
except tomllib.TOMLDecodeError:
raise SystemExit(1)
# Parsing is not enough: a line edit can land inside a multi-line value, which still
# parses while leaving the key unset. Require every key to have landed on the value the
# blob asks for, in its own section — skipping sections this file does not carry.
blob_doc = tomllib.loads(text)
for header, key, _ in wanted:
try:
expected = lookup(blob_doc, header, key)
except (KeyError, TypeError):
raise SystemExit(1)
try:
got = lookup(doc, header, key)
except (KeyError, TypeError):
if header is None:
raise SystemExit(1)
continue
if got != expected:
raise SystemExit(1)
# Pid-suffixed: two launches at once must not write the same scratch path.
tmp = target + ".raccoon-tmp." + str(os.getpid())
try:
with open(tmp, "w", encoding="utf-8") as fh:
fh.write(out)
if mode is not None:
os.chmod(tmp, mode)
os.replace(tmp, target)
except OSError:
try:
os.unlink(tmp)
except OSError:
pass
raise SystemExit(1)
'; then
echo "harness-setup: $id config keys refreshed -> $target" >&2
fi
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
}

View File

@@ -1,418 +0,0 @@
"""harness_registry.py — Python loader for ``scripts/harness-registry.toml``.
The ONE loader for the registry: TS callers shell into ``resolve_harness.py`` rather than
parse the TOML themselves, which is why the toolkit ships no TOML parser for TS (its
package.json has no zod/smol-toml).
This module supersedes ``benchmark_models_lib``'s ``HARNESS_BY_IMPORT_PATH`` and
``LEGACY_BARE_MODEL_AGENTS``; those should read from here rather than keep private
copies.
Harbor-free and dependency-free (stdlib ``tomllib``) so it can be imported from a
sandbox agent, a plain unit test, or the devcontainer python alike.
"""
from __future__ import annotations
import tomllib
from dataclasses import dataclass, field
from pathlib import Path
REGISTRY_PATH = Path(__file__).resolve().parent.parent / "harness-registry.toml"
MODEL_ID_SHAPES = frozenset({"bare", "provider/model", "provider:model"})
@dataclass(frozen=True)
class Harness:
"""One harness, as declared in harness-registry.toml."""
id: str
label: str
agent_import_path: str
model_id_shape: str
writes_atif: bool
capture: bool
seed_native: bool
seed_atif: bool
agent_import_path_single_turn: str | None = None
# Browser-opt-in variants (`[metadata] browser = true`). A harness that has no variant
# keeps its normal class: codex, for instance, gains the browser and its disclosure but
# has no `Read` equivalent to switch toolsets for.
agent_import_path_browser: str | None = None
agent_import_path_single_turn_browser: str | None = None
import_path_aliases: tuple[str, ...] = ()
legacy_bare_model_rows: bool = False
default_model: str | None = None
effort_kwarg: str = ""
effort_default: str | None = None
# Agent kwarg that opts a trial into the harness's fast/priority serving mode
# (claude-code: fast mode). Empty means the harness has none and --fast refuses.
fast_kwarg: str = ""
key_env: str | None = None
base_url_env: str | None = None
proxy_path: str | None = None
flaky_hangs: bool = False
enabled: bool = True
# Worker-container fields; see the registry header.
authoring: bool = False
cli: str | None = None
install: str | None = None
skills_dir: str | None = None
auth_path: str | None = None
auth_key_env: str | None = None
explore_launch: str | None = None
config_path: str | None = None
# Config the harness needs wherever it runs, trial sandbox included.
agent_config: str | None = None
# Config for both worker containers (explore and authoring).
container_config: str | None = None
# Config for the EXPLORE container only — the capture hooks, whose commands ship in
# explore/plugins/. Writing them in authoring would register hooks against files that
# are not there, firing on every prompt.
explore_config: str | None = None
# Fields added for a later phase, kept verbatim so this loader doesn't have to
# be edited in lockstep with the schema.
extra: dict = field(default_factory=dict, compare=False)
def agent_import_path_for(self, *, multi_turn: bool, browser: bool = False) -> str:
"""Agent class to launch. Multi-turn tasks need the resuming class; a
single-turn task given it would try to resume a session that isn't there.
``browser`` selects the opt-in variant, which for claude also carries the ``Read``
built-in — a different toolset is a different agent, so it is a different class with
its own name rather than a flag on the canonical one. Harnesses without a variant fall
through to their normal class."""
if browser:
variant = (
self.agent_import_path_browser
if multi_turn
else (self.agent_import_path_single_turn_browser or self.agent_import_path_browser)
)
if variant:
return variant
if multi_turn:
return self.agent_import_path
return self.agent_import_path_single_turn or self.agent_import_path
def row_label(self, model: str) -> str:
"""Row identity for one trial: bare model for legacy harnesses (so
published manifests keep their labels), else ``<harness>:<model>``."""
return model if self.legacy_bare_model_rows else f"{self.id}:{model}"
def agent_config_overrides(self) -> dict[str, str]:
"""``agent_config`` as flat ``dotted.key -> value`` pairs in CLI-override form.
Values are rendered bare — ``disabled``, not ``"disabled"``. Every consumer
interpolates these into a shell command, which would strip the quotes anyway;
emitting them would only make the result depend on how many shell layers the
string crosses. Bare is what the CLIs document (``-c model="o3"`` reaches the
binary as ``model=o3``).
These settings ride the command line as ``-c dotted.key=value`` everywhere the
harness runs, never a config file. A trial sandbox rules the file out: the
harness's own runner appends root keys to it, and TOML has no way back to the
root scope once a table has opened, so a table we appended would swallow them.
Overrides compose in any order and beat the file, so the same rendering serves
the explore launcher too — one declaration, one mechanism.
"""
if not self.agent_config:
return {}
try:
parsed = tomllib.loads(self.agent_config)
except tomllib.TOMLDecodeError as exc:
raise HarnessRegistryError(
f"{self.id}: agent_config is not valid TOML ({exc})"
) from exc
flat: dict[str, str] = {}
def walk(node: dict, prefix: str) -> None:
for key, value in node.items():
path = f"{prefix}{key}"
if isinstance(value, dict):
walk(value, f"{path}.")
elif isinstance(value, bool):
flat[path] = "true" if value else "false"
elif isinstance(value, (int, float)):
flat[path] = str(value)
elif isinstance(value, str):
if value != value.strip() or any(c in value for c in " \"'\\"):
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has a value needing "
"shell quoting, which the -c override form cannot carry"
)
flat[path] = value
else:
raise HarnessRegistryError(
f"{self.id}: agent_config key {path!r} has type "
f"{type(value).__name__}, which has no -c override form"
)
walk(parsed, "")
return flat
def container_config_text(self, *, surface: str) -> str | None:
"""Config file body for a worker container. `surface` is "explore" or
"authoring"; explore additionally gets `explore_config`. Root keys come from
`container_config` first, so appending a table section stays valid TOML."""
parts = [self.container_config]
if surface == "explore":
parts.append(self.explore_config)
kept = [part.strip("\n") for part in parts if part and part.strip()]
return "\n\n".join(kept) + "\n" if kept else None
def agent_config_flags(self) -> str:
"""``agent_config`` as a ``-c key=value`` command-line string."""
return " ".join(
f"-c {key}={value}"
for key, value in sorted(self.agent_config_overrides().items())
)
def explore_launch_command(self) -> str | None:
"""``explore_launch`` with the registry's own values substituted in.
The worker's Explore session and the trial must run the same agent, so the
model, effort and reductions are declared once here and rendered into both.
A literal in the launch string would be a second declaration, and the two
would drift the first time one of them was updated alone.
Only these three placeholders are substituted; ``$@`` and
``$RACCOON_TOOLSET_NOTE`` stay for the launcher's own shell to expand.
"""
if not self.explore_launch:
return None
return (
self.explore_launch.replace("$RACCOON_AGENT_FLAGS", self.agent_config_flags())
.replace("$RACCOON_MODEL", self.default_model or "")
.replace("$RACCOON_EFFORT", self.effort_default or "")
)
def known_import_paths(self) -> tuple[str, ...]:
paths = [self.agent_import_path, *self.import_path_aliases]
if self.agent_import_path_single_turn:
paths.append(self.agent_import_path_single_turn)
return tuple(paths)
_KNOWN_FIELDS = frozenset(
{
"id",
"label",
"agent_import_path",
"agent_import_path_single_turn",
"agent_import_path_browser",
"agent_import_path_single_turn_browser",
"import_path_aliases",
"legacy_bare_model_rows",
"default_model",
"model_id_shape",
"effort_kwarg",
"effort_default",
"fast_kwarg",
"key_env",
"base_url_env",
"proxy_path",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
"flaky_hangs",
"enabled",
"authoring",
"cli",
"install",
"skills_dir",
"auth_path",
"auth_key_env",
"explore_launch",
"config_path",
"agent_config",
"container_config",
"explore_config",
}
)
_REQUIRED_FIELDS = (
"id",
"label",
"agent_import_path",
"model_id_shape",
"writes_atif",
"capture",
"seed_native",
"seed_atif",
)
class HarnessRegistryError(ValueError):
"""Malformed registry. Raised rather than tolerated: a broken registry is a
broken deployment, and silently defaulting would pick the wrong agent."""
def _references_agent_flags(launch: str) -> bool:
return "$RACCOON_AGENT_FLAGS" in launch or "${RACCOON_AGENT_FLAGS}" in launch
@dataclass(frozen=True)
class HarnessRegistry:
version: int
harnesses: tuple[Harness, ...]
def all(self) -> tuple[Harness, ...]:
return self.harnesses
def enabled(self) -> tuple[Harness, ...]:
return tuple(h for h in self.harnesses if h.enabled)
def authoring(self) -> tuple[Harness, ...]:
"""Harnesses a worker can author with — what the worker containers install.
Narrower than enabled(): a harness can be runnable in a trial without having
an authoring story (no CLI to converse with, or no capture)."""
return tuple(h for h in self.harnesses if h.enabled and h.authoring)
def find(self, harness_id: str) -> Harness | None:
return next((h for h in self.harnesses if h.id == harness_id), None)
def require(self, harness_id: str) -> Harness:
harness = self.find(harness_id)
if harness is not None:
return harness
available = ", ".join(sorted(h.id for h in self.enabled()))
raise HarnessRegistryError(
f'Unknown harness "{harness_id}". Available: {available}'
)
def by_import_path(self, agent: str) -> Harness | None:
"""Resolve an agent identity — a ``name()`` or import path from
``result.json`` ``config.agent``, or a manifest row — to its harness."""
needle = (agent or "").strip()
if not needle:
return None
for harness in self.harnesses:
if needle == harness.id or needle in harness.known_import_paths():
return harness
return None
def _build(entry: dict, index: int) -> Harness:
for name in _REQUIRED_FIELDS:
if name not in entry:
raise HarnessRegistryError(
f"harness[{index}]: missing required field '{name}'"
)
shape = entry["model_id_shape"]
if shape not in MODEL_ID_SHAPES:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): model_id_shape {shape!r} not one of "
f"{sorted(MODEL_ID_SHAPES)}"
)
# These three reach `eval` in setup-harnesses.sh, which is how they support the
# `${CODEX_HOME:-$HOME/.codex}` default-value syntax that python's expandvars cannot
# express. Under eval a backtick or $( would EXECUTE, so refuse them here — the registry
# is ours, but "ours" is not an argument that survives a careless future edit.
for shell_field in ("config_path", "auth_path", "skills_dir"):
value = entry.get(shell_field)
if not isinstance(value, str):
continue
# A backtick or $( executes outright. A double quote closes the string these are
# interpolated into, and a semicolon then starts a new command inside it — same
# outcome, one step removed.
bad = [t for t in ("`", "$(", '"', ";") if t in value]
if bad:
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): {shell_field} contains "
f"{', '.join(repr(t) for t in bad)} ({value!r}). This value is shell-"
f"expanded, so that would execute; use plain $VAR or ${{VAR:-default}} only."
)
launch = entry.get("explore_launch")
if entry.get("agent_config") and launch and not _references_agent_flags(launch):
raise HarnessRegistryError(
f"harness[{index}] ({entry['id']}): declares agent_config but its "
"explore_launch does not pass $RACCOON_AGENT_FLAGS. The worker's session "
"would then run with a different toolset than the trial it is authoring "
"for, which is the drift agent_config exists to prevent."
)
return Harness(
id=entry["id"],
label=entry["label"],
agent_import_path=entry["agent_import_path"],
agent_import_path_single_turn=entry.get("agent_import_path_single_turn"),
agent_import_path_browser=entry.get("agent_import_path_browser"),
agent_import_path_single_turn_browser=entry.get("agent_import_path_single_turn_browser"),
import_path_aliases=tuple(entry.get("import_path_aliases", ())),
legacy_bare_model_rows=bool(entry.get("legacy_bare_model_rows", False)),
default_model=entry.get("default_model"),
model_id_shape=shape,
effort_kwarg=entry.get("effort_kwarg", ""),
effort_default=entry.get("effort_default"),
fast_kwarg=entry.get("fast_kwarg", ""),
key_env=entry.get("key_env"),
base_url_env=entry.get("base_url_env"),
proxy_path=entry.get("proxy_path"),
writes_atif=bool(entry["writes_atif"]),
capture=bool(entry["capture"]),
seed_native=bool(entry["seed_native"]),
seed_atif=bool(entry["seed_atif"]),
flaky_hangs=bool(entry.get("flaky_hangs", False)),
enabled=bool(entry.get("enabled", True)),
authoring=bool(entry.get("authoring", False)),
cli=entry.get("cli"),
install=entry.get("install"),
skills_dir=entry.get("skills_dir"),
auth_path=entry.get("auth_path"),
auth_key_env=entry.get("auth_key_env"),
explore_launch=entry.get("explore_launch"),
config_path=entry.get("config_path"),
agent_config=entry.get("agent_config"),
container_config=entry.get("container_config"),
explore_config=entry.get("explore_config"),
extra={k: v for k, v in entry.items() if k not in _KNOWN_FIELDS},
)
_cache: dict[Path, HarnessRegistry] = {}
def load_harness_registry(path: Path | str = REGISTRY_PATH) -> HarnessRegistry:
"""Parse and validate the registry. Raises HarnessRegistryError on a malformed
file, a duplicate id, or an import path claimed by two harnesses (which would
make ``by_import_path`` depend on declaration order)."""
resolved = Path(path).resolve()
if resolved in _cache:
return _cache[resolved]
with open(resolved, "rb") as handle:
doc = tomllib.load(handle)
if "version" not in doc:
raise HarnessRegistryError("harness-registry: missing 'version'")
entries = doc.get("harness") or []
if not entries:
raise HarnessRegistryError("harness-registry: no [[harness]] entries")
harnesses = tuple(_build(entry, i) for i, entry in enumerate(entries))
seen_ids: set[str] = set()
for harness in harnesses:
if harness.id in seen_ids:
raise HarnessRegistryError(
f"harness-registry: duplicate harness id: {harness.id}"
)
seen_ids.add(harness.id)
owners: dict[str, str] = {}
for harness in harnesses:
for import_path in harness.known_import_paths():
owner = owners.get(import_path)
if owner is not None and owner != harness.id:
raise HarnessRegistryError(
f'harness-registry: import path "{import_path}" claimed by both '
f'"{owner}" and "{harness.id}"'
)
owners[import_path] = harness.id
registry = HarnessRegistry(version=int(doc["version"]), harnesses=harnesses)
_cache[resolved] = registry
return registry

View File

@@ -1,351 +0,0 @@
/**
* input-checksums.ts — capture and compare sha256 checksums of the task inputs
* that reference runs and detector reports depend on.
*
* A reference run is only meaningful for the task inputs it actually ran
* against: the prompt (instruction.md), the snapshot session
* (environment/session.jsonl), the workspace patch
* (environment/workspace.patch), and the gitref the workspace is built from
* (task.toml `[metadata].commit`). Detector reports likewise assess a specific
* revision of instruction.md + the task's holistic rubric
* (tests/holistic-rubric.md on current tasks; tests/grader-guidance-consolidated.md
* on tasks created before the rename) — and, for the rubric detectors, the
* task's atomic rubric (tests/atomic-rubric.yaml; tests/rubrics.yaml on tasks
* converted before the rename) and tests/grader-context.md. When any of those
* change after the artifact was produced, the artifact is stale — it describes
* an older revision of the task than the one being packaged.
*
* This module is the single source of truth for WHAT gets checksummed and how
* captures are compared. Capture sites (copy-reference-run.ts,
* record-detector-inputs.ts) write a {@link TaskInputChecksums} record next to
* the artifact; submit-task.ts re-captures at packaging time and diffs.
* Content hashes rather than mtimes: a re-clone / whole-tree touch can fake or
* mask an mtime, but can't change a sha256.
*
* Lives in the worker toolkit's shipped file set (static/scripts/lib/), so in
* a packed toolkit it sits at scripts/lib/ next to both consumers. Repo-side
* callers go through the scripts/lib/input-checksums.ts re-export shim, which
* occupies the same relative path there (mirroring the check-devcontainer
* pattern).
*
* Related but deliberately separate: `computeDeliveryHash` (repo-side
* delivery script — grep the internal repo for it; not shipped with the
* toolkit) hashes an overlapping input set for delivery idempotency. It is
* NOT built on this module because its hash format is load-bearing (a
* changed hash re-delivers every task); if you change WHAT counts as a task
* input here, check whether the delivery hash needs the same change.
*/
import { createHash } from 'node:crypto';
import { existsSync, readFileSync } from 'node:fs';
import { join } from 'node:path';
/** Bump when the record shape changes incompatibly. */
export const INPUT_CHECKSUMS_VERSION = 1;
/** Filename of the record inside a reference-run directory. */
export const INPUT_CHECKSUMS_FILENAME = 'input-checksums.json';
/**
* Where in the artifact lifecycle a capture happened. The moment matters for
* how much a "fresh" verdict can be trusted:
*
* - 'run' — at trial launch (scripts/harbor-run stamps the trial dir).
* The strongest evidence: the record is what the agent ran
* against, whatever got edited afterwards.
* - 'copy' — at copy-reference-run time, the fallback when a trial carries
* no run-time stamp. An input edited between harbor-run and the
* copy is recorded at its post-edit state, so a stale run can
* read fresh.
* - 'stamp' — right after a detector report is written
* (record-detector-inputs.ts in a worker checkout; the
* internal repo's detector save path stamps the same way).
* - 'mirror' — retired: written by the repo-side flow that re-materialized
* canonical detector reports to disk back when reports had a
* remote canonical store. Reports are local-only now, so no
* current code writes it; the member stays so old stamps keep
* their recorded method when read.
* - 'regrade' — a re-grade of an existing run. Present on records already on
* disk; no current code path writes it.
*
* Absent on records written before this field existed.
*/
export type TaskInputCaptureMethod = 'run' | 'copy' | 'stamp' | 'mirror' | 'regrade';
/** Runtime mirror of {@link TaskInputCaptureMethod}, for validating a record read from disk. */
export const TASK_INPUT_CAPTURE_METHODS = Object.freeze([
'run',
'copy',
'stamp',
'mirror',
'regrade',
] as const satisfies readonly TaskInputCaptureMethod[]);
/**
* The checksums of a task's inputs as they stood at capture time. Every hash
* field is a sha256 hex digest, or `null` when the file didn't exist (a null
* that later becomes a hash — or vice versa — is a change like any other).
* `gitref` is the raw `[metadata].commit` string, recorded verbatim rather
* than hashed so a mismatch message can show it.
*/
export interface TaskInputChecksums {
readonly version: number;
/** ISO-8601 timestamp of the capture. */
readonly capturedAt: string;
/** Lifecycle point of the capture ({@link TaskInputCaptureMethod}). */
readonly capturedBy?: TaskInputCaptureMethod;
/**
* The task slug the capture was taken from (harbor-tasks/<slug>). Written
* by launch-time captures so stamping can be scoped to the right task's
* trial dirs when several harbor-runs share a cwd — and so a mis-routed
* stamp is detectable after the fact.
*/
readonly taskSlug?: string;
/**
* Set when the finalize flow re-stamped the prompt/graderGuidance hashes
* after the harbor-path scrub deliberately rewrote those docs
* (scripts/restamp-task-inputs.ts) — the run/report still reflects the
* task; only the doc bytes were normalized.
*/
readonly restampedAt?: string;
readonly inputs: {
readonly prompt: string | null;
readonly graderGuidance: string | null;
readonly sessionJsonl: string | null;
readonly workspacePatch: string | null;
readonly gitref: string | null;
/**
* tests/grader-guidance-consolidated.md — the holistic rubric under its
* pre-rename filename, which every task created before the rename keeps;
* hashed when present, null otherwise. Absent (undefined) on records
* captured before the field existed; comparisons skip a field the record
* predates, so old captures stay fresh until they are re-stamped.
*/
readonly graderGuidanceConsolidated?: string | null;
/**
* tests/holistic-rubric.md — the holistic rubric under its current
* filename (each task carries exactly one of this and the pre-rename
* name above). Hashed when present, null otherwise. Absent (undefined)
* on records captured before the field existed; comparisons skip a
* field the record predates. The task-checksum digest serializes fields
* in this order and appends new fields at the end (see
* scripts/lib/grader-run-checksums.ts).
*/
readonly holisticRubric?: string | null;
/**
* tests/atomic-rubric.yaml — the task's atomic rubric under its current
* filename. Hashed when present, null otherwise. Absent (undefined) on
* records captured before the field existed; comparisons skip a field
* the record predates.
*/
readonly atomicRubric?: string | null;
/**
* tests/rubrics.yaml — the atomic rubric under its pre-rename filename,
* which tasks converted before the rename keep. Hashed when present,
* null otherwise; absent (undefined) on records captured before the
* field existed.
*/
readonly rubricsYaml?: string | null;
/**
* tests/grader-context.md — the context document the rubric grader modes
* read beside the atomic rubric. Hashed when present, null otherwise;
* absent (undefined) on records captured before the field existed. Last
* in field order per the append-at-the-end digest rule above.
*/
readonly graderContext?: string | null;
};
}
export type TaskInputName = keyof TaskInputChecksums['inputs'];
/** Human-readable component names, used verbatim in staleness warnings. Frozen:
* its key set is the runtime source of truth for the task-input axes. */
export const INPUT_LABELS = Object.freeze({
prompt: 'prompt (instruction.md)',
graderGuidance: 'legacy-era grader guidance (tests/grader-guidance.md)',
graderGuidanceConsolidated: 'holistic rubric (tests/grader-guidance-consolidated.md)',
sessionJsonl: 'session snapshot (environment/session.jsonl)',
workspacePatch: 'workspace patch (environment/workspace.patch)',
gitref: 'gitref (task.toml commit)',
holisticRubric: 'holistic rubric (tests/holistic-rubric.md)',
atomicRubric: 'atomic rubric (tests/atomic-rubric.yaml)',
rubricsYaml: 'atomic rubric (tests/rubrics.yaml)',
graderContext: 'grader context (tests/grader-context.md)',
} as const satisfies Record<TaskInputName, string>);
/**
* The inputs that shape what the AGENT saw and did. Changing any of them means
* a captured reference run no longer reflects the task being packaged, and
* only re-running the agent can fix that. The holistic-rubric files are
* deliberately NOT in this set: editing the rubric stales the run's GRADE,
* not the run itself, and `scripts/harbor-regrade` re-derives grades without
* re-running the agent.
*/
export const REFERENCE_RUN_INPUTS = Object.freeze([
'prompt',
'sessionJsonl',
'workspacePatch',
'gitref',
] as const satisfies readonly TaskInputName[]);
/**
* The inputs a detector report assesses — instruction.md plus whichever
* holistic-rubric files the task directory carries (tests/holistic-rubric.md
* on current tasks, tests/grader-guidance-consolidated.md on tasks created
* before the rename, plus the legacy-era plain-named file when a task
* authored on an earlier generation carries one). An absent file hashes to
* null on both sides and never diffs. Compared by content.
*
* The atomic-rubric files are deliberately NOT in this set: fifteen of the
* seventeen detectors never open them, so writing an atomic rubric after
* running the detectors would stale every one of those reports over files
* they never read. The two that do read them use
* {@link RUBRIC_DETECTOR_REPORT_INPUTS}.
*/
export const DETECTOR_REPORT_INPUTS = Object.freeze([
'prompt',
'graderGuidance',
'graderGuidanceConsolidated',
'holisticRubric',
] as const satisfies readonly TaskInputName[]);
/**
* The inputs the two rubric detectors assess: {@link DETECTOR_REPORT_INPUTS}
* plus the atomic-rubric package (tests/atomic-rubric.yaml, the pre-rename
* tests/rubrics.yaml, and tests/grader-context.md), which they compare
* against the holistic rubric.
*/
export const RUBRIC_DETECTOR_REPORT_INPUTS = Object.freeze([
...DETECTOR_REPORT_INPUTS,
'atomicRubric',
'rubricsYaml',
'graderContext',
] as const satisfies readonly TaskInputName[]);
/** Detectors that read the atomic-rubric package, and so are staled by it. */
export const ATOMIC_RUBRIC_DETECTORS: readonly string[] = Object.freeze([
'detector-rubric-coverage',
'detector-rubric-form',
]);
/** The input set a named detector's report is judged against. */
export function detectorReportInputs(detectorName: string): readonly TaskInputName[] {
return ATOMIC_RUBRIC_DETECTORS.includes(detectorName)
? RUBRIC_DETECTOR_REPORT_INPUTS
: DETECTOR_REPORT_INPUTS;
}
/** sha256 hex digest of a file's bytes, or null when it doesn't exist. */
function sha256File(filePath: string): string | null {
if (!existsSync(filePath)) return null;
return createHash('sha256').update(readFileSync(filePath)).digest('hex');
}
/**
* The gitref (`[metadata].commit`) from a task.toml, or null when the file is
* missing, unreadable, or has no commit line — a read failure downgrades to
* "absent" rather than crashing a capture or a validation sweep.
*
* Deliberately a regex, not a TOML parser: this module ships in the worker
* toolkit, where a new runtime dep would break packaging for every worker
* whose container predates the dep (npm install runs only on container
* create, and containers survive toolkit upgrades). build-workspace.sh reads
* the same key with the same grep-a-`commit`-line approach. The one `commit`
* key in a task.toml is `[metadata].commit`, so anchoring to the first
* `commit = "…"` line is exact in practice.
*/
function readGitref(taskDir: string): string | null {
const tomlPath = join(taskDir, 'task.toml');
if (!existsSync(tomlPath)) return null;
try {
const match = /^\s*commit\s*=\s*(?:"([^"\n]+)"|'([^'\n]+)')\s*(?:#.*)?$/m.exec(
readFileSync(tomlPath, 'utf-8')
);
const commit = match?.[1] ?? match?.[2];
return commit && commit.length > 0 ? commit : null;
} catch {
return null;
}
}
/** Checksum the task inputs as they stand right now under `taskDir`. */
export function captureTaskInputs(
taskDir: string,
capturedBy?: TaskInputCaptureMethod
): TaskInputChecksums {
return Object.freeze({
version: INPUT_CHECKSUMS_VERSION,
capturedAt: new Date().toISOString(),
...(capturedBy ? { capturedBy } : {}),
inputs: Object.freeze({
prompt: sha256File(join(taskDir, 'instruction.md')),
graderGuidance: sha256File(join(taskDir, 'tests', 'grader-guidance.md')),
sessionJsonl: sha256File(join(taskDir, 'environment', 'session.jsonl')),
workspacePatch: sha256File(join(taskDir, 'environment', 'workspace.patch')),
gitref: readGitref(taskDir),
graderGuidanceConsolidated: sha256File(
join(taskDir, 'tests', 'grader-guidance-consolidated.md')
),
holisticRubric: sha256File(join(taskDir, 'tests', 'holistic-rubric.md')),
atomicRubric: sha256File(join(taskDir, 'tests', 'atomic-rubric.yaml')),
rubricsYaml: sha256File(join(taskDir, 'tests', 'rubrics.yaml')),
graderContext: sha256File(join(taskDir, 'tests', 'grader-context.md')),
}),
});
}
/**
* Read a previously captured record. Returns null when the file is missing or
* doesn't look like a capture (pre-tracking artifact, hand-edited JSON, a
* future incompatible version) — callers treat null as "staleness unknowable",
* never as an error. No zod here: this module ships in the worker toolkit,
* whose dependency set stays minimal, so the guard is manual.
*/
export function readTaskInputChecksums(filePath: string): TaskInputChecksums | null {
if (!existsSync(filePath)) return null;
let parsed: unknown;
try {
parsed = JSON.parse(readFileSync(filePath, 'utf-8'));
} catch {
return null;
}
if (typeof parsed !== 'object' || parsed === null) return null;
const record = parsed as TaskInputChecksums;
if (record.version !== INPUT_CHECKSUMS_VERSION) return null;
if (typeof record.inputs !== 'object' || record.inputs === null) return null;
for (const name of Object.keys(INPUT_LABELS) as TaskInputName[]) {
const value = record.inputs[name];
// undefined = the record predates this input field; still a valid capture.
if (value !== undefined && value !== null && typeof value !== 'string') return null;
}
// An unrecognized capture method is dropped, not rejected: the field is
// provenance colour, and rejecting would flip the whole run to "unknowable".
const method: unknown = record.capturedBy;
const isKnown = TASK_INPUT_CAPTURE_METHODS.includes(method as TaskInputCaptureMethod);
if (method !== undefined && !isKnown) {
const { capturedBy: _dropped, ...rest } = record;
return rest;
}
return record;
}
/**
* Which of `names` changed between a recorded capture and the current state?
* Returns the human-readable labels ({@link INPUT_LABELS}) of every component
* whose value differs — including absent→present and present→absent flips. A
* field the recorded capture predates (the key is not in the record at all)
* is skipped: freshness on that axis is unknowable, and flagging every old
* record the moment a new axis ships would drown the real signal. (The
* task-checksum fold folds absence as 'null' instead — it only ever reads a
* fresh capture, which is total, so the two never disagree in practice.)
*/
export function diffTaskInputs(
recorded: TaskInputChecksums,
current: TaskInputChecksums,
names: readonly TaskInputName[]
): string[] {
return names
.filter((name) => name in recorded.inputs)
.filter((name) => recorded.inputs[name] !== current.inputs[name])
.map((name) => INPUT_LABELS[name]);
}

View File

@@ -1,13 +0,0 @@
/**
* Wrap a notice in a banner loud enough to survive a scrollback.
*
* Yellow only when stderr is a terminal, so piped logs stay clean.
*/
export function banner(message: string, headline: string): string {
const RULE = '#'.repeat(78);
const pad = ' '.repeat(Math.max(0, Math.floor((78 - headline.length) / 2)));
const body = [RULE, `${pad}${headline}`, RULE, '', message, RULE].join('\n');
const color = process.stderr.isTTY ? ['\u001b[33m', '\u001b[39m'] : ['', ''];
return `${color[0]}${body}${color[1]}`;
}

View File

@@ -1,96 +0,0 @@
# shellcheck shell=bash
#
# resolve_pin — turn a task's pinned commit into a SHA that exists in the repo,
# translating through a commit map when history has been rewritten under it.
#
# A task pins a commit in task.toml. If that repo's history is later rewritten
# (to strip something that should never have shipped, say), every rewritten
# commit gets a new SHA and the pin stops resolving — including on machines we
# cannot reach, holding tasks we cannot edit. A commit map lets those pins keep
# working: `<old-sha> <new-sha>` per line, at task-shared/commit-maps/<member>.map,
# <member> being the task's `repo` key — a standalone toolkit checks its repo out
# at repo/, so the directory name is not the member name and cannot be the key.
#
# The map is only consulted when the pin does not resolve, so it carries only
# rewritten commits — an unchanged commit resolves on its own and its identity
# row could never be read.
#
# Usage (source, then call):
# . "$(dirname "$0")/lib/resolve-pin.sh"
# sha=$(resolve_pin "$REPO_DIR" "$COMMIT" "$TOOLKIT_ROOT/task-shared/commit-maps" "$MEMBER") || exit 1
#
# Writes the resolved SHA to stdout, notes on stderr. Returns non-zero if the
# pin cannot be resolved, having explained why.
RESOLVE_PIN_MAX_HOPS="${RESOLVE_PIN_MAX_HOPS:-25}"
_RESOLVE_PIN_ZERO='0000000000000000000000000000000000000000'
# Look one hop: echo the successor of $1 in map $2, or nothing. Fails if the
# prefix is ambiguous, which would otherwise pick an arbitrary commit.
_resolve_pin_hop() {
local from="$1" map="$2" hits
hits=$(awk -v p="$from" '
/^#/ || NF < 2 { next }
index($1, p) == 1 { print $2 }
' "$map" | sort -u)
[ -z "$hits" ] && return 1
if [ "$(printf '%s\n' "$hits" | wc -l | tr -d ' ')" -gt 1 ]; then
echo " pin $from is ambiguous in $(basename "$map") — use a longer SHA" >&2
return 2
fi
printf '%s\n' "$hits"
}
resolve_pin() {
local repo_dir="$1" commit="$2" map_dir="${3:-}" key="${4:-}" sha map cur hops next rc
# Present in the repo: nothing to translate.
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$commit^{commit}" 2>/dev/null); then
printf '%s\n' "$sha"
return 0
fi
map=""
if [ -n "$map_dir" ]; then
if [ -n "$key" ] && [ -f "$map_dir/$key.map" ]; then
map="$map_dir/$key.map"
elif [ -f "$map_dir/$(basename "$repo_dir").map" ]; then
map="$map_dir/$(basename "$repo_dir").map"
fi
fi
if [ -z "$map" ]; then
echo "Error: pinned commit $commit is not in $repo_dir, and no commit map is available." >&2
echo " The repo may be a shallow or partial copy — try a full clone." >&2
return 1
fi
# Follow the chain: a commit rewritten more than once maps forward a hop per
# rewrite, so keep going until the SHA exists or the trail ends.
cur="$commit"
hops=0
while [ "$hops" -lt "$RESOLVE_PIN_MAX_HOPS" ]; do
next=$(_resolve_pin_hop "$cur" "$map"); rc=$?
[ "$rc" -eq 2 ] && return 1
if [ "$rc" -ne 0 ]; then
echo "Error: pinned commit $commit is not in $repo_dir and is not in $(basename "$map")." >&2
echo " It predates the map, or came from a repo copy this toolkit was not built from." >&2
return 1
fi
if [ "$next" = "$_RESOLVE_PIN_ZERO" ]; then
echo "Error: pinned commit $commit was deleted by a history rewrite, not rewritten." >&2
echo " Re-pin this task to a commit that still exists." >&2
return 1
fi
if sha=$(git -C "$repo_dir" rev-parse --quiet --verify "$next^{commit}" 2>/dev/null); then
echo " Pin $commit was rewritten; using $sha" >&2
printf '%s\n' "$sha"
return 0
fi
cur="$next"
hops=$((hops + 1))
done
echo "Error: pinned commit $commit did not settle after $RESOLVE_PIN_MAX_HOPS hops." >&2
echo " $(basename "$map") may contain a cycle." >&2
return 1
}

View File

@@ -1,431 +0,0 @@
/**
* task-infra-integrity.ts — detect edits to toolkit-managed task files.
*
* `environment/Dockerfile`, `tests/test.sh`, and
* `tests/grader-system-prompt-consolidated.md` come from `task-shared/` and
* are the same in every task: they decide how the
* trial runs and how the grade is produced. An edit makes a task's reference
* runs incomparable to every other task's, and the scores still look normal,
* so nothing downstream notices.
*
* A task is compared against itself as created. {@link writeManagedStamp} records
* a sha256 of each managed file into `<task>/.toolkit-managed.json` at task
* creation, so a later mismatch is an edit made since. Tasks created before
* stamping have no record and fall back to matching the copies this toolkit
* ships — see {@link IntegrityStatus}.
*
* The toolkit appends to a task's Dockerfile itself (session staging, the
* reference-data corpus). Those blocks are wrapped in
* `# >>> toolkit-managed: <name> >>>` sentinels and stripped before hashing or
* comparing, so they never read as edits.
*/
import { createHash } from 'crypto';
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
import { basename, join } from 'path';
import { banner } from './notice-banner.js';
/**
* Every status is advisory. Nothing here stops a trial or a submission: an author
* who changed one of these files did it because they didn't know we'd rather they
* didn't, and refusing to package their work punishes a misunderstanding. The job
* is to say so clearly, and to record it so a reviewer sees it too.
*
* `ok` — identical to a copy this toolkit ships, or unchanged since the
* task was created.
* `outdated` — unchanged since creation, but the toolkit has shipped a newer
* copy since. Nobody's mistake; it does mean this task's runs
* aren't directly comparable to one built today.
* `modified` — matches neither its baseline nor anything shipped: an edit.
* `unverifiable` — no recorded baseline and matches nothing shipped, so an edit
* and an older release are indistinguishable.
* `missing` — the task doesn't have the file.
* `placeholder` — still the polyglot scaffold placeholder, so no base image has
* been selected yet.
*/
export type IntegrityStatus =
| 'ok'
| 'outdated'
| 'modified'
| 'unverifiable'
| 'missing'
| 'placeholder';
export interface FileVerdict {
/** Task-relative path, e.g. `environment/Dockerfile`. */
taskPath: string;
status: IntegrityStatus;
/** Command that restores the managed version, on `modified` / `unverifiable`. */
restore?: string;
}
export interface IntegrityReport {
/**
* False when this isn't a worker toolkit, or when the task was authored on
* a different toolkit generation (see {@link isTaskFromThisToolkitGeneration})
* — callers should skip silently.
*/
checked: boolean;
files: FileVerdict[];
/** Looks like an edit: matches neither a baseline nor anything shipped. */
modified: FileVerdict[];
/** Can't be told apart from an older release. */
unverifiable: FileVerdict[];
/** Unchanged, but a newer copy has shipped since. */
outdated: FileVerdict[];
}
interface ManagedFile {
taskPath: string;
/** Matches the candidate pristine filenames under `task-shared/`. */
baselinePattern: RegExp;
}
/** The Dockerfile pattern accepts `Dockerfile` and every `Dockerfile.<member>`. */
const MANAGED_FILES: ManagedFile[] = [
{ taskPath: 'environment/Dockerfile', baselinePattern: /^Dockerfile(\.[\w.-]+)?$/ },
{ taskPath: 'tests/test.sh', baselinePattern: /^test\.sh$/ },
{
taskPath: 'tests/grader-system-prompt-consolidated.md',
baselinePattern: /^grader-system-prompt-consolidated\.md$/,
},
];
const SENTINEL_OPEN = /^#\s*>>>\s*toolkit-managed:.*>>>\s*$/;
const SENTINEL_CLOSE = /^#\s*<<<\s*toolkit-managed\s*<<<\s*$/;
/**
* Line shapes from toolkit releases that predate the sentinels. Deliberately
* narrow: each is a literal line the toolkit wrote, not a general "ignore COPY
* lines" rule an edit could hide behind.
*/
const LEGACY_MANAGED_LINES: RegExp[] = [
/^# Stage session files for the snapshot agent adapter to install at runtime\.$/,
/^COPY session\.jsonl \/tmp\/snapshot-session\/session\.jsonl$/,
/^COPY session\/ \/tmp\/snapshot-session\/session\/$/,
/^RUN echo '[0-9a-fA-F-]+' > \/tmp\/snapshot-session\/uuid\.txt$/,
/^# Reference-data corpus at \/data\/zeta-corpus \(staged by build-workspace\)\.$/,
/^COPY corpus\/ \/data\/zeta-corpus\/$/,
];
/** Marker identifying the polyglot scaffold's deliberately-failing placeholder. */
const PLACEHOLDER_MARKER = 'POLYGLOT TOOLKIT';
/** Per-task stamp of the managed files as created. Lives in the task directory. */
export const STAMP_FILENAME = '.toolkit-managed.json';
interface ManagedStamp {
version: number;
stampedAt: string;
/** taskPath → sha256 of the stripped content. */
files: Record<string, string>;
}
/**
* Remove toolkit-appended content so only author-authored differences remain.
* Trailing blank lines go too — an editor adding or trimming a final newline is
* not something to fail a trial over.
*/
export function stripManagedBlocks(content: string): string {
const out: string[] = [];
let inBlock = false;
// Normalize CRLF before anything else: a Windows editor or a checkout with
// core.autocrlf rewrites every line ending, and that must not read as an edit.
for (const line of content.replace(/\r\n/g, '\n').split('\n')) {
if (!inBlock && SENTINEL_OPEN.test(line)) {
inBlock = true;
continue;
}
if (inBlock) {
if (SENTINEL_CLOSE.test(line)) inBlock = false;
continue;
}
if (LEGACY_MANAGED_LINES.some((re) => re.test(line))) continue;
out.push(line);
}
return out.join('\n').replace(/\s+$/, '');
}
export function sha256(content: string): string {
return createHash('sha256').update(content).digest('hex');
}
/** Pristine `task-shared/` filenames matching a managed file's baseline pattern. */
function baselineCandidates(sharedDir: string, pattern: RegExp): string[] {
if (!existsSync(sharedDir)) return [];
return readdirSync(sharedDir)
.filter((f) => pattern.test(f))
.sort();
}
/**
* Record the managed files, so later edits are detectable. Call at task creation
* and after a managed file is first put in place.
*
* A file earns a baseline only by matching a copy this toolkit ships, and an
* entry already recorded is never rewritten. Together those mean a stamp can
* only ever describe a pristine file: re-running this can't turn an author's
* edit into the new baseline, and a file dropped in later (the polyglot
* Dockerfile, which is the scaffold's placeholder at first stamp) still gets a
* baseline once it's in place.
*
* Returns true if anything was recorded.
*/
export function writeManagedStamp(taskDir: string, toolkitRoot: string): boolean {
const sharedDir = join(toolkitRoot, 'task-shared');
const existing = readStamp(taskDir);
const files: Record<string, string> = { ...(existing?.files ?? {}) };
let added = false;
for (const managed of MANAGED_FILES) {
if (files[managed.taskPath]) continue;
const p = join(taskDir, managed.taskPath);
if (!existsSync(p)) continue;
const raw = readFileSync(p, 'utf-8');
// Not a baseline: the author still has to drop in their member's base image.
if (raw.includes(PLACEHOLDER_MARKER)) continue;
const stripped = stripManagedBlocks(raw);
if (!matchesShipped(sharedDir, managed, stripped)) continue;
files[managed.taskPath] = sha256(stripped);
added = true;
}
if (!added) return false;
const stamp: ManagedStamp = {
version: 1,
stampedAt: new Date().toISOString(),
files,
};
writeFileSync(join(taskDir, STAMP_FILENAME), `${JSON.stringify(stamp, null, 2)}\n`);
return true;
}
function readStamp(taskDir: string): ManagedStamp | null {
const stampPath = join(taskDir, STAMP_FILENAME);
if (!existsSync(stampPath)) return null;
try {
const parsed = JSON.parse(readFileSync(stampPath, 'utf-8')) as ManagedStamp;
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
} catch {
// Treat a corrupt stamp as no stamp rather than blocking a trial over it.
return null;
}
}
/**
* How to restore a managed file, or undefined when this toolkit ships no copy to
* restore from. Only the Dockerfile can have several candidates (one per member).
*/
function restoreCommand(taskPath: string, candidates: string[], slug: string): string | undefined {
const dest = `harbor-tasks/${slug}/${taskPath}`;
if (candidates.length === 1) return `cp task-shared/${candidates[0]} ${dest}`;
if (candidates.length > 1) {
return `cp task-shared/Dockerfile.<your-member> ${dest} (list them: ls task-shared/Dockerfile.*)`;
}
// Never guess. Emitting the multi-candidate Dockerfile line here would tell an
// author to copy a Dockerfile over their grader prompt.
return undefined;
}
/** Render a restore line, saying so plainly when there is nothing to restore from. */
function restoreLine(f: FileVerdict): string {
return f.restore
? ` ${f.restore}`
: ` (no copy of ${f.taskPath} ships in task-shared/ — re-extract the toolkit zip)`;
}
/** Does this content match a pristine copy the toolkit ships? */
function matchesShipped(sharedDir: string, managed: ManagedFile, stripped: string): boolean {
return baselineCandidates(sharedDir, managed.baselinePattern).some(
(c) => stripManagedBlocks(readFileSync(join(sharedDir, c), 'utf-8')) === stripped
);
}
/**
* Was this task created by this toolkit generation? Task creation (the packed
* scaffold's task.toml and snapshot-to-task) writes `[metadata].toolkit_version`;
* a task directory without the key was authored on a different toolkit
* generation and grades with the assets frozen in its own tests/ directory, so
* comparing those against this toolkit's copies would report drift that is not
* an edit. Presence-based on purpose: wall-clock stamps cannot separate the
* generations, because tasks from an earlier generation are completed after
* later kits ship.
*
* A regex rather than a TOML parser, for the same shipped-dependency reason as
* input-checksums.ts readGitref: the one `toolkit_version` key in a task.toml
* is `[metadata].toolkit_version`.
*/
export function isTaskFromThisToolkitGeneration(taskDir: string): boolean {
const tomlPath = join(taskDir, 'task.toml');
if (!existsSync(tomlPath)) return false;
try {
return /^\s*toolkit_version\s*=/m.test(readFileSync(tomlPath, 'utf-8'));
} catch {
return false;
}
}
/**
* Compare a task's managed files against its creation-time stamp.
*
* @param taskDir Absolute path to `harbor-tasks/<slug>`.
* @param toolkitRoot Absolute path to the toolkit root (holds `task-shared/`).
*/
export function checkTaskInfraIntegrity(taskDir: string, toolkitRoot: string): IntegrityReport {
const sharedDir = join(toolkitRoot, 'task-shared');
// Without task-shared/ there is nothing to compare against; report "not
// checked" so callers no-op rather than reporting three phantom failures.
if (!existsSync(sharedDir)) {
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
}
// A task authored on a different toolkit generation grades with the assets
// frozen in its own tests/ directory. Comparing those against this toolkit's
// copies would report drift that is not an edit — and the printed remedy
// (restore the current copy) would change how that task grades. Skip it.
if (!isTaskFromThisToolkitGeneration(taskDir)) {
return { checked: false, files: [], modified: [], unverifiable: [], outdated: [] };
}
const slug = basename(taskDir);
const stamp = readStamp(taskDir);
const files: FileVerdict[] = [];
for (const managed of MANAGED_FILES) {
const taskFile = join(taskDir, managed.taskPath);
if (!existsSync(taskFile)) {
files.push({ taskPath: managed.taskPath, status: 'missing' });
continue;
}
const raw = readFileSync(taskFile, 'utf-8');
const candidates = baselineCandidates(sharedDir, managed.baselinePattern);
const restore = restoreCommand(managed.taskPath, candidates, slug);
const stripped = stripManagedBlocks(raw);
// FIRST: is this byte-for-byte something the toolkit ships right now? If so it
// cannot be an author edit, whatever the stamp says — and asking the stamp first
// is what used to make restoring the current copy (which is exactly what we tell
// authors to do) look like an edit, with no way out.
if (matchesShipped(sharedDir, managed, stripped)) {
files.push({ taskPath: managed.taskPath, status: 'ok' });
continue;
}
const expected = stamp?.files[managed.taskPath];
if (expected) {
// Matches its baseline but nothing shipped: untouched by the author, and the
// toolkit has moved on since. Worth saying, nobody's fault.
const status = sha256(stripped) === expected ? 'outdated' : 'modified';
files.push({ taskPath: managed.taskPath, status, restore });
continue;
}
// Checked after the stamp so that adding this marker to a file that HAS a
// baseline can't exempt it from the comparison.
if (raw.includes(PLACEHOLDER_MARKER)) {
files.push({ taskPath: managed.taskPath, status: 'placeholder' });
continue;
}
files.push({ taskPath: managed.taskPath, status: 'unverifiable', restore });
}
return {
checked: true,
files,
modified: files.filter((f) => f.status === 'modified'),
unverifiable: files.filter((f) => f.status === 'unverifiable'),
outdated: files.filter((f) => f.status === 'outdated'),
};
}
/**
* Human-readable report. Returns '' when there is nothing worth saying, so callers
* can `if (msg) print(msg)`.
*
* Deliberately not phrased as a refusal. An author who changed one of these files
* almost always did it to get unstuck, not knowing we'd rather they told us — so
* this explains what it means for their task and what restoring would do, and then
* lets them get on with it.
*/
export function formatIntegrityReport(report: IntegrityReport): string {
const sections: string[] = [];
if (report.modified.length > 0) {
sections.push(
[
'These files look edited since this task was created, and the toolkit manages',
'them — they set up how the trial runs and how the grade is produced, so they',
"have to be identical across every task. Yours aren't, which makes this task's",
"runs hard to compare with everyone else's:",
'',
...report.modified.map((f) => ` ${f.taskPath}`),
'',
'Restoring the shipped version puts that right:',
...report.modified.map(restoreLine),
'',
'If you changed one to work around a problem — a missing package, a grader that',
"wouldn't run — please tell us about the problem instead. It almost certainly",
'affects other authors too, and the fix belongs in the toolkit, not in one task.',
'Nothing here stops you running trials or submitting.',
].join('\n')
);
}
if (report.outdated.length > 0) {
sections.push(
[
'These files are unchanged, but the toolkit has shipped newer copies since this',
'task was created:',
'',
...report.outdated.map((f) => ` ${f.taskPath}`),
'',
"You haven't done anything wrong. It does mean this task was run and graded with",
"older versions than a task built today, so its scores aren't directly",
'comparable. To line them up, restore the current copies and re-run your trials:',
...report.outdated.map(restoreLine),
].join('\n')
);
}
if (report.unverifiable.length > 0) {
sections.push(
[
"These files don't match the copies this toolkit ships, and this task has no",
'record of what they looked like when it was created:',
'',
...report.unverifiable.map((f) => ` ${f.taskPath}`),
'',
'Two things look like this and we cannot tell them apart: a task created on an',
'earlier toolkit release (nothing to fix, though its scores are not directly',
'comparable to a task built today), or a file that was edited. Either way,',
'restoring the current copy and re-running your trials is what makes this task',
"comparable to everyone else's:",
...report.unverifiable.map(restoreLine),
].join('\n')
);
}
return sections.join('\n\n');
}
/**
* Wrap a report in a banner loud enough to survive a scrollback.
*
* Nothing blocks any more, so this notice is the entire mechanism — and an
* unframed paragraph among build output is one a reasonable person scrolls past.
*/
export function bannerize(message: string, report: IntegrityReport): string {
return banner(
message,
report.modified.length > 0
? '!! TOOLKIT-MANAGED FILES LOOK EDITED — PLEASE READ !!'
: '!! TOOLKIT-MANAGED FILES NEED A LOOK — PLEASE READ !!'
);
}

View File

@@ -1,217 +0,0 @@
/**
* toolkit-script-integrity.ts — detect edits to the toolkit's own scripts.
*
* Sibling of task-infra-integrity.ts, which covers a task's managed files. This
* covers `scripts/`. The scripts never ship with a task, so an edit can't reach
* the delivered workspace — but their OUTPUT does: `build-workspace.sh` alone
* stages `tests/test-commands.sh` (the deterministic checks behind the
* correctness score), writes the Dockerfile's toolkit-managed blocks, and
* records the managed stamp and input checksums. Nothing downstream re-derives
* those, and the reference runs can't be re-derived at all.
*
* The baseline is a manifest written at package time ({@link writeScriptManifest}),
* so it ships in the same zip as the scripts it describes. That removes the
* ambiguity a task's managed files have: there is no "created on an older
* release" case to tell apart, so a hash mismatch is an edit. Files absent from
* the manifest are ignored, which keeps a worker's own helper script — or a
* `__pycache__` left by a harbor run — from ever being reported.
*/
import { createHash } from 'crypto';
import { existsSync, readFileSync, readdirSync, writeFileSync } from 'fs';
import { join, relative } from 'path';
import { banner } from './notice-banner.js';
/** Manifest of the shipped `scripts/` tree. Lives at the toolkit root. */
export const SCRIPT_MANIFEST_FILENAME = '.toolkit-scripts.json';
const MANIFEST_VERSION = 1;
/** Runtime droppings, never part of the shipped tree. */
const IGNORED_DIRS = new Set(['__pycache__', 'node_modules', '.git']);
const IGNORED_FILES = /\.(pyc|pyo)$/;
/**
* `modified` — content differs from what shipped: an edit.
* `missing` — shipped, but no longer on disk.
* `ok` — unchanged.
*/
export type ScriptStatus = 'ok' | 'modified' | 'missing';
export interface ScriptVerdict {
/** Toolkit-relative path, e.g. `scripts/build-workspace.sh`. */
path: string;
status: ScriptStatus;
}
export interface ScriptIntegrityReport {
/** False when no manifest ships — callers should skip silently. */
checked: boolean;
files: ScriptVerdict[];
modified: ScriptVerdict[];
missing: ScriptVerdict[];
}
interface ScriptManifest {
version: number;
generatedAt: string;
/** Toolkit-relative path → sha256 of the normalized content. */
files: Record<string, string>;
}
/**
* Line endings and trailing whitespace are normalized away: a Windows editor, a
* checkout with core.autocrlf, or a formatter trimming a final newline must not
* read as an edit.
*/
function hashContent(content: string): string {
return createHash('sha256')
.update(content.replace(/\r\n/g, '\n').replace(/\s+$/, ''))
.digest('hex');
}
/** Every shipped file under `scripts/`, as toolkit-relative paths. */
function walkScripts(dir: string, toolkitRoot: string): string[] {
if (!existsSync(dir)) return [];
const out: string[] = [];
for (const entry of readdirSync(dir, { withFileTypes: true }).sort((a, b) =>
a.name.localeCompare(b.name)
)) {
const abs = join(dir, entry.name);
if (entry.isDirectory()) {
if (!IGNORED_DIRS.has(entry.name)) out.push(...walkScripts(abs, toolkitRoot));
continue;
}
if (!entry.isFile() || IGNORED_FILES.test(entry.name)) continue;
out.push(relative(toolkitRoot, abs));
}
return out;
}
/**
* Record the shipped `scripts/` tree. Call at package time, once the tree is
* fully staged — anything written to `scripts/` afterwards reads as an edit.
*
* Returns the number of files recorded.
*/
export function writeScriptManifest(toolkitRoot: string): number {
const files: Record<string, string> = {};
for (const rel of walkScripts(join(toolkitRoot, 'scripts'), toolkitRoot)) {
files[rel] = hashContent(readFileSync(join(toolkitRoot, rel), 'utf-8'));
}
const manifest: ScriptManifest = {
version: MANIFEST_VERSION,
generatedAt: new Date().toISOString(),
files,
};
writeFileSync(
join(toolkitRoot, SCRIPT_MANIFEST_FILENAME),
`${JSON.stringify(manifest, null, 2)}\n`
);
return Object.keys(files).length;
}
function readManifest(toolkitRoot: string): ScriptManifest | null {
const p = join(toolkitRoot, SCRIPT_MANIFEST_FILENAME);
if (!existsSync(p)) return null;
try {
const parsed = JSON.parse(readFileSync(p, 'utf-8')) as ScriptManifest;
if (parsed?.version !== MANIFEST_VERSION) return null;
return parsed?.files && typeof parsed.files === 'object' ? parsed : null;
} catch {
// A corrupt manifest is treated as no manifest rather than blocking a trial.
return null;
}
}
/**
* Compare the toolkit's `scripts/` tree against the manifest it shipped with.
*
* @param toolkitRoot Absolute path to the toolkit root (holds `scripts/`).
*/
export function checkToolkitScriptIntegrity(toolkitRoot: string): ScriptIntegrityReport {
const manifest = readManifest(toolkitRoot);
if (!manifest) return { checked: false, files: [], modified: [], missing: [] };
const files: ScriptVerdict[] = Object.entries(manifest.files).map(([path, expected]) => {
const abs = join(toolkitRoot, path);
if (!existsSync(abs)) return { path, status: 'missing' as const };
const status = hashContent(readFileSync(abs, 'utf-8')) === expected ? 'ok' : 'modified';
return { path, status };
});
return {
checked: true,
files,
modified: files.filter((f) => f.status === 'modified'),
missing: files.filter((f) => f.status === 'missing'),
};
}
/**
* Human-readable report. Returns '' when there is nothing worth saying, so callers
* can `if (msg) print(msg)`.
*
* Deliberately not phrased as a refusal, for the same reason the managed-file
* notice isn't: an author who changed one of these did it to get unstuck, and the
* fix they needed almost certainly belongs in the toolkit rather than in their copy.
*/
export function formatScriptIntegrityReport(report: ScriptIntegrityReport): string {
const sections: string[] = [];
if (report.modified.length > 0) {
sections.push(
[
'These toolkit scripts look edited:',
'',
...report.modified.map((f) => ` ${f.path}`),
'',
"They aren't part of any task, so an edit is easy to miss — but what they write",
'is. Building a task stages its deterministic checks, fills in parts of its',
'Dockerfile, and records the checksums a reviewer reads; a script that does any of',
'that differently produces a task that looks normal and behaves differently from',
'every other one.',
'',
'Re-extracting the toolkit zip over your copy restores them. Your tasks, snapshots',
'and reference runs are untouched by that.',
'',
'If you changed one to work around a problem — a build that would not run, a',
'missing dependency — please tell us about the problem instead. It almost',
'certainly affects other authors too, and the fix belongs in the toolkit.',
'Nothing here stops you running trials or submitting.',
].join('\n')
);
}
if (report.missing.length > 0) {
sections.push(
[
'These toolkit scripts shipped with this release but are no longer here:',
'',
...report.missing.map((f) => ` ${f.path}`),
'',
'Something that depends on one will fail partway through rather than up front.',
'Re-extract the toolkit zip over your copy to put them back.',
].join('\n')
);
}
return sections.join('\n\n');
}
/** The full notice, bannered and ready to write to stderr, or '' if all is well. */
export function scriptIntegrityNotice(report: ScriptIntegrityReport): string {
const message = formatScriptIntegrityReport(report);
if (!message) return '';
const headline =
report.modified.length > 0
? '!! TOOLKIT SCRIPTS LOOK EDITED — PLEASE READ !!'
: '!! TOOLKIT SCRIPTS ARE MISSING — PLEASE READ !!';
return banner(message, headline);
}

View File

@@ -1,304 +0,0 @@
/**
* Tests for tree-permissions.ts.
*
* The load-bearing case is the one from the field report: a directory that came
* across without its search bit makes `tar` fail with `Cannot stat` on the files
* *inside* it, so the repair has to fix directory modes, not just ownership.
* These tests run unprivileged, so they exercise the mode axis for real and the
* ownership axis only as far as an unprivileged process can (target resolution +
* graceful EPERM), which is the same shape CI runs in. One case needs root and
* skips otherwise; the rest hold under either uid, which is why the fixtures that
* must look human-owned say so with `ownedByHuman` instead of relying on the
* caller's uid.
*/
import assert from 'node:assert/strict';
import {
chmodSync,
chownSync,
mkdirSync,
rmSync,
statSync,
symlinkSync,
writeFileSync,
} from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { test } from 'node:test';
import {
didRepair,
manualRepairHint,
normalizeTreePermissions,
resolveWorkspaceOwner,
} from './tree-permissions';
function scratch(name: string): string {
const dir = join(tmpdir(), `tree-perms-${name}-${process.pid}`);
rmSync(dir, { recursive: true, force: true });
mkdirSync(dir, { recursive: true });
return dir;
}
const RUNNING_AS_ROOT = process.getuid?.() === 0;
const HUMAN_UID = RUNNING_AS_ROOT ? 1000 : (process.getuid?.() ?? 0);
const HUMAN_GID = RUNNING_AS_ROOT ? 1000 : (process.getgid?.() ?? 0);
/** Give a fixture a non-root owner, so the repair sees a tree it can hand back. */
function ownedByHuman(path: string): string {
chownSync(path, HUMAN_UID, HUMAN_GID);
return path;
}
test('restores the search bit on a directory that lost it', () => {
const root = scratch('searchbit');
const models = join(root, 'agent-output', 'app', 'models');
mkdirSync(models, { recursive: true });
writeFileSync(join(models, 'bill.rb'), 'class Bill; end\n');
// r-- : readdir works, so tar can NAME the file, but stat is refused.
chmodSync(models, 0o400);
const report = normalizeTreePermissions(root);
assert.equal(statSync(models).mode & 0o700, 0o700, 'owner rwx restored on the directory');
assert.ok(report.modeFixed.some((p) => p === models));
assert.ok(didRepair(report));
rmSync(root, { recursive: true, force: true });
});
test('recurses into a directory it had to widen first', () => {
const root = scratch('recurse');
const inner = join(root, 'locked', 'deeper');
mkdirSync(inner, { recursive: true });
const leaf = join(inner, 'leaf.rb');
writeFileSync(leaf, 'x\n');
chmodSync(leaf, 0o000);
chmodSync(inner, 0o400);
chmodSync(join(root, 'locked'), 0o400);
const report = normalizeTreePermissions(root);
// Only reachable if the walk widened each parent before descending.
assert.equal(statSync(leaf).mode & 0o600, 0o600, 'leaf became owner-readable');
assert.ok(report.modeFixed.includes(leaf));
rmSync(root, { recursive: true, force: true });
});
test('leaves already-correct trees untouched', () => {
const root = scratch('noop');
mkdirSync(join(root, 'sub'), { recursive: true });
writeFileSync(join(root, 'sub', 'f.txt'), 'hi\n');
const report = normalizeTreePermissions(root);
assert.deepEqual(report.modeFixed, [], 'no mode changes');
assert.deepEqual(report.ownerFixed, [], 'no owner changes (already ours)');
assert.deepEqual(report.failures, []);
assert.equal(didRepair(report), false);
rmSync(root, { recursive: true, force: true });
});
test('does not widen group/other beyond what was already there', () => {
const root = ownedByHuman(scratch('narrow'));
const f = join(root, 'secret.txt');
writeFileSync(f, 'x\n');
ownedByHuman(f);
chmodSync(f, 0o000);
normalizeTreePermissions(root, { ownerRef: root });
const mode = statSync(f).mode & 0o777;
assert.equal(mode, 0o600, 'owner rw only — group/other stay closed');
rmSync(root, { recursive: true, force: true });
});
test('ignores symlinks rather than following them out of the tree', () => {
const root = scratch('symlink');
const outside = scratch('symlink-outside');
const victim = join(outside, 'victim.txt');
writeFileSync(victim, 'x\n');
chmodSync(victim, 0o000);
symlinkSync(outside, join(root, 'link'));
const report = normalizeTreePermissions(root);
assert.equal(statSync(victim).mode & 0o777, 0o000, 'target outside the tree untouched');
assert.deepEqual(report.failures, []);
rmSync(root, { recursive: true, force: true });
rmSync(outside, { recursive: true, force: true });
});
test('never throws on a missing root, and reports it', () => {
const report = normalizeTreePermissions(join(tmpdir(), 'definitely-not-here-xyz'));
assert.equal(report.failures.length, 1);
assert.equal(report.failures[0].reason, 'ENOENT');
});
test('resolveWorkspaceOwner reads the reference path, not the caller', () => {
const root = scratch('owner');
const owner = resolveWorkspaceOwner(root);
assert.ok(owner, 'resolved');
const st = statSync(root);
assert.equal(owner.uid, st.uid);
assert.equal(owner.gid, st.gid);
assert.equal(resolveWorkspaceOwner(join(tmpdir(), 'nope-xyz')), null);
rmSync(root, { recursive: true, force: true });
});
test('never chowns TO root, even when the owner ref is root-owned', () => {
// The regression this guards: workspace root owned by root (unzipped with
// sudo) while the task files are correctly owned by the human. Chowning to the
// ref's owner would inflict the very lockout this module prevents. `/` is
// root-owned on every platform we run on, so it's a stable stand-in.
const root = scratch('root-ref');
const f = join(root, 'mine.txt');
writeFileSync(f, 'x\n');
const beforeUid = statSync(f).uid;
const report = normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(report.target?.uid, 0, 'resolved a root target');
assert.deepEqual(report.ownerFixed, [], 'declined to chown anything to root');
assert.deepEqual(report.failures, [], 'and did not fail trying');
assert.equal(statSync(f).uid, beforeUid, 'owner untouched');
rmSync(root, { recursive: true, force: true });
});
test('still normalizes modes when the chown target is root', () => {
const root = scratch('root-ref-modes');
const sub = join(root, 'sub');
mkdirSync(sub, { recursive: true });
writeFileSync(join(sub, 'f.txt'), 'x\n');
chmodSync(sub, 0o400);
const report = normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(statSync(sub).mode & 0o700, 0o700, 'mode axis still applied');
assert.ok(report.modeFixed.includes(sub));
rmSync(root, { recursive: true, force: true });
});
test('keeps modes narrow for files that have a real owner, even under a root ref', () => {
// The complement of the case below: we declined to chown, but these entries are
// already the human's, so owner bits reach them and nothing should be widened.
const root = ownedByHuman(scratch('root-ref-narrow'));
const f = join(root, 'mine.txt');
writeFileSync(f, 'x\n');
ownedByHuman(f);
chmodSync(f, 0o600);
normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(statSync(f).mode & 0o077, 0, 'group/other untouched');
rmSync(root, { recursive: true, force: true });
});
test(
'grants read+search to group and other on files stranded root-owned',
{ skip: process.getuid?.() !== 0 ? 'needs root to create root-owned files' : false },
() => {
// The worker authoring container: root process, root-owned workspace. The chown
// is declined, so owner bits land on root and the human — a different uid in
// Explore and on a WSL host — is still locked out of a --w------- capture.
const root = scratch('stranded');
const sub = join(root, 'agent-output');
mkdirSync(sub, { recursive: true });
const f = join(sub, 'answer.md');
writeFileSync(f, 'x\n');
chmodSync(f, 0o200);
chmodSync(sub, 0o300);
normalizeTreePermissions(root, { ownerRef: '/' });
assert.equal(
statSync(f).mode & 0o777,
0o644,
'file readable by everyone, writable by none but root'
);
assert.equal(statSync(sub).mode & 0o777, 0o755, 'directory searchable');
}
);
test('walks a tree as deep as the filesystem allows', () => {
const root = scratch('deep');
// PATH_MAX caps how deep a tree can physically get (~300 levels at these name
// lengths — building deeper fails with ENAMETOOLONG), which is well inside any
// call-stack limit. So this isn't a stack test; it just pins that a deep,
// narrow tree walks cleanly end to end.
let path = root;
for (let i = 0; i < 250; i++) {
path = join(path, `d${i}`);
}
mkdirSync(path, { recursive: true });
writeFileSync(join(path, 'leaf.txt'), 'x\n');
chmodSync(join(path, 'leaf.txt'), 0o000);
const report = normalizeTreePermissions(root);
assert.deepEqual(report.failures, [], 'walked the whole depth cleanly');
assert.equal(statSync(join(path, 'leaf.txt')).mode & 0o600, 0o600, 'reached the deepest leaf');
rmSync(root, { recursive: true, force: true });
});
test('a failure in one subtree does not abandon the rest', () => {
const root = scratch('partial');
const good = join(root, 'good');
mkdirSync(good, { recursive: true });
const goodFile = join(good, 'f.txt');
writeFileSync(goodFile, 'x\n');
chmodSync(goodFile, 0o000);
// A dangling symlink and a vanished path both produce per-entry trouble.
symlinkSync(join(root, 'nowhere'), join(root, 'dangling'));
const report = normalizeTreePermissions(root);
assert.equal(statSync(goodFile).mode & 0o600, 0o600, 'the healthy subtree was still repaired');
assert.ok(report.modeFixed.includes(goodFile));
rmSync(root, { recursive: true, force: true });
});
test('reports rather than throws when the root is a file, not a directory', () => {
const root = scratch('file-root');
const f = join(root, 'lonely.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
const report = normalizeTreePermissions(f);
assert.equal(statSync(f).mode & 0o600, 0o600);
assert.deepEqual(report.failures, []);
rmSync(root, { recursive: true, force: true });
});
test('RACCOON_SKIP_PERMISSION_REPAIR=1 makes it a total no-op', () => {
const root = scratch('killswitch');
const sub = join(root, 'sub');
mkdirSync(sub, { recursive: true });
const f = join(sub, 'f.txt');
writeFileSync(f, 'x\n');
chmodSync(f, 0o000);
chmodSync(sub, 0o400);
const prev = process.env.RACCOON_SKIP_PERMISSION_REPAIR;
process.env.RACCOON_SKIP_PERMISSION_REPAIR = '1';
try {
const report = normalizeTreePermissions(root);
assert.equal(report.skipped, true);
assert.deepEqual(report.modeFixed, []);
assert.deepEqual(report.ownerFixed, []);
assert.deepEqual(report.failures, []);
assert.equal(didRepair(report), false);
assert.equal(statSync(sub).mode & 0o777, 0o400, 'directory left exactly as it was');
} finally {
if (prev === undefined) delete process.env.RACCOON_SKIP_PERMISSION_REPAIR;
else process.env.RACCOON_SKIP_PERMISSION_REPAIR = prev;
}
chmodSync(sub, 0o700);
rmSync(root, { recursive: true, force: true });
});
test('manual hint repairs both axes, ownership first', () => {
const hint = manualRepairHint('harbor-tasks/my-slug');
assert.match(hint, /chown -R/);
assert.match(hint, /chmod -R u\+rwX/);
assert.ok(hint.indexOf('chown') < hint.indexOf('chmod'), 'chown before chmod');
});

View File

@@ -1,126 +0,0 @@
/**
* tree-permissions.ts — make a copied tree readable by whoever owns the workspace.
*
* Files captured from a task run can arrive owned by another user, or with a
* directory missing the permission needed to walk into it. Packaging then fails
* with `Cannot stat: Permission denied`. This repairs both.
*
* Grants owner rwX only, never group or other. Never throws, and never hands
* files to root. Set `RACCOON_SKIP_PERMISSION_REPAIR=1` to turn it off.
*/
import { chmodSync, chownSync, lstatSync, readdirSync, statSync } from 'fs';
import { join } from 'path';
export interface NormalizeReport {
/** Paths whose owner was changed. */
ownerFixed: string[];
/** Paths whose mode gained owner rwX. */
modeFixed: string[];
/** Paths we wanted to change but could not, with the errno. */
failures: { path: string; reason: string }[];
/** Resolved target owner, or null if it couldn't be determined. */
target: { uid: number; gid: number } | null;
/** Set when disabled via RACCOON_SKIP_PERMISSION_REPAIR. */
skipped?: boolean;
}
/** Owner a workspace tree should have: whoever owns `ownerRef`. */
export function resolveWorkspaceOwner(ownerRef: string): { uid: number; gid: number } | null {
try {
const st = statSync(ownerRef);
return { uid: st.uid, gid: st.gid };
} catch {
return null;
}
}
/**
* Owner-rwX mode, preserving every other bit. Dirs also need the search bit.
*
* `stranded` means the file stays root-owned because we have no non-root owner to
* give it to. Owner bits then help nobody — whoever has to read it is a different
* user — so read and search are granted more widely. Never write, never +x on files.
*/
function withOwnerAccess(mode: number, isDir: boolean, stranded: boolean): number {
const owner = isDir ? 0o700 : 0o600;
return mode | owner | (stranded ? (isDir ? 0o055 : 0o044) : 0);
}
/**
* Give every entry under `root` to the workspace owner and make sure that owner
* can read and traverse it. Symlinks are skipped. Repairs what it can and
* reports what it couldn't; it never throws and never blocks its caller.
*/
export function normalizeTreePermissions(
root: string,
options: { ownerRef?: string } = {}
): NormalizeReport {
if (process.env.RACCOON_SKIP_PERMISSION_REPAIR === '1') {
return { ownerFixed: [], modeFixed: [], failures: [], target: null, skipped: true };
}
const target = resolveWorkspaceOwner(options.ownerRef ?? process.cwd());
const report: NormalizeReport = { ownerFixed: [], modeFixed: [], failures: [], target };
// Never hand files to root — that would lock the owner out rather than help.
const chownTarget = target && target.uid !== 0 ? target : null;
try {
const stack: string[] = [root];
while (stack.length > 0) {
const path = stack.pop() as string;
let st;
try {
st = lstatSync(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ELSTAT' });
continue;
}
if (st.isSymbolicLink()) continue;
const isDir = st.isDirectory();
// Mode first: a directory we can't search is one we can't descend into.
const wanted = withOwnerAccess(st.mode, isDir, chownTarget === null && st.uid === 0);
if (wanted !== st.mode) {
try {
chmodSync(path, wanted);
report.modeFixed.push(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHMOD' });
}
}
if (chownTarget && (st.uid !== chownTarget.uid || st.gid !== chownTarget.gid)) {
try {
chownSync(path, chownTarget.uid, chownTarget.gid);
report.ownerFixed.push(path);
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'ECHOWN' });
}
}
if (!isDir) continue;
try {
for (const entry of readdirSync(path)) stack.push(join(path, entry));
} catch (err) {
report.failures.push({ path, reason: (err as NodeJS.ErrnoException).code ?? 'EREADDIR' });
}
}
} catch (err) {
report.failures.push({ path: root, reason: (err as NodeJS.ErrnoException).code ?? 'EWALK' });
}
return report;
}
/** True when something was actually repaired. */
export function didRepair(report: NormalizeReport): boolean {
return report.ownerFixed.length > 0 || report.modeFixed.length > 0;
}
/** The command to run on your host if we couldn't fix it ourselves. */
export function manualRepairHint(path: string): string {
return `sudo chown -R "$(id -un):$(id -gn)" ${path} && chmod -R u+rwX ${path}`;
}

View File

@@ -1,80 +0,0 @@
/**
* record-detector-inputs.ts — stamp detector report(s) with the checksums of
* the task inputs they assessed.
*
* Run this right after a detector skill writes (or rewrites)
* harbor-tasks/<slug>/detectors/<detector-name>.md. It records a sha256
* capture of the task inputs next to the report, as
* detectors/<detector-name>.inputs.json, so submit-task.ts can tell by
* content — not by file timestamp — whether the report still matches the
* task being packaged.
*
* Usage:
* npx tsx scripts/record-detector-inputs.ts <task-slug> <detector-name> [detector-name...]
* npx tsx scripts/record-detector-inputs.ts my-cool-task detector-rubric-clarity
*/
import { existsSync, mkdirSync, writeFileSync } from 'fs';
import { join } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import { captureTaskInputs } from './lib/input-checksums';
const argv = yargs(hideBin(process.argv))
.usage(
'$0 <slug> <detectors...>',
'Record the task-input checksums a detector report assessed',
(y) =>
y
.positional('slug', { type: 'string', demandOption: true, describe: 'Task slug' })
.positional('detectors', {
type: 'string',
array: true,
demandOption: true,
describe: 'Detector name(s), e.g. detector-rubric-clarity',
})
)
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.help()
.parseSync();
const log = pino(
{ name: 'record-detector-inputs', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
const slug = argv.slug as string;
const detectors = argv.detectors as string[];
const taskDir = join(process.cwd(), 'harbor-tasks', slug);
if (!existsSync(taskDir)) {
log.fatal({ taskDir }, 'Task directory not found');
process.exit(1);
}
// One capture serves every report stamped in this invocation — they all
// assessed the same on-disk revision of the task.
const capture = captureTaskInputs(taskDir, 'stamp');
const detectorsDir = join(taskDir, 'detectors');
mkdirSync(detectorsDir, { recursive: true });
for (const name of detectors) {
const report = join(detectorsDir, `${name}.md`);
if (!existsSync(report)) {
// Stamp anyway — skills sometimes stamp before the final write lands —
// but say so, since a stamp with no report usually means a typo'd name.
log.warn({ report: `detectors/${name}.md` }, 'No report found for this detector name');
}
const stampPath = join(detectorsDir, `${name}.inputs.json`);
writeFileSync(stampPath, JSON.stringify(capture, null, 2) + '\n');
log.info({ stamp: `detectors/${name}.inputs.json` }, 'Recorded task-input checksums');
}

View File

@@ -1,166 +0,0 @@
"""Pure (no-harbor) helpers for inspecting a captured reference run.
Kept separate from ``replay_agent.py`` (which imports ``harbor``) so this logic
can be unit-tested with the plain devcontainer Python and shipped in the worker
toolkit alongside the replay agent.
The one safety-critical helper here is :func:`captured_mutating_tools`: it tells
the replay agent whether a run with no ``agent-output/`` is a harmless advisory
run (the agent only read + answered in chat) or a genuine capture loss (the
agent edited files but they weren't preserved). The replay agent grades the
former from the captured transcript and refuses the latter.
Structured edit tools (``Write``/``Edit``/``MultiEdit``/``NotebookEdit``) are
obvious. ``Bash`` is the subtle one: a shell call can mutate the workspace
(``rm``, ``mv``, ``sed -i``, ``echo … > f`` …) just as easily as it can read it.
So a ``Bash`` call is treated as **potentially mutating unless the command is
verifiably read-only** (:func:`bash_mutates`) — the safe direction: an unknown
command counts as a mutation, so we never silently grade a run that lost edits.
"""
from __future__ import annotations
import json
import re
from pathlib import Path
# Structured tools that always mutate the workspace.
MUTATING_TOOLS = frozenset({"Write", "Edit", "MultiEdit", "NotebookEdit"})
# Base commands that only read (or touch non-workspace state like cwd). Anything
# NOT here — or any file-writing redirection, or `sed -i`, or a non-read-only git
# subcommand — is treated as potentially mutating.
_READONLY_BASH = frozenset({
"ls", "cat", "head", "tail", "grep", "egrep", "fgrep", "rg", "ag", "find",
"fd", "wc", "echo", "printf", "file", "stat", "pwd", "tree", "sort", "uniq",
"cut", "tr", "awk", "jq", "yq", "less", "more", "diff", "cmp", "basename",
"dirname", "realpath", "readlink", "true", "false", "test", "[", "date",
"env", "printenv", "which", "type", "command", "column", "nl", "od", "xxd",
"hexdump", "comm", "paste", "fold", "expand", "tac", "du", "df", "seq",
"sleep", ":", "cd", "pushd", "popd", "dirs", "whoami", "hostname", "uname",
"id", "cksum", "md5sum", "sha1sum", "sha256sum", "strings", "wc",
})
# git subcommands that don't write the repo/workspace.
_READONLY_GIT_SUB = frozenset({
"log", "diff", "status", "show", "blame", "grep", "ls-files", "ls-tree",
"cat-file", "rev-parse", "describe", "shortlog", "reflog", "rev-list",
"for-each-ref", "name-rev", "symbolic-ref", "whatchanged", "var", "help",
"show-ref", "merge-base", "cherry", "count-objects", "verify-pack",
})
# fd-dups (2>&1, >&2, 1>&-) and /dev/null sinks are harmless; strip them before
# looking for a real file-writing redirection.
_HARMLESS_REDIR = re.compile(r"[0-9&]*>>?\s*(?:&\s*[0-9-]+|/dev/null)")
# Split a command line into segments on shell separators + substitutions.
_SEG_SPLIT = re.compile(r"\|\||&&|[|;&\n]|\$\(|`")
_ASSIGN = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*=")
def bash_mutates(command: str) -> bool:
"""Heuristic: does this shell command potentially write the workspace?
Conservative by design — errs toward True (an unrecognized command, a
file-writing redirect, `sed -i`, or a non-read-only git subcommand all count
as mutating). Read-only exploration (ls/cat/grep/find/wc/… piped together,
with `2>/dev/null` / `>/dev/null`) returns False."""
if not command or not command.strip():
return False
# Drop fd-dups (2>&1) + /dev/null sinks up front, so they neither look like a
# file write nor split a segment on their `&` (2>&1 → bogus "1" command).
stripped = _HARMLESS_REDIR.sub(" ", command)
# 1. Any remaining redirection now writes a real file.
if ">" in stripped:
return True
# 2. The leading command of every segment must be read-only.
for seg in _SEG_SPLIT.split(stripped):
toks = seg.split()
idx = 0
while idx < len(toks) and _ASSIGN.match(toks[idx]): # skip VAR=val prefixes
idx += 1
if idx >= len(toks):
continue
cmd = toks[idx].rsplit("/", 1)[-1]
rest = toks[idx + 1:]
if cmd == "sed" and any(t == "-i" or t.startswith("-i") for t in rest):
return True
if cmd == "git":
sub = next((t for t in rest if not t.startswith("-")), "")
if sub and sub not in _READONLY_GIT_SUB:
return True
continue
if cmd and cmd not in _READONLY_BASH:
return True
return False
def _bash_command(call_args) -> str:
if isinstance(call_args, dict):
return str(call_args.get("command", "") or "")
return ""
def _scan_trajectory(trajectory_path: Path) -> set[str]:
"""Mutating tool names in an ATIF agent/trajectory.json
(steps[].tool_calls[].function_name; Bash inspected by command)."""
found: set[str] = set()
if not trajectory_path.exists():
return found
try:
data = json.loads(trajectory_path.read_text())
except (json.JSONDecodeError, OSError):
return found
for step in data.get("steps", []):
for call in step.get("tool_calls") or []:
name = call.get("function_name")
if name in MUTATING_TOOLS:
found.add(name)
elif name == "Bash" and bash_mutates(_bash_command(call.get("arguments"))):
found.add("Bash")
return found
def _scan_stream_json(stream_path: Path) -> set[str]:
"""Mutating tool names in a raw stream-json claude-code.txt
(one JSON object per line, message.content[].tool_use; Bash by input)."""
found: set[str] = set()
if not stream_path.exists():
return found
try:
lines = stream_path.read_text(errors="ignore").splitlines()
except OSError:
return found
for line in lines:
line = line.strip()
if not line:
continue
try:
obj = json.loads(line)
except json.JSONDecodeError:
continue
message = obj.get("message") if isinstance(obj, dict) else None
content = message.get("content") if isinstance(message, dict) else None
if not isinstance(content, list):
continue
for block in content:
if not (isinstance(block, dict) and block.get("type") == "tool_use"):
continue
name = block.get("name")
if name in MUTATING_TOOLS:
found.add(name)
elif name == "Bash" and bash_mutates(_bash_command(block.get("input"))):
found.add("Bash")
return found
def captured_mutating_tools(reference_run_dir: Path | str) -> set[str]:
"""Return the file-mutating tool names found in a captured run's transcript.
Checks the ATIF ``agent/trajectory.json`` first, then falls back to the raw
stream-json ``agent/claude-code.txt``, so a lossy/partial trajectory can't
hide a real edit. ``Bash`` is included only when its command isn't verifiably
read-only (see :func:`bash_mutates`). An empty result means the agent made no
workspace edits — i.e. a missing ``agent-output/`` is an advisory no-op.
"""
ref = Path(reference_run_dir)
return _scan_trajectory(ref / "agent" / "trajectory.json") | _scan_stream_json(
ref / "agent" / "claude-code.txt"
)

View File

@@ -1,54 +0,0 @@
#!/bin/bash
# Rewrite the auth FILES harnesses read their key from — and the base URL beside them —
# off the live .env, then exec "$@".
#
# codex reads its key from ${CODEX_HOME:-$HOME/.codex}/auth.json, which container-create
# wrote once from the .env of that moment — so a key rotated afterwards never reached it
# and needed a rebuild. claude needs none of this: it has an apiKeyHelper that re-reads
# .env per request. Interactive launches route through here so each one re-derives first.
#
# The base URL never rotates, so the case that matters is the one where container-create
# could not derive it at all (no .env yet) and wrote no config: the key then refreshes
# fine while codex still has no proxy URL and talks to the provider directly.
#
# Trials are unaffected either way: harbor-run re-derives OPENAI_API_KEY per invocation
# and harbor's codex agent authenticates the sandbox from that env var, not from this file.
set -uo pipefail
_scripts_dir="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
# Subshell, and every failure swallowed: a refresh that cannot run must never stop the
# agent from starting. The auth file already on disk is the PREVIOUS key, not nothing, so
# failing open leaves the worker exactly where they were before this wrapper existed.
(
set -a
# shellcheck disable=SC1090
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true
set +a
# shellcheck disable=SC1091
HARNESS_SCRIPTS_DIR="$_scripts_dir" . "$_scripts_dir/lib/harness-credentials.sh" || exit 0
harness_setup_credentials
harness_write_auth
harness_refresh_config_keys
) >/dev/null 2>&1 || true
# No args is a valid call: refresh only, for a lifecycle hook.
[ "$#" -gt 0 ] || exit 0
# Outside the subshell, because these have to reach the exec'd command: codex now reads
# its key from $ANTHROPIC_API_KEY per request, and a non-login shell sourced neither
# .bashrc (the key, the call origin) nor the profile that puts the CLI on PATH.
# Failures stay swallowed — an unreadable .env must not stop the agent starting.
export PATH="$HOME/.local/bin:$PATH"
if [ -f "${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
set -a
# shellcheck disable=SC1090
. "${RACCOON_ENV_FILE:-/workspace/.env}" 2>/dev/null || true
set +a
fi
if [ -f "$HOME/.raccoon-call-origin" ]; then
# shellcheck disable=SC1091
. "$HOME/.raccoon-call-origin" 2>/dev/null || true
fi
exec "$@"

View File

@@ -1,201 +0,0 @@
"""
Harbor agent adapter that re-grades an existing reference run by replaying
its captured workspace state — no model calls, no agent work.
Reads a `reference_run_dir` pointing at a `reference-runs/<id>/` directory
captured during a prior real trial. Inside that dir:
agent/trajectory.json — the grader reads this via the symlink
/tmp/outputs/task_transcript.txt → /logs/agent/trajectory.json
agent-output/ — files the agent created or modified (captured by
tests/test.sh after the agent ran)
agent-output/_HARBOR_DELETIONS.txt
— list of tracked files the agent deleted (one path
per line). Empty/absent when nothing was deleted.
On run() (after the trial's docker env is up and /workspace has the base
state from the Dockerfile's COPY workspace/), the adapter:
1. Uploads agent-output/ into /workspace — overlays the agent's surviving
edits on top of the base workspace.
2. Applies _HARBOR_DELETIONS.txt by `rm -f`'ing each listed path under
/workspace. (The marker file itself was uploaded in step 1; it gets
removed too so the verifier's re-capture doesn't pick it up as
untracked content.)
3. Uploads trajectory.json to /logs/agent/trajectory.json so the grader
sees the same transcript it would have on the original run.
Verifier then runs as it would for any real trial — exact same code path,
exact same artifacts, just with the agent phase replaced by a deterministic
file-overlay. See scripts/harbor-regrade for the host-side wrapper.
Older reference runs captured before the deletion-capture line shipped
(verifier-deletion-capture PR #219) won't have _HARBOR_DELETIONS.txt; the
deletion-replay step is a no-op in that case. Deletions made in those runs
remain lost.
"""
from pathlib import Path, PurePosixPath
from harbor.agents.base import BaseAgent
from harbor.environments.base import BaseEnvironment
from harbor.models.agent.context import AgentContext
from harbor.models.trial.paths import EnvironmentPaths
# Sibling module (shipped alongside in the worker toolkit; on PYTHONPATH via
# harbor-regrade). Kept harbor-free so its logic stays unit-testable.
from reference_run_capture import captured_mutating_tools
_WORKSPACE = PurePosixPath("/workspace")
_DELETIONS_MARKER = "_HARBOR_DELETIONS.txt"
class ReplayAgent(BaseAgent):
"""Replays a captured reference run so the verifier can be re-graded
without invoking the model again."""
SUPPORTS_WINDOWS: bool = False # paths below assume POSIX /workspace.
def __init__(
self,
logs_dir,
reference_run_dir: str,
source_agent_import_path: str | None = None,
source_model_name: str | None = None,
**kwargs,
):
# `source_*` are provenance, not behaviour: harbor-regrade reads them off the
# source run and passes them so THIS replay's result.json records which
# harness and model produced the trajectory being graded. A replay reports
# `replay_agent:ReplayAgent` with model_name null, and the source run is
# usually deleted (a regrade is copied back over what it regraded), so a bare
# reference_run_dir pointer does not survive as provenance.
#
# They must be accepted here rather than left in **kwargs: harbor records the
# trial config's agent kwargs regardless of what the agent does with them, but
# BaseAgent would reject the unknown keys and take every regrade down with it.
self._source_agent_import_path = source_agent_import_path
self._source_model_name = source_model_name
super().__init__(logs_dir=logs_dir, **kwargs)
ref = Path(reference_run_dir).expanduser().resolve()
if not ref.is_dir():
raise FileNotFoundError(f"reference_run_dir does not exist: {ref}")
self._reference_run_dir = ref
self._agent_output_dir = ref / "agent-output"
self._trajectory_path = ref / "agent" / "trajectory.json"
@staticmethod
def name() -> str:
return "replay"
def version(self) -> str:
return "1.0.0"
async def setup(self, environment: BaseEnvironment) -> None:
# No installation needed; the verifier brings everything it requires.
return
async def run(
self,
instruction: str,
environment: BaseEnvironment,
context: AgentContext,
) -> None:
# Advisory tasks — the agent only reads and answers in chat — make NO
# workspace edits, so a faithful capture of one has an empty (or, in
# older pipelines, absent) agent-output/. That is not a data gap: the
# deliverable is the agent's final message, captured in
# agent/trajectory.json, which the grader reads via
# /tmp/outputs/task_transcript.txt. So overlay captured edits when
# present; otherwise grade the base workspace + transcript, exactly
# what the original advisory grading saw.
if not self._agent_output_dir.is_dir():
mutating = captured_mutating_tools(self._reference_run_dir)
if mutating:
# The agent edited files but they weren't captured — grading the
# base workspace would silently score the wrong state. Refuse.
raise FileNotFoundError(
f"reference_run_dir {self._reference_run_dir} has no "
f"agent-output/ but its captured transcript shows "
f"file-mutating tool calls {sorted(mutating)}. The agent's "
f"workspace edits were lost at capture time, so this run "
f"cannot be faithfully re-graded — re-capture it."
)
self.logger.warning(
"reference_run %s has no agent-output/ and made no "
"file-mutating tool calls — treating it as an advisory run and "
"grading the base workspace + captured transcript.",
self._reference_run_dir,
)
# Skip the overlay/deletion steps; fall through to trajectory upload.
await self._upload_trajectory(environment)
return
# 1. Overlay captured agent edits onto the base /workspace.
await environment.upload_dir(
source_dir=str(self._agent_output_dir),
target_dir=str(_WORKSPACE),
)
# 2. Apply captured deletions, if present. Read the marker from the
# host so we don't have to shell into the container to parse it,
# then issue per-path rm's plus a final cleanup of the marker
# itself (which was uploaded in step 1).
host_marker = self._agent_output_dir / _DELETIONS_MARKER
if host_marker.exists():
deletion_paths = [
line.strip()
for line in host_marker.read_text().splitlines()
if line.strip()
]
for raw in deletion_paths:
self._validate_relative_path(raw)
await environment.exec(
command=f'rm -f -- "/workspace/{raw}"',
user="root",
)
await environment.exec(
command=f'rm -f -- "/workspace/{_DELETIONS_MARKER}"',
user="root",
)
# 3. Materialize the captured trajectory at the path the grader's
# test.sh symlinks to /tmp/outputs/task_transcript.txt.
await self._upload_trajectory(environment)
async def _upload_trajectory(self, environment: BaseEnvironment) -> None:
"""Upload agent/trajectory.json to the path the grader's test.sh
symlinks to /tmp/outputs/task_transcript.txt. Shared by the normal
(overlay) path and the advisory (no agent-output) path."""
if self._trajectory_path.exists():
env_paths = EnvironmentPaths.for_os(environment.os)
await environment.upload_file(
source_path=str(self._trajectory_path),
target_path=str(env_paths.agent_dir / "trajectory.json"),
)
else:
# The grader's test.sh reads /tmp/outputs/task_transcript.txt,
# which symlinks to trajectory.json. Without the file the symlink
# dangles and the grader sees an empty transcript — so the regrade
# will look like the agent did nothing. Yell via harbor's own
# logger (self.logger is a child of harbor.utils.logger) so the
# warning lands in trial.log, not a stray "replay-agent" logger
# nothing's wired to.
self.logger.warning(
"reference_run %s has no agent/trajectory.json — the grader "
"will see an empty transcript. Investigate whether the source "
"run was produced by an older harbor that didn't write the "
"ATIF file (or by snapshot_agent before the multi-JSONL fix).",
self._reference_run_dir,
)
@staticmethod
def _validate_relative_path(raw: str) -> None:
"""Guard against absolute paths and `..` traversal in the deletions
manifest. The marker should only list paths *under* the workspace
root; anything else is a captured-data integrity problem worth
failing loudly on."""
if not raw or raw.startswith("/"):
raise ValueError(f"refusing absolute path in {_DELETIONS_MARKER}: {raw!r}")
if ".." in PurePosixPath(raw).parts:
raise ValueError(f"refusing `..` traversal in {_DELETIONS_MARKER}: {raw!r}")

View File

@@ -1,459 +0,0 @@
#!/usr/bin/env python3
"""resolve_harness.py — turn a harness id + task dir into the flags a trial needs.
``scripts/harbor-run`` is bash and cannot parse the TOML registry, so it shells in
here and evals the result::
RESOLVED="$(python3 scripts/resolve_harness.py --task-dir "$TASK_DIR")" || exit 1
eval "$RESOLVED"
Python rather than TS on purpose: this ships in the worker toolkit, whose
package.json has no ``zod``/``smol-toml``, and ``tomllib`` is stdlib — so the
toolkit gains a harness-aware harbor-run with zero new dependencies. There is no TS
loader: TS callers (submit-task) shell in here, so both the schema and the selection
policy exist exactly once and there is nothing to drift.
Output is POSIX ``KEY='value'`` assignments (single-quoted, embedded quotes
escaped) on stdout; everything human-facing goes to stderr, so the eval only ever
sees assignments. A non-zero exit means "do not launch" — the point is to fail in a
second rather than burn agent minutes on a trial that cannot produce a usable grade.
Refuses to resolve when:
- the harness id is unknown or disabled
- the harness writes no ATIF trajectory (the grader would have no transcript)
- the task ships a session to resume but the harness cannot resume one. This is
the important one: it is the only failure here that would otherwise look like
SUCCESS, with the agent answering a prompt whose conversation it never saw.
- the harness's credential env var is unset
``--check-model`` additionally asks the proxy whether the model is granted. Opt-in
on purpose: it is a network call, and one in every run's critical path trades a fast
local failure for a new way to hang. The credential check, which is free, always runs.
"""
from __future__ import annotations
import argparse
import json
import os
import re
import shlex
import sys
import tomllib
import urllib.error
import urllib.request
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent / "lib"))
from harness_registry import ( # noqa: E402
Harness,
HarnessRegistryError,
load_harness_registry,
)
# Harness used when nothing selects one. Keeps every existing caller on today's
# behaviour, so adding harness selection changes no current run.
DEFAULT_HARNESS = "claude-code"
MODELS_TIMEOUT_SEC = 20
def warn(message: str) -> None:
print(f"resolve-harness: {message}", file=sys.stderr)
def fail(message: str) -> "None":
print(f"resolve-harness: ERROR: {message}", file=sys.stderr)
raise SystemExit(1)
def is_multi_turn(task_dir: str | None) -> bool:
"""A task is multi-turn when it ships a NON-EMPTY session to resume. Empty is
the documented one-shot-snapshot fallback and must run cold, so size is the
test, not existence."""
if not task_dir:
return False
session = Path(task_dir) / "environment" / "session.jsonl"
return session.is_file() and session.stat().st_size > 0
def wants_browser(task_dir: str | None) -> bool:
"""True when task.toml opts into a browser (`[metadata] browser = true`).
Read straight from the file rather than via tomllib: this must agree with
build-workspace.sh, which decides whether the IMAGE gets Playwright using the same
text match. If the two ever disagree the agent is told about a browser the image
lacks, which is the one failure the disclosure is designed to make impossible.
Accepts the quoted form for the same reason build-workspace.sh does."""
if not task_dir:
return False
toml_path = Path(task_dir) / "task.toml"
if not toml_path.is_file():
return False
try:
text = toml_path.read_text(encoding="utf-8")
except OSError:
return False
return re.search(r'^[ \t]*browser[ \t]*=[ \t]*"?true"?[ \t]*$', text, re.M) is not None
def harness_from_task_toml(task_dir: str | None) -> str | None:
"""The task's own `[agent] harness` — the authoritative record of which harness
this task was authored against.
This is where the worker's choice lands: the snapshot flow stamps it from the CLI
that produced the snapshot, and a manual author writes it themselves. Either way
it is set at task-creation time, BEFORE any trial, so nothing here depends on a
trial's output.
Parsed with tomllib rather than a grep: a regex would happily match a commented
line or the wrong table, and picking the wrong harness is a silent
wrong-agent-runs bug.
Returns None when the field is simply absent — the normal case for every task
finalized before harness selection existed — so the caller falls through to the
toolkit default.
But an UNPARSEABLE task.toml refuses outright rather than falling back. Those are
different situations and treating them alike is how the wrong harness runs
quietly: the most likely way to break this file is adding a second `[agent]`
table instead of a `harness` line inside the existing one (tasks already carry
`[agent] timeout_sec`), and TOML rejects a duplicate table. Falling back there
would run claude against a task its author wrote for codex and grade it as if
nothing were wrong.
"""
if not task_dir:
return None
path = Path(task_dir) / "task.toml"
if not path.is_file():
return None
try:
with open(path, "rb") as handle:
doc = tomllib.load(handle)
except (OSError, tomllib.TOMLDecodeError) as exc:
fail(
f"{path} could not be parsed ({exc}). Refusing to guess a harness — fix "
f'the file. If you were adding a harness, put `harness = "..."` inside '
f"the EXISTING [agent] table rather than starting a second one."
)
harness = (doc.get("agent") or {}).get("harness")
return harness if isinstance(harness, str) and harness else None
def normalize_model(harness: Harness, model: str) -> str:
"""Model id on the wire, per the harness's declared shape."""
if harness.model_id_shape == "provider:model":
return model.replace("/", ":")
return model
def granted_models(harness: Harness) -> list[str] | None:
"""Model ids the key is granted, or None when the check couldn't run."""
base_url = os.environ.get(harness.base_url_env or "")
key = os.environ.get(harness.key_env or "")
if not base_url or not key:
warn("--check-model skipped: base URL or key env is unset")
return None
request = urllib.request.Request(
f"{base_url.rstrip('/')}/models", headers={"Authorization": f"Bearer {key}"}
)
try:
with urllib.request.urlopen(request, timeout=MODELS_TIMEOUT_SEC) as response:
body = json.loads(response.read().decode("utf-8"))
except (urllib.error.URLError, TimeoutError, ValueError, OSError) as exc:
warn(f"--check-model skipped: /models unreachable ({exc})")
return None
return [m["id"] for m in body.get("data", []) if isinstance(m.get("id"), str)]
def assert_model_granted(harness: Harness, model: str) -> None:
granted = granted_models(harness)
if granted is None:
return
# The proxy LISTS ids provider-prefixed ("openai/gpt-5.6-sol") but 400s on that
# form — requests take the bare id. Accept either spelling.
bare = {g.split("/")[-1] for g in granted}
if model not in granted and model not in bare:
shown = ", ".join(granted[:12]) + (", …" if len(granted) > 12 else "")
fail(
f'Model "{model}" is not granted for this key. Granted ({len(granted)}): {shown}'
)
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
parser.add_argument(
"--harness",
help=f"harness id (default: the task's [agent] harness, else {DEFAULT_HARNESS})",
)
parser.add_argument(
"--task-dir",
help="task directory; decides multi-turn from environment/session.jsonl",
)
parser.add_argument("--model", help="override the harness's default model")
parser.add_argument(
"--fast",
action="store_true",
help="run the trial agent in the harness's fast serving mode (higher token "
"rate, faster output). Refuses on a harness that has none.",
)
parser.add_argument(
"--check-model",
action="store_true",
help="also ask the proxy whether the model is granted (network call)",
)
parser.add_argument(
"--authoring-installs",
action="store_true",
help="print '<id>\\t<cli>\\t<install>' for each harness a worker can author "
"with, and exit. Consumed by scripts/setup-harnesses.sh so the worker "
"containers install from the registry rather than from hardcoded lists that "
"drift.",
)
parser.add_argument(
"--container-configs",
action="store_true",
help="print '<id>\\t<config_path>\\t<base64 container_config>' for each "
"authoring harness that declares one, and exit. Base64 because the config is "
"multi-line TOML and these query modes are line-oriented.",
)
parser.add_argument(
"--surface",
choices=("authoring", "explore"),
default="authoring",
help="which worker container --container-configs is for; explore additionally "
"gets the capture hooks, whose commands only ship there.",
)
parser.add_argument(
"--defaults",
action="store_true",
help="print '<id>\\t<default_model>\\t<effort_default>' for every harness, and "
"exit. For recording what a task was authored against; nothing reads it back.",
)
parser.add_argument(
"--explore-launchers",
action="store_true",
help="print '<id>\\t<cli>\\t<launch command>' for each authoring harness, and "
"exit. The launch command has the registry's model, effort and agent_config "
"already substituted, so Explore and a trial cannot disagree about them. "
"Consumed by setup-harnesses.sh to write one launcher per harness.",
)
parser.add_argument(
"--skills-dirs",
action="store_true",
help="print '<id>\t<skills_dir>' for each authoring harness that discovers "
"skills from a directory, and exit. Lets setup-harnesses.sh install the "
"snapshot skill for harnesses that have no plugin system.",
)
parser.add_argument(
"--auth-files",
action="store_true",
help="print '<id>\t<auth_path>\t<auth_key_env>' for each authoring harness that "
"authenticates from a file rather than the environment, and exit.",
)
parser.add_argument(
"--authoring-credentials",
action="store_true",
help="print '<id>\\t<key_env>\\t<base_url_env>\\t<proxy_path>' for each "
"authoring harness, and exit. Lets the containers point every harness at the "
"same proxy key on its own provider path.",
)
parser.add_argument(
"--declared-harness",
action="store_true",
help="print ONLY the harness --task-dir's task.toml declares (empty if it "
"declares none) and exit. Unlike the default mode this applies no fallback, so "
"a caller can tell 'declared' from 'defaulted'. Exists so consumers without a "
"TOML parser never hand-roll one: a regex would match a commented line or the "
"wrong table, and the duplicate-[agent] shape is exactly the likely mistake.",
)
parser.add_argument(
"--resolve-identity",
action="append",
default=None,
metavar="AGENT",
help="resolve agent identities (a result.json config.agent import_path or name) "
"to harness ids and exit; repeatable. Prints one TAB-separated "
"'<identity>\\t<harness-id>' line each, with an empty id when nothing claims it. "
"Lets callers that cannot import the registry (the worker toolkit has no "
"zod/smol-toml) still resolve through the one source of truth.",
)
parser.add_argument(
"--list",
action="store_true",
help="print the selectable harnesses and exit (what task.toml's [agent] harness accepts)",
)
parser.add_argument("--registry", default=None, help="registry path (tests)")
args = parser.parse_args(argv)
try:
registry = (
load_harness_registry(args.registry)
if args.registry
else load_harness_registry()
)
except HarnessRegistryError as exc:
fail(str(exc))
# --- read-only query modes: answer and exit, never emit assignments -------
if args.authoring_installs:
for harness in registry.authoring():
print(f"{harness.id}\t{harness.cli or ''}\t{harness.install or ''}")
return 0
if args.container_configs:
import base64
for harness in registry.authoring():
config = harness.container_config_text(surface=args.surface)
if not (harness.config_path and config):
continue
blob = base64.b64encode(config.encode()).decode()
print(f"{harness.id}\t{harness.config_path}\t{blob}")
return 0
if args.defaults:
for harness in registry.all():
print(
f"{harness.id}\t{harness.default_model or ''}\t"
f"{harness.effort_default or ''}"
)
return 0
if args.explore_launchers:
for harness in registry.authoring():
print(
f"{harness.id}\t{harness.cli or ''}\t"
f"{harness.explore_launch_command() or ''}"
)
return 0
if args.skills_dirs:
for harness in registry.authoring():
if harness.skills_dir:
print(f"{harness.id}\t{harness.skills_dir}")
return 0
if args.auth_files:
for harness in registry.authoring():
if harness.auth_path and harness.auth_key_env:
print(f"{harness.id}\t{harness.auth_path}\t{harness.auth_key_env}")
return 0
if args.authoring_credentials:
for harness in registry.authoring():
print(
f"{harness.id}\t{harness.key_env or ''}\t"
f"{harness.base_url_env or ''}\t{harness.proxy_path or ''}"
)
return 0
if args.declared_harness:
print(harness_from_task_toml(args.task_dir) or "")
return 0
if args.resolve_identity:
for identity in args.resolve_identity:
harness = registry.by_import_path(identity)
print(f"{identity}\t{harness.id if harness else ''}")
return 0
if args.list:
# Printed on stdout because it is the requested output here, not the
# eval-able assignments — this mode is for a human, and never shelled into.
for harness in registry.enabled():
turns = (
"multi-turn + single-turn"
if harness.seed_native
else "single-turn only"
)
model = harness.default_model or "(pass --model)"
print(f"{harness.id:<16} {harness.label:<20} {turns:<24} {model}")
return 0
# --- selection ------------------------------------------------------------
# Precedence: an explicit --harness (a benchmark, or a deliberate override) beats the
# task's own record, which is what its author chose. Everything else — every task
# finalized before harness selection existed — is the default.
requested = args.harness or harness_from_task_toml(args.task_dir) or DEFAULT_HARNESS
try:
harness = registry.require(requested)
except HarnessRegistryError as exc:
fail(str(exc))
if not harness.enabled:
fail(
f'Harness "{harness.id}" is disabled in the registry (never verified here). '
f"Enable it in scripts/harness-registry.toml once a trial has been run with it."
)
if not harness.writes_atif:
fail(
f'Harness "{harness.id}" writes no ATIF trajectory, so the grader would have '
f"no transcript and its rewards would be meaningless."
)
multi_turn = is_multi_turn(args.task_dir)
if multi_turn and not harness.seed_native:
fail(
f'Task ships a session to resume, but harness "{harness.id}" cannot resume '
f"one. Running anyway would look like a success while the agent answered a "
f"prompt whose conversation it never saw."
)
if harness.key_env and not os.environ.get(harness.key_env):
fail(f'{harness.key_env} is unset — required by harness "{harness.id}".')
if args.fast and not harness.fast_kwarg:
fail(
f'Harness "{harness.id}" has no fast serving mode (no fast_kwarg in the '
f"registry). Drop --fast or pick a harness that declares one."
)
model = args.model or harness.default_model
if not model:
fail(
f'Harness "{harness.id}" has no default_model in the registry; pass --model '
f"explicitly."
)
if args.check_model:
assert_model_granted(harness, model)
# Every assignment here becomes a harbor flag. Nothing else: the caller is bash, and
# anything it would only echo back at the worker is said below instead.
browser = wants_browser(args.task_dir)
assignments = {
"AGENT_IMPORT_PATH": harness.agent_import_path_for(multi_turn=multi_turn, browser=browser),
"MODEL": normalize_model(harness, model),
"EFFORT_KWARG": harness.effort_kwarg,
"EFFORT_VALUE": (harness.effort_default or "") if harness.effort_kwarg else "",
"FAST_KWARG": harness.fast_kwarg if args.fast else "",
}
warn(
f"{harness.label} · model={assignments['MODEL']} · "
f"{'multi-turn' if multi_turn else 'single-turn'} · "
f"{'browser · ' if browser else ''}"
f"{'fast · ' if args.fast else ''}"
f"agent={assignments['AGENT_IMPORT_PATH']}"
)
if browser and not harness.agent_import_path_browser:
# Not a failure: the image still gets Playwright and the agent is still told about
# it. Only the Read-enabled toolset swap is claude-specific, and saying so beats
# letting someone infer from a log line that the opt-in was ignored entirely.
warn(
f'"{harness.id}" has no browser-specific agent, so it runs its usual toolset. '
f"The browser and its disclosure are unaffected."
)
if harness.flaky_hangs:
warn(
f"{harness.label} is known to hang with no client-side timeout on a small "
f"fraction of trials. A silent, output-less trial is that, not a task defect."
)
for key, value in assignments.items():
print(f"{key}={shlex.quote(value)}")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -1,260 +0,0 @@
/**
* Strip machine-identifying filesystem paths, and optional keywords, from a session
* transcript. Pure: raw JSONL in, JSONL out, no I/O.
*/
export const DEFAULT_PLACEHOLDER = '~/repo';
export const HOME_DIR_PLACEHOLDER = '~';
export const REDACTION_PLACEHOLDER = '[redacted]';
export interface SanitizeOptions {
/** Replacement for the cwd-prefix. Its dash-encoded form is derived from it. */
placeholder?: string;
/** Keyword regexes to redact. Empty by default, leaving a pure path-scrubber. */
forbiddenMarkers?: readonly RegExp[];
/**
* Exact prefix to strip. An inferred one is only the repo root when some cwd sat
* there, so callers that know the root pass it here.
*/
cwdPrefix?: string;
/** Several roots at once (a session spanning two checkouts). Wins over `cwdPrefix`. */
cwdPrefixes?: readonly string[];
/**
* Also strip home-rooted paths in the CONTENT: a sandbox-recorded session has a
* sandbox `cwd`, so the cwd passes never see the local checkout it still mentions.
*/
scrubEmbeddedHomePaths?: boolean;
}
export interface SanitizeResult {
sanitized: string;
prefixStripped: string | null;
encodedPrefixStripped: string | null;
homeDirStripped: string | null;
encodedHomeDirStripped: string | null;
embeddedPrefixStripped: string | null;
embeddedHomeDirStripped: string | null;
/** Replacement count per marker, keyed by the regex's source string. */
markersScrubbed: Record<string, number>;
}
/** Longest common prefix by path COMPONENT: `/a/bb` and `/a/b` share `/a`, not `/a/b`.
* Returns `''` when only the root `/` is common. */
export function findLongestCommonPathPrefix(paths: Iterable<string>): string {
const arr = Array.from(paths);
if (arr.length === 0) return '';
const splits = arr.map((p) => p.split('/'));
const minLen = Math.min(...splits.map((s) => s.length));
let lastShared = 0;
for (let i = 0; i < minLen; i++) {
const c = splits[0][i];
if (splits.some((s) => s[i] !== c)) break;
lastShared = i + 1;
}
// Only the leading empty piece matched → just the root, not useful.
if (lastShared <= 1) return '';
return splits[0].slice(0, lastShared).join('/');
}
/** The home-dir portion of an absolute path, or `null` for an unrecognized shape —
* better to skip the home pass than strip what may be repo content. */
export function extractHomeDir(cwdPrefix: string): string | null {
if (!cwdPrefix.startsWith('/')) return null;
// Windows-under-WSL shapes first: the generic drive shape below would stop at the
// drive letter and leave the account name in. A volume or drive root carries no
// identity by itself, so those take the directory under it.
const patterns: RegExp[] = [
/^\/mnt\/host\/[^/]+\/Users\/[^/]+/,
/^\/mnt\/[^/]+\/Users\/[^/]+/,
/^\/Users\/[^/]+/,
/^\/home\/[^/]+/,
/^\/Volumes\/[^/]+\/[^/]+/,
/^\/mnt\/[^/]+\/[^/]+/,
/^\/var\/root(?=\/|$)/,
/^\/root(?=\/|$)/,
];
for (const re of patterns) {
const m = cwdPrefix.match(re);
if (m) return m[0];
}
return null;
}
/** Every distinct `cwd` in the transcript. Read at the top level (Claude Code) and
* under `payload` (codex), so both harnesses are covered. Bad lines are skipped. */
export function collectCwds(raw: string): Set<string> {
const out = new Set<string>();
const add = (v: unknown) => {
if (typeof v === 'string' && v.startsWith('/')) out.add(v);
};
for (const line of raw.split('\n')) {
if (!line.trim()) continue;
let parsed: unknown;
try {
parsed = JSON.parse(line);
} catch {
continue;
}
if (typeof parsed !== 'object' || parsed === null) continue;
const rec = parsed as { cwd?: unknown; payload?: unknown };
add(rec.cwd);
if (typeof rec.payload === 'object' && rec.payload !== null) {
add((rec.payload as { cwd?: unknown }).cwd);
}
}
return out;
}
/** One path segment: stops at `/`, whitespace, quotes and JSON punctuation. */
const COMP = String.raw`[^/\s"'\\,:;)\]}<>]+`;
// macOS/Windows display names can contain spaces, but only consume them while
// more path follows, so a bare home-dir mention doesn't swallow trailing prose.
const USER_WITH_SPACES = `${COMP}(?:(?: +${COMP})+(?=/))?`;
const EMBEDDED_HOME_RE = new RegExp(
'(?:' +
String.raw`\/home\/${COMP}` +
'|' +
String.raw`\/Users\/${USER_WITH_SPACES}` +
'|' +
String.raw`\/mnt\/c\/Users\/${USER_WITH_SPACES}` +
'|' +
// Component boundary, so these don't match inside `/rootfs` or `/root_ca.pem`.
String.raw`\/var\/root(?![^/])` +
'|' +
String.raw`\/root(?![^/])` +
')' +
String.raw`(?:\/${COMP})*`,
'g'
);
export function collectEmbeddedHomePaths(raw: string): Set<string> {
const out = new Set<string>();
for (const m of raw.matchAll(EMBEDDED_HOME_RE)) out.add(m[0]);
return out;
}
function literalReplaceAll(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
return haystack.split(needle).join(replacement);
}
/** Can `ch` continue a path component? A `.` counts only mid-component, so `…/repo.git`
* is one component but `…/repo.` ending a sentence is not. */
function continuesComponent(text: string, at: number): boolean {
const ch = text[at];
if (ch === undefined) return false;
if (/[A-Za-z0-9_-]/.test(ch)) return true;
return ch === '.' && at + 1 < text.length && /[A-Za-z0-9_-]/.test(text[at + 1]);
}
/** Replace `needle` only where it ends at a component boundary, so stripping `…/wt/repo`
* can't turn `…/wt/repo-backup` into `<replacement>-backup`. Skipped ones go to the home pass. */
function replacePrefixAtBoundary(haystack: string, needle: string, replacement: string): string {
if (!needle) return haystack;
let out = '';
let from = 0;
for (;;) {
const i = haystack.indexOf(needle, from);
if (i === -1) return out + haystack.slice(from);
const end = i + needle.length;
out += haystack.slice(from, i) + (continuesComponent(haystack, end) ? needle : replacement);
from = end;
}
}
/** Replace a prefix and its dash-encoded form (`.claude/projects/<encoded>/`). */
function stripBothForms(haystack: string, needle: string, replacement: string): string {
const out = literalReplaceAll(haystack, needle, replacement);
return literalReplaceAll(out, needle.replace(/\//g, '-'), replacement.replace(/\//g, '-'));
}
export function sanitizeSessionJsonl(raw: string, opts: SanitizeOptions = {}): SanitizeResult {
const placeholder = opts.placeholder ?? DEFAULT_PLACEHOLDER;
const markers = opts.forbiddenMarkers ?? [];
const cwds = collectCwds(raw);
let working = raw;
let prefixStripped: string | null = null;
let encodedPrefixStripped: string | null = null;
let homeDirStripped: string | null = null;
let encodedHomeDirStripped: string | null = null;
let embeddedPrefixStripped: string | null = null;
let embeddedHomeDirStripped: string | null = null;
const requested = opts.cwdPrefixes?.length
? [...opts.cwdPrefixes]
: opts.cwdPrefix
? [opts.cwdPrefix]
: cwds.size > 0
? [findLongestCommonPathPrefix(cwds)]
: [];
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const prefixes = [...new Set(requested.filter(Boolean))].sort((a, b) => b.length - a.length);
// EVERY root before ANY home dir: a home pass run between roots would rewrite a
// sibling root's own prefix, leaving it unmatched when its turn came.
for (const prefix of prefixes) {
const encodedPrefix = prefix.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, prefix, placeholder);
working = literalReplaceAll(working, encodedPrefix, placeholder.replace(/\//g, '-'));
prefixStripped ??= prefix;
encodedPrefixStripped ??= encodedPrefix;
}
// Only catches what is left outside the roots, e.g. `/home/<user>/.claude/projects/`.
const homeDirs = new Set(
prefixes
.map((p) => extractHomeDir(p))
.filter((h): h is string => h !== null && !prefixes.includes(h))
);
for (const homeDir of homeDirs) {
const encodedHomeDir = homeDir.replace(/\//g, '-');
working = replacePrefixAtBoundary(working, homeDir, HOME_DIR_PLACEHOLDER);
working = literalReplaceAll(working, encodedHomeDir, HOME_DIR_PLACEHOLDER.replace(/\//g, '-'));
homeDirStripped ??= homeDir;
encodedHomeDirStripped ??= encodedHomeDir;
}
if (opts.scrubEmbeddedHomePaths) {
const embedded = collectEmbeddedHomePaths(working);
if (embedded.size > 0) {
// Take each path's own shortest `/repo`-terminated prefix rather than a
// common prefix, which mis-collapses when paths diverge above the root.
const repoRoots = new Set<string>();
const homeDirs = new Set<string>();
for (const p of embedded) {
const h = extractHomeDir(p);
if (h) homeDirs.add(h);
const m = p.match(/^(.*?\/repo)(?:\/|$)/);
if (m) repoRoots.add(m[1]);
}
// Longest first, so a shorter root sharing a prefix can't partly clobber a nested one.
const sortedRoots = [...repoRoots].sort((a, b) => b.length - a.length);
for (const root of sortedRoots) working = stripBothForms(working, root, placeholder);
for (const h of homeDirs) working = stripBothForms(working, h, HOME_DIR_PLACEHOLDER);
embeddedPrefixStripped = sortedRoots[0] ?? null;
embeddedHomeDirStripped = [...homeDirs][0] ?? null;
}
}
const markersScrubbed: Record<string, number> = {};
for (const re of markers) {
let count = 0;
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
const global = new RegExp(re.source, flags);
working = working.replace(global, () => {
count++;
return REDACTION_PLACEHOLDER;
});
if (count > 0) markersScrubbed[re.source] = count;
}
return {
sanitized: working,
prefixStripped,
encodedPrefixStripped,
homeDirStripped,
encodedHomeDirStripped,
embeddedPrefixStripped,
embeddedHomeDirStripped,
markersScrubbed,
};
}

View File

@@ -1,34 +0,0 @@
/**
* Shared helper for finding a Claude Code session id inside a JSONL or
* stream-json `.txt` file. Both formats embed the id under one of two keys:
*
* - `session_id` — stream-json output (`claude-code.txt` from `--print
* --output-format=stream-json`).
* - `sessionId` — Claude Code's internal session log (the resumable JSONL
* under `~/.claude/projects/<key>/<id>.jsonl`).
*
* Real files only use one key, but if a future format ever emits both we
* shouldn't have two callers picking different winners — so this helper is
* the single source of truth.
*/
import { readFileSync } from 'fs';
export function readSessionId(filePath: string): string | null {
const lines = readFileSync(filePath, 'utf8').split('\n');
for (const line of lines) {
if (!line) continue;
let parsed: unknown;
try {
parsed = JSON.parse(line);
} catch {
continue;
}
if (typeof parsed !== 'object' || parsed === null) continue;
const obj = parsed as Record<string, unknown>;
const camel = typeof obj.sessionId === 'string' ? obj.sessionId : null;
const snake = typeof obj.session_id === 'string' ? obj.session_id : null;
const id = camel ?? snake;
if (id) return id;
}
return null;
}

View File

@@ -1,353 +0,0 @@
#!/bin/bash
# Install the harnesses a worker can author with, from scripts/harness-registry.toml.
#
# Source it, then call unpiped — it exports credentials, which a subshell would lose:
#
# . /workspace/scripts/setup-harnesses.sh
# harness_setup_all
#
# Registry reading and credential derivation live in lib/harness-credentials.sh, sourced
# below, because `harbor-run` needs those and nothing else here.
#
# No -e here — but this file is SOURCED, and shell options belong to the caller's shell:
# both post-creates run with -e, so that is what is in force. An unguarded failure below
# therefore aborts container creation, which is why every failure site is individually
# guarded (`|| true`, `if !`) rather than relying on this line.
set -uo pipefail
_HARNESS_REGISTRY_DIR="${HARNESS_SCRIPTS_DIR:-/workspace/scripts}"
if [ ! -f "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh" ]; then
echo "harness-setup: FATAL — $_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh is" >&2
echo "harness-setup: missing, so nothing here can read the registry. Every step below" >&2
echo "harness-setup: would report a missing interpreter instead of this." >&2
return 1 2>/dev/null || exit 1
fi
# shellcheck disable=SC1091
. "$_HARNESS_REGISTRY_DIR/lib/harness-credentials.sh"
# Every setup step reads the registry through _harness_query, and each call suppresses
# stderr so one bad row can't abort the container. That means a BROKEN interpreter turns
# the whole of setup into a silent no-op: no credentials, no CLIs, no config, no
# launchers, and no error anywhere. Check it once, loudly, before any of that.
harness_preflight() {
local err py found=yes
py=$(_raccoon_python) || { py=python3; found=no; }
if ! err=$("$py" "$_HARNESS_REGISTRY_DIR/resolve_harness.py" --list 2>&1 >/dev/null); then
echo "harness-setup: FATAL — cannot read the harness registry, so no agent CLI" >&2
echo "harness-setup: would be installed. Nothing below will run." >&2
echo "harness-setup: interpreter: $(command -v "$py" || echo MISSING) ($("$py" -V 2>&1))" >&2
if [ "$found" = no ]; then
echo "harness-setup: no python3.11+ with tomllib found; set RACCOON_PYTHON to override" >&2
fi
echo "harness-setup: registry: $_HARNESS_REGISTRY_DIR/harness-registry.toml" >&2
printf 'harness-setup: %s\n' "$err" >&2
return 1
fi
}
# claude installs into $HOME/.local/bin, which is not on PATH during post-create.
case ":$PATH:" in
*":$HOME/.local/bin:"*) ;;
*) export PATH="$HOME/.local/bin:$PATH" ;;
esac
# --- installs ----------------------------------------------------------------
harness_install_clis() {
local id cli install
while IFS=$'\t' read -r id cli install; do
[ -n "$install" ] || continue
if command -v "$cli" >/dev/null 2>&1; then
echo "harness-setup: $cli already installed — skipping" >&2
continue
fi
echo "harness-setup: installing $id ($cli)" >&2
# Reported as unavailable below rather than fatal.
if ! bash -c "$install" >&2; then
echo "harness-setup: WARNING $id failed to install — $cli will be unavailable" >&2
fi
done < <(_harness_query --authoring-installs 2>/dev/null || true)
}
# Report which CLIs are usable. Non-zero when NONE are: one harness missing is survivable
# (a worker uses the other), but zero means the container cannot author anything at all,
# and that must stop setup rather than read as a couple of warnings.
harness_report() {
local id cli install ready=0 missing=0
while IFS=$'\t' read -r id cli install; do
[ -n "$cli" ] || continue
if command -v "$cli" >/dev/null 2>&1; then
echo " $cli — ready" >&2
ready=$((ready + 1))
else
echo " $cli — NOT AVAILABLE (install failed; see above)" >&2
missing=$((missing + 1))
fi
done < <(_harness_query --authoring-installs 2>/dev/null || true)
# A CLI on PATH with no key is worse than a missing one: it starts, then fails at the
# first request with the harness's own auth error, which says nothing about setup.
local id key_env base_url_env proxy_path
while IFS=$'\t' read -r id key_env base_url_env proxy_path; do
[ -n "$key_env" ] || continue
if [ -z "${!key_env:-}" ]; then
echo " $id — installed but NO CREDENTIALS: $key_env is unset." >&2
echo " Derived from ANTHROPIC_BASE_URL + ANTHROPIC_API_KEY; set both in .env." >&2
fi
done < <(_harness_query --authoring-credentials 2>/dev/null || true)
if [ "$ready" -eq 0 ]; then
echo "harness-setup: FATAL — no agent CLI installed ($missing attempted)." >&2
echo "harness-setup: This container cannot author a task. Check the install" >&2
echo "harness-setup: output above: the CLIs download over the network, so a" >&2
echo "harness-setup: proxy, DNS or upstream change breaks every one at once." >&2
return 1
fi
[ "$missing" -gt 0 ] && echo "harness-setup: $missing harness(es) unavailable; $ready usable" >&2
return 0
}
# --- Explore launchers -------------------------------------------------------
# One `raccoon-explore-<cli>` per harness, aliased to its `cli`.
harness_install_launchers() {
local bin="$HOME/.local/bin"
mkdir -p "$bin"
# Read at launcher run time so the note stays a file, not a baked-in copy.
local note_src="${HARNESS_TOOLSET_NOTE:-/workspace/scripts/toolset_note.md}"
local browser_note_src="${note_src%.md}_browser.md"
local read_note_src="${note_src%.md}_read.md"
local agent_cli_dir="${AGENT_CLI_DIR:-/opt/agent-cli}"
# Which harnesses keep their key in a file rather than reading $ENV per request. Those
# launchers refresh it first: the file dates from container create, so a key rotated in
# .env since then would otherwise reach the harness only after a rebuild.
local file_auth_ids="" aid apath akey
while IFS=$'\t' read -r aid apath akey; do
[ -n "$apath" ] || continue
file_auth_ids="${file_auth_ids:+$file_auth_ids }$aid"
done < <(_harness_query --auth-files 2>/dev/null || true)
local id cli launch switchable refresh_line
while IFS=$'\t' read -r id cli launch; do
[ -n "$cli" ] && [ -n "$launch" ] || continue
# `|| true` twice over (here and inside the script): the launcher runs under
# `set -e`, and a failed refresh must not cost the worker their agent.
if [[ " $file_auth_ids " == *" $id "* ]]; then
refresh_line="\"$_HARNESS_REGISTRY_DIR/refresh-harness-auth\" || true"
else
refresh_line=""
fi
# Whether RACCOON_BROWSER_TASK can change THIS harness's toolset, read off the
# registry rather than hardcoded: a launch line that interpolates $RACCOON_TOOLS
# can, and one that doesn't cannot. codex is the second case — it ships view_image,
# so a browser task needs nothing added and the flag has nothing to switch.
# Match the whole variable name: a substring test also hits RACCOON_TOOLSET_NOTE,
# which every launch line references, and every harness would look switchable.
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
switchable=1
else
switchable=0
fi
cat > "$bin/raccoon-explore-$cli" <<LAUNCHER
#!/bin/bash
# GENERATED by scripts/setup-harnesses.sh from harness-registry.toml — do not edit.
set -euo pipefail
# A harness that reads its key from \$ENV per request needs .env in its environment,
# and only an interactive shell sources .bashrc — which is also where PATH picks up
# ~/.local/bin, where the CLI itself lives. Both are set here so a launch works the
# same either way, with the key .env holds right now.
export PATH="\$HOME/.local/bin:\$PATH"
if [ -f "\${RACCOON_ENV_FILE:-/workspace/.env}" ]; then
set -a
. "\${RACCOON_ENV_FILE:-/workspace/.env}"
set +a
fi
if [ -f "\$HOME/.raccoon-call-origin" ]; then
. "\$HOME/.raccoon-call-origin"
fi
if [ -f "$note_src" ]; then
RACCOON_TOOLSET_NOTE="\$(sed "s#/opt/agent-cli#$agent_cli_dir#g" "$note_src")"
else
RACCOON_TOOLSET_NOTE=""
fi
# RACCOON_BROWSER_TASK=1 explores with the toolset a \`browser = true\` task runs under.
# Named for the flag it mirrors: one word, \`browser\`, whether it's set in task.toml or
# here. Per invocation, not per container — authoring a browser task shouldn't need a
# rebuild, and neither should changing your mind. Default off, so ordinary exploring
# still mirrors an ordinary trial.
#
# The correction must be appended AFTER the base note, which says there is no Read tool.
RACCOON_TOOLS="Bash"
if [ "\${RACCOON_BROWSER_TASK:-0}" = "1" ] && [ "$switchable" = "1" ] && [ -f "$read_note_src" ]; then
RACCOON_TOOLS="Bash,Read"
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
\$(cat "$read_note_src")"
fi
export RACCOON_TOOLS
# Only mention the browser on an image that actually has one — most don't. Probed at
# launch, not baked in, so the same launcher is correct in whichever container it runs.
#
# Exported two ways because the harnesses take extra instructions differently: claude
# appends the whole toolset note to --append-system-prompt, while codex has no equivalent
# and takes -c developer_instructions=. codex must NOT get the claude-shaped toolset note
# (it has no str_replace_editor), so the browser part is exported on its own too.
RACCOON_BROWSER_NOTE=""
RACCOON_BROWSER_FLAGS=()
if command -v pw >/dev/null 2>&1 && [ -f "$browser_note_src" ]; then
RACCOON_BROWSER_NOTE="\$(cat "$browser_note_src")"
RACCOON_TOOLSET_NOTE="\${RACCOON_TOOLSET_NOTE}
\${RACCOON_BROWSER_NOTE}"
RACCOON_BROWSER_FLAGS=(-c "developer_instructions=\${RACCOON_BROWSER_NOTE}")
fi
export RACCOON_TOOLSET_NOTE RACCOON_BROWSER_NOTE
export RACCOON_HARNESS="$id"
# These launchers exist only in explore, and a refresh that has to CREATE a config
# needs the surface to know the capture hooks belong in it.
export RACCOON_SURFACE=explore
# No RACCOON_SNAPSHOT_DATA here on purpose. capture-snapshot.mjs and save-session-info.mjs
# already share the same default ($HOME/.raccoon/snapshot-data), which is what codex needs
# — it has no CLAUDE_PLUGIN_* to fall back to. Exporting it ALSO overrode the dir for
# claude, whose slash command pins --plugin-data to the plugin dir, so the hook wrote one
# place and capture read another and the recorded session was silently ignored.
$refresh_line
$launch
LAUNCHER
chmod +x "$bin/raccoon-explore-$cli"
echo "harness-setup: launcher raccoon-explore-$cli" >&2
done < <(_harness_query --explore-launchers 2>/dev/null || true)
}
# Alias lines for ~/.bashrc.
harness_alias_lines() {
local id cli launch switchable
local browser_clis=""
while IFS=$'\t' read -r id cli launch; do
[ -n "$cli" ] && [ -n "$launch" ] || continue
echo "alias $cli=\"raccoon-explore-$cli\""
# Same derivation as the launcher: only a harness whose launch line takes
# $RACCOON_TOOLS has a toolset the flag can change.
if [[ "$launch" =~ \$\{?RACCOON_TOOLS\}?([^A-Za-z0-9_]|$) ]]; then
browser_clis="${browser_clis:+$browser_clis }$cli"
fi
done < <(_harness_query --explore-launchers 2>/dev/null || true)
# The browser hint belongs at the shell prompt, not in the launcher. Claude Code takes the
# alternate screen buffer, so anything printed just before exec is hidden for the whole
# session and resurfaces only after quitting — advice arriving exactly too late. Here it
# lands in ordinary scrollback, before any TUI exists, and there is nothing to quit yet.
#
# `pw` is probed at shell start, so one ~/.bashrc is correct in a container with a browser
# and in one without.
[ -n "$browser_clis" ] || return 0
local first="${browser_clis%% *}"
cat <<HINT
if [[ \$- == *i* ]] && [ "\${RACCOON_BROWSER_TASK:-0}" != "1" ] && command -v pw >/dev/null 2>&1; then
echo "browser available (Playwright + Chromium, \\\`pw <script.js>\\\`)."
echo "Authoring a \\\`browser = true\\\` task? Start it with: RACCOON_BROWSER_TASK=1 $first"
fi
HINT
}
# Write each harness's config file from the registry, replacing whatever was there.
#
# The file is OWNED, not merged: TOML has no way to return to the document root after a
# table header, so appending or prepending around foreign content silently reparents
# root-level keys into whichever table happens to precede them. Owning it also means a
# registry change actually reaches a container that was already set up.
harness_write_configs() {
local id config_path blob target tmp
while IFS=$'\t' read -r id config_path blob; do
[ -n "$config_path" ] && [ -n "$blob" ] || continue
# Guarded: a bare failing assignment exits the caller's `set -e` post-create with
# no explanation. A path this cannot expand is one harness's problem, not the
# container's.
target=$(eval "printf '%s' \"$config_path\"") || {
echo "harness-setup: WARNING $id config_path could not be expanded — skipping" >&2
continue
}
mkdir -p "$(dirname "$target")"
tmp="$target.raccoon-tmp"
# Expansion is strict: an unset var would otherwise be written through as the
# literal ${VAR}, which surfaces much later as an unparseable value.
if ! {
echo "# Generated from harness-registry.toml — edits here are overwritten."
printf '%s' "$blob" | base64 -d | python3 -c '
import os, re, sys
text = sys.stdin.read()
missing = sorted(
{m.group(1) for m in re.finditer(r"\$\{(\w+)\}", text) if m.group(1) not in os.environ}
)
if missing:
sys.stderr.write("unset: " + ", ".join(missing) + "\n")
raise SystemExit(1)
sys.stdout.write(os.path.expandvars(text))
'
} > "$tmp"; then
rm -f "$tmp"
echo "harness-setup: WARNING $id config NOT written — a value it needs is unset." >&2
echo "harness-setup: run harness_setup_credentials first (harness_setup_all does)." >&2
continue
fi
mv "$tmp" "$target"
echo "harness-setup: $id config -> $target" >&2
done < <(_harness_query --container-configs --surface "${RACCOON_SURFACE:-authoring}" 2>/dev/null || true)
}
# Link every available skill into each harness's skills_dir, for harnesses that declare one.
# Both container layouts are covered: the explore container holds the snapshot skill under
# plugins/, the authoring container holds the authoring skills under .claude/skills. Whichever
# directories exist here are the ones this container has.
harness_install_skills() {
local sources="${RACCOON_SKILL_SOURCE_DIRS:-/workspace/plugins/create-snapshot/skills /workspace/.claude/skills}"
local id dir target src skill name installed
while IFS=$'\t' read -r id dir; do
[ -n "$dir" ] || continue
target=$(eval "printf '%s' \"$dir\"") || {
echo "harness-setup: WARNING $id skills_dir could not be expanded — skipping" >&2
continue
}
mkdir -p "$target"
installed=0
for src in $sources; do
[ -d "$src" ] || continue
for skill in "$src"/*/; do
[ -f "$skill/SKILL.md" ] || continue
name=$(basename "$skill")
ln -sfn "${skill%/}" "$target/$name"
installed=$((installed + 1))
done
done
echo "harness-setup: $id skills -> $target ($installed linked)" >&2
done < <(_harness_query --skills-dirs 2>/dev/null || true)
}
# The lines that explain a setup failure are printed as it happens, and the devcontainer
# CLI's own stack trace lands on top of them. Close with a banner so the worker has
# something to look for, and something to send us.
_harness_fatal_banner() {
echo "" >&2
echo " ============================================================" >&2
echo " HARNESS SETUP FAILED — this container has no agent CLI." >&2
echo "" >&2
echo " The harness-setup: lines above say why. Anything the" >&2
echo " devcontainer prints after this is a consequence, not the" >&2
echo " cause; send us the harness-setup: lines." >&2
echo " ============================================================" >&2
echo "" >&2
}
harness_setup_all() {
harness_preflight || { _harness_fatal_banner; return 1; }
harness_setup_credentials
harness_write_auth
harness_install_clis
harness_write_configs
harness_install_skills
# Launchers are NOT installed here. They are an Explore concern (that container aliases
# `claude`/`codex` to them), and it passes its own AGENT_CLI_DIR — installing them here
# too wrote every launcher twice, the first time with the wrong editor path, and left an
# unused one in the authoring container.
echo "harness-setup: authoring harnesses" >&2
harness_report || { _harness_fatal_banner; return 1; }
}

View File

@@ -1,821 +0,0 @@
/**
* snapshot-to-task: Create a harbor task scaffold from a snapshot.
*
* Usage:
* npx tsx scripts/snapshot-to-task.ts --snapshot <dir>
*/
import { execFileSync, execSync } from 'child_process';
import {
chmodSync,
copyFileSync,
existsSync,
mkdirSync,
readFileSync,
readdirSync,
statSync,
writeFileSync,
} from 'fs';
import { basename, join, resolve } from 'path';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
import { stripAuthoringScaffolding, truncationIndex, turnsFromLines } from './harness-session.mjs';
// This script must not call cpSync — it fails EACCES on a macOS docker bind mount.
import { HOLISTIC_RUBRIC_SCAFFOLD } from './holistic-rubric-scaffold';
import { copyTree } from './lib/copy-tree';
import { collectCwds, sanitizeSessionJsonl } from './sanitize-session-jsonl';
// --- CLI ---
const argv = yargs(hideBin(process.argv))
.option('snapshot', {
type: 'string',
describe: 'Path to the snapshot directory',
demandOption: true,
})
.option('json', {
type: 'boolean',
describe: 'Output structured JSON logs',
default: false,
})
.strict()
.help()
.parseSync();
const log = pino(
{ name: 'snapshot-to-task', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
// --- Read snapshot data ---
const snapshotDir = argv.snapshot;
if (!existsSync(snapshotDir)) {
log.fatal(
{ path: snapshotDir },
'Snapshot directory not found. Check that the path points to a directory inside explore/snapshots/.'
);
process.exit(1);
}
interface SnapshotMetadata {
slug: string;
session_uuid: string;
/** Absent on snapshots captured before harness selection existed. */
harness?: string;
original_cwd: string;
commit: string | null;
branch: string | null;
remote_url: string | null;
timestamp: string;
plugin_version: string;
}
interface Annotation {
what_trying: string;
what_hoping: string;
what_happened: string;
[key: string]: string;
}
const metadata = JSON.parse(
readFileSync(join(snapshotDir, 'metadata.json'), 'utf8')
) as SnapshotMetadata;
const annotation = JSON.parse(
readFileSync(join(snapshotDir, 'annotation.json'), 'utf8')
) as Annotation;
if (!metadata.slug) {
log.fatal(
'No slug found in snapshot metadata.json. This snapshot may have been created by an older version of the plugin. Please file a bug.'
);
process.exit(1);
}
const slug = metadata.slug;
// --- Locate harbor infrastructure ---
function findRepoRoot(): string | null {
let dir = process.cwd();
while (dir !== resolve(dir, '..')) {
if (existsSync(join(dir, 'harbor-tasks'))) return dir;
dir = resolve(dir, '..');
}
return null;
}
const maybeRepoRoot = findRepoRoot();
if (!maybeRepoRoot) {
log.fatal(
"Could not find harbor-tasks/ directory. Make sure you're running this from the toolkit root (the Authoring container). Please file a bug if this persists."
);
process.exit(1);
}
const repoRoot: string = maybeRepoRoot;
const harborTasks = join(repoRoot, 'harbor-tasks');
const sharedCandidates = [join(harborTasks, 'raccoon-shared'), join(repoRoot, 'task-shared')];
const sharedDir = sharedCandidates.find((d) => existsSync(d));
const taskDir = join(harborTasks, slug);
if (existsSync(taskDir)) {
log.fatal(
{ path: taskDir },
`Task directory already exists. To recreate it, delete it first: rm -rf ${taskDir}`
);
process.exit(1);
}
if (!sharedDir) {
log.fatal(
'Shared infrastructure (Dockerfile, test.sh, etc.) not found. The toolkit may be corrupted. Please file a bug.'
);
process.exit(1);
}
// --- Detect repo name ---
interface ToolkitConfig {
repo: string;
defaultCommit: string;
/** The packed kit's release version (git describe at pack time). */
version?: string;
}
function readToolkitConfig(): ToolkitConfig | null {
const configPath = join(repoRoot, 'toolkit.json');
if (!existsSync(configPath)) return null;
return JSON.parse(readFileSync(configPath, 'utf8')) as ToolkitConfig;
}
function repoNameFromRemote(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const match = remoteUrl.match(/\/([^/]+?)(?:\.git)?$/);
return match ? match[1] : null;
}
function findSubmoduleDir(remoteUrl: string | null): string | null {
if (!remoteUrl) return null;
const reposDir = join(repoRoot, 'repos');
if (!existsSync(reposDir)) return null;
const normalize = (url: string) =>
url
.replace(/\.git$/, '')
.replace(/^git@github\.com:/, 'https://github.com/')
.toLowerCase();
for (const entry of readdirSync(reposDir)) {
const repoPath = join(reposDir, entry, 'repo');
if (!existsSync(repoPath)) continue;
try {
const remote = execSync('git remote get-url origin', {
cwd: repoPath,
encoding: 'utf8',
stdio: ['pipe', 'pipe', 'pipe'],
}).trim();
if (normalize(remote) === normalize(remoteUrl)) return entry;
} catch {
continue;
}
}
return null;
}
const toolkitConfig = readToolkitConfig();
// A polyglot toolkit's toolkit.json has repos[] + polyglot:true (no top-level .repo).
// Derive which member this task targets from the snapshot's original_cwd basename,
// validated against the member list.
const polyglotMember = (() => {
const cfg = toolkitConfig as { polyglot?: boolean; repos?: Array<{ repo: string }> } | null;
if (!cfg?.polyglot || !Array.isArray(cfg.repos)) return null;
const base = metadata.original_cwd?.split('/').filter(Boolean).pop() ?? null;
const members = cfg.repos.map((r) => r.repo);
return base && members.includes(base) ? base : null;
})();
const repoName =
polyglotMember ??
toolkitConfig?.repo ??
findSubmoduleDir(metadata.remote_url) ??
repoNameFromRemote(metadata.remote_url);
if (!repoName) {
log.fatal(
'Could not determine repo name. The toolkit may be missing toolkit.json. Please file a bug.'
);
process.exit(1);
}
const commitShort = metadata.commit ? metadata.commit.slice(0, 9) : 'unknown';
const sessionUuid = metadata.session_uuid;
log.info({ slug, repo: repoName, commit: commitShort }, 'Creating harbor task');
// --- Create task directory structure ---
mkdirSync(join(taskDir, 'environment'), { recursive: true });
mkdirSync(join(taskDir, 'tests'), { recursive: true });
mkdirSync(join(taskDir, 'reference-runs'), { recursive: true });
// --- Copy shared infrastructure ---
// The complete grader asset set test.sh depends on: the grader system prompt
// and the renderer (test.sh exits without the renderer). Sources missing from
// task-shared/ are skipped by the existsSync guard below.
const sharedFiles = [
{ src: 'test.sh', dest: 'tests/test.sh' },
{
src: 'grader-system-prompt-consolidated.md',
dest: 'tests/grader-system-prompt-consolidated.md',
},
// test.sh execs this to render the grade; without it the verifier writes no reward
// file and the trial errors out rather than scoring.
{ src: 'render-grade-consolidated.py', dest: 'tests/render-grade-consolidated.py' },
];
for (const { src, dest } of sharedFiles) {
const srcPath = join(sharedDir, src);
const destPath = join(taskDir, dest);
if (existsSync(srcPath)) {
copyFileSync(srcPath, destPath);
if (src === 'test.sh') chmodSync(destPath, 0o755);
log.debug({ src, dest }, 'Copied shared file');
} else {
log.warn({ src }, 'Shared file not found');
}
}
// Deterministic checks (tests/typecheck/lint). test.sh sources these and hands
// their output to the grader as evidence for the CORRECTNESS score, so without
// them a code task's correctness is never signal-backed — the grader falls back
// to reading the diff alone. Same per-member-then-generic resolution as the
// Dockerfile below: a polyglot toolkit ships test-commands.<member>.sh per
// member, a single-repo toolkit ships the lone test-commands.sh.
const perMemberTestCommands = join(sharedDir, `test-commands.${repoName.toLowerCase()}.sh`);
const genericTestCommands = join(sharedDir, 'test-commands.sh');
const testCommandsSrc = existsSync(perMemberTestCommands)
? perMemberTestCommands
: genericTestCommands;
if (existsSync(testCommandsSrc)) {
const testCommandsDest = join(taskDir, 'tests', 'test-commands.sh');
copyFileSync(testCommandsSrc, testCommandsDest);
chmodSync(testCommandsDest, 0o755);
log.debug({ src: testCommandsSrc }, 'Copied deterministic checks');
} else {
// Not fatal: the grader still scores correctness by walking the changed code.
log.info(
'No test-commands.sh for this repo — expected when it has no runnable suite. The grader scores correctness by reading the changed code instead; say so in your holistic rubric.'
);
}
// --- Write Dockerfile with session resume support ---
//
// Read the per-repo task-shared/Dockerfile (Ruby/Postgres/Node for ZenBill,
// TS-Node/Postgres/pnpm for Palolo) from the toolkit and append session-
// staging COPY/RUN steps. Session staging happens after the original CMD —
// COPY and RUN are layer ops independent of CMD, so the original
// `CMD ["sleep", "infinity"]` remains active after the appended layers.
//
// Falls back to a bare debian Dockerfile if no task-shared/Dockerfile is
// present (toolkit corruption, or a repo without a per-repo Dockerfile).
// Polyglot toolkits ship a per-member task-shared/Dockerfile.<member>; a graded task
// targets one member, so prefer its Dockerfile. Single-repo toolkits use the lone
// task-shared/Dockerfile. Fall back to the generic one if the per-member file is absent.
const perMemberDockerfile = join(repoRoot, 'task-shared', `Dockerfile.${repoName.toLowerCase()}`);
const taskSharedDockerfile = existsSync(perMemberDockerfile)
? perMemberDockerfile
: join(repoRoot, 'task-shared', 'Dockerfile');
let baseDockerfile: string;
if (existsSync(taskSharedDockerfile)) {
baseDockerfile = readFileSync(taskSharedDockerfile, 'utf-8');
log.debug({ dockerfile: taskSharedDockerfile }, 'Loaded base Dockerfile');
} else {
log.warn(
'task-shared/Dockerfile not found; falling back to bare debian. The harbor task container will lack any language runtime — agents will not be able to execute code in the repo.'
);
baseDockerfile = `FROM debian:bookworm-slim
RUN apt-get update && apt-get install -y \\
git \\
python3 \\
curl \\
jq \\
&& rm -rf /var/lib/apt/lists/*
# Install Claude Code globally (needed by the grader in test.sh)
RUN curl -fsSL https://claude.ai/install.sh | bash && \\
cp /root/.claude-code/claude /usr/local/bin/claude 2>/dev/null || \\
cp /root/.local/bin/claude /usr/local/bin/claude 2>/dev/null || \\
ln -sf $(find /root -name claude -type f 2>/dev/null | head -1) /usr/local/bin/claude
WORKDIR /workspace
COPY workspace/ .
# Block network tools — agent should only read code and write documents
RUN mkdir -p .claude && \\
echo '{"permissions":{"deny":["WebFetch","WebSearch"]}}' > .claude/settings.json
RUN git init && \\
git config user.email "dev@agent" && \\
git config user.name "Dev" && \\
git add -A && \\
git commit -m "initial" --quiet
CMD ["sleep", "infinity"]
`;
}
// Wrapped in toolkit-managed sentinels so check-task-infra reads this as the
// toolkit's own append rather than an edit to the Dockerfile.
// Only Claude Code produces the sibling session/ directory (subagents, tool results).
// A COPY of an empty directory fails the build outright — buildkit does not carry empty
// directories in the context, so the layer errors with `"/session": not found`.
// Read the SNAPSHOT, not the task dir: the Dockerfile is generated before the session
// files are copied into environment/, so the task-side copy is not there yet.
const sessionSiblingDir = join(snapshotDir, 'session');
const hasSessionSibling =
existsSync(sessionSiblingDir) && readdirSync(sessionSiblingDir).length > 0;
const sessionStaging = `
# >>> toolkit-managed: snapshot-session >>>
# Stage session files for the snapshot agent adapter to install at runtime.
COPY session.jsonl /tmp/snapshot-session/session.jsonl
${hasSessionSibling ? 'COPY session/ /tmp/snapshot-session/session/\n' : ''}RUN echo '${sessionUuid}' > /tmp/snapshot-session/uuid.txt
# <<< toolkit-managed <<<
`;
const dockerfile = baseDockerfile.trimEnd() + '\n' + sessionStaging;
writeFileSync(join(taskDir, 'environment', 'Dockerfile'), dockerfile);
log.debug('Wrote Dockerfile (per-repo base + session staging)');
// --- Copy snapshot.patch as workspace.patch ---
const snapshotPatch = join(snapshotDir, 'snapshot.patch');
if (existsSync(snapshotPatch)) {
copyFileSync(snapshotPatch, join(taskDir, 'environment', 'workspace.patch'));
log.debug('Copied snapshot.patch -> workspace.patch');
}
// --- Scrub the worker's filesystem layout out of the session ---
// In Explore the recorded `cwd` is the worker's HOST checkout (explore/repo is an absolute
// symlink); rewriting the repo root to /workspace both drops the leak and matches the trial.
const WORKSPACE_MOUNT = '/workspace';
const escapeRegExp = (v: string) => v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
/** Member names when this toolkit is polyglot; empty means single-repo. */
const MEMBER_NAMES: readonly string[] = (() => {
const dir = join(repoRoot, 'repos');
if (!existsSync(dir)) return [];
try {
return readdirSync(dir, { withFileTypes: true })
.filter((e) => e.isDirectory())
.map((e) => e.name);
} catch {
return [];
}
})();
/** The repo root within a cwd — the prefix a trial mounts at /workspace. `/repos/<member>`
* anchors only on a polyglot toolkit, so a personal `~/repos/…` above it can't win. */
function repoRootOf(cwd: string): string | null {
if (MEMBER_NAMES.length > 0) {
// A real member of THIS toolkit wins; the generic shape covers a member whose
// directory the toolkit no longer has (an older snapshot, a renamed member).
for (const name of MEMBER_NAMES) {
const hit = cwd.match(new RegExp(`^(.*?/repos/${escapeRegExp(name)})(?:/|$)`));
if (hit) return hit[1];
}
const generic = cwd.match(/^(.*?\/repos\/[^/]+)(?:\/|$)/);
if (generic) return generic[1];
}
// `/repo` needs a component boundary, so it never matches inside `/repos/`.
const m = cwd.match(/^(.*?\/repo)(?:\/|$)/);
return m ? m[1] : null;
}
/** Rewrite every checkout root to /workspace, and the home dir each sits under to `~`. The
* `repo/` anchor needs no host-root list; the home pass still keys off extractHomeDir. */
function scrubWorkerPaths(raw: string): { text: string; roots: string[] } {
// Each cwd contributes its own root, longest first, so a nested root isn't clobbered
// and a session spanning two checkouts is scrubbed rather than skipped.
const roots = [...new Set([...collectCwds(raw)].map(repoRootOf))]
.filter((r): r is string => r !== null)
.sort((a, b) => b.length - a.length);
const { sanitized } = sanitizeSessionJsonl(raw, {
cwdPrefixes: roots,
placeholder: WORKSPACE_MOUNT,
});
return { text: sanitized, roots };
}
// --- Copy session files for --resume ---
//
// The full session.jsonl (including any post-end_turn entries) goes into the
// task root for reference. A truncated version — keeping everything up to
// and including the last assistant entry with stop_reason="end_turn" — goes
// into environment/ for the container. Stopping on a clean assistant turn
// avoids Claude Code's synthetic "No response requested." injection when
// the session is resumed with --fork-session and a new --print prompt.
const sessionJsonl = join(snapshotDir, 'session.jsonl');
if (existsSync(sessionJsonl)) {
// Fail-open: a session this can't scrub ships exactly as it was, because a
// leaked path is a smaller problem than a task that can't be created.
let sessionText = readFileSync(sessionJsonl, 'utf8');
try {
const { text, roots } = scrubWorkerPaths(sessionText);
if (roots.length > 0) {
sessionText = text;
log.info(
{ roots, mountedAt: WORKSPACE_MOUNT },
'Rewrote the authoring checkout path to the trial mount point'
);
} else {
log.debug('No worker-rooted cwd to rewrite; session used as-is');
}
} catch (err) {
log.warn(
{ err: err instanceof Error ? err.message : String(err) },
'Could not rewrite paths in the session; using it as-is'
);
}
// Full version for reference
writeFileSync(join(taskDir, 'session-full.jsonl'), sessionText);
log.debug('Wrote full session.jsonl to task root');
// Truncated version for the container: strip everything from the last
// user text turn onwards. This drops the failure-eliciting question
// (which `--print` will redeliver to the trial agent as the new prompt)
// AND the failure response itself (so the trial agent doesn't see its
// previous answer), while preserving conversational context up to the
// last clean assistant `end_turn`.
//
// Algorithm (refined Option B):
// 1. Find U = index of the last user-text turn that is NOT a slash
// command (use the same command-marker filter as
// extractLastUserMessage).
// 2. Walk backwards from U - 1 to find the last `assistant` entry
// with stop_reason: "end_turn".
// 3. Truncate slice(0, lastEndTurnIndex + 1).
//
// If U doesn't exist or no end_turn assistant precedes U, write an
// empty session.jsonl — the snapshot agent adapter detects this and
// skips --resume entirely, starting fresh from --print.
const sessionLines = sessionText.trimEnd().split('\n');
// A non-Claude session is not a Claude transcript, so the scan below finds no
// `stop_reason: "end_turn"` and would silently write an empty session. Its reader
// applies the same rule in that harness's own format.
const harness = metadata.harness ?? 'claude-code';
const isClaude = harness === 'claude-code';
let lastUserTextIndex = -1;
for (let i = 0; i < sessionLines.length; i++) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type !== 'user' || typeof entry.message?.content !== 'string') continue;
// Compaction summaries are synthetic user turns whose text often quotes
// earlier /create-snapshot:snapshot runs — never the command turn itself,
// so they must not trip the break below.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
// Mirror extractLastUserMessage: skip the snapshot command itself
// and any slash-command / local-command marker turns.
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserTextIndex = i;
} catch {
continue;
}
}
let lastEndTurnIndex = -1;
if (lastUserTextIndex > 0) {
for (let i = lastUserTextIndex - 1; i >= 0; i--) {
try {
const entry = JSON.parse(sessionLines[i]) as {
type?: string;
message?: { stop_reason?: unknown };
};
if (entry.type === 'assistant' && entry.message?.stop_reason === 'end_turn') {
lastEndTurnIndex = i;
break;
}
} catch {
continue;
}
}
}
if (!isClaude) {
const cut = truncationIndex(turnsFromLines(harness, sessionLines));
const kept = cut >= 0 ? sessionLines.slice(0, cut + 1) : [];
const truncated = stripAuthoringScaffolding(harness, kept);
writeFileSync(
join(taskDir, 'environment', 'session.jsonl'),
truncated.length ? truncated.join('\n') + '\n' : ''
);
log.debug(
{ harness, fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (harness reader)'
);
} else if (lastEndTurnIndex >= 0) {
const truncated = sessionLines.slice(0, lastEndTurnIndex + 1);
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), truncated.join('\n') + '\n');
log.debug(
{ fullLines: sessionLines.length, truncatedLines: truncated.length },
'Wrote truncated session.jsonl to environment/ (strips last user turn + failure response, keeps through last clean assistant end_turn)'
);
} else {
writeFileSync(join(taskDir, 'environment', 'session.jsonl'), '');
if (lastUserTextIndex < 0) {
log.warn(
'No user text turn found in session — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
} else {
log.warn(
'No assistant entry with stop_reason="end_turn" found before the last user turn (one-shot snapshot) — wrote empty session.jsonl. The snapshot agent adapter will skip --resume and start fresh.'
);
}
}
}
const sessionDir = join(snapshotDir, 'session');
if (existsSync(sessionDir) && statSync(sessionDir).isDirectory()) {
copyTree(sessionDir, join(taskDir, 'environment', 'session'));
// Claude Code writes subagent files write-only (--w-------). Fix them so
// Harbor's dirhash can read them during environment setup.
execSync(`chmod -R +r "${join(taskDir, 'environment', 'session')}"`, { stdio: 'pipe' });
log.debug('Copied session/');
} else {
mkdirSync(join(taskDir, 'environment', 'session'), { recursive: true });
}
// The harness that captured the snapshot; the trial runs this one.
const harness =
typeof metadata.harness === 'string' && metadata.harness ? metadata.harness : 'claude-code';
/**
* The model and effort this harness defaulted to when the task was authored, recorded
* for reference only — nothing reads these back, and a trial still resolves both from
* the registry at run time. Best-effort: a task is not worth failing over a note.
*/
function authoredDefaults(harnessId: string): { model: string; effort: string } | null {
try {
const resolver = join(repoRoot, 'scripts', 'resolve_harness.py');
// Same interpreter search as `_raccoon_python` in scripts/lib/harness-credentials.sh
// and `pythonWithTomllib` in submit-task.ts: `python3` is not always 3.11+, and the
// registry needs tomllib. Best-effort, so a miss just omits the note.
let python = '';
for (const candidate of [
process.env.RACCOON_PYTHON,
'python3',
'python3.13',
'python3.12',
'python3.11',
]) {
if (!candidate) continue;
try {
execFileSync(candidate, ['-c', 'import tomllib'], { stdio: 'ignore' });
python = candidate;
break;
} catch {
continue;
}
}
if (!python) return null;
const rows = execFileSync(python, [resolver, '--defaults'], {
encoding: 'utf-8',
stdio: ['ignore', 'pipe', 'ignore'],
});
for (const line of rows.split('\n')) {
const [id, model, effort] = line.split('\t');
if (id === harnessId && model) return { model, effort: effort ?? '' };
}
} catch {
// registry unreadable here — omit the note
}
return null;
}
const authored = authoredDefaults(harness);
// --- Write task.toml ---
// The reference-data corpus is included in every zeta task (build-workspace decides from the repo),
// so there's nothing to set here.
const taskToml = `version = "1.0"
[metadata]
program = "raccoon"
author = "rl-env-coding"
category = "sdlc/technical-writing"
repo = "${repoName}"
commit = "${commitShort}"
# The toolkit release this task was created with. Written by the toolkit —
# leave it in place: task tooling reads it to know which toolkit's assets
# this task grades with.
toolkit_version = "${toolkitConfig?.version ?? 'unknown'}"
snapshot = "${basename(snapshotDir)}"
session_uuid = "${sessionUuid}"
# Set true for a task about a UI: the trial gets Playwright + Chromium (\`pw <script.js>\`),
# and on claude the \`Read\` tool so the agent can view a screenshot it takes.
browser = false
${authored ? `authored_model = "${authored.model}"\nauthored_effort = "${authored.effort}"\n` : ''}
[verifier]
timeout_sec = 7200.0
[agent]
harness = "${harness}"
timeout_sec = 18000.0
[environment]
build_timeout_sec = 6000.0
cpus = 2
memory_mb = 4096
storage_mb = 10240
gpus = 0
allow_internet = true
[verifier.env]
ANTHROPIC_API_KEY = "\${ANTHROPIC_API_KEY}"
ANTHROPIC_BASE_URL = "\${ANTHROPIC_BASE_URL}"
[solution.env]
`;
writeFileSync(join(taskDir, 'task.toml'), taskToml);
log.debug('Wrote task.toml');
// --- Extract instruction from session transcript ---
function extractLastUserMessage(sessionPath: string, harness: string): string | null {
if (!existsSync(sessionPath)) return null;
const lines = readFileSync(sessionPath, 'utf8').trimEnd().split('\n');
// A non-Claude session has no `type: "user"` records, so the scan below finds nothing
// and the worker silently gets a placeholder instruction. Its reader applies the same
// rule — last real user turn, ignoring command invocations — in that harness's format.
if (harness !== 'claude-code') {
const userTurns = turnsFromLines(harness, lines).filter(
(t) => t.role === 'user' && !t.isCommand && t.text.trim()
);
return userTurns.length ? userTurns[userTurns.length - 1].text : null;
}
let lastUserMessage: string | null = null;
for (const line of lines) {
try {
const entry = JSON.parse(line) as {
type?: string;
isCompactSummary?: boolean;
message?: { content?: unknown };
};
if (entry.type === 'user' && typeof entry.message?.content === 'string') {
// Synthetic compaction summary — not a real user turn, and its text
// often quotes earlier /create-snapshot:snapshot runs.
if (entry.isCompactSummary) continue;
const content = entry.message.content;
if (content.includes('create-snapshot:snapshot')) break;
if (
content.includes('<command-name>') ||
content.includes('<command-message>') ||
content.includes('<local-command-caveat>')
) {
continue;
}
lastUserMessage = content;
}
} catch {
continue;
}
}
return lastUserMessage;
}
const lastUserMessage = extractLastUserMessage(
join(snapshotDir, 'session.jsonl'),
metadata.harness ?? 'claude-code'
);
const instructionHeader =
'# Replace this with your refined task instruction\n\n' +
"<!-- The text below was auto-extracted from your snapshot's last user message.\n" +
' Refine, condense, or rewrite to focus on the behavior you want to elicit. -->\n\n';
if (lastUserMessage) {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader + lastUserMessage.trimEnd() + '\n'
);
log.info('Wrote instruction.md (from last user message in session)');
} else {
writeFileSync(
join(taskDir, 'instruction.md'),
instructionHeader +
'<!-- Could not extract user message from session. Write the instruction manually. -->\n'
);
log.warn('Could not extract instruction from session — needs manual editing');
}
// --- Scaffold holistic-rubric.md ---
const holisticRubricMd = `<!--
Snapshot: ${basename(snapshotDir)}
Session: ${metadata.session_uuid}
Repo: ${metadata.remote_url}
Commit: ${metadata.commit}
What happened in the snapshot conversation
The worker was trying to: ${annotation.what_trying}
They hoped Claude would: ${annotation.what_hoping}
Instead, Claude: ${annotation.what_happened}
Draft this file with /write-holistic-rubric, or point your agent at it,
session-full.jsonl, and task-shared/grading-standard.md. Delete this comment
when you are done.
-->
${HOLISTIC_RUBRIC_SCAFFOLD}`;
writeFileSync(join(taskDir, 'tests', 'holistic-rubric.md'), holisticRubricMd);
log.info('Scaffolded tests/holistic-rubric.md (needs manual editing)');
// --- Build workspace ---
const buildScript = join(repoRoot, 'scripts', 'build-workspace.sh');
if (existsSync(buildScript)) {
log.info({ repo: repoName, commit: commitShort }, 'Building workspace');
try {
execSync(`bash "${buildScript}" "${slug}" "${commitShort}"`, {
cwd: repoRoot,
encoding: 'utf8',
stdio: 'inherit',
// build-workspace does a bulk-file write burst (git archive|tar of the
// repo tree + a throwaway git add/commit to apply the patch, and for zeta
// toolkits a hardlink-stage of the ~126k-file reference-data corpus that
// falls back to a full copy across filesystems). On a slow bind mount
// (Docker Desktop non-VirtioFS, or WSL2 with the toolkit on a Windows/9p
// path) that legitimately runs into minutes, so a tight cap false-fails a
// working-but-slow build as "not runnable". Keep this generous — it's only
// a backstop against a true hang; the real Harbor build downstream budgets
// build_timeout_sec = 6000.
timeout: 1_200_000,
});
} catch (e: unknown) {
const msg = e instanceof Error ? e.message : String(e);
log.fatal({ error: msg }, 'Workspace build failed — task is not runnable');
log.fatal(` Retry manually: bash scripts/build-workspace.sh ${slug}`);
log.fatal(` Then: scripts/harbor-run harbor-tasks/${slug}`);
process.exit(1);
}
} else {
log.fatal('scripts/build-workspace.sh not found. Please file a bug.');
process.exit(1);
}
try {
execSync('bash -ic "_ev task_created 2>/dev/null" 2>/dev/null', {
stdio: 'ignore',
timeout: 5000,
});
} catch {
// best-effort
}
// --- Done ---
log.info({ taskDir: resolve(taskDir) }, 'Task scaffolded');
log.info('Next steps:');
log.info(' 1. Review instruction.md');
log.info(' 2. Edit tests/holistic-rubric.md — write the rubric');
log.info(' 3. Run calibration trials to validate scoring tiers');

File diff suppressed because it is too large Load Diff

View File

@@ -1,295 +0,0 @@
/**
* stage-atomic-rubric.ts — stage a task's atomic rubric into the grading
* copies the rubric grader modes read, so
* `HARBOR_GRADER_MODE=rubric-trinary scripts/harbor-regrade …` can run inside
* this toolkit.
*
* Source of truth: the task's own atomic rubric —
* `tests/atomic-rubric.yaml` (the current name), or `tests/rubrics.yaml` on a
* task converted before the rename. A task carrying BOTH names with different
* content is a hard error: silently preferring either file could stage a
* rubric that is not the one just edited, and the grader would grade the
* wrong criteria with no signal.
*
* Staged into `harbor-tasks/<slug>/tests/`:
* - `rubric-criteria.md` — the criteria text shown to the grader: one
* "### Criterion: <id>" section per criterion, guideline and elaboration
* only. Category, severity, and dimensions are stripped, so the grader
* stays severity-blind.
* - `rubric-criteria.json` — {task, criteria: [{id, category, severity,
* dimensions}]} for `render-rubric-grade.py` (criterion-id validation and
* severity-weighted aggregation). The grader never sees this file.
* - `render-rubric-grade.py` — synced from `task-shared/` when the task's
* copy is missing or differs from the shared source.
*
* `tests/grader-context.md` is part of the task's own package — this script
* checks that it exists and never writes it. Write it alongside the rubric;
* the `/write-atomic-rubric` skill covers both files.
*
* Staged files are derived from the rubric. Re-run this script after every
* rubric edit, and run `--restore` to remove the staged copies. This script
* validates structure only (readable YAML, unique criterion ids, at most two
* Crux criteria); the `/detector-rubric-coverage` and `/detector-rubric-form`
* skills are the content review.
*
* Usage:
* npx tsx scripts/stage-atomic-rubric.ts <task-slug>
* npx tsx scripts/stage-atomic-rubric.ts <task-slug> --restore
*/
import { createHash } from 'node:crypto';
import { copyFileSync, existsSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
import { createRequire } from 'node:module';
import { basename, dirname, isAbsolute, join, relative, resolve } from 'node:path';
import { fileURLToPath } from 'node:url';
import pino from 'pino';
import pinoPretty from 'pino-pretty';
import yargs from 'yargs';
import { hideBin } from 'yargs/helpers';
const __dirname = dirname(fileURLToPath(import.meta.url));
const TOOLKIT_ROOT = resolve(__dirname, '..');
const SHARED_DIR = join(TOOLKIT_ROOT, 'task-shared');
const argv = yargs(hideBin(process.argv))
.usage('Usage: $0 <task> [options]')
.positional('task', { type: 'string', describe: 'Task slug, or a path to harbor-tasks/<slug>' })
.option('restore', {
type: 'boolean',
default: false,
describe: 'Remove the files a previous staging created',
})
.option('json', { type: 'boolean', default: false, describe: 'Structured JSON logs' })
.demandCommand(1, 'Name the task to stage: npx tsx scripts/stage-atomic-rubric.ts <task-slug>')
.strict()
.help()
.parseSync();
const log = pino(
{ name: 'stage-atomic-rubric', level: 'info' },
argv.json
? process.stdout
: pinoPretty({ colorize: true, translateTime: 'HH:MM:ss', ignore: 'pid,hostname' })
);
/** The atomic-rubric filenames, current name first. */
const RUBRIC_NAMES = ['atomic-rubric.yaml', 'rubrics.yaml'] as const;
/** Record of exactly what staging created, so --restore removes only that. */
const STAGE_MANIFEST = '.rubric-staged.json';
/** The rubric shape this script needs. Validation is structural only — the
* rubric detectors and the repo-side validator own the content rules. */
interface RubricCriterion {
readonly id: string;
readonly category: string;
readonly severity?: string | null;
readonly guideline: string;
readonly elaboration?: string | null;
readonly dimensions?: readonly string[];
}
interface RubricsDoc {
readonly task: string;
readonly criteria: readonly RubricCriterion[];
}
function fail(message: string): never {
log.error(message);
process.exit(1);
}
/** harbor-tasks/<slug> from a slug or a path, mirroring check-task-infra.ts. */
function resolveTaskDir(task: string): string {
const candidate = isAbsolute(task) ? task : resolve(process.cwd(), task);
if (existsSync(join(candidate, 'task.toml'))) return candidate;
const bySlug = join(TOOLKIT_ROOT, 'harbor-tasks', task);
if (existsSync(join(bySlug, 'task.toml'))) return bySlug;
return fail(`No task found at ${task} or harbor-tasks/${task} (expected a task.toml inside).`);
}
const sha256 = (p: string): string => createHash('sha256').update(readFileSync(p)).digest('hex');
/** The task's rubric file — current name first, pre-rename name honored, both
* present with different content refused. */
function resolveRubricPath(testsDir: string): string {
const present = RUBRIC_NAMES.map((name) => join(testsDir, name)).filter((p) => existsSync(p));
if (present.length === 0) {
return fail(
`No atomic rubric found: expected tests/atomic-rubric.yaml ` +
`(or tests/rubrics.yaml on a task converted before the rename). ` +
`Write it with the /write-atomic-rubric skill first.`
);
}
if (present.length > 1 && new Set(present.map(sha256)).size > 1) {
return fail(
`Both tests/atomic-rubric.yaml and tests/rubrics.yaml exist with different content. ` +
`Keep exactly one; tests/atomic-rubric.yaml is the current name.`
);
}
return present[0];
}
/** Structural gate: the properties the staged outputs are built from. */
function toRubricsDoc(raw: unknown, sourceName: string): RubricsDoc {
if (typeof raw !== 'object' || raw === null || Array.isArray(raw)) {
return fail(`${sourceName} is not a YAML mapping with task and criteria keys.`);
}
const doc = raw as { task?: unknown; criteria?: unknown };
if (typeof doc.task !== 'string' || doc.task.length === 0) {
return fail(`${sourceName} is missing the top-level task key.`);
}
if (!Array.isArray(doc.criteria) || doc.criteria.length === 0) {
return fail(`${sourceName} has no criteria list.`);
}
const seen = new Set<string>();
let cruxCount = 0;
for (const [index, entry] of doc.criteria.entries()) {
const criterion = entry as Partial<RubricCriterion> | null;
if (typeof criterion !== 'object' || criterion === null) {
return fail(`${sourceName} criteria[${index}] is not a mapping.`);
}
if (typeof criterion.id !== 'string' || criterion.id.length === 0) {
return fail(`${sourceName} criteria[${index}] has no id.`);
}
if (seen.has(criterion.id)) {
return fail(`${sourceName} has a duplicate criterion id: ${criterion.id}.`);
}
seen.add(criterion.id);
if (typeof criterion.guideline !== 'string' || criterion.guideline.trim().length === 0) {
return fail(`${sourceName} criterion ${criterion.id} has no guideline.`);
}
if (typeof criterion.category !== 'string' || criterion.category.length === 0) {
return fail(`${sourceName} criterion ${criterion.id} has no category.`);
}
if (criterion.severity === 'crux') cruxCount += 1;
}
if (cruxCount > 2) {
return fail(`${sourceName} designates ${cruxCount} Crux criteria; the cap is two.`);
}
return doc as RubricsDoc;
}
/** Grader-facing view: guideline and elaboration only, severity-blind. */
function renderCriteriaMarkdown(doc: RubricsDoc): string {
const sections = doc.criteria.map((criterion) => {
const parts = [`### Criterion: ${criterion.id}`, criterion.guideline.trim()];
if (criterion.elaboration?.trim()) parts.push(criterion.elaboration.trim());
return parts.join('\n\n');
});
return sections.join('\n\n') + '\n';
}
function stage(taskDir: string): void {
const testsDir = join(taskDir, 'tests');
if (!existsSync(testsDir)) {
return fail(`${relative(TOOLKIT_ROOT, taskDir)} has no tests/ directory.`);
}
const rubricPath = resolveRubricPath(testsDir);
const sourceName = `tests/${basename(rubricPath)}`;
let parsed: unknown;
try {
parsed = parseYaml(readFileSync(rubricPath, 'utf8'));
} catch (error) {
return fail(`${sourceName} is not readable YAML: ${(error as Error).message}`);
}
const doc = toRubricsDoc(parsed, sourceName);
const created: string[] = [];
const writeStaged = (name: string, content: string, mode?: number): void => {
writeFileSync(join(testsDir, name), content, mode ? { mode } : undefined);
created.push(name);
};
writeStaged('rubric-criteria.md', renderCriteriaMarkdown(doc));
writeStaged(
'rubric-criteria.json',
JSON.stringify(
{
task: doc.task,
// severity feeds render-rubric-grade.py's severity-weighted
// aggregation; dimensions ride along for offline slicing. The grader
// never sees this file — severity-blindness lives in
// rubric-criteria.md.
criteria: doc.criteria.map((criterion) => ({
id: criterion.id,
category: criterion.category,
severity: criterion.severity ?? null,
dimensions: criterion.dimensions ?? [],
})),
},
null,
2
) + '\n'
);
// The renderer is a shared asset. Sync it so the regrade runs the current
// weights; scripts/harbor-regrade performs the same self-heal.
const rendererSource = join(SHARED_DIR, 'render-rubric-grade.py');
const rendererDest = join(testsDir, 'render-rubric-grade.py');
if (!existsSync(rendererSource)) {
return fail('task-shared/render-rubric-grade.py is missing from this toolkit.');
}
if (!existsSync(rendererDest) || sha256(rendererDest) !== sha256(rendererSource)) {
copyFileSync(rendererSource, rendererDest);
created.push('render-rubric-grade.py');
}
writeFileSync(join(testsDir, STAGE_MANIFEST), JSON.stringify({ created }, null, 2) + '\n');
if (!existsSync(join(testsDir, 'grader-context.md'))) {
log.warn(
'tests/grader-context.md is missing. The rubric grader modes read it beside the ' +
'criteria; write it before running a rubric-mode regrade or packaging the task.'
);
}
log.info(
{ source: sourceName, criteria: doc.criteria.length, staged: created },
'Staged the atomic-rubric grading copies.'
);
}
function restore(taskDir: string): void {
const testsDir = join(taskDir, 'tests');
const manifestPath = join(testsDir, STAGE_MANIFEST);
if (!existsSync(manifestPath)) {
log.warn('No staging manifest found; nothing to restore.');
return;
}
let names: string[] = [];
try {
const manifest = JSON.parse(readFileSync(manifestPath, 'utf8')) as { created?: unknown };
if (Array.isArray(manifest.created)) {
names = manifest.created.filter((n): n is string => typeof n === 'string');
}
} catch {
return fail(`${STAGE_MANIFEST} is unreadable; remove the staged files by hand.`);
}
for (const name of names) {
// Only ever files this script wrote into tests/ — refuse anything else.
if (name.includes('/') || name.includes('..')) continue;
const filePath = join(testsDir, name);
if (existsSync(filePath)) rmSync(filePath);
}
rmSync(manifestPath);
log.info({ removed: names }, 'Removed the staged grading copies.');
}
// The yaml package reaches containers created after it joined package.json;
// a container created earlier has every other dependency but not this one,
// so resolve it at run time and say what to do instead of crashing.
let parseYaml: (src: string) => unknown;
try {
const requireFromHere = createRequire(fileURLToPath(import.meta.url));
({ parse: parseYaml } = requireFromHere('yaml') as { parse: (src: string) => unknown });
} catch {
fail(
'The yaml package is not installed in this container. Rebuild the Authoring ' +
'container ("Dev Containers: Rebuild Container"), or run npm install in the toolkit root.'
);
}
const taskDir = resolveTaskDir(String(argv._[0]));
if (argv.restore) restore(taskDir);
else stage(taskDir);

View File

@@ -1,146 +0,0 @@
/**
* stamp-trial-inputs.ts — record, at trial LAUNCH time, the checksums of the
* task inputs a harbor run is about to execute against, and stamp them into
* the trial directories the run produces.
*
* Why launch time: copy-reference-run.ts used to capture checksums at COPY
* time, which misses the headline staleness ordering — run trials, edit the
* prompt, then copy the runs — and records the post-edit hashes (a genuinely
* stale run then reads `fresh`). Harbor creates trial dirs itself (and, on
* the daytona backend, populates them only at download after the trial), so
* the earliest host-side point to capture is the moment `scripts/harbor-run`
* launches: `capture` snapshots the inputs to a temp file before harbor
* starts, and `apply` copies that snapshot into each trial dir once the job
* directory exists. copy-reference-run.ts then prefers this run-time record
* over its own capture-at-copy fallback.
*
* Usage (normally invoked by scripts/harbor-run, not by hand):
* npx tsx scripts/stamp-trial-inputs.ts capture <task-dir> --out <file>
* npx tsx scripts/stamp-trial-inputs.ts apply <capture-file> <job-dir> [job-dir...]
*
* Ships in the worker toolkit (see raccoon-worker-toolkit/package-worker-toolkit.ts),
* so it must only import from its shipped file set — same constraint as
* copy-reference-run.ts.
*/
import './lib/check-devcontainer';
import { existsSync, readFileSync, readdirSync, statSync, writeFileSync } from 'fs';
import { fileURLToPath } from 'node:url';
import { basename, join, resolve } from 'path';
import {
INPUT_CHECKSUMS_FILENAME,
captureTaskInputs,
readTaskInputChecksums,
} from './lib/input-checksums';
function usage(): never {
console.error(
[
'Usage:',
' npx tsx scripts/stamp-trial-inputs.ts capture <task-dir> --out <file>',
' npx tsx scripts/stamp-trial-inputs.ts apply <capture-file> <job-dir> [job-dir...]',
].join('\n')
);
process.exit(1);
}
/** Snapshot the task inputs as they stand at launch; write the record to `outFile`. */
export function captureCommand(taskDir: string, outFile: string): void {
if (!existsSync(taskDir)) {
console.error(`Error: task dir ${taskDir} does not exist`);
process.exit(1);
}
// taskSlug scopes `apply` to this task's trial dirs (concurrent harbor-runs
// share harbor-jobs/) and makes any mis-routed stamp diagnosable later.
const record = { ...captureTaskInputs(taskDir, 'run'), taskSlug: basename(resolve(taskDir)) };
writeFileSync(outFile, JSON.stringify(record, null, 2) + '\n');
}
/**
* Does this trial dir belong to the task the capture was taken from? Two
* signals, strongest first:
*
* 1. result.json `task_name` — written per-trial by harbor with the FULL,
* unambiguous slug (org-prefixed for hub-published tasks). Authoritative
* when readable, exactly as copy-reference-run.ts resolves trials.
* 2. The trial DIRNAME's `<prefix>__<trialId>` prefix — harbor TRUNCATES
* long slugs here, so the test is "the prefix is a truncation of the
* slug", not equality. (Two tasks sharing a truncated prefix are told
* apart by signal 1; the dirname alone can't distinguish them.)
*/
function trialBelongsToTask(trialDir: string, entry: string, slug: string): boolean {
const sep = entry.lastIndexOf('__');
if (sep === -1) return false;
const resultPath = join(trialDir, 'result.json');
if (existsSync(resultPath)) {
try {
const taskName = (JSON.parse(readFileSync(resultPath, 'utf-8')) as { task_name?: unknown })
.task_name;
if (typeof taskName === 'string' && taskName.length > 0) {
return taskName.replace(/^[^/]+\//, '') === slug;
}
} catch {
// Unparseable result.json — fall through to the dirname prefix.
}
}
const prefix = entry.substring(0, sep);
return prefix === slug || slug.startsWith(prefix);
}
/**
* Copy a launch-time capture into the capture's OWN task's trial dirs under
* the given harbor job dir(s). Trial dirs are the `<slug>__<trialId>`
* subdirectories harbor creates; anything else (stray files, harbor's own
* metadata) is skipped — and so is any trial belonging to a DIFFERENT task:
* several harbor-run invocations can share a cwd, and stamping another
* task's trials with this capture's hashes would fabricate `capturedBy:
* 'run'` evidence for inputs that task never ran against. An existing record
* is left alone — it can only be from an earlier stamp of the same trial.
*/
export function applyCommand(captureFile: string, jobDirs: string[]): number {
const record = readTaskInputChecksums(captureFile);
if (!record || typeof record.taskSlug !== 'string' || record.taskSlug.length === 0) {
console.error(
`Error: ${captureFile} is not a readable input-checksums capture with a taskSlug`
);
process.exit(1);
}
const slug = record.taskSlug;
const raw = readFileSync(captureFile, 'utf-8');
let stamped = 0;
for (const jobDir of jobDirs) {
if (!existsSync(jobDir)) continue;
for (const entry of readdirSync(jobDir)) {
const trialDir = join(jobDir, entry);
if (!entry.includes('__') || !statSync(trialDir).isDirectory()) continue;
if (!trialBelongsToTask(trialDir, entry, slug)) continue;
const dest = join(trialDir, INPUT_CHECKSUMS_FILENAME);
if (existsSync(dest)) continue;
writeFileSync(dest, raw);
stamped++;
console.log(`Stamped ${dest}`);
}
}
return stamped;
}
// Main. Guarded so the test file can import the commands without running them.
if (process.argv[1] && fileURLToPath(import.meta.url) === resolve(process.argv[1])) {
const [command, ...rest] = process.argv.slice(2);
if (command === 'capture') {
const outIdx = rest.indexOf('--out');
const taskDir = rest.filter((a, i) => i !== outIdx && i !== outIdx + 1)[0];
const outFile = outIdx !== -1 ? rest[outIdx + 1] : undefined;
if (!taskDir || !outFile) usage();
captureCommand(taskDir, outFile);
} else if (command === 'apply') {
const [captureFile, ...jobDirs] = rest;
if (!captureFile || jobDirs.length === 0) usage();
const stamped = applyCommand(captureFile, jobDirs);
console.log(`Stamped ${stamped} trial dir(s) with launch-time input checksums`);
} else {
usage();
}
}

View File

@@ -1,93 +0,0 @@
#!/usr/bin/env python3
"""str_replace_editor — CLI-as-MCP wrapper around the vendored EditTool.
This is the "CLI-as-MCP" delivery of the `str_replace_editor` tool: the agent
(which has ONLY the bash tool) invokes this script and passes the tool's
arguments as one JSON object on stdin. The actual editing logic is the vendored
`EditTool` under str_replace_editor_vendor/ (see VENDORED.md) — we add no
behavior, we only:
* instantiate it with run_command_preexec_fn=None (the class's own documented
way to skip its uid/gid-1000 demotion, which would break writes in our
sandbox where the workspace is owned by the agent user); and
* adapt structured stdin-JSON <-> a bash-invokable CLI.
stdin: one JSON object, e.g.
{"command":"view","path":"/workspace/app/models/x.rb"}
{"command":"view","path":"/workspace/x.rb","view_range":[1,40]}
{"command":"str_replace","path":"/workspace/x.rb","old_str":"a","new_str":"b"}
{"command":"create","path":"/workspace/new.rb","file_text":"..."}
{"command":"insert","path":"/workspace/x.rb","insert_line":10,"insert_text":"..."}
stdout: the tool's result text (exit 0). stderr + exit 1: a tool error message.
"""
import asyncio
import json
import os
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from str_replace_editor_vendor.base import ToolError # noqa: E402
from str_replace_editor_vendor.edit import EditTool # noqa: E402
# The keyword-only params the vendored EditTool.__call__ accepts.
_ACCEPTED = {
"command", "path", "file_text", "view_range",
"old_str", "new_str", "insert_text", "insert_line",
}
async def _run(payload: dict):
# Reject unknown keys instead of silently dropping them: a typo like
# `old_string` (vs `old_str`) should be a clear argument error, not a
# confusing failure deeper inside EditTool with the param silently missing.
unknown = set(payload) - _ACCEPTED
if unknown:
raise ToolError(
f"unknown argument(s): {', '.join(sorted(unknown))}. "
f"accepted keys: {', '.join(sorted(_ACCEPTED))}."
)
kwargs = dict(payload)
if "command" not in kwargs or "path" not in kwargs:
raise ToolError("Both `command` and `path` are required.")
# run_command_preexec_fn=None → no uid/gid demotion (see module docstring).
tool = EditTool(run_command_preexec_fn=None)
return await tool(**kwargs)
def main() -> int:
raw = sys.stdin.read()
if not raw.strip():
sys.stderr.write("str_replace_editor: expected a JSON object on stdin\n")
return 2
try:
payload = json.loads(raw)
except json.JSONDecodeError as e:
sys.stderr.write(f"str_replace_editor: invalid JSON on stdin: {e}\n")
return 2
if not isinstance(payload, dict):
sys.stderr.write("str_replace_editor: stdin JSON must be an object\n")
return 2
try:
result = asyncio.run(_run(payload))
except ToolError as e:
sys.stderr.write((e.message or "tool error") + "\n")
return 1
except TypeError as e:
# e.g. an unexpected/duplicate kwarg shape — surface like a tool error.
sys.stderr.write(f"str_replace_editor: bad arguments: {e}\n")
return 1
# EditTool returns a (CLI)Result with .output / .error / .base64_image / .system
if getattr(result, "error", None):
sys.stderr.write(result.error if result.error.endswith("\n") else result.error + "\n")
if getattr(result, "system", None):
sys.stderr.write(f"[system] {result.system}\n")
out = getattr(result, "output", None) or ""
if getattr(result, "base64_image", None):
out += "\n(image content omitted in CLI mode)"
if out:
sys.stdout.write(out if out.endswith("\n") else out + "\n")
return 1 if getattr(result, "error", None) else 0
if __name__ == "__main__":
sys.exit(main())

View File

@@ -1 +0,0 @@
"""Vendored verbatim — do not edit. See VENDORED.md for provenance."""

View File

@@ -1,49 +0,0 @@
from dataclasses import dataclass, fields, replace
@dataclass(kw_only=True, frozen=True)
class ToolResult:
"""Represents the result of a tool execution."""
output: str | None = None
error: str | None = None
base64_image: str | None = None
system: str | None = None
def __bool__(self):
return any(getattr(self, field.name) for field in fields(self))
def __add__(self, other: "ToolResult"):
def combine_fields(field: str | None, other_field: str | None, concatenate: bool = True):
if field and other_field:
if concatenate:
return field + other_field
raise ValueError("Cannot combine tool results")
return field or other_field
return ToolResult(
output=combine_fields(self.output, other.output),
error=combine_fields(self.error, other.error),
base64_image=combine_fields(self.base64_image, other.base64_image, False),
system=combine_fields(self.system, other.system),
)
def replace(self, **kwargs):
"""Returns a new ToolResult with the given fields replaced."""
return replace(self, **kwargs)
# QUESTION(simon): What's our intent behind differentiating here?
class CLIResult(ToolResult):
"""A ToolResult that can be rendered as a CLI output."""
class ToolFailure(ToolResult):
"""A ToolResult that represents a failure."""
class ToolError(Exception):
"""Raised when a tool encounters an error."""
def __init__(self, message):
self.message = message

View File

@@ -1,476 +0,0 @@
import asyncio
import base64
import shlex
from collections import deque
from dataclasses import dataclass
from pathlib import Path
from typing import Literal, get_args
from .base import CLIResult, ToolError, ToolResult
from .run import demote, maybe_truncate, run
TRUNCATED_MESSAGE: str = "<response clipped><NOTE>To save on context only part of this file has been shown to you. You should retry this tool after you have searched inside the file with `grep -n` in order to find the line numbers of what you are looking for.</NOTE>"
Command = Literal[
"view",
"create",
"str_replace",
"insert",
]
SNIPPET_LINES: int = 4
MAX_RESPONSE_LEN: int = 16000
class EditTool:
"""
An filesystem editor tool that allows the agent to view, create, and edit files.
The tool parameters are defined by Anthropic and are not editable.
"""
def __init__(self, run_command_preexec_fn=demote):
"""
Initialize the EditTool.
Args:
run_command_preexec_fn: Function to run in child process before executing
shell commands via the run() utility.
Defaults to demote() which drops privileges to uid/gid 1000.
Pass None to skip preexec, or any callable for custom behavior.
"""
self._run_command_preexec_fn = run_command_preexec_fn
async def __call__(
self,
*,
command: Command,
path: str,
file_text: str | None = None,
view_range: list[int] | None = None,
old_str: str | None = None,
new_str: str | None = None,
insert_text: str | None = None,
insert_line: int | None = None,
):
_path = Path(path)
self.validate_path(command, _path)
if command == "view":
return await self.view(_path, view_range)
elif command == "create":
if file_text is None:
raise ToolError("Parameter `file_text` is required for command: create")
await self.write_file(_path, file_text)
return ToolResult(output=f"File created successfully at: {_path}")
elif command == "str_replace":
if old_str is None:
raise ToolError("Parameter `old_str` is required for command: str_replace")
return await self.str_replace(_path, old_str, new_str)
elif command == "insert":
if insert_line is None:
raise ToolError("Parameter `insert_line` is required for command: insert")
if insert_text is None:
raise ToolError("Parameter `insert_text` is required for command: insert")
return await self.insert(_path, insert_line, insert_text)
raise ToolError(
f"Unrecognized command {command}. The allowed commands for the {self.name} tool are: {', '.join(get_args(Command))}"
)
def validate_path(self, command: str, path: Path):
"""
Check that the path/command combination is valid.
"""
# Check if its an absolute path
if not path.is_absolute():
suggested_path = Path("") / path
raise ToolError(
f"The path {path} is not an absolute path, it should start with `/`. Maybe you meant {suggested_path}?"
)
# Check if path exists
if not path.exists() and command != "create":
raise ToolError(f"The path {path} does not exist. Please provide a valid path.")
if path.exists() and command == "create":
raise ToolError(f"File already exists at: {path}. Cannot overwrite files using command `create`.")
# Check if the path points to a directory
if path.is_dir():
if command != "view":
raise ToolError(
f"The path {path} is a directory and only the `view` command can be used on directories"
)
async def view(self, path: Path, view_range: list[int] | None = None):
"""Implement the view command"""
if path.is_dir():
if view_range:
raise ToolError("The `view_range` parameter is not allowed when `path` points to a directory.")
_, stdout, stderr = await run(
rf"find {path} -maxdepth 2 -not -path '*/\.*'", preexec_fn=self._run_command_preexec_fn
)
if not stderr:
stdout = f"Here's the files and directories up to 2 levels deep in {path}, excluding hidden items:\n{stdout}\n"
return CLIResult(output=stdout, error=stderr)
image_extensions = {'.png', '.jpg', '.jpeg', '.gif', '.bmp', '.tiff', '.tif', '.webp', '.svg', '.ico'}
if path.suffix.lower() in image_extensions:
if view_range:
raise ToolError("The `view_range` parameter is not allowed when `path` points to an image file.")
try:
image_bytes = path.read_bytes()
base64_encoded = base64.b64encode(image_bytes).decode()
return CLIResult(
output=f"Displaying image file: {path}",
base64_image=base64_encoded
)
except Exception as e:
raise ToolError(f"Failed to read image file {path}: {e}") from None
file_content = await self.read_file(path, truncate_after=None)
file_text_lines = file_content.splitlines(keepends=True)
n_lines_file = len(file_text_lines) + (1 if file_content.endswith(("\n", "\r\n", "\r")) else 0)
if view_range:
if len(view_range) != 2 or not all(isinstance(i, int) for i in view_range):
raise ToolError("Invalid `view_range`. It should be a list of two integers.")
init_line, final_line = view_range
if init_line < 1 or init_line > n_lines_file:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its first element `{init_line}` should be within the range of lines of the file: {[1, n_lines_file]}"
)
if final_line > n_lines_file:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its second element `{final_line}` should be smaller than the number of lines in the file: `{n_lines_file}`"
)
if final_line != -1 and final_line < init_line:
raise ToolError(
f"Invalid `view_range`: {view_range}. Its second element `{final_line}` should be larger or equal than its first `{init_line}`"
)
# Extract only the requested lines
if final_line != -1:
selected_lines = file_text_lines[max(view_range[0] - 1, 0) : view_range[1]]
else:
selected_lines = file_text_lines[max(view_range[0] - 1, 0) :]
# Join without modifying the original line endings
file_content = "".join(selected_lines)
file_content = process_view_output_str(
file_text=file_content,
path=str(path),
total_path_lines=n_lines_file,
max_resp_ln=MAX_RESPONSE_LEN,
view_range=(view_range[0], view_range[1]) if view_range else None,
)
return CLIResult(output=file_content)
async def str_replace(self, path: Path, old_str: str, new_str: str | None):
"""Implement the str_replace command, which replaces old_str with new_str in the file content"""
# Read the file content
file_content = await self.read_file(path, truncate_after=None)
new_str = new_str if new_str is not None else ""
# Check if old_str is unique in the file
occurrences = file_content.count(old_str)
if occurrences == 0:
raise ToolError(f"No replacement was performed, old_str `{old_str}` did not appear verbatim in {path}.")
elif occurrences > 1:
file_content_lines = file_content.split("\n")
lines = [idx + 1 for idx, line in enumerate(file_content_lines) if old_str in line]
raise ToolError(
f"No replacement was performed. Multiple occurrences of old_str `{old_str}` in lines {lines}. Please ensure it is unique"
)
# Replace old_str with new_str
new_file_content = file_content.replace(old_str, new_str)
# Write the new content to the file
await self.write_file(path, new_file_content)
# Create a snippet of the edited section
replacement_line = file_content.split(old_str)[0].count("\n")
start_line = max(0, replacement_line - SNIPPET_LINES)
end_line = replacement_line + SNIPPET_LINES + new_str.count("\n")
snippet = "\n".join(new_file_content.split("\n")[start_line : end_line + 1])
# Prepare the success message
success_msg = f"The file {path} has been edited. "
success_msg += self._make_output(snippet, f"a snippet of {path}", start_line + 1)
success_msg += "Review the changes and make sure they are as expected. Edit the file again if necessary."
return CLIResult(output=success_msg)
async def insert(self, path: Path, insert_line: int, new_str: str):
"""Implement the insert command, which inserts new_str at the specified line in the file content."""
file_text = await self.read_file(path, truncate_after=None)
file_text_lines = file_text.split("\n")
n_lines_file = len(file_text_lines)
if insert_line < 0 or insert_line > n_lines_file:
raise ToolError(
f"Invalid `insert_line` parameter: {insert_line}. It should be within the range of lines of the file: {[0, n_lines_file]}"
)
new_str_lines = new_str.split("\n")
new_file_text_lines = file_text_lines[:insert_line] + new_str_lines + file_text_lines[insert_line:]
snippet_lines = (
file_text_lines[max(0, insert_line - SNIPPET_LINES) : insert_line]
+ new_str_lines
+ file_text_lines[insert_line : insert_line + SNIPPET_LINES]
)
new_file_text = "\n".join(new_file_text_lines)
snippet = "\n".join(snippet_lines)
await self.write_file(path, new_file_text)
success_msg = f"The file {path} has been edited. "
success_msg += self._make_output(
snippet,
"a snippet of the edited file",
max(1, insert_line - SNIPPET_LINES + 1),
)
success_msg += "Review the changes and make sure they are as expected (correct indentation, no duplicate lines, etc). Edit the file again if necessary."
return CLIResult(output=success_msg)
async def read_file(self, path: Path, truncate_after: int | None = MAX_RESPONSE_LEN):
"""Read the content of a file from a given path; raise a ToolError if an error occurs."""
try:
code, out, err = await run(
f"cat {shlex.quote(str(path))}", truncate_after=truncate_after, preexec_fn=self._run_command_preexec_fn
)
if code != 0:
raise ToolError(f"Ran into {err} while trying to read {path}")
return out
except Exception as e:
print(e)
raise ToolError(f"Ran into {e} while trying to read {path}") from None
async def write_file(self, path: Path, file: str):
"""Write the content of a file to a given path; raise a ToolError if an error occurs."""
try:
# Write using stdin to avoid argument size limits
process = await asyncio.create_subprocess_shell(
f"cat > {shlex.quote(str(path))}",
stdin=asyncio.subprocess.PIPE,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
preexec_fn=self._run_command_preexec_fn,
)
stdout, stderr = await asyncio.wait_for(
process.communicate(input=file.encode('utf-8')),
timeout=120.0
)
if process.returncode != 0:
raise ToolError(f"Ran into {stderr.decode()} while trying to write to {path}")
except asyncio.TimeoutError:
raise ToolError(f"Timed out while trying to write to {path}")
except Exception as e:
raise ToolError(f"Ran into {e} while trying to write to {path}") from None
def _make_output(
self,
file_content: str,
file_descriptor: str,
init_line: int = 1,
expand_tabs: bool = True,
):
"""Generate output for the CLI based on the content of a file."""
file_content = maybe_truncate(file_content)
if expand_tabs:
file_content = file_content.expandtabs()
file_content = "\n".join([f"{i + init_line:6}\t{line}" for i, line in enumerate(file_content.split("\n"))])
return f"Here's the result of running `cat -n` on {file_descriptor}:\n" + file_content + "\n"
### AUX utilities
def add_line_numbers(text: str, includes_final_line: bool, n_first_line: int = 1) -> str:
"""
Given a string, returns the string with line numbers prepended to each line.
This function:
- Preserves the original line endings (CR, LF, or CRLF) of each line
- Adds a tab-separated line number prefix to each line
- If the text ends with any newline character (\n, \r\n, or \r), adds an
additional empty numbered line to represent the terminal empty line
"""
lines_with_endings = text.splitlines(keepends=True)
result = [f"{ind + n_first_line:6}\t{line_with_ending}" for ind, line_with_ending in enumerate(lines_with_endings)]
# Add an extra empty line with line number if original text ends with newline
if includes_final_line and text.endswith(("\n", "\r\n", "\r")):
result.append(f"{len(lines_with_endings) + n_first_line:6}\t")
return "".join(result)
def process_view_output_str(
file_text: str,
path: str,
total_path_lines: int,
max_resp_ln: int,
view_range: tuple[int, int] | None = None,
) -> str:
# Get header
header = f"Here's the content of {path} with line numbers"
if total_path_lines is not None and view_range is not None:
header += f" (which has a total of {total_path_lines} lines) with view_range={list(view_range)}"
# See if final line is included in the view_range
if view_range is None or view_range[1] == -1 or view_range[1] == total_path_lines:
includes_final_line = True
else:
includes_final_line = False
n_first_line = view_range[0] if view_range is not None else 1
# Truncate if needed
maybe_truncated_str = truncate_from_middle_v2(ss=file_text, max_len=max_resp_ln, n_line_offset=n_first_line - 1)
if isinstance(maybe_truncated_str, str):
# No truncation
file_text_with_line_numbers = add_line_numbers(
file_text,
includes_final_line=includes_final_line,
n_first_line=n_first_line,
)
else:
# Truncation occurred
before_with_line_numbers = add_line_numbers(
text="".join(maybe_truncated_str.as_str(maybe_truncated_str.before_lines)),
includes_final_line=False,
n_first_line=n_first_line,
)
if maybe_truncated_str.single_line:
file_text_with_line_numbers = before_with_line_numbers
else:
after_with_line_numbers = add_line_numbers(
text="".join(maybe_truncated_str.as_str(maybe_truncated_str.after_lines)),
includes_final_line=includes_final_line,
n_first_line=1 + maybe_truncated_str.truncated_end_line,
)
file_text_with_line_numbers = (
before_with_line_numbers + f"\t{maybe_truncated_str.truncation_msg}" + after_with_line_numbers
)
# Add context-aware truncation message
if view_range is not None:
# User already using view_range, suggest adjusting it
truncation_note = "\n<response clipped><NOTE>To save on context only part of the view range has been shown. You can adjust the view_range parameters or use `grep -n` to find specific content.</NOTE>"
else:
# User viewing whole file, suggest view_range or grep
truncation_note = "\n<response clipped><NOTE>To save on context only part of this file has been shown to you. You can use view_range=[start_line, end_line] to see specific sections, or use `grep -n` to find what you're looking for.</NOTE>"
file_text_with_line_numbers += truncation_note
return f"{header}:\n{file_text_with_line_numbers}"
@dataclass
class TruncatedString:
# Blocks
before_lines: list[str]
middle_lines: list[str]
after_lines: list[str]
# Line numbers (starting from 1)
truncated_start_line: int
truncated_end_line: int
# Truncation msg
truncation_msg: str
single_line: bool
def as_str(self, lines: list[str]) -> str:
return "".join(lines)
@property
def full_truncated_str(self) -> str:
return "".join(self.before_lines + [self.truncation_msg] + self.after_lines)
def truncate_from_middle_v2(ss: str, max_len: int, n_line_offset: int = 0) -> "str | TruncatedString":
"""
If no truncation is needed, returns the original string.
If truncation is needed, returns TruncatedString
"""
# No truncation needed
if len(ss) <= max_len:
return ss
# Single line
lines_with_endings = ss.splitlines(True)
if len(lines_with_endings) == 1:
chars_per_side = max(1, max_len // 2)
truncated_char_count = len(ss) - (chars_per_side * 2)
truncation_msg = f"...< truncated {truncated_char_count} characters >..."
before_lines = [ss[:chars_per_side] + truncation_msg + ss[-chars_per_side:]]
return TruncatedString(
before_lines=before_lines,
middle_lines=[],
after_lines=[],
truncated_start_line=1 + n_line_offset,
truncated_end_line=1 + n_line_offset,
truncation_msg=truncation_msg,
single_line=True,
)
# Line truncation
current_len = 0
before_lines = []
middle_lines = deque(lines_with_endings)
after_lines = deque([])
while current_len < max_len and len(middle_lines) > 1:
# Before
before_candidate_line = middle_lines[0]
if len(before_candidate_line) + current_len <= max_len:
before_lines.append(middle_lines.popleft())
current_len += len(before_candidate_line)
else:
break
# After
if len(middle_lines) > 1:
after_candidate_line = middle_lines[-1]
if len(after_candidate_line) + current_len <= max_len:
after_lines.appendleft(middle_lines.pop())
current_len += len(after_candidate_line)
else:
break
# Find truncated lines
first_truncated_line = 1 + len(before_lines) + n_line_offset
last_truncated_line = first_truncated_line + len(middle_lines) - 1
if ss.endswith(("\n", "\r", "\r\n")) and len(after_lines) == 0:
last_truncated_line += 1
# Create truncation msg
if first_truncated_line == last_truncated_line:
truncation_msg = f"< truncated line {first_truncated_line} >"
else:
truncation_msg = f"< truncated lines {first_truncated_line}-{last_truncated_line} >"
if len(after_lines) != 0:
if before_lines[0].endswith("\r\n"):
truncation_msg += "\r\n"
elif before_lines[0].endswith("\r"):
truncation_msg += "\r"
else:
truncation_msg += "\n"
return TruncatedString(
# Blocks
before_lines=before_lines,
middle_lines=list(middle_lines),
after_lines=list(after_lines),
# Line numbers (starting from 1)
truncated_start_line=first_truncated_line,
truncated_end_line=last_truncated_line,
# Truncation msg
truncation_msg=truncation_msg,
single_line=False,
)

View File

@@ -1,66 +0,0 @@
"""Utility to run shell commands asynchronously with a timeout."""
import asyncio # noqa -- swapping to trio would be beneficial, but not blocking atm
import os
TRUNCATED_MESSAGE: str = "<response clipped><NOTE>To save on context only part of this file has been shown to you. You should retry this tool after you have searched inside the file with `grep -n` in order to find the line numbers of what you are looking for.</NOTE>"
MAX_RESPONSE_LEN: int = 16000
def maybe_truncate(content: str, truncate_after: int | None = MAX_RESPONSE_LEN):
"""Truncate content and append a notice if content exceeds the specified length."""
return (
content
if not truncate_after or len(content) <= truncate_after
else content[:truncate_after] + TRUNCATED_MESSAGE
)
def demote():
"""Drop privileges to uid/gid 1000 for security.
This function is intended to be used as a preexec_fn in subprocess calls
to ensure commands run with reduced privileges.
"""
os.setgid(1000)
os.setuid(1000)
async def run(
cmd: str,
timeout: float | None = 120.0, # seconds # noqa: ASYNC109
truncate_after: int | None = MAX_RESPONSE_LEN,
preexec_fn=demote,
):
"""Run a shell command asynchronously with a timeout.
Args:
cmd: Command to execute
timeout: Command timeout in seconds
truncate_after: Maximum response length before truncation
preexec_fn: Function to run in child process before exec (default: demote).
Pass None to skip preexec, or any callable for custom behavior.
Returns:
Tuple of (return_code, stdout, stderr)
"""
process = await asyncio.create_subprocess_shell(
cmd,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
preexec_fn=preexec_fn,
)
try:
stdout, stderr = await asyncio.wait_for(process.communicate(), timeout=timeout)
return (
process.returncode or 0,
maybe_truncate(stdout.decode(), truncate_after=truncate_after),
maybe_truncate(stderr.decode(), truncate_after=truncate_after),
)
except TimeoutError as exc:
try:
process.kill()
except ProcessLookupError:
pass
raise TimeoutError(f"Command '{cmd}' timed out after {timeout} seconds") from exc

File diff suppressed because it is too large Load Diff

View File

@@ -1,20 +0,0 @@
# Your actual toolset (this overrides any earlier tool guidance above)
This harness gives you exactly two ways to act, both through the `Bash` tool:
1. **Shell commands** for everything read-only and for running things: view and search files with `cat`, `sed -n`, `grep -rn`, `find`, `ls`; run tests; run `git`; etc.
2. **A `str_replace_editor` file editor**, which you invoke from Bash by piping ONE JSON object on stdin to `/opt/agent-cli/str_replace_editor`. Use a quoted heredoc so backslashes and quotes survive:
`/opt/agent-cli/str_replace_editor <<'EDITOR'` then a line of JSON then `EDITOR`
The JSON `"command"` field selects the operation:
- `view` — view a file (optionally `"view_range":[start,end]`) or list a directory: `{"command":"view","path":"/abs/file.rb"}`
- `create` — create a NEW file (fails if it exists): `{"command":"create","path":"/abs/new.rb","file_text":"..."}`
- `str_replace` — replace a UNIQUE substring: `{"command":"str_replace","path":"/abs/file.rb","old_str":"...","new_str":"..."}`
- `insert` — insert text after a line: `{"command":"insert","path":"/abs/file.rb","insert_line":N,"insert_text":"..."}`
Paths must be absolute. Inside JSON strings, escape newlines as `\n` and double-quotes as `\"`.
There are **no** `Read`, `Grep`, `Glob`, `Edit`, `Write`, `MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite`, or `AskUserQuestion` tools — `Bash` is your only built-in tool. So disregard the earlier "Prefer the dedicated file/search tools over shell commands" guidance and the Memory section's "use the Write tool" instruction: those tools are not available in this harness. Search and read with shell commands; view, create, and edit files with `str_replace_editor`.
There is also no tool for asking the user an interactive question. If you need to ask the user something, or raise a concern about the request before acting on it, put it in your normal text response.

View File

@@ -1,4 +0,0 @@
## Browser
Chromium is available in this environment via Playwright. `pw <script.js>` runs Node with
`require("playwright")` resolvable (CommonJS — `import` will not find it).

View File

@@ -1,7 +0,0 @@
## Correction to the toolset above: you also have `Read`
This task runs with `Read` in addition to `Bash`, so the statement above that there is no `Read`
tool does not apply here. `Read` renders images — use it to look at a screenshot you have
written to disk. Everything else above still holds: no `Grep`, `Glob`, `Edit`, `Write`,
`MultiEdit`, `NotebookEdit`, `Task`, `TodoWrite` or `AskUserQuestion`, and you still create and
edit files with `str_replace_editor`.

View File

@@ -1,88 +0,0 @@
#!/usr/bin/env python3
"""validate_task_dir.py — say WHY harbor will not accept a task directory.
``harbor run -p <dir>`` silently reinterprets a directory that fails task
validation as a *dataset* of tasks, finds none inside, and dies with
``ValueError: Either datasets or tasks must be provided.`` — a message naming
neither the path nor the missing file. ``scripts/harbor-run`` calls this first so
the author reads "tests/test.sh is missing" instead.
Must run under HARBOR'S interpreter (its uv-tool venv), not any python3.11+: it
imports harbor to reuse ``Task.is_valid_dir``, the exact predicate the CLI
branches on, so the two cannot drift.
Usage: validate_task_dir.py <task-dir> [--disable-verification]
Prints ``verdict=valid`` or ``verdict=invalid`` on stdout; the reason goes to
stderr. Callers must gate on the stdout verdict, never on the exit code alone —
an interpreter that cannot run this file at all also exits non-zero.
Exit 0 = valid, 1 = invalid, 2 = the check could not run.
"""
import sys
from pathlib import Path
def reason(task_dir: Path, disable_verification: bool) -> str | None:
"""Return why harbor rejects task_dir, or None if it accepts it."""
from harbor.models.task.config import TaskConfig
from harbor.models.task.paths import TaskPaths
from harbor.models.task.task import Task
if Task.is_valid_dir(task_dir, disable_verification=disable_verification):
return None
paths = TaskPaths(task_dir)
if not paths.config_path.exists():
return f"{paths.config_path} is missing."
if not paths.environment_dir.exists():
return f"{paths.environment_dir} is missing."
try:
config = TaskConfig.model_validate_toml(paths.config_path.read_text())
except Exception as exc:
return f"{paths.config_path} does not parse as a task config: {exc}"
# A stepped task carries no root instruction.md, so only the shape harbor
# checks may be asserted here — hence steps first, root instruction last.
if disable_verification:
for step in config.steps or []:
if not paths.step_dir(step.name).exists():
return f"{paths.step_dir(step.name)} is missing."
if not paths.step_instruction_path(step.name).exists():
return f"{paths.step_instruction_path(step.name)} is missing."
if not config.steps and not paths.instruction_path.exists():
return f"{paths.instruction_path} is missing."
else:
# Private, but it owns the instruction/test diagnostics is_valid_dir discards.
try:
Task._validate_tests(config, paths)
except FileNotFoundError as exc:
return str(exc)
except AttributeError:
pass
return f"{task_dir} is not a task directory harbor recognizes."
def main() -> int:
args = sys.argv[1:]
disable_verification = "--disable-verification" in args
positional = [a for a in args if not a.startswith("-")]
if len(positional) != 1:
print(
f"usage: {sys.argv[0]} <task-dir> [--disable-verification]", file=sys.stderr
)
return 2
try:
why = reason(Path(positional[0]), disable_verification)
except Exception as exc:
print(f"validate_task_dir: check did not run ({exc})", file=sys.stderr)
return 2
if why is None:
print("verdict=valid")
return 0
print("verdict=invalid")
print(why, file=sys.stderr)
return 1
if __name__ == "__main__":
sys.exit(main())

View File

@@ -1,41 +0,0 @@
#!/bin/bash
# Welcome banner for raccoon dev containers
CYAN='\033[1;36m'
YELLOW='\033[1;33m'
GRAY='\033[0;90m'
RESET='\033[0m'
CONTAINER_TYPE="${1:-explore}"
if [ "$CONTAINER_TYPE" = "explore" ]; then
COLOR="$CYAN"
else
COLOR="$YELLOW"
fi
cat << 'RACCOON'
.----------------. .----------------. .----------------. .----------------. .----------------. .----------------. .-----------------.
| .--------------. || .--------------. || .--------------. || .--------------. || .--------------. || .--------------. || .--------------. |
| | _______ | || | __ | || | ______ | || | ______ | || | ____ | || | ____ | || | ____ _____ | |
| | |_ __ \ | || | / \ | || | .' ___ | | || | .' ___ | | || | .' `. | || | .' `. | || ||_ \|_ _| | |
| | | |__) | | || | / /\ \ | || | / .' \_| | || | / .' \_| | || | / .--. \ | || | / .--. \ | || | | \ | | | |
| | | __ / | || | / ____ \ | || | | | | || | | | | || | | | | | | || | | | | | | || | | |\ \| | | |
| | _| | \ \_ | || | _/ / \ \_ | || | \ `.___.'\ | || | \ `.___.'\ | || | \ `--' / | || | \ `--' / | || | _| |_\ |_ | |
| | |____| |___| | || ||____| |____|| || | `._____.' | || | `._____.' | || | `.____.' | || | `.____.' | || ||_____|\____| | |
| | | || | | || | | || | | || | | || | | || | | |
| '--------------' || '--------------' || '--------------' || '--------------' || '--------------' || '--------------' || '--------------' |
'----------------' '----------------' '----------------' '----------------' '----------------' '----------------' '----------------'
__ .-.
.-"` .`'. /\\|
_(\-/)_" , . ,\ /\\\/
{(#b^d#)} . ./, |/\\\/
`-.(Y).-` , | , |\.-`
/~/,_/~~~\,__.-`
////~ // ~\\
==`==` ==` ==`
------------------------------------------------
RACCOON